diff --git a/failure-recovery-37143203751-c70.md b/failure-recovery-37143203751-c70.md new file mode 100644 index 0000000000..d87ddcd9cd --- /dev/null +++ b/failure-recovery-37143203751-c70.md @@ -0,0 +1,63 @@ +# Failure recovery — Run Sweep 37143203751 agentic c70 + +## Class + +**recipe** (admission): `override_c70` `max-num-seqs` too high for DCP PyNCCL `kv_gather` under AgentX c70 on B300 DSXE. + +## Evidence + +| Field | Value | +| --- | --- | +| Tip SHA | `c9ffe2b12d30d909b0e3ef97cdb0010ce6cb38d2` | +| Run | [37143203751](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37143203751) attempt 1 RED | +| Job | `111276873706` (agentic c70) | +| Slurm | `7087` on `b300-dsxe_07` / `dsxe-sa-b300-prd0-gpu-15` | +| server_logs artifact | `11285248802` (`server_logs_kimik3_tp8_conc70_...`) | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=48`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `20:01:07` (false positive). +- Real rail: `Mooncake rail: ibp198s0f0` / `Patched device_name='ibp198s0f0'`. +- `Application startup complete`; Mooncake `failed_keys=0` through serve. +- Warmup completed `772/774`; profiling started `21:06:27` and kept `115` requests (not warmup-only die). + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `21:17:09`: + +```text +[Rank 7] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=353066, OpType=_ALLGATHER_BASE, NumelIn=3142656, +NumelOut=25141248, Timeout(ms)=600000) ran for 600016 ms +PG ID 3: last enqueued work: 353081, last started work: -1, +last completed work: 353065 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +_context_parallel_compute_prefill_context → _forward_prefill_fused +→ DistBackendError / terminate / EngineDead / ProfileAborted +``` + +Hang window starts ~`21:07:09` (600s before watchdog). Last healthy engine line then stall: + +- `21:07:12` Running: 10, Waiting: 2, Deferred: 2, GPU KV: **21.3%**, gen 232 tok/s +- `21:07:22` Running: 10, Waiting: 2, Deferred: 2, GPU KV: 21.3%, gen **0.0** tok/s + +No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. + +### Same-run packing (context, not the kill line) + +Earlier in the same profile, admission still packed the pool: max GPU KV **99.6%** at `21:00:12` (Running: 0, Waiting: 48, Deferred: 46); 99 windows with Running≈0 and KV≥85% from `20:38:42`–`21:01:32`. Engine recovered and resumed generating before the ALLGATHER hang. Distinct from tip `1d937c73e` c70 deferred-admission starve under `max-num-seqs=70` (0 kept, `sample_tokens` 1800s). + +Same PyNCCL `kv_gather` family as tip `fd61acb0` c56 / `ebc4eb499` c16. + +## Fix + +Smallest supported knob only: + +- `override_c70.max-num-seqs`: **48 → 32** +- Match `cudagraph_capture_sizes` to max 32 (same shape as `override_c32`) +- Do **not** re-enable `KV_GATHER` / `Q_GATHER` +- Do **not** set `compact_group_io`, `MC_MAX_MR_SIZE`, or `load_async=false` +- Do **not** touch #3632 + +`origin/main` already merged (0 behind); no merge required for this push. diff --git a/failure-recovery-37155636981-c24.md b/failure-recovery-37155636981-c24.md new file mode 100644 index 0000000000..8344683347 --- /dev/null +++ b/failure-recovery-37155636981-c24.md @@ -0,0 +1,66 @@ +# Failure recovery — Run Sweep 37155636981 agentic c24 + +## Class + +**recipe** (admission): `override_c24` `max-num-seqs` too high (2×=48) for DCP PyNCCL `kv_gather` under AgentX c24 on B300 DSXE. + +## Evidence + +| Field | Value | +| --- | --- | +| Tip SHA | `f3997a93aaea2663ead720910a9fbb178d006c64` | +| Run | [37155636981](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37155636981) attempt 1 RED | +| Job | `111311962571` (agentic c24) | +| Slurm | `7107` on `b300-dsxe_07` / `dsxe-sa-b300-prd0-gpu-12` | +| server_logs artifact | `11288576572` (`server_logs_kimik3_tp8_conc24_...`) | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=48`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `23:05:03` (false positive). +- Real rail / Mooncake path healthy through serve; `failed_keys=0` in KV transfer metrics. +- `Application startup complete`; startup args confirm `max_num_seqs: 48`. +- Profiling kept `730/1077` (265 warmup, 288 error dropped) before kill; not warmup-only die. + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `23:56:57`: + +```text +[Rank 1] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=224109, OpType=_ALLGATHER_BASE, NumelIn=4755456, +NumelOut=38043648, Timeout(ms)=600000) ran for 600006 ms +PG ID 3: last enqueued work: 224148, last started work: -1, +last completed work: 224108 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +_context_parallel_compute_prefill_context → _forward_prefill_fused +→ DistBackendError / terminate / worker_crash:8 / EngineDead / +ProfileAborted (82/812 = 10.099%) +``` + +Hang window starts ~`23:46:57` (600s before watchdog). Last healthy then stall: + +- `23:46:50` Running: 22, Waiting: 0, GPU KV: **78.8%**, gen 202 tok/s +- `23:47:00` Running: 17, Waiting: 0, GPU KV: 75.7%, gen 402 tok/s +- `23:47:10` Running: 17, Waiting: 0, GPU KV: 75.7%, gen **0.0** tok/s + +No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. Do not re-enable KV_GATHER or Q_GATHER. + +### Same-run packing (context) + +Earlier in the same profile, peak GPU KV **91.1%** at `23:41:20` (Running: 19, Waiting: 1, Deferred: 1). Aggregate result also reported GPU KV usage **93.3%**. c24 was the last mid-conc still at 2× admission. + +Same PyNCCL `kv_gather` family as tip `c9ffe2b12` c70 / `fd61acb0` c56 / `ebc4eb499` c16. + +## Fix + +Smallest supported knob only: + +- `override_c24.max-num-seqs`: **48 → 24** +- Match `cudagraph_capture_sizes` to max 24 +- Header comment: admission is 2× through CONC 8 only; CONC 16–48 use 1× +- Keep A2A=1; KV_GATHER=0; Q_GATHER=0; no compact_group_io / MC_MAX_MR_SIZE + +## New tip + +(filled after push) diff --git a/failure-recovery-37164230642-c40.md b/failure-recovery-37164230642-c40.md new file mode 100644 index 0000000000..93268ab59a --- /dev/null +++ b/failure-recovery-37164230642-c40.md @@ -0,0 +1,63 @@ +# Failure recovery — Run Sweep 37164230642 agentic c40 + +## Class + +**recipe** (admission): `override_c40` `max-num-seqs` too high (1×=40) for DCP PyNCCL `kv_gather` under AgentX c40 on B300 DSXE. + +## Evidence + +| Field | Value | +| --- | --- | +| Tip SHA | `71080871410ca24104005a5c4df313430726c1b5` | +| Run | [37164230642](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37164230642) attempt 1 RED | +| Job | `111336354457` (agentic c40) | +| Slurm | `7123` on `b300-dsxe_06` / `dsxe-sa-b300-prd0-gpu-09` | +| server_logs artifact | `11292012067` (`server_logs_kimik3_tp8_conc40_...`) | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=40`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `01:41:33` (false positive). +- Real rail / Mooncake path healthy through serve. +- `Application startup complete`; runtime args confirm `--max-num-seqs 40`. +- Profiling kept `838/1376` (444 warmup, 449 error dropped) before kill; not warmup-only die. +- Canary + all agentic evals SUCCESS on this tip; sole agentic fail = c40 (fail-fast cancelled siblings). + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `02:49:41`: + +```text +[Rank 7] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=345118, OpType=_ALLGATHER_BASE, NumelIn=4755456, +NumelOut=38043648, Timeout(ms)=600000) ran for 600000 ms +PG ID 3: last enqueued work: 345144, last started work: -1, +last completed work: 345117 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +DistBackendError / terminate / worker_crash:8 / ProfileAborted +(94/932 = 10.086%) +``` + +Hang window starts ~`02:39:41` (600s before watchdog). Last healthy then stall: + +- `02:39:48` Running: 17, Waiting: 20, Deferred: 20, GPU KV: **71.5%**, gen 200 tok/s +- `02:39:58` Running: 17, Waiting: 20, Deferred: 20, GPU KV: 71.5%, gen **0.0** tok/s + +Earlier packing in the same profile: peak GPU KV **~99.9%** with elevated Waiting/Deferred (e.g. Running: 2 / Waiting: 35 / Deferred: 21 at `02:33:28`; Running: 7 / Waiting: 27 / Deferred: 23 at `02:38:38`). Aggregate result also reported GPU KV usage **100.0%**. + +No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. Do not re-enable KV_GATHER or Q_GATHER. + +Same PyNCCL `kv_gather` family as tip `f3997a93` c24 / `c9ffe2b12` c70 / `fd61acb0` c56 / `ebc4eb499` c16. Tip already at 1× CONC for c40; further admission cut required (same pattern as c56 56→48 and c70 48→32). + +## Fix + +Smallest supported knob only: + +- `override_c40.max-num-seqs`: **40 → 32** +- Match `cudagraph_capture_sizes` to max 32 +- Header comment: note c40 caps at 32 (not 1×) +- Keep A2A=1; KV_GATHER=0; Q_GATHER=0; no compact_group_io / MC_MAX_MR_SIZE + +## New tip + +(filled after push) diff --git a/failure-recovery-37173109032-c48.md b/failure-recovery-37173109032-c48.md new file mode 100644 index 0000000000..9a6f600ada --- /dev/null +++ b/failure-recovery-37173109032-c48.md @@ -0,0 +1,65 @@ +# Failure recovery — Run Sweep 37173109032 agentic c48 + +## Class + +**recipe** (admission): `override_c48` `max-num-seqs` too high (1×=48) for DCP PyNCCL `kv_gather` under AgentX c48 on B300 DSXE. + +## Evidence + +| Field | Value | +| --- | --- | +| Tip SHA | `767f0b3b2eefc9c1058c52c87156fd4be7678882` | +| Run | [37173109032](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37173109032) attempt 1 RED | +| Job | `111362024256` (agentic c48) | +| Slurm | `7147` on `b300-dsxe_04` / `dsxe-sa-b300-prd0-gpu-10` | +| server_logs artifact | `11295365223` | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=48`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `04:35:49` (`first_ts=last_ts`; false positive for the kill). +- Real rail / Mooncake path healthy through serve. +- `Application startup complete`; runtime args confirm `--max-num-seqs 48`. +- Profiling kept `452/1033` (531 warmup, 467 error dropped) before kill; not warmup-only die. +- SIGTERM only at shutdown after DistBackend (`05:45:00`), not the first boundary. + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `05:43:54`: + +```text +[Rank 3] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=334427, OpType=_ALLGATHER_BASE, NumelIn=4755456, +NumelOut=38043648, Timeout(ms)=600000) ran for 600017 ms +PG ID 3: last enqueued work: 334452, last started work: -1, +last completed work: 334426 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +DistBackendError / terminate / worker_crash:8 / EngineDead / +ProfileAborted (50/499 = 10.0%) +``` + +Hang window starts ~`05:33:54` (600s before watchdog). Last live then stall: + +- `05:33:18` Running: 10, Waiting: 39, Deferred: 28, GPU KV: **100.0%** +- `05:33:58` Running: 13, Waiting: 38, Deferred: 35, GPU KV: **98.1%**, gen 145.5 tok/s +- `05:34:08` Running: 13, Waiting: 38, Deferred: 35, GPU KV: 98.1%, gen **0.0** tok/s +- `05:34:56` first mid-profile `shm_broadcast` starvation line + +Earlier packing in the same profile also hit GPU KV **99.8%** / **99.7%**. No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. Do not re-enable KV_GATHER or Q_GATHER. + +Same PyNCCL `kv_gather` family as tip `71080871` c40 / `f3997a93` c24 / `c9ffe2b12` c70 / `fd61acb0` c56 / `ebc4eb499` c16. Tip already at 1× CONC for c48; further admission cut required (same pattern as c40 40→32 and c70 48→32). + +Eval c32 xgrammar ISE on this ledger remains infra/flake (prior dig); left alone. + +## Fix + +Smallest supported knob only: + +- `override_c48.max-num-seqs`: **48 → 32** +- Match `cudagraph_capture_sizes` to max 32 +- Header comment: note c48 caps at 32 (not 1×) +- Keep A2A=1; KV_GATHER=0; Q_GATHER=0; no compact_group_io / MC_MAX_MR_SIZE + +## New tip + +`cefbd21826df0eee931eeaf8664660b642388f7e` — `override_c48` `max-num-seqs` 48→32; gathers remain gated. diff --git a/failure-recovery-37181441604-c70.md b/failure-recovery-37181441604-c70.md new file mode 100644 index 0000000000..a596af7b96 --- /dev/null +++ b/failure-recovery-37181441604-c70.md @@ -0,0 +1,67 @@ +# Failure recovery — Run Sweep 37181441604 agentic c70 + +## Class + +**recipe** (admission): `override_c70` `max-num-seqs` too high (32) for DCP PyNCCL `kv_gather` under AgentX c70 on B300 DSXE. + +## Evidence + +| Field | Value | +| --- | --- | +| Failed tip SHA | `141fbb8567c8dfc14616512fd74bc97806400ada` | +| Run | [37181441604](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37181441604) attempt 1 RED | +| Job | `111386701892` (agentic c70) | +| Slurm | `7166` on `b300-dsxe_09` / `dsxe-sa-b300-prd0-gpu-09` | +| server_logs artifact | `11298073431` (`server_logs_kimik3_tp8_conc70_...`) | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=32`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `07:28:28` (`first_ts=last_ts`; false positive for the kill). +- Real rail: `Mooncake rail: ibp198s0f0` / `Patched mooncake_store_config device_name='ibp198s0f0'`. +- `Application startup complete` at `07:35:50` window; runtime args confirm `max_num_seqs: 32`. +- Mooncake `failed_keys=0` through serve. +- Warmup progressed (`284/774` returned, `67` in flight at hang) then stalled; profiling never started (`warmup_failure`). Not a GHA wrapper-only fail. +- Canary + all agentic evals + collect-evals SUCCESS on this tip; sole agentic fail = c70 (fail-fast cancelled siblings). + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `08:18:49`: + +```text +[Rank 5] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=224761, OpType=_ALLGATHER_BASE, NumelIn=3032064, +NumelOut=24256512, Timeout(ms)=600000) ran for 600011 ms +PG ID 3: last enqueued work: 224769, last started work: -1, +last completed work: 224760 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +_context_parallel_compute_prefill_context → _forward_prefill_fused +→ DistBackendError / terminate / worker_crash:8 / EngineDead / +ProfileAborted (warmup_failure; 0 kept / 683 total) +``` + +Hang window starts ~`08:08:49` (600s before watchdog). Last live then stall: + +- `08:05:41` peak GPU KV **94.6%** (Running: 0, Waiting: 70, Deferred: 29) +- `08:08:51` Running: 1, Waiting: 65, Deferred: 65, GPU KV: **89.9%**, gen 0.2 tok/s +- `08:09:01` Running: 1, Waiting: 65, Deferred: 65, GPU KV: 89.9%, gen **0.0** tok/s +- `08:09:50` first mid-hang `shm_broadcast` starvation line + +No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. SIGTERM only at shutdown after DistBackend (`08:19:54`), not the first boundary. Do not re-enable KV_GATHER or Q_GATHER. + +Same PyNCCL `kv_gather` family as tip `cefbd2182` c48 / `71080871` c40 / `f3997a93` c24 / `c9ffe2b12` c70 / `fd61acb0` c56 / `ebc4eb499` c16. Tip already at max-num-seqs=32 for c70; further admission cut required. + +Eval c32 xgrammar ISE on superseded tip `767f0b3b` / 37173109032 remains deferred; left alone. + +## Fix + +Smallest supported knob only: + +- `override_c70.max-num-seqs`: **32 → 24** +- Match `cudagraph_capture_sizes` to max 24 (same shape as `override_c24`) +- Header comment: note c70 caps at 24 +- Keep A2A=1; KV_GATHER=0; Q_GATHER=0; no compact_group_io / MC_MAX_MR_SIZE + +## New tip + +`35ef5fad78634955c4b7edd771ed2b2ac50a7bb2` — `override_c70` `max-num-seqs` 32→24; gathers remain gated. diff --git a/failure-recovery-37189444890-c56.md b/failure-recovery-37189444890-c56.md new file mode 100644 index 0000000000..b2daec50ac --- /dev/null +++ b/failure-recovery-37189444890-c56.md @@ -0,0 +1,54 @@ +# Failure recovery — Run Sweep 37189444890 agentic c56 + +## Class + +**recipe** (admission): `override_c56` `max-num-seqs` too high (48) for packed GPU KV under AgentX c56 on B300 DSXE. Same packed-KV family as prior c40@40 and c48@48 admissions. + +## Evidence + +| Field | Value | +| --- | --- | +| Failed tip SHA | `bdcf0a0b0beb049af94c950390985b93a235578b` | +| Run | [37189444890](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37189444890) attempt 1 RED | +| Job | `111411443417` (agentic c56) | +| Slurm | `7189` on `b300-dsxe_15` / `dsxe-sa-b300-prd0-gpu-09` | +| server_logs artifact | `11302089911` (`server_logs_kimik3_tp8_conc56_...`) | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=48`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileMetricCoverageError`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `10:04:38` (`first_ts=last_ts`; false positive for the kill). +- Real rail: `Mooncake rail: ibp198s0f0` / `Patched mooncake_store_config device_name='ibp198s0f0'`. +- `Application startup complete`; runtime args confirm `--max-num-seqs 48`. +- Mooncake `failed_keys=0` through serve. +- Warmup completed `619/619` in `2180.67s`; profiling then kept `1296` (619 warmup dropped from coverage accounting). Not a GHA wrapper-only fail. +- Canary + agentic evals on this tip are green; fail-fast RED is this c56 throughput job. + +### First real kill boundary + +**Packed-KV stall mid-profile** (no ALLGATHER watchdog this time). Last live engine log at `11:36:32`: + +```text +Running: 23 reqs, Waiting: 32 reqs, Deferred: 32 reqs, +GPU KV cache usage: 98.2%, Avg generation throughput: 0.0 tokens/s +``` + +Then silence until SIGTERM at shutdown `11:55:07`. Peak GPU KV **100.0%** at `10:40:52` during warmup (Running: 0, Waiting: 54, Deferred: 51). Profiling duration 3600s timed out with `grace_period_timeout=True`; AIPerf then raised `ProfileMetricCoverageError` (TTFT 71.4% / ITL 71.5%; neither signal in the final 180s). + +No multimem timeout, no CUDA OOM, no `sample_tokens` timeout, no DistBackendError. Do not re-enable KV_GATHER or Q_GATHER. + +Same packed-KV family as tip `767f0b3b2` c48 / `71080871` c40 / `cefbd2182` c48 recovery. Tip already at max-num-seqs=48 for c56; further admission cut required. + +## Fix + +Smallest supported knob only: + +- `override_c56.max-num-seqs`: **48 → 32** +- Match `cudagraph_capture_sizes` to max 32 (same shape as `override_c32` / `override_c48`) +- Header comment: note c56 caps at 32 +- Keep A2A=1; KV_GATHER=0; Q_GATHER=0; no compact_group_io / MC_MAX_MR_SIZE +- Do not touch `override_c40` / `override_c48` / `override_c70` + +## New tip + +`39259bb2c629328c2fe10b49e0ba7b7a70d1357c` — `override_c56` `max-num-seqs` 48→32; gathers remain gated. diff --git a/failure-recovery-37201050113-c32.md b/failure-recovery-37201050113-c32.md new file mode 100644 index 0000000000..57f43809d7 --- /dev/null +++ b/failure-recovery-37201050113-c32.md @@ -0,0 +1,66 @@ +# Failure recovery — Run Sweep 37201050113 agentic c32 + +## Class + +**recipe** (admission): `override_c32` `max-num-seqs` too high (32) for DCP PyNCCL `kv_gather` under AgentX c32 on B300 DSXE. Same packed-KV / PyNCCL family as prior mid- and high-conc cells; c32 at 1x was green on superseded tip `bdcf0a0b` and is further admission pressure on this tip. + +## Evidence + +| Field | Value | +| --- | --- | +| Failed tip SHA | `1e365aae596adc51039c54745b219c8eb4c7b6a2` | +| Run | [37201050113](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37201050113) attempt 1 RED | +| Job | `111446959460` (agentic c32) | +| Slurm | `7216` on `b300-dsxe_04` / `dsxe-sa-b300-prd0-gpu-13` | +| server_logs artifact | `11306955758` (`server_logs_kimik3_tp8_conc32_...`) | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=32`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `13:37:57` (`first_ts=last_ts`; false positive for the kill). +- Real rail: `Mooncake rail: ibp198s0f0` / `Patched mooncake_store_config device_name='ibp198s0f0'`. +- `Application startup complete` after `13:45:15`; runtime args confirm `max_num_seqs: 32`. +- Mooncake `failed_keys=0` through serve. +- Profiling progressed (`1162` successful / `1646` total; `354` warmup and `398` error dropped) then aborted at `130/1292 = 10.062%`. Not a GHA wrapper-only fail. +- Canary + agentic evals c1–c70 plus dup c70 + collect-evals/collect-results SUCCESS on this tip; sole agentic fail = c32 (fail-fast cancelled siblings, so c56@32 is unproven). + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `14:47:08`: + +```text +[Rank 2] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=357100, OpType=_ALLGATHER_BASE, NumelIn=4755456, +NumelOut=38043648, Timeout(ms)=600000) ran for 600017 ms +PG ID 3: last enqueued work: 357132, last started work: -1, +last completed work: 357099 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +_context_parallel_compute_prefill_context → _forward_prefill_fused +→ DistBackendError / terminate / worker_crash:8 / EngineDead / +ProfileAborted (130/1292 = 10.062%) +``` + +Hang window starts ~`14:37:08` (600s before watchdog). Last live then stall: + +- `14:13:15` peak GPU KV **100.0%** (Running: 9, Waiting: 23, Deferred: 23) +- `14:37:15` Running: 25, Waiting: 0, GPU KV: **45.4%**, gen 28.4 tok/s +- `14:37:25` Running: 25, Waiting: 0, GPU KV: 45.4%, gen **0.0** tok/s +- `14:38:09` first mid-hang `shm_broadcast` starvation line + +No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. SIGTERM only at shutdown after DistBackend, not the first boundary. Do not re-enable KV_GATHER or Q_GATHER. + +Same PyNCCL `kv_gather` family as tip `141fbb856` c70 / `767f0b3b2` c48 / `71080871` c40 / `f3997a93` c24. Tip already at max-num-seqs=32 for c32; further admission cut required. Do not change `override_c56` or other conc cells. + +## Fix + +Smallest supported knob only: + +- `override_c32.max-num-seqs`: **32 → 24** +- Match `cudagraph_capture_sizes` to max 24 (same shape as `override_c24` / `override_c70`) +- Header comment: note c32 caps at 24 +- Keep A2A=1; KV_GATHER=0; Q_GATHER=0; no compact_group_io / MC_MAX_MR_SIZE +- Do not touch `override_c16` / `override_c24` / `override_c40` / `override_c48` / `override_c56` / `override_c70` + +## New tip + +`1c1a9b29f6944afb8ecb717a373ef69b9c52f5e5` — `override_c32` `max-num-seqs` 32→24; gathers remain gated. diff --git a/inferencex-e2e/benchmarks/benchmark_lib.sh b/inferencex-e2e/benchmarks/benchmark_lib.sh index 98750ea1a0..46664a74b0 100644 --- a/inferencex-e2e/benchmarks/benchmark_lib.sh +++ b/inferencex-e2e/benchmarks/benchmark_lib.sh @@ -135,6 +135,35 @@ run_amd_multinode_after_preflight() { --kill-on-bad-exit=1 --signal=TERM@30 --unbuffered "$@" } +# Setup scripts (for example kimik3-b300-mooncake.sh) source this library with +# --validation-only, so the Mooncake rail helper must stay above that gate. +select_mooncake_rdma_device() { + local sysfs_root="${1:-/sys/class/infiniband}" + local device + MOONCAKE_RAIL="" + for device in "$sysfs_root"/*; do + # DSXE has both EFA and Mellanox adapters. The latter may be renamed + # ibp*, so identify the driver rather than assuming an mlx5_* name. + [[ "$(readlink "$device/device/driver" 2>/dev/null)" == */mlx5_core ]] || continue + grep -qx '4: ACTIVE' "$device/ports/1/state" 2>/dev/null || continue + case "$(cat "$device/ports/1/link_layer" 2>/dev/null)" in + InfiniBand) MC_GID_INDEX=0 ;; + Ethernet) MC_GID_INDEX=3 ;; + *) continue ;; + esac + MOONCAKE_RAIL="${device##*/}" + if [[ -z "$MOONCAKE_RAIL" || "$MOONCAKE_RAIL" == "*" ]]; then + echo "Error: resolved an empty Mooncake RDMA rail name from $device" >&2 + return 1 + fi + export MOONCAKE_RAIL + export MC_GID_INDEX + return 0 + done + echo "Error: no active Mellanox RDMA rail; Mooncake cannot initialise" >&2 + return 1 +} + # Launchers may load only input validation, without benchmark initialization. if [[ "${1-}" == "--validation-only" ]]; then return 0 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh index 243b3d2111..152b7e202e 100755 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh @@ -1,6 +1,12 @@ #!/usr/bin/env bash -# Pin the worker's Mooncake client and point its store at one active RDMA rail. -set -euo pipefail +# Pin the worker's Mooncake client, backport load-failure recovery, and point +# the store at one active Mellanox RDMA rail (driver-selected, including ibp*). +set -eo pipefail + +ws=/infmax-workspace +# Temporary upstream #55297 backport; docs/waiver/3088.md is pending review. +python3 "$ws/runners/patch_kimik3_mooncake_recovery.py" + pip_install=(python3 -m pip install) if python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages; then pip_install+=(--break-system-packages) @@ -10,33 +16,42 @@ fi python3 -c "from mooncake.store import MooncakeDistributedStore" >/dev/null # Rail-isolated nodes: two RNICs cannot reach each other even within a node, so -# every rank uses one rail. mlx5_0 is down on some nodes, and topology discovery -# then finds no HCA, so take the first active rail at runtime. DSXE nodes name -# their rails rdmap*. -rail="" -for device in mlx5_0 mlx5_1 mlx5_2 mlx5_3 mlx5_4 mlx5_5 mlx5_8 mlx5_9 \ - mlx5_10 mlx5_11 mlx5_16 mlx5_17 mlx5_20 mlx5_21 mlx5_22 mlx5_23 \ - $(ls /sys/class/infiniband 2>/dev/null | grep '^rdmap' | sort -V); do - if grep -q ACTIVE "/sys/class/infiniband/$device/ports/1/state" 2>/dev/null; then - rail="$device" - break - fi -done +# every rank uses one Mellanox rail. Identify by driver (mlx5_core) rather than +# assuming mlx5_* names — DSXE may rename them ibp*. EFA rails are skipped. +# shellcheck source=/dev/null +source "$ws/benchmarks/benchmark_lib.sh" --validation-only +select_mooncake_rdma_device +rail="$MOONCAKE_RAIL" if [[ -z "$rail" ]]; then - echo "Error: no active RDMA rail on $(hostname); Mooncake cannot initialise" >&2 - for state in /sys/class/infiniband/*/ports/*/state; do - echo "$state: $(cat "$state" 2>&1)" >&2 - done + echo "Error: select_mooncake_rdma_device returned an empty rail" >&2 + exit 1 +fi +echo "Mooncake rail: $rail (MC_GID_INDEX=$MC_GID_INDEX)" + +# The enroot EFA hook binds the host libibverbs over the image's, and +# libibverbs only loads providers built against its own private ABI, so the +# image's mlx5 provider never loads and the Mellanox rail vanishes. +# runners.yaml mounts the host library directory at /host-usr-lib; libibverbs +# appends its own -rdmavNN suffix to an absolute RDMAV_DRIVERS entry. +if [[ ! -d /host-usr-lib/libibverbs ]]; then + echo "Error: /host-usr-lib/libibverbs missing; host mlx5 provider required" >&2 exit 1 fi +export RDMAV_DRIVERS=/host-usr-lib/libibverbs/libmlx5 + config="${MOONCAKE_CONFIG_PATH:-/logs/mooncake_store_config.json}" python3 - "$config" "$rail" <<'PY' import json, sys path, rail = sys.argv[1:] +if not rail.strip(): + raise SystemExit("Error: refusing to write empty Mooncake device_name") with open(path) as handle: config = json.load(handle) config["device_name"] = rail with open(path, "w") as handle: json.dump(config, handle, indent=2) +written = json.load(open(path)) +if not str(written.get("device_name") or "").strip(): + raise SystemExit(f"Error: mooncake store config still has empty device_name: {written!r}") +print(f"Patched mooncake_store_config device_name={written['device_name']!r}") PY -echo "Mooncake rail: $rail" diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index e8d5e7039b..41dc0b3b20 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -26,9 +26,12 @@ base: interval_seconds: 10 max_attempts: 360 # Embedded Mooncake: each TP rank contributes TOTAL_CPU_DRAM_GB / 8 GB. The - # setup script pins the client to the master's version and fills in the - # node's active RDMA rail. DSXE rails are InfiniBand without a netdev, so the - # transfer engine picks its own GID (a RoCE v2 index 3 does not exist). + # setup script pins the client, backports load-failure recovery, selects one + # active Mellanox rail by driver (ibp* or mlx5_*), sets MC_GID_INDEX from the + # link layer, requires the host mlx5 provider via RDMAV_DRIVERS at + # /host-usr-lib, and refuses an empty device_name. Master lease is 60s. + # device_name starts empty; kimik3-b300-mooncake.sh patches it before engines + # start (srt-slurm's "Wrote mooncake_store_config" line is the pre-patch dump). setup_script: kimik3-b300-mooncake.sh services: - name: mooncake-master @@ -36,7 +39,7 @@ base: preamble: >- python3 -m pip install --break-system-packages --quiet --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.11.post1 - args: ["--eviction_high_watermark_ratio=0.95", "--eviction_ratio=0.10"] + args: ["--eviction_high_watermark_ratio=0.95", "--eviction_ratio=0.10", "--default_kv_lease_ttl=60000"] options: store_config: mode: embedded @@ -65,7 +68,10 @@ base: load-format: fastsafetensors moe-backend: auto no-enable-flashinfer-autotune: true - enable-cumem-allocator: true + # Keep cuMem/VMM off on this DSXE Mooncake path (LMCache-style caution for + # VMM buffers). The register_buffer -600 / AddressNotRegistered storm on + # tip 15820fd2e/1d9c54d was driven by MC_MAX_MR_SIZE=4GiB (removed below), + # not by cumem alone — pre-MC_MAX_MR tips registered with cumem on. enable-prefix-caching: true prefix-match-unit: 128 kv-cache-dtype: fp8 @@ -73,17 +79,47 @@ base: attention-backend: TOKENSPEED_MLA attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' disable-uvicorn-access-log: true - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + # Do NOT enable compact_group_io on DSXE single-rail Mooncake: tip ee33eeba + # c8 (run 36838397438) enabled it and immediately stormed TRANSFER_FAIL on + # compact-group-io-v1 ~25MiB puts (7177 fails), then hung at 0 tok/s until + # the 1800s sample_tokens cap. Keep max_load_batch_keys=1 after tips + # 860c1ccf c48 / e51c58f5 c32 DCP PYNCCL _ALLGATHER_BASE hangs under + # ~98-100% GPU KV. Tip ea88d652 / sweep 36935627690 canary c1 crashed + # immediately with AssertionError "load_async must be True for better + # performance" in mooncake store worker get_finished when load_async was + # false — restore load_async=true (required by this vLLM Mooncake path). + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"max_load_batch_keys":1,"enable_offload":false}}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' VLLM_USE_V2_MODEL_RUNNER: '1' - # These default to auto; name them so the measured DCP a2a path runs. + # Tip 68cdcc58 c24 (Slurm 6814): with A2A=0 the hang was PyNCCL + # ALLTOALL_BASE in dcp_a2a_lse_reduce (dcp.py:765; last started work: + # -1) after ~43m serve. Enable direct A2A (GB300 DCP8 / B200 stock). + # load_async must stay true. VLLM_USE_DIRECT_DCP_A2A: '1' - VLLM_USE_DIRECT_DCP_Q_GATHER: '1' - VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + # Tip 1f837c46 eval-only c8 hung ~8.5m after dcp:0 then failed EP + # PyNccl ncclCommInitRank with Q_GATHER=0, so Q was restored. Tip + # db6caecc c24 (run 37109541419, Slurm 7019) then died in the direct + # Q kernel itself: "direct DCP q-gather multimem timeout source=7 + # epoch=1107073" → CUDA unspecified launch failure in + # KVCacheStoreSendingThread → EngineDead / ProfileAborted (105/1041). + # Mooncake failed_keys=0; GPU KV ~28–42%; not admission starvation. + # Stock gate off leaves the trapping multimem path. Keep A2A on. + VLLM_USE_DIRECT_DCP_Q_GATHER: '0' + # Tip bed9f1ce c40 hung in PyNCCL kv_gather _ALLGATHER_BASE when this + # was 0, so it was restored. Tip f7be12ed c56 (run 37086876838, Slurm + # 6989) with it on died in the direct kernel itself: "direct DCP + # kv-gather multimem timeout source=1 epoch=494017" then asm trap → + # CUDA unspecified launch failure (Mooncake memcpy -800 is the dead + # device, not the first boundary). Stock gate off selects + # all_gather_into_tensor. Keep A2A on; Q gather also gated off above. + VLLM_USE_DIRECT_DCP_KV_GATHER: '0' VLLM_ENGINE_READY_TIMEOUT_S: '3600' VLLM_RPC_TIMEOUT: '600000' + # Mooncake loads can block inside execute_model past the 300s default; + # c40 on run 36800796192 went silent ~279s then EngineDead at the cap. + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' PYTHONNOUSERSITE: '1' @@ -95,6 +131,9 @@ base: MC_STORE_MEMCPY: '1' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' MC_SLICE_SIZE: '1048576' + # Do NOT set MC_MAX_MR_SIZE on DSXE: tip ee33eeba+ (4GiB) made every rank + # register_buffer fail (-600) on the ~40GiB KV region, then AddressNotRegistered + # TRANSFER_FAIL. Pre-MC_MAX_MR tips (c40/c48) registered cleanly without it. MC_WORKERS_PER_CTX: '4' WITH_NVIDIA_PEERMEM: '0' VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' @@ -108,11 +147,76 @@ base: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2249' -# One variant per point. Admission is 2x CONC; CONC 56 and 70 keep more memory -# headroom. Graphs capture (1 + drafts) x 1..min(2x CONC, 128) tokens, then the -# larger powers of two to 8192. Throughput runs switch DSpark to synthetic -# rejection at the golden acceptance length; points above CONC 16 do not draft -# and keep the matrix's mtp label. +# One variant per point. Admission is 2x CONC through CONC 8; CONC 16/24/32 +# use 1x CONC; c40/c48/c56 cap at 32; c70 caps at 24. Tip ebc4eb499 c16 (run 37134266476, +# Slurm 7062) under 2x=32 packed GPU KV to ~98–99.7% during AgentX warmup +# (Running: 3, Waiting: 7, Deferred: 7 at 17:44:47) then hung PyNCCL +# _ALLGATHER_BASE 600s (SeqNum=89905, last started work: -1, PG ID 3) → +# DistBackendError / VllmWorker-7 died / EngineDead / ProfileAborted (176 +# warmup dropped / 0 kept). Cap c16 at 1x=16. Tip f3997a93 c24 (run +# 37155636981, Slurm 7107) under 2x=48 hung mid-profile in PyNCCL kv_gather +# _ALLGATHER_BASE (dcp.py:1413, SeqNum=224109, last started work: -1) after +# peak GPU KV 91.1% earlier in the same profile (Running: 17–25 near hang); +# cap c24 at 1x=24. Tip 71080871 c40 (run 37164230642, Slurm 7123) under +# 1x=40 packed GPU KV to ~99.9% (Running: 2–30, Waiting: 16–35, Deferred: +# 11–34) then hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE +# (dcp.py:1413, SeqNum=345118, last started work: -1) → DistBackendError / +# worker_crash:8 / ProfileAborted (94/932 = 10.086%); cap c40 at 32. Tip +# 767f0b3b2 c48 (run 37173109032, Slurm 7147) under 1x=48 packed GPU KV to +# 100.0% then hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE +# (dcp.py:1413, SeqNum=334427, last started work: -1) → DistBackendError / +# worker_crash:8 / ProfileAborted (50/499 = 10.0%); cap c48 at 32. Tip +# 031de17bf c56 (Slurm 6848) with A2A/Q/KV all direct, +# util 0.85, and max-num-seqs=112 packed GPU KV to ~99.7% (Running: 1, +# Waiting: 12, Deferred: 12) then hung workers for the full +# VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 — shm_broadcast starvation from +# 15:28, then TimeoutError: RPC call to sample_tokens timed out (no Watchdog / +# ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only +# ibv_query_port_speed WARN). Tip 861a1512 c32 (Slurm 6916) repeated the same +# kill under max-num-seqs=64 (2x): GPU KV pinned ~99–100% for tens of minutes, +# then hung at 0 tok/s (Running: 23, Waiting: 11, Deferred: 11) with +# shm_broadcast starvation from 21:03 through the 1800s sample_tokens timeout +# → EngineDead / ProfileAborted (20 empty streams + 62 ClientConnectorError +# after shutdown; ingest nccl_error:16 again init-only). Cap CONC 32+ admission +# so prefix+deferred loads cannot pin the pool before decode. Tip fd61acb0 c56 +# (run 37118728189, Slurm 7040) still packed GPU KV to 98.2% at 1x=56 +# (Running: 8, Waiting: 53, Deferred: 38) then hung PyNCCL kv_gather +# _ALLGATHER_BASE (dcp.py:1413, last started work: -1) for 600s. Cap c56 at +# 48. Tip bdcf0a0b c56 (run 37189444890, Slurm 7189, job 111411443417) +# under max-num-seqs=48 packed GPU KV to 100.0% (Running: 0, Waiting: 54, +# Deferred: 51 at 10:40:52 in warmup) then stalled mid-profile at 11:36:32 +# (Running: 23, Waiting: 32, Deferred: 32, GPU KV 98.2%, gen 0 tok/s) with +# no further engine metrics until SIGTERM 11:55:07 → +# ProfileMetricCoverageError (TTFT 71.4% / ITL 71.5%; 619 warmup dropped / +# 1296 kept). nccl_error:16 is init-only ibv_query_port_speed WARN. Cap +# c56 at 32. Tip c9ffe2b12 c70 (run 37143203751, Slurm 7087) under max-num-seqs=48 +# still hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE (SeqNum=353066, +# last started work: -1) after packing GPU KV to 99.6% earlier in the same +# profile; cap c70 at 32. Tip 141fbb856 c70 (run 37181441604, Slurm 7166) +# under max-num-seqs=32 still hung during AgentX warmup in PyNCCL kv_gather +# _ALLGATHER_BASE (dcp.py:1413, SeqNum=224761, last started work: -1) after +# packing GPU KV to 94.6% (Running: 0, Waiting: 70, Deferred: 29) → +# DistBackendError / worker_crash:8 / ProfileAborted (profiling never +# started); cap c70 at 24. Tip 1e365aae c32 (run 37201050113, Slurm 7216, +# job 111446959460) under 1x=32 packed GPU KV to 100.0% (Running: 9, +# Waiting: 23, Deferred: 23 at 14:13:15) then hung mid-profile in PyNCCL +# kv_gather _ALLGATHER_BASE (dcp.py:1413, SeqNum=357100, last started +# work: -1) at 14:47:08 after last live 14:37:25 (Running: 25, Waiting: 0, +# GPU KV 45.4%, gen 0 tok/s) → DistBackendError / worker_crash:8 / +# ProfileAborted (130/1292 = 10.062%); cap c32 at 24. CONC 8+ use +# gpu-memory-utilization 0.85: with +# VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 (Slurm 6737) +# over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once CUDA graphs +# are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB needed, +# ~2.3 GiB free) ~4m after Application startup. Tip 5ab41690 c8 (Slurm 6877) +# then soft-OOMed at util 0.92 during warmup (CUDACachingAllocator ~3.03 GiB +# alloc with ~1.16 GiB free on ranks 1–7) → empty streamed responses → +# InvalidInferenceResultError / ProfileAborted at 2/11 > 10% (engine stayed +# up; ingest nccl_error:16 again init-only WARN). Keep 0.92 only on c1–c4. +# Graphs capture every size up to max-num-seqs, then the larger powers of two +# to 8192. Throughput runs switch DSpark to synthetic rejection at the golden +# acceptance length; points above CONC 16 do not draft and keep the matrix's +# mtp label. override_c1: roles: agg: @@ -154,7 +258,7 @@ override_c8: agg: args: max-num-seqs: 16 - gpu-memory-utilization: 0.92 + gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,32,40,48,56,64,72,80,88,96,104,112,120,128,256,512,1024,2048,4096,8192]}' speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' benchmark: @@ -165,9 +269,11 @@ override_c16: roles: agg: args: - max-num-seqs: 32 - gpu-memory-utilization: 0.92 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,256,512,1024,2048,4096,8192]}' + # Tip ebc4eb499 c16 (Slurm 7062) under 2x=32 packed GPU KV ~99% in + # warmup then hung PyNCCL _ALLGATHER_BASE 600s (last started work: -1). + max-num-seqs: 16 + gpu-memory-utilization: 0.85 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,128,256,512,1024,2048,4096,8192]}' speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":3,"attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' benchmark: env: @@ -177,9 +283,12 @@ override_c24: roles: agg: args: - max-num-seqs: 48 - gpu-memory-utilization: 0.92 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,64,128,256,512,1024,2048,4096,8192]}' + # Tip f3997a93 c24 (Slurm 7107) under 2x=48 hung mid-profile in + # PyNCCL kv_gather _ALLGATHER_BASE 600s (dcp.py:1413, SeqNum=224109, + # last started work: -1) after peak GPU KV 91.1% earlier. + max-num-seqs: 24 + gpu-memory-utilization: 0.85 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '24' @@ -189,9 +298,13 @@ override_c32: roles: agg: args: - max-num-seqs: 64 - gpu-memory-utilization: 0.92 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,128,256,512,1024,2048,4096,8192]}' + # Tip 1e365aae c32 (run 37201050113, Slurm 7216, job 111446959460) + # under 1x=32 packed GPU KV to 100.0% then hung mid-profile in + # PyNCCL kv_gather _ALLGATHER_BASE 600s (dcp.py:1413, SeqNum=357100, + # last started work: -1). Cap at 24. + max-num-seqs: 24 + gpu-memory-utilization: 0.85 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '32' @@ -201,9 +314,12 @@ override_c40: roles: agg: args: - max-num-seqs: 80 - gpu-memory-utilization: 0.92 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,128,256,512,1024,2048,4096,8192]}' + # Tip 71080871 c40 (Slurm 7123) under 1x=40 packed GPU KV to ~99.9% + # then hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE 600s + # (dcp.py:1413, SeqNum=345118, last started work: -1). Cap at 32. + max-num-seqs: 32 + gpu-memory-utilization: 0.85 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '40' @@ -213,9 +329,12 @@ override_c48: roles: agg: args: - max-num-seqs: 96 - gpu-memory-utilization: 0.92 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,128,256,512,1024,2048,4096,8192]}' + # Tip 767f0b3b2 c48 (Slurm 7147) under 1x=48 packed GPU KV to 100.0% + # then hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE 600s + # (dcp.py:1413, SeqNum=334427, last started work: -1). Cap at 32. + max-num-seqs: 32 + gpu-memory-utilization: 0.85 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '48' @@ -225,9 +344,17 @@ override_c56: roles: agg: args: - max-num-seqs: 112 - gpu-memory-utilization: 0.9 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,128,256,512,1024,2048,4096,8192]}' + # Tip fd61acb0 c56 (Slurm 7040) under 1x (56) packed GPU KV to 98.2% + # (Running: 8, Waiting: 53, Deferred: 38) then hung kv_gather + # _ALLGATHER_BASE 600s (last started work: -1). Cap at 48. Tip + # bdcf0a0b c56 (run 37189444890, Slurm 7189, job 111411443417) under + # 48 packed GPU KV to 100.0% then stalled mid-profile (11:36:32 + # Running: 23, Waiting: 32, Deferred: 32, GPU KV 98.2%, gen 0 tok/s) + # → ProfileMetricCoverageError (TTFT 71.4%; 619 dropped / 1296 kept). + # Cap at 32. + max-num-seqs: 32 + gpu-memory-utilization: 0.85 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '56' @@ -237,9 +364,15 @@ override_c70: roles: agg: args: - max-num-seqs: 140 - gpu-memory-utilization: 0.9 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128,256,512,1024,2048,4096,8192]}' + # Tip 1d937c73e c70 under 1x (70) deferred-admission starved + # (Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% → sample_tokens + # 1800s). Cap to 48 then 32 still hung: tip 141fbb856 c70 (run + # 37181441604, Slurm 7166) packed GPU KV to 94.6% during warmup then + # hung PyNCCL kv_gather _ALLGATHER_BASE 600s (dcp.py:1413, + # SeqNum=224761, last started work: -1). Cap at 24. + max-num-seqs: 24 + gpu-memory-utilization: 0.85 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '70' diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index d0f2de0803..4f28423e55 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -668,6 +668,11 @@ clusters: single-node-models: staged container-aliases: [dynamo-trtllm, dynamo-sglang, dynamo-vllm] nginx-aliases: [nginx-sqsh] + # Host libibverbs ABI for Mooncake RDMA: enroot EFA hook overlays the + # image provider, so mount the host library tree and load mlx5 via + # RDMAV_DRIVERS in kimik3-b300-mooncake.sh. + mounts: + /usr/lib/x86_64-linux-gnu: /host-usr-lib gb200-nv: gpus-per-node: 4 available-cpu-dram-mib: 860_160 diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index f594426d43..09f6d8d6c5 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -267,7 +267,7 @@ All ten AgentX throughput points use DSpark K6 (target verification length 7) and the committed golden AL 3.77. C1/2/4/8/16 use TP8/EP1; C48/64/96/128/256 use TP8/DPA8/EP8 with native RCCL. Each point runs for 3600 seconds. The C256 full GSM8K eval omits forced acceptance. Keep the -pinned `rocm/atom-dev:nightly_202609291501` image and GPU-only KV. C1 through C16 use +pinned `rocm/atom-dev:nightly_202609161445` image and GPU-only KV. C1 through C16 use BF16 KV, while C48 and above retain FP8 KV; all points use the FP4 index cache, 8192-token checkpoints and DEP dense FULL graph ladder. Fixed q7 graphs are captured in each new server; confirm target and DSpark draft capture in @@ -280,12 +280,8 @@ record model/source identity and requested settings. Successful startup, graph capture and requests require runtime log evidence. The pinned image is the official ATOM nightly -`rocm/atom-dev:nightly_202609291501` (ATOM `0.1.7.dev46+g74fd942b0`, ROCm 7.2.4), -which includes the merged +`rocm/atom-dev:nightly_202609161445`, which includes the merged [ROCm/ATOM#2233](https://github.com/ROCm/ATOM/pull/2233) inference-mode fix. -From this image ATOM stores the checkpoint's `ue8m0` FP8 block scales as E8M0 -on gfx950 by default ([ROCm/ATOM#2419](https://github.com/ROCm/ATOM/pull/2419)); -the powers-of-two scales are represented exactly. The recipe does not patch AITER source at runtime; TP communication fusion, DSpark K6 and graph capture use the implementation shipped in the image. @@ -333,6 +329,68 @@ The H200 DSpark recipe uses the same minimum capture size and preserves the same The B300 DSpark recipe sets explicit capture sizes per point, described below. +Kimi-K3 on B300 selects one active Mellanox adapter by its sysfs driver, including +DSXE `ibp*` names; EFA devices are excluded from this RDMA recipe. The embedded +Mooncake ranks share that adapter. InfiniBand uses GID index 0 and RoCE retains +index 3. If no compatible active adapter exists, or if the host mlx5 provider mount +is missing, startup fails before serving. The recipe YAML may leave `device_name` +empty; `kimik3-b300-mooncake.sh` patches a real rail into the store config and +refuses to continue with an empty name (srt-slurm's earlier "Wrote +mooncake_store_config" line is the pre-patch dump). On DSXE the container's +libibverbs comes from the host through the enroot EFA hook, so +`configs/runners.yaml` mounts the host library directory at `/host-usr-lib` and +the setup script requires `RDMAV_DRIVERS` to load its mlx5 provider. Keep +`max_load_batch_keys: 1` and `load_async: true` (tips 860c1ccf c48 / +e51c58f5 c32 hung in DCP PYNCCL `_ALLGATHER_BASE` under ~98–100% GPU KV with +async loads and clean Mooncake metrics; `last started work: -1`. Tip ea88d652 +canary c1 crashed with Mooncake `AssertionError: load_async must be True for +better performance` when `load_async` was set false, so restore the required +stock true), but do not enable `compact_group_io` on this DSXE single-rail path +(it storm-failed ~25 MiB compact-group puts at c8). Keep +`VLLM_USE_DIRECT_DCP_A2A=1`, and set both `VLLM_USE_DIRECT_DCP_Q_GATHER=0` and +`VLLM_USE_DIRECT_DCP_KV_GATHER=0` (tip 68cdcc58 c24 hung in PyNCCL +`ALLTOALL_BASE` inside `dcp_a2a_lse_reduce` with A2A off. Tip bed9f1ce c40 +hung in PyNCCL `kv_gather` `_ALLGATHER_BASE` when KV gather was off, which is +why that gate was restored; tip f7be12ed c56, run 37086876838, then died on +the direct KV kernel itself — `direct DCP kv-gather multimem timeout source=1 +epoch=494017` followed by an `asm trap` that is the CUDA unspecified launch +failure — so KV gather is gated off again. Tip db6caecc c24, run 37109541419, +then died on the direct Q kernel — `direct DCP q-gather multimem timeout +source=7 epoch=1107073` → CUDA unspecified launch failure in +`KVCacheStoreSendingThread` → EngineDead / ProfileAborted — so Q gather is +gated off too. Tip 1f837c46 eval-only c8 previously hung ~8.5m after `dcp:0` +then failed EP `ncclCommInitRank` with Q gather off; watch eval init on this +gate. Mooncake GPU memcpy `-800` was the poisoned device, not the first +boundary.). +Do not set `MC_MAX_MR_SIZE` here: with 4GiB every rank hit +`register_buffer failed ... -600` on the ~40 GiB KV region and stormed +`AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips registered +cleanly. Keep `enable-cumem-allocator` off on this path. Keep +`gpu-memory-utilization` at 0.85 for CONC 8+ (tip b70e4260a c48 reached +Application startup with KV 45.2 GiB at 0.92 under +`VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0`, then OOMed in flashinfer FP4 MoE +`prepare_moe` allocating ~2.89 GiB with ~2.3 GiB free; vLLM suggested ~36.78 GiB +KV once CUDA graphs are counted. Tip 5ab41690 c8 then soft-OOMed at util 0.92 +during warmup — CUDACachingAllocator failed a ~3.03 GiB alloc with ~1.16 GiB +free — yielding empty streams and ProfileAborted at 2/11 > 10%; keep 0.92 only +on c1–c4). Cap `max-num-seqs` at 1×CONC for CONC 16 and CONC 32–48, and at 48 +for CONC 56 and CONC 70 (tip 031de17bf c56 with A2A/Q/KV all direct and util 0.85 packed GPU KV to +~99.7% under 2× admission, then hung workers through the 1800s +`sample_tokens` RPC timeout with no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; +ingest `nccl_error:16` was init-only `ibv_query_port_speed` WARN. Tip 861a1512 +c32 repeated the same kill under `max-num-seqs=64`: GPU KV pinned ~99–100%, +then hung at 0 tok/s with shm_broadcast starvation from 21:03 through the +1800s `sample_tokens` timeout → EngineDead / ProfileAborted; keep 2× on CONC +8 and CONC 24. Tip ebc4eb499 c16 under 2×=`32` packed GPU KV to ~99% during +warmup (Running: 3, Waiting: 7, Deferred: 7) then hung PyNCCL +`_ALLGATHER_BASE` 600s (`last started work: -1`) → DistBackendError / +EngineDead / ProfileAborted with 176 warmup dropped / 0 kept; drop c16 to +1×=`16`. Tip 1d937c73e c70 still starved under 1×=`70`: Running≈0 / +Waiting≈66 / Deferred≈50–67 / KV≈86–91% for ~30m then the same 1800s +`sample_tokens` timeout; drop c70 to `max-num-seqs=48`. Tip fd61acb0 c56 +under 1×=`56` packed GPU KV to 98.2% then hung the same `_ALLGATHER_BASE` +watchdog; drop c56 to `max-num-seqs=48`). + The B200 entry uses `vllm/vllm-openai:nightly-dev-x86_64-cu130-ac9126e58aa7` with FlashInfer autotuning. TP4 covers concurrency 1–128. DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) covers 8–32 and DEP4 (TP1 x DP4 + EP4, MegaMoE) covers 64–128, both behind a consistent-hash vLLM Router. DEP2 keeps about diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 4ce5ab04f7..c6bbb92729 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -247,7 +247,7 @@ DSpark Markov/confidence head、全部 66 个分片的 header 与 payload 边界 全部十个 AgentX 性能点使用 DSpark K6(target 验证长度为 7)和已提交的 golden AL 3.77。C1/2/4/8/16 使用 TP8/EP1;C48/64/96/128/256 使用 TP8/DPA8/EP8 原生 RCCL。每个性能点运行 3600 秒。C256 全量 GSM8K 不传强制 -接受率参数。保留固定的 `rocm/atom-dev:nightly_202609291501` 镜像和 GPU KV;C1 至 C16 +接受率参数。保留固定的 `rocm/atom-dev:nightly_202609161445` 镜像和 GPU KV;C1 至 C16 使用 BF16 KV,C48 及以上继续使用 FP8 KV,所有任务均使用 FP4 index cache、 8192-token checkpoint 和 DEP dense FULL graph 阶梯。每个新服务进程重新捕获固定 q7 图;必须从 `server.log` 确认 target 和 DSpark draft capture 完成。confidence @@ -258,11 +258,8 @@ schedule 和 ragged verification 保持关闭。 `runtime_manifest.json` 和 `server_command.txt` 保存模型/源码身份及请求的配置。 成功启动、graph capture 和请求执行仍需运行时日志证明。 -固定镜像为官方 ATOM nightly `rocm/atom-dev:nightly_202609291501`(ATOM `0.1.7.dev46+g74fd942b0`, -ROCm 7.2.4),已包含已合入的 +固定镜像为官方 ATOM nightly `rocm/atom-dev:nightly_202609161445`,已包含已合入的 [ROCm/ATOM#2233](https://github.com/ROCm/ATOM/pull/2233) inference-mode 修复。 -自该镜像起,ATOM 在 gfx950 上默认以 E8M0 存储检查点的 `ue8m0` FP8 block scale -([ROCm/ATOM#2419](https://github.com/ROCm/ATOM/pull/2419)),2 的幂次 scale 可被精确表示。 配方不再在运行时修改 AITER 源码;TP 通信融合、DSpark K6 和 graph capture 直接使用镜像内实现。 @@ -313,6 +310,64 @@ H200 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工 B300 的 DSpark 配方按测试点显式设置捕获尺寸,详见下文。 +B300 上的 Kimi-K3 按 sysfs 驱动选择一块活动的 Mellanox 网卡,包括 DSXE 的 +`ibp*` 命名;本 RDMA 配方排除 EFA。嵌入式 Mooncake 各 rank 共用该网卡。 +InfiniBand 使用 GID 索引 0,RoCE 保留索引 3。若没有可用的活动适配器,或缺少 +主机 mlx5 provider 挂载,则在服务启动前失败。配方 YAML 可将 `device_name` 留空; +`kimik3-b300-mooncake.sh` 会在引擎启动前把真实 rail 写入 store config,并拒绝 +空名称(srt-slurm 较早的 `Wrote mooncake_store_config` 行是 patch 前的转储)。 +DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库,因此 +`configs/runners.yaml` 将主机库目录挂到 `/host-usr-lib`,setup 脚本通过 +`RDMAV_DRIVERS` 强制加载其 mlx5 provider。保留 `max_load_batch_keys: 1` 且 +`load_async: true`(tip 860c1ccf 的 c48 / tip e51c58f5 的 c32 在异步 load、 +Mooncake 指标干净时,于约 98–100% GPU KV 下挂起在 DCP PYNCCL +`_ALLGATHER_BASE`,`last started work: -1`;tip ea88d652 的 canary c1 在 +`load_async: false` 时于 Mooncake `get_finished` 直接 +`AssertionError: load_async must be True for better performance`,故恢复必需的 +stock true),但不要在该 DSXE 单 rail 路径上启用 +`compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 +保持 `VLLM_USE_DIRECT_DCP_A2A=1`,并将 `VLLM_USE_DIRECT_DCP_Q_GATHER` 与 +`VLLM_USE_DIRECT_DCP_KV_GATHER` 均设为 0(tip 68cdcc58 的 c24 在关闭 A2A 时于 +PyNCCL `ALLTOALL_BASE` / `dcp_a2a_lse_reduce` 挂起。tip bed9f1ce 的 c40 在关闭 +KV gather 时于 PyNCCL `kv_gather` `_ALLGATHER_BASE` 挂起,因此该开关曾被重新 +打开;tip f7be12ed 的 c56(运行 37086876838)随后死在 direct KV kernel +本身——`direct DCP kv-gather multimem timeout source=1 epoch=494017`,接着 +`asm trap`,即 CUDA unspecified launch failure——故 KV gather 再次关闭。tip +db6caecc 的 c24(运行 37109541419)随后死在 direct Q kernel——`direct DCP +q-gather multimem timeout source=7 epoch=1107073` → +`KVCacheStoreSendingThread` 中 CUDA unspecified launch failure → EngineDead / +ProfileAborted——故 Q gather 一并关闭。tip 1f837c46 的 eval-only c8 曾在关闭 Q +gather 后于 `dcp:0` 之后挂起约 8.5 分钟并导致 EP `ncclCommInitRank` 失败;此门 +禁下需关注 eval 初始化。Mooncake GPU memcpy `-800` 是设备已损坏后的后果,不是 +第一故障边界。)。 +不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 +`register_buffer failed ... -600`,并引发 `AddressNotRegistered` +TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。此路径保持关闭 +`enable-cumem-allocator`。CONC 8+ 将 `gpu-memory-utilization` 保持为 0.85 +(tip b70e4260a 的 c48 在 `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0` 与 0.92 +下以 45.2 GiB KV 完成 Application startup,随后在 flashinfer FP4 MoE +`prepare_moe` 申请约 2.89 GiB 时仅剩约 2.3 GiB 空闲而 OOM;vLLM 在计入 CUDA +graph 后建议约 36.78 GiB KV。tip 5ab41690 的 c8 在 util 0.92 的 warmup 中软 +OOM——CUDACachingAllocator 申请约 3.03 GiB 时仅剩约 1.16 GiB——导致空流与 +ProfileAborted(2/11 > 10%);仅 c1–c4 保留 0.92)。CONC 16 与 CONC 32–48 将 +`max-num-seqs` 限制为 1×CONC,CONC 56 与 CONC 70 限制为 48(tip 031de17bf 的 c56 在 +A2A/Q/KV 均已 direct 且 util 0.85 时,2× 准入把 GPU KV 堆到约 99.7%,随后 +worker 挂满 1800 秒 `sample_tokens` RPC 超时,无 Watchdog / ALLGATHER / +ALLTOALL / CUDA OOM;ingest 的 `nccl_error:16` 仅为初始化期 +`ibv_query_port_speed` WARN。tip 861a1512 的 c32 在 `max-num-seqs=64` 下重现 +同一杀伤:GPU KV 钉在约 99–100%,随后以 0 tok/s 挂起,自 21:03 起 +shm_broadcast 饥饿直至满 1800 秒 `sample_tokens` 超时 → EngineDead / +ProfileAborted;仅 CONC 8 与 CONC 24 保留 2×。tip ebc4eb499 的 c16 在 2×=`32` +下 warmup 把 GPU KV 堆到约 99%(Running: 3,Waiting: 7,Deferred: 7),随后 +PyNCCL `_ALLGATHER_BASE` 挂起 600 秒(`last started work: -1`)→ +DistBackendError / EngineDead / ProfileAborted(176 warmup 丢弃 / 0 保留); +将 c16 降为 1×=`16`。tip 1d937c73e 的 c70 在 1×=`70` +下仍饥饿:Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% 约 30 分钟后 +同样满 1800 秒 `sample_tokens` 超时;将 c70 降为 `max-num-seqs=48`。tip +fd61acb0 的 c56 在 1×=`56` 下把 GPU KV 堆到 98.2% 后同样挂起 `_ALLGATHER_BASE` +watchdog;将 c56 降为 `max-num-seqs=48`)。 + + 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 [`nvidia-master.yaml`](../configs/nvidia-master.yaml) 中按 SKU 固定的 `image`(最初为 `vllm/vllm-openai:deepseekv41-flash-0909`),在 Blackwell SKU 上采用 TP4、原生五 token DSpark、 概率采样草稿。吞吐测试使用[已提交的黄金 AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样和自适应验证。 diff --git a/inferencex-e2e/docs/waiver/3088.md b/inferencex-e2e/docs/waiver/3088.md new file mode 100644 index 0000000000..5f694f0e91 --- /dev/null +++ b/inferencex-e2e/docs/waiver/3088.md @@ -0,0 +1,78 @@ +# Inference-engine patch waiver — PR #3088 + +**English** | [中文](./3088_zh.md) + +**Status: proposed; not approved.** The maintainer's sign-off must explicitly link +this waiver under the engine-patch item of the [review checklist](../PR_REVIEW_CHECKLIST.md). + +## Scope and provenance + +- Config: `kimik3-fp4-b300-vllm-agentic-dspark` in `configs/nvidia-master.yaml`. +- Image: `vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77`, vLLM source + `3696c772aae308f2420a8f307b0971e6986c4818`. +- Mooncake remains `mooncake-transfer-engine-cuda13==0.3.11.post1`. +- Entrypoint: `benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh` + (the `setup_script` of `benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml`) + invokes `runners/patch_kimik3_mooncake_recovery.py` before pinning Mooncake and + selecting the RDMA rail. Other hardware recipes do not invoke it. +- Backport: [vLLM #55297](https://github.com/vllm-project/vllm/pull/55297), pinned + commit `f3a831c2015d9eb6f7e600dbd2ef565166d64437`; the upstream PR is still open. + +## Why the shipped image needs this change + +[PR run 34818543261](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34818543261) +failed its C8 benchmark after repeated Mooncake `LEASE_EXPIRED` reads. The server +logged 688 core KV-load recoveries; one logged rank/key failed 126 times. At +08:36:37 UTC the reported get duration averaged 20.551 seconds, with 587/587 keys +failed, while the master lease was 5 seconds. The benchmark failed its unchanged +95% metric-coverage gate; valid partial power does not qualify that performance point. + +In this image, async first-block failure can reset a request to token zero, then +repeat the same external lookup on the next scheduling attempt. Upstream #55297 +addresses exactly that unbounded recovery path. Its author reports an integration +negative control and successful local recomputation after the fix. That upstream +evidence is not current-head B300 qualification. Existing logs establish repeated +key failures but omit request IDs and per-rank completion counters, so they do not +prove this is the only cause of the C8 stall. The independent C8 eval Slurm launch +socket timeout is outside this waiver. + +## Exact patch and limits + +The backport adds the upstream `on_load_failure` callback to the connector base, +MooncakeStore connector/scheduler, MultiConnector, and core scheduler. Only async +failures reset to zero request a bypass. That request skips external lookups until +local allocation succeeds; partial-prefix and synchronous recovery keep their +existing behavior. Global cache metadata and other requests remain available. + +Only two insertion contexts were adapted for the older source layout; the upstream +recovery additions are unchanged. The helper verifies all five pristine or patched +source hashes before any writes. Unknown and mixed states fail the launch. + +This does not change the pinned image, Mooncake version, transport, external KV +tier, workload, speculative-decoding settings, 11-point concurrency curve, or 95% +gate. The recipe separately raises the master lease TTL from 5 s to 60 s +(`--default_kv_lease_ttl=60000`); the patch itself does not touch leases. Recovery +changes serving behavior and must be visible in results provenance; patched results +must not be represented as an unmodified upstream image. + +## Validation and removal + +Eight local patch-application and fail-closed checks pass. In the cached pinned +image, [runtime run 34832901970](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34832901970) +reproduced the original scheduler regression (two passes, one expected failure). +The candidate passed 51 core/error/Store/Multi tests; one Store test failed because +its block fixture lacked the pinned API. After correcting only that fixture, +[its single-case run](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34833354601) +passed, giving 52 distinct passing candidate cases across the two runs. Both runs +verified the same five patched source hashes; no production patch changed between them. + +These are runtime CPU regression checks, without model loading. The affected full +PR sweep/evals and maintainer waiver approval remain required before sign-off; +the original three successful performance points do not qualify the changed runtime. + +After #55297 or an equivalent reviewed fix lands and ships in an official vLLM +image, update this config to that pinned image, remove the helper and invocation, +and retire this waiver and its translation in the same PR. Requalify the complete +existing B300 curve and applicable evals; do not replace the curve with a diagnostic +subset. If upstream changes the proposed behavior before merge, review and update +this exact backport before adopting those changes. diff --git a/inferencex-e2e/docs/waiver/3088_zh.md b/inferencex-e2e/docs/waiver/3088_zh.md new file mode 100644 index 0000000000..b7cd317a61 --- /dev/null +++ b/inferencex-e2e/docs/waiver/3088_zh.md @@ -0,0 +1,58 @@ +# 推理引擎补丁豁免 — PR #3088 + +[English](./3088.md) | **中文** + +**状态:待审,尚未批准。** 维护者签核时必须在[评审清单](../PR_REVIEW_CHECKLIST.md) +的引擎补丁项明确链接这份豁免。 + +## 范围与来源 + +- 配置:`configs/nvidia-master.yaml` 中的 `kimik3-fp4-b300-vllm-agentic-dspark`。 +- 镜像保持 `vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77`,vLLM 源码为 + `3696c772aae308f2420a8f307b0971e6986c4818`。 +- Mooncake 保持 `mooncake-transfer-engine-cuda13==0.3.11.post1`。 +- 入口:`benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh` + (`b300-fp4-mtp/agentic.yaml` 的 `setup_script`)在启动服务前调用 + `runners/patch_kimik3_mooncake_recovery.py`;其他硬件配方不调用。 +- 回移 [vLLM #55297](https://github.com/vllm-project/vllm/pull/55297),固定提交为 + `f3a831c2015d9eb6f7e600dbd2ef565166d64437`;该上游 PR 目前尚未合并。 + +## 原镜像的问题 + +[PR run 34818543261](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34818543261) +的 C8 benchmark 在反复出现 Mooncake `LEASE_EXPIRED` 后失败。日志有688次核心调度恢复, +一个已记录的 rank/key 失败126次。08:36:37 UTC 的 get 平均耗时20.551秒,587/587个 key +失败,而 master 租约只有5秒。性能未通过保持不变的95%信号覆盖门槛;部分功耗有效不能使该性能点合格。 + +此镜像在异步首块加载失败后可能将请求退回 token 0,再次命中同一外部缓存,形成无界恢复循环。 +上游 #55297 专门修复该路径,作者报告了集成负对照及修复后成功本地重算。这不等于本PR当前头 +已经通过 B300 验证。旧日志缺 request ID 和各 rank 完成计数,尚不能证明循环是 C8 停滞的唯一原因。 +C8 eval 的 Slurm 启动 socket timeout 是独立问题,不在本豁免范围内。 + +## 补丁与限制 + +回移上游 `on_load_failure` 回调,涉及 connector 基类、MooncakeStore connector/scheduler、 +MultiConnector 和核心 scheduler。只有异步失败且退回0的请求触发绕过;该请求暂时跳过外部查询, +直到成功分配本地重算资源。部分前缀与同步恢复保持原行为,全局缓存元数据和其他请求仍可使用。 + +仅为旧版本调整了两处方法插入上下文,上游恢复逻辑新增内容不变。写入前检查全部5个文件的原始或 +完整补丁 SHA;未知版本、部分已修改状态均停止启动。 + +镜像、Mooncake 版本、传输、外部 KV 层、工作负载、推测解码配置、11点并发曲线和95%门槛 +均不改变。配方另将 master 租约 TTL 从 5 秒提高到 60 秒(`--default_kv_lease_ttl=60000`),补丁本身不涉及租约。恢复行为属于运行时改动,结果来源必须明确包含此补丁,不能描述为未修改的上游镜像。 + +## 验证与移除 + +8 项本地补丁应用及拒绝未知状态的检查通过。固定缓存镜像中的 +[运行 34832901970](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34832901970) +复现原版调度器回归(2 项通过、1 项预期失败)。候选的核心、错误传播、Store 和 Multi 共 51 项通过; +另一个 Store 用例因 blocks 测试夹具不符合旧版 API 失败。仅修正该夹具后, +[单点运行](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34833354601)通过, +两个运行合计覆盖 52 个不同的候选用例。两次均验证同一组 5 个补丁源码 SHA,生产补丁未改变。 + +这些是未加载模型的运行时 CPU 回归。请求签核前仍需完整 PR sweep/eval 和维护者豁免批准; +原先三个成功性能点不代表修改后运行时已获资格。 + +待 #55297 或同等已评审修复合入并进入官方 vLLM 镜像后,在同一PR中更新配置镜像、移除补丁工具和调用、 +退役本豁免及翻译,再验证完整现有 B300 曲线及适用 eval。诊断子集不能替换完整曲线。若上游合入前修改 +方案,应先复核并更新这里固定的回移内容。 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index db5662f5b8..23a2f3a279 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -1739,7 +1739,6 @@ - "Add --gpu-memory-utilization 0.9 to server launch" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/1133 - - config-keys: - dsv4-fp8-h200-vllm description: @@ -6100,7 +6099,6 @@ - "Add GB200 DeepSeek-V4-Pro FP4 Dynamo-vLLM AgentX mirroring the GB300 PR #2571 MTP tuning, with every GB300 4-GPU worker sized to 8 GPUs on GB200." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2636 - - config-keys: - qwen3.5-fp8-h200-sglang-agentic-hicache-mtp scenario-type: @@ -6492,7 +6490,6 @@ - "Add HiCache host-DRAM KV tier arms at TP4 concurrency 40, 48, 56, and 64 and TP2 concurrency 20, 24, 28, and 32, using hicache ratio 1.5 with write_through, direct io, and page_first_direct layout." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2693 - - config-keys: - kimik2.6-fp4-b200-dynamo-vllm - dsv4-fp4-b200-dynamo-vllm @@ -9276,6 +9273,176 @@ - "Run TP4 at concurrency 1-16 and replace TP2 with DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) at concurrency 8-192 behind a consistent-hash vLLM Router." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3652 +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + description: + - "Discover active Mellanox RDMA devices by driver on DSXE, select the link-layer GID, and load the host mlx5 provider so Mooncake sees the rail" + - "Backport vLLM #55297 for request-local recompute after failed asynchronous Mooncake loads on B300; see docs/waiver/3088.md" + - "Raise the Mooncake master lease from the 5s default to 60s (--default_kv_lease_ttl=60000): at concurrency >= 32 gets took 10-30s at p90, so 60-75% of transferred keys were discarded as LEASE_EXPIRED and recomputed; 60s matches the client's own 60s transfer cap. vLLM's execute_model timeout stays at its 300s default" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Move select_mooncake_rdma_device above benchmark_lib.sh's --validation-only gate so kimik3-b300-mooncake.sh can resolve the helper (exit 127 / command not found previously cancelled the fail-fast canary)" + - "将 select_mooncake_rdma_device 移到 benchmark_lib.sh 的 --validation-only 门禁之上,使 kimik3-b300-mooncake.sh 能解析该辅助函数(此前 exit 127 / command not found 导致 fail-fast canary 取消)" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Restore VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 on the B300 Mooncake AgentX recipe. Tip fail-fast run 36800796192 killed agentic c40 after ~279s of engine silence (shm_broadcast 60s waits from 04:57:58-05:00:58 PT window, then EngineCore TimeoutError / EngineDeadError at 05:01:56) under the default 300s execute_model cap; AIPerf then ProfileAborted at 30/292 = 10.3% failed requests. Keep the existing 60s Mooncake master lease (--default_kv_lease_ttl=60000). Sibling Kimi Mooncake recipes already set this timeout." + - "将 B300 Mooncake AgentX 配方的 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS 恢复为 1800。tip fail-fast 运行 36800796192 在默认 300 秒 execute_model 上限下,于约 279 秒引擎静默后杀死 agentic c40(shm_broadcast 60 秒等待,随后 EngineCore TimeoutError / EngineDeadError);AIPerf 随即以 30/292 = 10.3% 失败请求触发 ProfileAborted。保留既有的 60 秒 Mooncake master 租约。同系列 Kimi Mooncake 配方已设置该超时。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Align B300 DCP8 Mooncake connector with proven GB300 DCP8 settings: compact_group_io + max_load_batch_keys=2 and MC_MAX_MR_SIZE=4GiB. Tip fail-fast run 36820538765 agentic c48 (job 110258065670, Slurm 6544) had a real rail (setup patched device_name to ibp198s0f0; srt-slurm Wrote line is pre-patch) then died with GPU memcpy fail / direct DCP kv-gather multimem timeout under high KV pressure. Also fail loud if the host mlx5 provider mount is missing or device_name would stay empty." + - "将 B300 DCP8 Mooncake 连接器与已验证的 GB300 DCP8 设置对齐:compact_group_io、max_load_batch_keys=2 以及 MC_MAX_MR_SIZE=4GiB。tip fail-fast 运行 36820538765 的 agentic c48(job 110258065670,Slurm 6544)已有真实 rail(setup 将 device_name 写成 ibp198s0f0;srt-slurm 的 Wrote 行为 patch 前转储),随后在高 KV 压力下因 GPU memcpy 失败 / direct DCP kv-gather multimem timeout 崩溃。主机 mlx5 provider 挂载缺失或 device_name 仍为空时改为明确失败。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Revert compact_group_io on B300 DSXE Mooncake (keep max_load_batch_keys=2) and set VLLM_USE_DIRECT_DCP_KV_GATHER=0. Tip ee33eeba fail-fast run 36838397438 agentic c8 patched device_name to ibp198s0f0 (GH Wrote line is pre-setup), then Compact Mooncake group I/O (~25MiB puts) stormed 7177 TRANSFER_FAIL / save_put_failed_keys up to 3782 and hung at 0 tok/s until the 1800s sample_tokens timeout. Prior tip c48 multimem/memcpy with KV_GATHER=1 motivates disabling direct KV gather." + - "撤销 B300 DSXE Mooncake 上的 compact_group_io(保留 max_load_batch_keys=2),并将 VLLM_USE_DIRECT_DCP_KV_GATHER 设为 0。tip ee33eeba fail-fast 运行 36838397438 的 agentic c8 已将 device_name 写成 ibp198s0f0(GH Wrote 行为 setup 前转储),随后 Compact Mooncake group I/O(约 25MiB put)触发 7177 次 TRANSFER_FAIL / save_put_failed_keys 最高 3782,并以 0 tok/s 挂起直至 1800 秒 sample_tokens 超时。此前 tip c48 在 KV_GATHER=1 下出现 multimem/memcpy,因此关闭 direct KV gather。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Drop enable-cumem-allocator on B300 DSXE Mooncake. Tip 15820fd2e fail-fast run 36852970338 agentic c2 (job 110367970533, Slurm 6595, gpu-04) patched device_name to ibp198s0f0 (GH Wrote line is pre-setup); every rank logged register_buffer failed for the ~39GiB KV region (-600), then AddressNotRegistered TRANSFER_FAIL on puts inside that span; after ~38m healthy serve the engine hung at 0 tok/s (Running: 2) until the 1800s sample_tokens timeout / EngineDead / ProfileAborted. nccl_error:16 was ibv_query_port_speed WARN only. Passing tip canary c1 had the same register_buffer -600 storm but survived via recompute. Same class as LMCache dropping cumem when VMM buffers cannot be registered/exported; keep max_load_batch_keys=2, no compact_group_io, KV_GATHER=0, MC_MAX_MR_SIZE=4GiB." + - "在 B300 DSXE Mooncake 上关闭 enable-cumem-allocator。tip 15820fd2e fail-fast 运行 36852970338 的 agentic c2(job 110367970533,Slurm 6595,gpu-04)已将 device_name 写成 ibp198s0f0(GH Wrote 行为 setup 前转储);各 rank 均对约 39GiB KV 区域报 register_buffer failed(-600),随后该地址范围内的 put 触发 AddressNotRegistered TRANSFER_FAIL;约 38 分钟健康服务后引擎以 0 tok/s(Running: 2)挂起直至 1800 秒 sample_tokens 超时 / EngineDead / ProfileAborted。nccl_error:16 仅为 ibv_query_port_speed WARN。通过的 tip canary c1 有相同的 register_buffer -600 风暴,但靠 recompute 存活。与 LMCache 在 VMM 缓冲无法注册/导出时关闭 cumem 同类;保留 max_load_batch_keys=2、不启用 compact_group_io、KV_GATHER=0、MC_MAX_MR_SIZE=4GiB。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Remove MC_MAX_MR_SIZE from B300 DSXE Mooncake. Tip 1d9c54d fail-fast run 36870807626 agentic c32 (job 110437058728, Slurm 6614, gpu-01) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; GH Wrote line is pre-setup) and cumem already off, but every rank still logged register_buffer failed for the ~43.7GiB KV region (-600) then ~128k AddressNotRegistered TRANSFER_FAIL; after ~40m of degraded serve an NCCL ALLGATHER watchdog (600s) killed VllmWorker-2 (worker_crash:8) → EngineDead / ProfileAborted. Pre-MC_MAX_MR tips (old c40/c48) had zero register_buffer -600; the 4GiB cap was added in ee33eeba GB300 alignment and is the verified driver. Keep no compact_group_io, max_load_batch_keys=2, KV_GATHER=0, cumem off." + - "从 B300 DSXE Mooncake 移除 MC_MAX_MR_SIZE。tip 1d9c54d fail-fast 运行 36870807626 的 agentic c32(job 110437058728,Slurm 6614,gpu-01)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched;GH Wrote 行为 setup 前转储)且 cumem 已关,但各 rank 仍对约 43.7GiB KV 区域报 register_buffer failed(-600),随后约 12.8 万次 AddressNotRegistered TRANSFER_FAIL;约 40 分钟降级服务后 NCCL ALLGATHER watchdog(600 秒)杀死 VllmWorker-2(worker_crash:8)→ EngineDead / ProfileAborted。加入 MC_MAX_MR 之前的 tip(旧 c40/c48)无 register_buffer -600;该 4GiB 上限来自 ee33eeba 的 GB300 对齐,是已验证根因。保留不启用 compact_group_io、max_load_batch_keys=2、KV_GATHER=0、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cut Mooncake max_load_batch_keys from 2 to 1 on B300 DSXE. Tip 860c1ccf fail-fast run 36894040300 agentic c48 (job 110510865864, Slurm 6639, gpu-03) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; GH Wrote line is pre-setup), zero register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem, and clean Mooncake load/save metrics (failed_keys=0) through ~35m of serve; then under GPU KV ~100% the engine hung at 0 tok/s (Running: 4, Waiting: 46) and NCCL _ALLGATHER_BASE on PG ID 3 timed out after 600s (last started work: -1) → VllmWorker-4 died (worker_crash:8) → EngineDead / ProfileAborted. Keep no compact_group_io, no MC_MAX_MR_SIZE, KV_GATHER=0, cumem off, load_async true." + - "将 B300 DSXE 上的 Mooncake max_load_batch_keys 从 2 降为 1。tip 860c1ccf fail-fast 运行 36894040300 的 agentic c48(job 110510865864,Slurm 6639,gpu-03)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched;GH Wrote 行为 setup 前转储),无 register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem,且 Mooncake load/save 指标干净(failed_keys=0)约 35 分钟;随后在 GPU KV ~100% 下引擎以 0 tok/s 挂起(Running: 4,Waiting: 46),PG ID 3 上的 NCCL _ALLGATHER_BASE 在 600 秒后超时(last started work: -1)→ VllmWorker-4 死亡(worker_crash:8)→ EngineDead / ProfileAborted。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、KV_GATHER=0、关闭 cumem、load_async true。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Disable Mooncake load_async on B300 DSXE (keep max_load_batch_keys=1, lookup_async). Tip e51c58f5 fail-fast run 36915805279 agentic c32 (job 110582923073, Slurm 6677, gpu-07) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; GH Wrote line is pre-setup), zero register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem, and clean Mooncake metrics (failed_keys=0); under GPU KV ~98% the engine hung at 0 tok/s (Running: 10, Waiting: 23) and NCCL _ALLGATHER_BASE on PG ID 3 timed out after 600s (last started work: -1) → VllmWorker-1 died (worker_crash:8) → EngineDead / ProfileAborted. Same signature as tip 860c1ccf c48 with batch_keys=2; cutting batch_keys to 1 alone did not stop the hang. Keep no compact_group_io, no MC_MAX_MR_SIZE, KV_GATHER=0, cumem off." + - "在 B300 DSXE 上关闭 Mooncake load_async(保留 max_load_batch_keys=1、lookup_async)。tip e51c58f5 fail-fast 运行 36915805279 的 agentic c32(job 110582923073,Slurm 6677,gpu-07)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched;GH Wrote 行为 setup 前转储),无 register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem,且 Mooncake 指标干净(failed_keys=0);在 GPU KV ~98% 下引擎以 0 tok/s 挂起(Running: 10,Waiting: 23),PG ID 3 上的 NCCL _ALLGATHER_BASE 在 600 秒后超时(last started work: -1)→ VllmWorker-1 死亡(worker_crash:8)→ EngineDead / ProfileAborted。与 tip 860c1ccf 的 c48(batch_keys=2)签名相同;仅将 batch_keys 降为 1 未能阻止挂起。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、KV_GATHER=0、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Restore Mooncake load_async=true on B300 DSXE (keep max_load_batch_keys=1). Tip ea88d652 fail-fast run 36935627690 canary c1 (job 110615439146, Slurm 6679, gpu-00) reached health OK then aborted warmup: WorkerProc AssertionError in mooncake/store/worker.py get_finished — 'load_async must be True for better performance' — then EngineCore Internal Server Error / ProfileAborted. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive), not runtime ALLGATHER. Tip 670fb9f0 disable load_async is incompatible with this vLLM Mooncake connector assert; restore stock true." + - "在 B300 DSXE 上恢复 Mooncake load_async=true(保留 max_load_batch_keys=1)。tip ea88d652 fail-fast 运行 36935627690 的 canary c1(job 110615439146,Slurm 6679,gpu-00)健康检查通过后 warmup 中止:WorkerProc 在 mooncake/store/worker.py get_finished 断言 AssertionError: load_async must be True for better performance,随后 EngineCore Internal Server Error / ProfileAborted。ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报),非运行时 ALLGATHER。tip 670fb9f0 关闭 load_async 与该 vLLM Mooncake 连接器断言不兼容;恢复 stock true。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Set VLLM_USE_DIRECT_DCP_Q_GATHER=0 on B300 DSXE (keep load_async=true, max_load_batch_keys=1, KV_GATHER=0). After tip ea88d652b canary c1 AssertionError forced restoring load_async, tip d7176e6 would otherwise re-run the Q_GATHER=1 + KV_GATHER=0 mix that hung tips 860c1ccf c48 / e51c58f5 c32 in DCP PYNCCL _ALLGATHER_BASE under ~98-100% GPU KV (last started work: -1; Mooncake metrics clean). Align Q gather off with KV gather (MI355X-style). Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上将 VLLM_USE_DIRECT_DCP_Q_GATHER 设为 0(保留 load_async=true、max_load_batch_keys=1、KV_GATHER=0)。tip ea88d652b canary c1 的 AssertionError 已强制恢复 load_async 后,tip d7176e6 否则会重跑 tip 860c1ccf c48 / e51c58f5 c32 在约 98–100% GPU KV 下挂起于 DCP PYNCCL _ALLGATHER_BASE 的 Q_GATHER=1 + KV_GATHER=0 混用路径(last started work: -1;Mooncake 指标干净)。将 Q gather 与 KV gather 一并关闭(对齐 MI355X)。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Restore VLLM_USE_DIRECT_DCP_Q_GATHER=1 and set VLLM_USE_DIRECT_DCP_A2A=0 on B300 DSXE. Tip 1f837c46 fail-fast run 36937918258 eval-only c8 (job 110646326761, Slurm 6692, gpu-09) had a real rail (Mooncake rail: ibp198s0f0 / Patched) then hung ~8.5m after logging dcp:0 PYNCCL and failed WorkerProc init on EP group PyNcclCommunicator ncclCommInitRank with RuntimeError: NCCL error: remote process exited or there was a network error → Engine core initialization failed. Same tip canary c1 completed ep:0 in ~2s; Q_GATHER=0 is the new tip knob vs prior green evals. Keep load_async=true, max_load_batch_keys=1, KV_GATHER=0, no compact_group_io, no MC_MAX_MR_SIZE, cumem off. A2A=0 is the next probe for prior high-conc DCP PYNCCL _ALLGATHER_BASE hangs." + - "在 B300 DSXE 上恢复 VLLM_USE_DIRECT_DCP_Q_GATHER=1,并将 VLLM_USE_DIRECT_DCP_A2A 设为 0。tip 1f837c46 fail-fast 运行 36937918258 的 eval-only c8(job 110646326761,Slurm 6692,gpu-09)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),在记录 dcp:0 PYNCCL 后挂起约 8.5 分钟,随后 EP 组 PyNcclCommunicator ncclCommInitRank 报 RuntimeError: NCCL error: remote process exited or there was a network error → Engine core initialization failed。同 tip 的 canary c1 约 2 秒内完成 ep:0;相对此前全绿 eval,新 tip 旋钮为 Q_GATHER=0。保留 load_async=true、max_load_batch_keys=1、KV_GATHER=0、不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。A2A=0 作为此前高并发 DCP PYNCCL _ALLGATHER_BASE 挂起的下一探针。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Lower gpu-memory-utilization to 0.85 on B300 DSXE CONC 24+ (keep load_async=true, max_load_batch_keys=1, KV_GATHER=0, Q_GATHER=1, A2A=0, ESTIMATE_CUDAGRAPHS=0). Tip b70e4260a fail-fast run 36952119306 agentic c48 (job 110686724979, Slurm 6737, gpu-05) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name), Application startup complete, then ~4m into AgentX serve CUDA OOM in flashinfer FP4 MoE prepare_moe (tried 2.89 GiB, ~2.3 GiB free; GPU KV 45.2 GiB at util 0.92; vLLM suggested ~36.78 GiB once CUDA graphs counted) → EngineDead / ProfileAborted. Zero register_buffer -600 / TRANSFER_FAIL / ALLGATHER. GHA 'engine never ready' was wrapper noise after serve-time OOM. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "将 B300 DSXE 上 CONC 24+ 的 gpu-memory-utilization 降至 0.85(保留 load_async=true、max_load_batch_keys=1、KV_GATHER=0、Q_GATHER=1、A2A=0、ESTIMATE_CUDAGRAPHS=0)。tip b70e4260a fail-fast 运行 36952119306 的 agentic c48(job 110686724979,Slurm 6737,gpu-05)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,随后 AgentX 服务约 4 分钟时在 flashinfer FP4 MoE prepare_moe 发生 CUDA OOM(申请 2.89 GiB,空闲约 2.3 GiB;util 0.92 下 GPU KV 45.2 GiB;计入 CUDA graph 后 vLLM 建议约 36.78 GiB)→ EngineDead / ProfileAborted。无 register_buffer -600 / TRANSFER_FAIL / ALLGATHER。GHA 的 engine never ready 为服务期 OOM 后的包装噪声。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Re-enable VLLM_USE_DIRECT_DCP_KV_GATHER=1 on B300 DSXE (keep A2A=0, Q_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 24+ util 0.85). Tip bed9f1ce fail-fast run 36963354550 agentic c40 (job 110720603462, Slurm 6765, gpu-13) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, Mooncake failed_keys=0; after ~42m serve under GPU KV ~86-100% hung at 0 tok/s (Running: 29, Waiting: 10) then PG ID 3 NCCL _ALLGATHER_BASE in dcp.py kv_gather timed out 600s (last started work: -1) → VllmWorker-3 died (worker_crash:8) → EngineDead / ProfileAborted. Stack is the PyNCCL path forced by KV_GATHER=0; Q gather already logged direct symmetric-memory. Prior multimem/memcpy with KV_GATHER=1 was under util 0.92 before the 0.85 headroom cut. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上重新启用 VLLM_USE_DIRECT_DCP_KV_GATHER=1(保留 A2A=0、Q_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 24+ util 0.85)。tip bed9f1ce fail-fast 运行 36963354550 的 agentic c40(job 110720603462,Slurm 6765,gpu-13)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,Mooncake failed_keys=0;约 42 分钟服务后在 GPU KV ~86–100% 下以 0 tok/s 挂起(Running: 29,Waiting: 10),随后 PG ID 3 上 dcp.py kv_gather 的 NCCL _ALLGATHER_BASE 超时 600 秒(last started work: -1)→ VllmWorker-3 死亡(worker_crash:8)→ EngineDead / ProfileAborted。堆栈为 KV_GATHER=0 强制的 PyNCCL 路径;Q gather 已记录使用 direct symmetric-memory。此前 KV_GATHER=1 的 multimem/memcpy 发生在 util 0.92、尚未降至 0.85 时。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Enable VLLM_USE_DIRECT_DCP_A2A=1 on B300 DSXE (keep Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 24+ util 0.85). Tip 68cdcc58 fail-fast run 36976589313 agentic c24 (job 110806435815, Slurm 6814, gpu-12) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, and logged both direct Q gather and direct chunked-context KV gather; Mooncake failed_keys=0. After ~43m serve hung at 0 tok/s (Running: 23, Waiting: 0, GPU KV ~68%) then PG ID 3 NCCL ALLTOALL_BASE in dcp.py dcp_a2a_lse_reduce timed out 600s (last started work: -1) → worker_crash:8 → EngineDead / ProfileAborted. Distinct from tip bed9f1ce c40 kv_gather _ALLGATHER_BASE (fixed by KV_GATHER=1). A2A=0 forced the PyNCCL ALLTOALL LSE-reduce path; enable direct A2A to match GB300 DCP8 / B200 stock. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上启用 VLLM_USE_DIRECT_DCP_A2A=1(保留 Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 24+ util 0.85)。tip 68cdcc58 fail-fast 运行 36976589313 的 agentic c24(job 110806435815,Slurm 6814,gpu-12)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,并已记录 direct Q gather 与 direct chunked-context KV gather;Mooncake failed_keys=0。约 43 分钟服务后以 0 tok/s 挂起(Running: 23,Waiting: 0,GPU KV ~68%),随后 PG ID 3 上 dcp.py dcp_a2a_lse_reduce 的 NCCL ALLTOALL_BASE 超时 600 秒(last started work: -1)→ worker_crash:8 → EngineDead / ProfileAborted。与 tip bed9f1ce c40 的 kv_gather _ALLGATHER_BASE(已由 KV_GATHER=1 修复)不同。A2A=0 强制走 PyNCCL ALLTOALL LSE-reduce 路径;启用 direct A2A 以对齐 GB300 DCP8 / B200 stock。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 1x CONC for CONC 48+ on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 24+ util 0.85). Tip 031de17bf fail-fast run 37010740215 agentic c56 (job 110882921963, Slurm 6848, gpu-12) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, all three direct DCP gathers logged, Mooncake failed_keys=0. Measured warmup packed GPU KV to ~99.7% (Running: 1, Waiting: 12, Deferred: 12) then hung workers: shm_broadcast starvation from 15:28 for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted. No Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Distinct from tip b70e4260a c48 OOM (util 0.92), tip bed9f1ce c40 kv_gather ALLGATHER (KV_GATHER=0), and tip 68cdcc58 c24 ALLTOALL dcp_a2a_lse_reduce (A2A=0). With all three directs already on, high-conc 2x admission was the remaining lever that pinned KV before decode. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上将 CONC 48+ 的 max-num-seqs 限制为 1×CONC(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 24+ util 0.85)。tip 031de17bf fail-fast 运行 37010740215 的 agentic c56(job 110882921963,Slurm 6848,gpu-12)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,三种 direct DCP gather 均已记录,Mooncake failed_keys=0。实测 warmup 将 GPU KV 堆到约 99.7%(Running: 1,Waiting: 12,Deferred: 12)后 worker 挂起:自 15:28 起 shm_broadcast 饥饿直至满 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted。无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。有别于 tip b70e4260a c48 OOM(util 0.92)、tip bed9f1ce c40 kv_gather ALLGATHER(KV_GATHER=0)、tip 68cdcc58 c24 ALLTOALL dcp_a2a_lse_reduce(A2A=0)。三种 direct 已全部开启时,高并发 2× 准入是仍会在 decode 前钉死 KV 的剩余杠杆。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Extend gpu-memory-utilization 0.85 to CONC 8+ on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 48+ max-num-seqs 1x). Tip 5ab41690 fail-fast run 37033506119 agentic c8 (job 110957197772, Slurm 6877, gpu-04) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, KV 38.89 GiB at util 0.92 under VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0; during warmup CUDACachingAllocator soft-OOMed on ranks 0-7 trying to allocate ~3.03 GiB with ~1.16 GiB free → empty streamed responses → InvalidInferenceResultError / ProfileAborted at 2/11 > 10% (engine stayed up; no Watchdog / ALLGATHER / ALLTOALL / hard EngineDead). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Same soft-OOM family as tip b70e4260a c48 at util 0.92; keep 0.92 only on c1-c4. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "将 B300 DSXE 上 gpu-memory-utilization 0.85 扩展到 CONC 8+(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 48+ max-num-seqs 1×)。tip 5ab41690 fail-fast 运行 37033506119 的 agentic c8(job 110957197772,Slurm 6877,gpu-04)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,在 VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0 与 util 0.92 下 KV 为 38.89 GiB;warmup 中 ranks 0–7 的 CUDACachingAllocator 软 OOM(申请约 3.03 GiB,空闲约 1.16 GiB)→ 空流 → InvalidInferenceResultError / ProfileAborted(2/11 > 10%;引擎未死;无 Watchdog / ALLGATHER / ALLTOALL / 硬 EngineDead)。ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。与 tip b70e4260a c48 在 util 0.92 下的软 OOM 同族;仅 c1–c4 保留 0.92。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 1x CONC for CONC 32+ on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85). Tip 861a1512 fail-fast run 37049360028 agentic c32 (job 111008983824, Slurm 6916, gpu-00) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, all three direct DCP gathers logged, Mooncake failed_keys=0. Under max-num-seqs=64 (2x) GPU KV pinned ~99-100% for tens of minutes, then hung at 0 tok/s (Running: 23, Waiting: 11, Deferred: 11, GPU KV 71.8%) with shm_broadcast starvation from 21:03 for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted (20 InvalidInferenceResultError empty streams + 62 ClientConnectorError after shutdown). No Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Same admission-pin family as tip 031de17bf c56; extend 1x CONC from CONC 48+ down through c32/c40 and keep 2x only through CONC 24. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上将 CONC 32+ 的 max-num-seqs 限制为 1×CONC(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 8+ util 0.85)。tip 861a1512 fail-fast 运行 37049360028 的 agentic c32(job 111008983824,Slurm 6916,gpu-00)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,三种 direct DCP gather 均已记录,Mooncake failed_keys=0。在 max-num-seqs=64(2×)下 GPU KV 钉在约 99–100% 数十分钟,随后以 0 tok/s 挂起(Running: 23,Waiting: 11,Deferred: 11,GPU KV 71.8%),自 21:03 起 shm_broadcast 饥饿直至满 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted(关机后 20 个空流 InvalidInferenceResultError + 62 个 ClientConnectorError)。无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。与 tip 031de17bf c56 同属准入钉死族;将 1×CONC 从 CONC 48+ 下扩到 c32/c40,仅 CONC 24 及以下保留 2×。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 48 for CONC 70 on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-56 max-num-seqs 1x). Tip 1d937c73e fail-fast run 37071755348 agentic c70 (job 111074417068, Slurm 6950, gpu-11) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, all three direct DCP gathers logged, Mooncake failed_keys=0. Under max-num-seqs=70 (1x) warmup packed Waiting≈63-70 / Deferred≈50-67 with Running≈0 and GPU KV ≈86-91% for ~30m, then hung with shm_broadcast starvation through VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted (771 warmup dropped / 0 kept; time_to_ready=None). No Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Same deferred-admission family as tip 861a1512 c32 / 031de17bf c56; 1x insufficient at c70 so drop below 1x to 48. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上将 CONC 70 的 max-num-seqs 限制为 48(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 8+ util 0.85、CONC 32–56 max-num-seqs 1×)。tip 1d937c73e fail-fast 运行 37071755348 的 agentic c70(job 111074417068,Slurm 6950,gpu-11)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,三种 direct DCP gather 均已记录,Mooncake failed_keys=0。在 max-num-seqs=70(1×)下 warmup 将 Waiting≈63–70 / Deferred≈50–67 堆满且 Running≈0、GPU KV ≈86–91% 约 30 分钟,随后 shm_broadcast 饥饿直至满 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted(丢弃 771 warmup / 保留 0;time_to_ready=None)。无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。与 tip 861a1512 c32 / 031de17bf c56 同属 deferred 准入族;c70 上 1× 不足,降至 48。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + - config-keys: - glm5.2-fp4-mi355x-sglang-agentic-mtp scenario-type: @@ -9292,3 +9459,93 @@ - "Update the B200 DeepSeek-V4.1-Flash vLLM AgentX image from nightly ddd6fbca to nightly-dev-x86_64-cu130-ac9126e58aa7 and enable FlashInfer autotuning." - "Run TP4 at concurrency 1-128 and replace TP2 with DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) at concurrency 8-32 and DEP4 (TP1 x DP4 + EP4, MegaMoE) at concurrency 64-128, both behind a consistent-hash vLLM Router. All points set gpu-memory-utilization 0.97." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3686 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Disable VLLM_USE_DIRECT_DCP_KV_GATHER on B300 DSXE (keep A2A=1, Q_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-56 max-num-seqs 1x, c70 max-num-seqs 48). Tip f7be12ed fail-fast run 37086876838 agentic c56 (job 111113692599, Slurm 6989, gpu-01) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; the srt-slurm Wrote line is the pre-setup dump), Application startup complete, and profiling kept 1429 requests. At 07:39:41 the direct kernel logged direct DCP kv-gather multimem timeout source=1 epoch=494017 then trapped the device (dcp_direct_kv_gather.cu wait_for_epoch plus asm trap) which is the CUDA unspecified launch failure observed in KVCacheStoreSendingThread, then GPU memcpy failed src_dev=-1 dst_dev=2 size=884736 / TRANSFER_FAIL -800, VllmWorker-2 died, EngineDead / ProfileAborted (143/1429 = 10.007%). Mooncake failed_keys stayed 0 until that instant. Not deferred-admission starvation: the prior 10s window was 347 tok/s with Running: 9 and there was no sample_tokens timeout. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Distinct from tip 1d937c73e c70 (Running about 0, 1800s sample_tokens, 0 kept). The stock gate leaves the trapping multimem kernel for all_gather_into_tensor. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Disable VLLM_USE_DIRECT_DCP_Q_GATHER on B300 DSXE (keep A2A=1, KV_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-56 max-num-seqs 1x, c70 max-num-seqs 48). Tip db6caecc fail-fast run 37109541419 agentic c24 (job 111177660118, Slurm 7019, gpu-17) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; the srt-slurm Wrote line is the pre-setup dump), Application startup complete, and profiling kept 1041 requests (dropped 265 warmup). At 10:41:47 the direct kernel logged direct DCP q-gather multimem timeout source=7 epoch=1107073 then CUDA unspecified launch failure in KVCacheStoreSendingThread across ranks, EngineDead / ProfileAborted (105/1041 = 10.086%). Mooncake failed_keys stayed 0; prior 10s windows were healthy (Running ~17-25, GPU KV ~28-42%, generation hundreds of tok/s). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same multimem-trap family as tip f7be12ed c56 kv-gather (KV_GATHER already 0); not deferred-admission starvation. CAUTION: tip 1f837c46 / sweep 36937918258 previously tried Q_GATHER=0 and hung ~8.5m after dcp:0 on agentic-eval c8 at EP PyNcclCommunicator ncclCommInitRank (rail ibp198s0f0) — Q was restored and A2A briefly disabled then restored; watch eval init under this gate with A2A=1. Stock gate leaves the trapping q-gather multimem kernel. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 48 for CONC 56 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-48 max-num-seqs 1x, c70 max-num-seqs 48). Tip fd61acb0 fail-fast run 37118728189 agentic c56 (job 111203259062, Slurm 7040, gpu-10) had a real rail, Application startup complete, Mooncake failed_keys=0, and warmup completed 619/619 with 0 errors. Profiling packed GPU KV to 98.2% (Running: 8, Waiting: 53, Deferred: 38) then hung at 0 tok/s at 15:08:49; first causal line is NCCL watchdog WorkNCCL SeqNum=342641 OpType=_ALLGATHER_BASE Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 15:18:37 → DistBackendError / EngineDead / ProfileAborted (62/392 = 15.816%). Direct A2A remained on (dcp.py:1283). Not A2A multimem/ALLTOALL and not a direct-gather multimem trap. Same packed-KV PyNCCL kv_gather family as tip bed9f1ce c40 / 860c1ccf c48; do not re-enable KV/Q gather (multimem traps on f7be12ed/db6caecc). 1x=56 still insufficient at c56 so drop below 1x to 48 like c70. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 1x CONC (16) for CONC 16 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-48 max-num-seqs 1x, c56/c70 max-num-seqs 48). Tip ebc4eb499 fail-fast run 37134266476 agentic c16 (job 111250499353, Slurm 7062, gpu-13) had a real rail, Application startup complete at 17:29:18Z, Mooncake failed_keys=0, and warmup returned 140/177 with 10 in flight from 17:44:47 at 0 tok/s while GPU KV stayed ~98-99.7% (Running: 3, Waiting: 7, Deferred: 7). First causal line is NCCL watchdog WorkNCCL SeqNum=89905 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 on PG ID 3 Rank 3 at 17:54:46 → DistBackendError / VllmWorker-7 died / EngineDead / ProfileAborted (176 warmup dropped / 0 kept; profiling never started). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same packed-KV PyNCCL _ALLGATHER_BASE family as tip fd61acb0 c56; not a direct-gather multimem trap and not setup. 2x=32 at c16 is the remaining 2x cell besides c8/c24; drop c16 to 1x=16. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 32 for CONC 70 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/32-48 max-num-seqs 1x, c56 max-num-seqs 48). Tip c9ffe2b12 fail-fast run 37143203751 agentic c70 (job 111276873706, Slurm 7087, gpu-15) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, Mooncake failed_keys=0, warmup completed 772/774, and profiling kept 115 requests. During profile GPU KV packed to 99.6% (Running: 0, Waiting: 48, Deferred: 46 at 21:00:12); after recovery the first causal kill is NCCL watchdog WorkNCCL SeqNum=353066 OpType=_ALLGATHER_BASE Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 21:17:09 (hang from ~21:07:09 with Running: 10, Waiting: 2, Deferred: 2, GPU KV 21.3%) → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted (82/115 InvalidInferenceResultError). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same PyNCCL kv_gather family as tip fd61acb0 c56 / ebc4eb499 c16; not multimem trap, not sample_tokens deferred-admission starve (distinct from tip 1d937c73e c70 at max-num-seqs=70). Cap 48 still insufficient at c70 so drop to 32. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 1x CONC (24) for CONC 24 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/32-48 max-num-seqs 1x, c56 max-num-seqs 48, c70 max-num-seqs 32). Tip f3997a93 fail-fast run 37155636981 agentic c24 (job 111311962571, Slurm 7107, gpu-12) had a real rail, Application startup complete, Mooncake failed_keys=0, warmup progressed, and profiling kept 730/1077 (265 warmup, 288 error dropped). Peak GPU KV 91.1% at 23:41:20 (Running: 19, Waiting: 1, Deferred: 1). First causal kill is NCCL watchdog WorkNCCL SeqNum=224109 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 23:56:57 → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted (82/812 = 10.099%). Hang from ~23:46:57 with Running: 17–25 and GPU KV ~76–84%. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same PyNCCL kv_gather family as tip c9ffe2b12 c70 / fd61acb0 c56 / ebc4eb499 c16; c24 was the last mid-conc still at 2x. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 32 for CONC 40 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/24/32/48 max-num-seqs 1x, c56 max-num-seqs 48, c70 max-num-seqs 32). Tip 71080871 fail-fast run 37164230642 agentic c40 (job 111336354457, Slurm 7123, b300-dsxe_06 / gpu-09) had a real rail, Application startup complete, Mooncake path healthy, warmup progressed, and profiling kept 838/1376 (444 warmup, 449 error dropped). Peak GPU KV ~99.9% with Waiting/Deferred queues elevated (e.g. Running: 2, Waiting: 35, Deferred: 21 at 02:33:28; Running: 7, Waiting: 27, Deferred: 23 at 02:38:38). First causal kill is NCCL watchdog WorkNCCL SeqNum=345118 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 02:49:41 → DistBackendError / worker_crash:8 / ProfileAborted (94/932 = 10.086%). Hang from ~02:39:41 with Running: 17, Waiting: 20, Deferred: 20, gen 0 tok/s. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same PyNCCL kv_gather family as tip f3997a93 c24 / c9ffe2b12 c70 / fd61acb0 c56 / ebc4eb499 c16; 1x=40 still insufficient at c40 so drop to 32. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 32 for CONC 48 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/24/32 max-num-seqs 1x, c40/c70 max-num-seqs 32, c56 max-num-seqs 48). Tip 767f0b3b2 fail-fast run 37173109032 agentic c48 (job 111362024256, Slurm 7147, b300-dsxe_04 / gpu-10) had a real rail, Application startup complete, Mooncake path healthy, warmup progressed, and profiling kept 452/1033 (531 warmup, 467 error dropped). Peak GPU KV 100.0% at 05:33:18 (Running: 10, Waiting: 39, Deferred: 28); hang onset ~05:33:54 with last live metrics at 05:33:58 (Running: 13, Waiting: 38, Deferred: 35, GPU KV 98.1%) then 0 tok/s. First causal kill is NCCL watchdog WorkNCCL SeqNum=334427 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 05:43:54 → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted (50/499 = 10.0%). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (first_ts=last_ts=04:35:49). SIGTERM only at shutdown after DistBackend. Same PyNCCL kv_gather family as tip 71080871 c40 / f3997a93 c24 / c9ffe2b12 c70 / fd61acb0 c56 / ebc4eb499 c16; 1x=48 still insufficient at c48 so drop to 32. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 24 for CONC 70 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/24/32 max-num-seqs 1x, c40/c48 max-num-seqs 32, c56 max-num-seqs 48). Tip 141fbb856 fail-fast run 37181441604 agentic c70 (job 111386701892, Slurm 7166, b300-dsxe_09 / gpu-09) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name='ibp198s0f0'), Application startup complete, Mooncake failed_keys=0, and runtime max_num_seqs=32. Warmup reached 284/774 returned with 67 in flight then stalled; profiling never started (ProfileAborted warmup_failure). Peak GPU KV 94.6% at 08:05:41 (Running: 0, Waiting: 70, Deferred: 29). First causal kill is NCCL watchdog WorkNCCL SeqNum=224761 OpType=_ALLGATHER_BASE NumelIn=3032064 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 08:18:49 → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted. Hang from ~08:08:49 with last live metrics at 08:08:51 (Running: 1, Waiting: 65, Deferred: 65, GPU KV 89.9%) then 0 tok/s and shm_broadcast starvation from 08:09:50. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (first_ts=last_ts=07:28:28). SIGTERM only at shutdown after DistBackend. Same PyNCCL kv_gather family as tip cefbd2182 c48 / 71080871 c40 / f3997a93 c24 / c9ffe2b12 c70 / fd61acb0 c56 / ebc4eb499 c16; max-num-seqs=32 still insufficient at c70 so drop to 24. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 32 for CONC 56 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/24/32 max-num-seqs 1x, c40/c48 max-num-seqs 32, c70 max-num-seqs 24). Tip bdcf0a0b fail-fast run 37189444890 agentic c56 (job 111411443417, Slurm 7189, b300-dsxe_15 / gpu-09) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name='ibp198s0f0'), Application startup complete, Mooncake failed_keys=0, and runtime max_num_seqs=48. Warmup completed 619/619 in 2180s then profiling kept 1296 (619 warmup dropped). Peak GPU KV 100.0% at 10:40:52 (Running: 0, Waiting: 54, Deferred: 51). Last live metrics 11:36:32 Running: 23, Waiting: 32, Deferred: 32, GPU KV 98.2%, gen 0 tok/s then silence until SIGTERM 11:55:07. Wrapper ProfileMetricCoverageError (TTFT 71.4% / ITL 71.5% over 3600s; neither signal in the final 180s). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (first_ts=last_ts=10:04:38). No ALLGATHER watchdog / DistBackend / EngineDead this time; same packed-KV admission family as tip 767f0b3b2 c48 / 71080871 c40. Cap c56 48→32. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 24 for CONC 32 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/24 max-num-seqs 1x, c40/c48/c56 max-num-seqs 32, c70 max-num-seqs 24). Tip 1e365aae fail-fast run 37201050113 agentic c32 (job 111446959460, Slurm 7216, b300-dsxe_04 / gpu-13) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name='ibp198s0f0'), Application startup complete, Mooncake failed_keys=0, and runtime max_num_seqs=32. Profiling kept 1162/1646 (354 warmup, 398 error dropped) then aborted at 130/1292 = 10.062%. Peak GPU KV 100.0% at 14:13:15 (Running: 9, Waiting: 23, Deferred: 23). Last live metrics 14:37:25 Running: 25, Waiting: 0, GPU KV 45.4%, gen 0 tok/s then shm_broadcast starvation from 14:38:09. First causal kill is NCCL watchdog WorkNCCL SeqNum=357100 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 14:47:08 → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (first_ts=last_ts=13:37:57). Same PyNCCL kv_gather family as tip 141fbb856 c70 / 767f0b3b2 c48; 1x=32 still insufficient at c32 (green on superseded tip bdcf0a0b) so drop to 24. Do not retune c56 or other conc cells. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + diff --git a/inferencex-e2e/runners/__init__.py b/inferencex-e2e/runners/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/inferencex-e2e/runners/patch_kimik3_mooncake_recovery.py b/inferencex-e2e/runners/patch_kimik3_mooncake_recovery.py new file mode 100644 index 0000000000..0e40484069 --- /dev/null +++ b/inferencex-e2e/runners/patch_kimik3_mooncake_recovery.py @@ -0,0 +1,228 @@ +#!/usr/bin/env python3 +"""Backport vLLM #55297 to the pinned Kimi-K3 B300 Mooncake image. + +Upstream: f3a831c2015d9eb6f7e600dbd2ef565166d64437 +Base: 3696c772aae308f2420a8f307b0971e6986c4818 +Only two hook insertion contexts differ from upstream; recovery logic is unchanged. +See docs/waiver/3088.md. Unknown or partially patched sources stop the launch. +""" + +from __future__ import annotations + +import hashlib +import importlib.util +import sys +from pathlib import Path + +# Relative package path, pristine SHA256, patched SHA256, exact edits. +PATCHES = ( + ('distributed/kv_transfer/kv_connector/v1/base.py', + 'bc1965431087676876f58360cd9cc07ab6c06febe6d747695f10b051fd85c412', + 'cbce0b160ca2d14c477103cf3b7f3433b2533eaa631439a693d1ffd06501342f', ( + (''' """ + return + + def update_connector_output(self, connector_output: KVConnectorOutput): + """ + Update KVConnector state from worker-side connectors output. +''', ''' """ + return + + def on_load_failure(self, request_ids: set[str]) -> None: + """Notify the connector before failed external KV loads are looked up again. + + Connectors may use this callback to make a failed external cache hit a + request-local miss on the next scheduling attempt. The default is a + no-op because not every connector needs special handling before the + affected tokens are recomputed. + """ + return + + def update_connector_output(self, connector_output: KVConnectorOutput): + """ + Update KVConnector state from worker-side connectors output. +'''), + )), + ('distributed/kv_transfer/kv_connector/v1/mooncake/store/connector.py', + 'd18e207bfce93ab53b2156902c8bdf7b223e1441c5070eddb07b4db2cb7b54b9', + '44ea4cda5ef3dbc8c7a32b824dc43280e264f4f63e956a5197d716c2fbaf66a3', ( + (''' def take_events(self) -> Iterable[KVCacheEvent]: +''', ''' def on_load_failure(self, request_ids: set[str]) -> None: + if self.connector_scheduler is not None: + self.connector_scheduler.on_load_failure(request_ids) + + def take_events(self) -> Iterable[KVCacheEvent]: +'''), + )), + ('distributed/kv_transfer/kv_connector/v1/mooncake/store/scheduler.py', + 'c75fc95a585ee4390963b01618c5ece1b52b30f8da95ff1a00a850948a943544', + '3fbd7807ea1b1d55558b604c208e6158d932e57778e3d8a30705edae22bdf7bb', ( + (''' + # Per-request state + self.load_specs: dict[str, LoadSpec] = {} # to be loaded + self._request_trackers: dict[str, RequestTracker] = {} # scheduled new requests + self._unfinished_requests: dict[str, tuple[Request, tuple[list[int], ...]]] = {} + self._unfinished_request_ids: set[str] = set() +''', ''' + # Per-request state + self.load_specs: dict[str, LoadSpec] = {} # to be loaded + # A failed load can rewind a request to token zero. Bypass lookups + # until local allocation succeeds so stale metadata cannot cause a livelock. + self._load_failure_bypass_req_ids: set[str] = set() + self._request_trackers: dict[str, RequestTracker] = {} # scheduled new requests + self._unfinished_requests: dict[str, tuple[Request, tuple[list[int], ...]]] = {} + self._unfinished_request_ids: set[str] = set() +'''), + (''' Returns ``(None, False)`` when an async lookup is still in flight, + signaling the scheduler to retry this request on a later step. + """ + if not self.enable_lookup: + return 0, False + +''', ''' Returns ``(None, False)`` when an async lookup is still in flight, + signaling the scheduler to retry this request on a later step. + """ + if request.request_id in self._load_failure_bypass_req_ids: + self.load_specs.pop(request.request_id, None) + logger.info( + "Skipping Mooncake lookup for request %s after KV load failure", + request.request_id, + ) + return 0, False + + if not self.enable_lookup: + return 0, False + +'''), + (''' + self._unfinished_requests[request.request_id] = (request, local_block_ids) + self._unfinished_request_ids.add(request.request_id) + + if request.request_id not in self.load_specs: + return +''', ''' + self._unfinished_requests[request.request_id] = (request, local_block_ids) + self._unfinished_request_ids.add(request.request_id) + self._load_failure_bypass_req_ids.discard(request.request_id) + + if request.request_id not in self.load_specs: + return +'''), + (''' + for finished_req_id in scheduler_output.finished_req_ids: + self.client.discard(finished_req_id) + self.load_specs.pop(finished_req_id, None) + self._request_trackers.pop(finished_req_id, None) + self._unfinished_requests.pop(finished_req_id, None) +''', ''' + for finished_req_id in scheduler_output.finished_req_ids: + self.client.discard(finished_req_id) + self._load_failure_bypass_req_ids.discard(finished_req_id) + self.load_specs.pop(finished_req_id, None) + self._request_trackers.pop(finished_req_id, None) + self._unfinished_requests.pop(finished_req_id, None) +'''), + (''' def update_connector_output(self, connector_output: KVConnectorOutput) -> None: +''', ''' def on_load_failure(self, request_ids: set[str]) -> None: + """Skip external lookups until requests are allocated for recompute.""" + self._load_failure_bypass_req_ids.update(request_ids) + + def update_connector_output(self, connector_output: KVConnectorOutput) -> None: +'''), + )), + ('distributed/kv_transfer/kv_connector/v1/multi_connector.py', + 'aafafbf4b0e3a43e7c864270fe56ffaa4b0f3dc343e77a0871db25f075d984dd', + '6f4fad0450ef91a75717b9631d668379d905d07fd876325da8d72eae17cca17e', ( + (''' for c in self._connectors: + c.on_new_request(request) + + def build_connector_meta( + self, scheduler_output: SchedulerOutput + ) -> MultiKVConnectorMetadata: +''', ''' for c in self._connectors: + c.on_new_request(request) + + def on_load_failure(self, request_ids: set[str]) -> None: + for c in self._connectors: + c.on_load_failure(request_ids) + + def build_connector_meta( + self, scheduler_output: SchedulerOutput + ) -> MultiKVConnectorMetadata: +'''), + )), + ('v1/core/sched/scheduler.py', + '6ba2a83c7fb6078e4d1c2af7a9a2bf820f83b0570f9e1e1908294a981bf1bace', + '9266f66969dbc0a6474bf53b4b3835c77a961ad4ffb9398ebcb368e1cb2eafc7', ( + (''' total_failed_tokens, + ) + + # Mark async requests with KV load failures for retry once loading completes + self.failed_recving_kv_req_ids |= async_failed_req_ids + # Return sync affected IDs to skip in update_from_output +''', ''' total_failed_tokens, + ) + + # Only async requests rewound to zero return to the waiting queue and + # run the connector lookup again. Notify the connector so a persistent + # external hit cannot make that request repeat the same failed load. + async_relookup_req_ids = { + req_id + for req_id in async_failed_req_ids + if self.requests[req_id].num_computed_tokens == 0 + } + if self.connector is not None and async_relookup_req_ids: + self.connector.on_load_failure(async_relookup_req_ids) + + # Mark async requests with KV load failures for retry once loading completes + self.failed_recving_kv_req_ids |= async_failed_req_ids + # Return sync affected IDs to skip in update_from_output +'''), + )), +) + + +def patch_mooncake(package_root: Path) -> bool: + """Preflight every source before writing; return False for the complete patch.""" + pending: list[tuple[Path, str]] = [] + already_patched = 0 + for relative, pristine_sha, patched_sha, edits in PATCHES: + path = package_root / relative + source = path.read_text() + digest = hashlib.sha256(source.encode()).hexdigest() + if digest == patched_sha: + already_patched += 1 + continue + if digest != pristine_sha: + raise RuntimeError(f"unsupported vLLM source: {path} (SHA256 {digest})") + patched = source + for old, new in edits: + if patched.count(old) != 1: + raise RuntimeError(f"unexpected patch context: {path}") + patched = patched.replace(old, new) + if hashlib.sha256(patched.encode()).hexdigest() != patched_sha: + raise RuntimeError(f"unexpected patched source: {path}") + pending.append((path, patched)) + if already_patched and pending: + raise RuntimeError("partially patched vLLM Mooncake recovery; refusing to serve") + for path, patched in pending: + path.write_text(patched) + return bool(pending) + + +def main() -> int: + try: + spec = importlib.util.find_spec("vllm") + if spec is None or not spec.submodule_search_locations: + raise RuntimeError("vllm package is not installed") + root = Path(next(iter(spec.submodule_search_locations))) + changed = patch_mooncake(root) + except (OSError, RuntimeError) as error: + print(f"ERROR: Kimi B300 Mooncake recovery patch: {error}", file=sys.stderr) + return 1 + print("Applied" if changed else "Already applied", "vLLM #55297 Mooncake recovery") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/inferencex-e2e/runners/test_kimik3_b300.py b/inferencex-e2e/runners/test_kimik3_b300.py new file mode 100644 index 0000000000..342a41b535 --- /dev/null +++ b/inferencex-e2e/runners/test_kimik3_b300.py @@ -0,0 +1,52 @@ +"""Unit coverage for the B300 Mooncake recovery patch.""" +from __future__ import annotations + +from pathlib import Path + +import pytest + + +@pytest.fixture +def mooncake_patch_fixture(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): + """Small package sources exercise the patcher's file-write contract.""" + import hashlib + from runners import patch_kimik3_mooncake_recovery as recovery + + original = {"first.py": "value = 1\n", "second.py": "value = 2\n"} + updated = {"first.py": "value = 3\n", "second.py": "value = 4\n"} + patches = [] + for name, source in original.items(): + (tmp_path / name).write_text(source) + patches.append(( + name, hashlib.sha256(source.encode()).hexdigest(), + hashlib.sha256(updated[name].encode()).hexdigest(), + ((source, updated[name]),), + )) + monkeypatch.setattr(recovery, "PATCHES", patches) + return recovery, tmp_path, original, updated + + +def test_mooncake_patch_is_complete_and_idempotent(mooncake_patch_fixture): + recovery, root, _, expected = mooncake_patch_fixture + assert recovery.patch_mooncake(root) is True + assert {name: (root / name).read_text() for name in expected} == expected + assert recovery.patch_mooncake(root) is False + assert {name: (root / name).read_text() for name in expected} == expected + + +@pytest.mark.parametrize("state", ["unknown", "partial", "missing"]) +def test_mooncake_patch_preflights_all_sources_before_writing( + mooncake_patch_fixture, state: str, +): + recovery, root, _, updated = mooncake_patch_fixture + second = root / "second.py" + if state == "unknown": + second.write_text("different upstream revision\n") + elif state == "partial": + second.write_text(updated["second.py"]) + else: + second.unlink() + before = {path.name: path.read_bytes() for path in root.iterdir()} + with pytest.raises((RuntimeError, OSError)): + recovery.patch_mooncake(root) + assert {path.name: path.read_bytes() for path in root.iterdir()} == before diff --git a/inferencex-e2e/runners/test_mooncake_rdma_device.py b/inferencex-e2e/runners/test_mooncake_rdma_device.py new file mode 100644 index 0000000000..be91705349 --- /dev/null +++ b/inferencex-e2e/runners/test_mooncake_rdma_device.py @@ -0,0 +1,126 @@ +"""Exercise rail selection with Linux sysfs layouts, including DSXE names.""" +import json +import subprocess +from pathlib import Path + +LIB = Path(__file__).resolve().parents[1] / "benchmarks/benchmark_lib.sh" +PATCH_CONFIG = r""" +import json, sys +path, rail = sys.argv[1:] +if not rail.strip(): + raise SystemExit("Error: refusing to write empty Mooncake device_name") +with open(path) as handle: + config = json.load(handle) +config["device_name"] = rail +with open(path, "w") as handle: + json.dump(config, handle, indent=2) +written = json.load(open(path)) +if not str(written.get("device_name") or "").strip(): + raise SystemExit(f"Error: mooncake store config still has empty device_name: {written!r}") +print(written["device_name"]) +""" + + +def add_device( + root: Path, name: str, *, driver: str = "mlx5_core", + state: str = "4: ACTIVE", layer: str = "InfiniBand", +) -> None: + device = root / name + port = device / "ports/1" + port.mkdir(parents=True) + (port / "state").write_text(state + "\n") + (port / "link_layer").write_text(layer + "\n") + (device / "device").mkdir() + (device / "device/driver").symlink_to("/sys/bus/pci/drivers/" + driver) + + +def select(root: Path) -> subprocess.CompletedProcess[str]: + # Match kimik3-b300-mooncake.sh: the helper must resolve under --validation-only. + return subprocess.run( + [ + "bash", + "-ec", + ( + 'source "$1" --validation-only; select_mooncake_rdma_device "$2"; ' + 'printf "%s %s\n" "$MOONCAKE_RAIL" "$MC_GID_INDEX"' + ), + "bash", + str(LIB), + str(root), + ], + text=True, + capture_output=True, + timeout=5, + ) + + +def patch_config(path: Path, rail: str) -> subprocess.CompletedProcess[str]: + """Same write+verify contract as kimik3-b300-mooncake.sh.""" + return subprocess.run( + ["python3", "-", str(path), rail], + input=PATCH_CONFIG, + text=True, + capture_output=True, + timeout=5, + ) + + +def test_renamed_infiniband_device_skips_efa_and_down_port(tmp_path: Path) -> None: + add_device(tmp_path, "a_efa", driver="efa", layer="Unknown") + add_device(tmp_path, "ibp198s0f0", state="1: DOWN") + add_device(tmp_path, "ibp199s0f0") + result = select(tmp_path) + assert result.returncode == 0, result.stderr + assert result.stdout == "ibp199s0f0 0\n" + + +def test_roce_keeps_gid_three(tmp_path: Path) -> None: + add_device(tmp_path, "mlx5_0", state="1: DOWN", layer="Ethernet") + add_device(tmp_path, "mlx5_1", layer="Ethernet") + result = select(tmp_path) + assert result.returncode == 0, result.stderr + assert result.stdout == "mlx5_1 3\n" + + +def test_no_usable_rail_fails(tmp_path: Path) -> None: + add_device(tmp_path, "mlx5_0", state="1: DOWN") + add_device(tmp_path, "rdmap86s0", driver="efa", layer="Unknown") + result = select(tmp_path) + assert result.returncode != 0 + assert result.stdout == "" + + +def test_select_then_patch_replaces_empty_device_name(tmp_path: Path) -> None: + add_device(tmp_path, "ibp198s0f0") + config = tmp_path / "mooncake_store_config.json" + config.write_text(json.dumps({ + "mode": "embedded", + "protocol": "rdma", + "device_name": "", + "enable_offload": False, + })) + selected = select(tmp_path) + assert selected.returncode == 0, selected.stderr + rail, gid = selected.stdout.strip().split() + assert rail == "ibp198s0f0" + assert gid == "0" + patched = patch_config(config, rail) + assert patched.returncode == 0, patched.stderr + assert patched.stdout.strip() == "ibp198s0f0" + written = json.loads(config.read_text()) + assert written["device_name"] == "ibp198s0f0" + assert written["protocol"] == "rdma" + + +def test_empty_rail_cannot_silently_patch_device_name(tmp_path: Path) -> None: + config = tmp_path / "mooncake_store_config.json" + config.write_text(json.dumps({ + "mode": "embedded", + "protocol": "rdma", + "device_name": "", + })) + for empty in ("", " "): + result = patch_config(config, empty) + assert result.returncode != 0, result.stdout + assert "empty" in (result.stderr + result.stdout).lower() + assert json.loads(config.read_text())["device_name"] == ""