From 1f066813518bdfdfbe8260a306ac6405c08c7f8d Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Wed, 30 Sep 2026 15:52:42 -0700 Subject: [PATCH 01/12] feat(config): enable MiniMax-M3 H100 FP8 indexer cache MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 启用 MiniMax-M3 H100 FP8 索引缓存,更新 vLLM 镜像并调整内存与 Mooncake 设置。 --- .../minimaxm3/vllm/h100-fp8-mtp/agentic.yaml | 16 +++++++++------- inferencex-e2e/configs/nvidia-master.yaml | 2 +- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 3 files changed, 19 insertions(+), 8 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml index dc2aa8f058..d212a5afb5 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml @@ -1,11 +1,11 @@ # MiniMax-M3 MXFP8 AgentX on H100 with vLLM EAGLE3 (the GQA draft head) and -# optional Mooncake DRAM KV offload. 26 GiB of weights per GPU live in host memory. +# optional Mooncake DRAM KV offload and an FP8 MSA indexer cache. base: schema: 2 name: minimaxm3-fp8-h100-vllm-agentic model: path: hf:MiniMaxAI/MiniMax-M3-MXFP8 - container: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4 + container: vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89 precision: fp8 resources: gpu_type: h100 @@ -35,11 +35,12 @@ base: served-model-name: MiniMaxAI/MiniMax-M3-MXFP8 tensor-parallel-size: 8 data-parallel-size: 1 - gpu-memory-utilization: 0.90 - cpu-offload-gb: 26 + gpu-memory-utilization: 0.93 + cpu-offload-gb: 5 attention-backend: TRITON_ATTN safetensors-load-strategy: lazy kv-cache-dtype: fp8 + attention-config: '{"indexer_kv_dtype":"fp8"}' block-size: 128 language-model-only: true enable-prefix-caching: true @@ -55,6 +56,7 @@ base: env: PYTHONNOUSERSITE: '1' VLLM_ENGINE_READY_TIMEOUT_S: '7200' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' benchmark: type: custom command: bash /infmax-workspace/benchmarks/srt_agentic.sh @@ -122,7 +124,7 @@ override_tp8_c5: # DRAM points offload KV to an embedded Mooncake store. Per rank: the host budget # (1731 GB = 1612 GiB) less the 414 GiB checkpoint page cache, over TP8, less the -# 26 GiB CPU weight offload and the 4 GiB local buffer = 119 GB. +# 5 GiB CPU weight offload and the 4 GiB local buffer = 140 GB. override_tp8_c6_dram: # The worker's Mooncake client and the master run the same pinned release. setup_script: vllm-mooncake-0.3.11.sh @@ -144,7 +146,7 @@ override_tp8_c6_dram: store_config: mode: embedded metadata_server: P2PHANDSHAKE - global_segment_size: 119GB + global_segment_size: 140GB local_buffer_size: 4GB protocol: rdma device_name: '' @@ -187,7 +189,7 @@ override_tp8_c8_dram: store_config: mode: embedded metadata_server: P2PHANDSHAKE - global_segment_size: 119GB + global_segment_size: 140GB local_buffer_size: 4GB protocol: rdma device_name: '' diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 63dcf8fff8..a009eb2b40 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -5367,7 +5367,7 @@ qwen3.5-fp4-b200-trt-mtp: srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml minimaxm3-fp8-h100-vllm-agentic-mtp: - image: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4 + image: vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89 model: MiniMaxAI/MiniMax-M3-MXFP8 model-prefix: minimaxm3 runner: cluster:h100-dgxc diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index b415cf56a5..0319275d67 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9158,3 +9158,12 @@ - "Port the DeepSeek-V4-Pro-0813 MI355X ATOM 1P1D AgentX config from the legacy amd_utils path to a native srt-slurm recipe (benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml), one override variant per point. Server flags, env, AToMesh routing policies and decode CUDA-graph capture sizes match the legacy server_atom.sh for every tier: TP8 at concurrency 1 and 16, DP attention with prefill TBO at 64 and 128, and DP attention with ATOM's in-process LMCache CPU offload (lmcache_offload, 187 GB per prefill rank) at 256. Golden acceptance 3.01 for DSpark with three draft tokens is now injected by the srt-slurm path, the same value the legacy models_atom.yaml hardcoded." - "The LMCache tier uses srt-slurm's extra-kv-connectors (NVIDIA/srt-slurm#507, in v2.36.0) to add lmcache_offload next to the generated Mooncake connector." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3543 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Update the MiniMax-M3 H100 vLLM image, enable the FP8 MSA indexer cache, adjust CPU weight offload and GPU memory settings, and resize the Mooncake store segments." + - "更新 MiniMax-M3 H100 的 vLLM 镜像,启用 FP8 MSA 索引缓存,调整 CPU 权重卸载与 GPU 内存设置,并调整 Mooncake 存储段大小。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX From 14a1e0081c0b5df8ac1bb29a80e7a77f6beffb35 Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Wed, 30 Sep 2026 15:53:46 -0700 Subject: [PATCH 02/12] chore(config): link MiniMax-M3 H100 changelog entry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 补充 MiniMax-M3 H100 变更日志条目的 PR 链接。 --- inferencex-e2e/perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 0319275d67..2c6393edfe 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9166,4 +9166,4 @@ description: - "Update the MiniMax-M3 H100 vLLM image, enable the FP8 MSA indexer cache, adjust CPU weight offload and GPU memory settings, and resize the Mooncake store segments." - "更新 MiniMax-M3 H100 的 vLLM 镜像,启用 FP8 MSA 索引缓存,调整 CPU 权重卸载与 GPU 内存设置,并调整 Mooncake 存储段大小。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 From 995dce32ef14f63000aa09b9a0be5c6b5fdc7c51 Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Wed, 30 Sep 2026 23:55:14 -0700 Subject: [PATCH 03/12] fix(config): route MiniMax-M3 H100 sweep to DSXE runners MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 MiniMax-M3 H100 扫描任务切换到 DSXE 运行器,并添加对应的 Slurm 集群配置。 --- .github/workflows/benchmark-tmpl.yml | 2 ++ inferencex-e2e/configs/nvidia-master.yaml | 2 +- inferencex-e2e/configs/runners.yaml | 40 +++++++++++++++++++++++ inferencex-e2e/perf-changelog.yaml | 9 +++++ 4 files changed, 52 insertions(+), 1 deletion(-) diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index 35548863fe..75c79fb14e 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -164,8 +164,10 @@ permissions: jobs: benchmark: + # DSXE runners currently advertise h100-dgxc-new without a cluster label. runs-on: >- ${{ fromJSON( + inputs.runner == 'cluster:h100-dsxe' && '["self-hosted","h100-dgxc-new"]' || vars.PRIORITY_SCHEDULER_ENABLED == 'true' && ( vars.NODE_SLOT_SCHEDULER_ENABLED == 'true' && diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index baadd0ccd1..a0916b9ed1 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -5370,7 +5370,7 @@ minimaxm3-fp8-h100-vllm-agentic-mtp: image: vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89 model: MiniMaxAI/MiniMax-M3-MXFP8 model-prefix: minimaxm3 - runner: cluster:h100-cw + runner: cluster:h100-dsxe precision: fp8 framework: vllm multinode: false diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index 8036d3de69..b4f2773b2b 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -2,6 +2,9 @@ labels: h100: - h100-cw_00 - h100-cw_01 + - h100-dgxc-new_00 + - h100-dgxc-new_01 + - h100-dgxc-new_02 - h100-dgxc-slurm_00 - h100-dgxc-slurm_01 - h100-dgxc-slurm_02 @@ -156,6 +159,14 @@ labels: cluster:h100-cw: - h100-cw_00 - h100-cw_01 + h100-dgxc-new: + - h100-dgxc-new_00 + - h100-dgxc-new_01 + - h100-dgxc-new_02 + cluster:h100-dsxe: + - h100-dgxc-new_00 + - h100-dgxc-new_01 + - h100-dgxc-new_02 cluster:h100-dgxc: - h100-dgxc-slurm_00 - h100-dgxc-slurm_01 @@ -326,6 +337,35 @@ clusters: network-interface: "" gpus-per-node-directive: false single-node-exclusive: false + h100-dsxe: + gpus-per-node: 8 + available-cpu-dram-mib: 1_998_848 + arch: x86_64 + models: + entries: + MiniMax-M3-MXFP8: {root: data-models, dir: MiniMax-M3-MXFP8} + scheduler: slurm + slurm: + partition: batch_1 + account: benchmark + exclusive: false + gres: "gpu:{gpus}" + volumes: + hf-hub-cache: {path: /data/home/sa-gha-runner/gharunners/hf-hub-cache} + aiperf-cache: {path: /data/home/sa-gha-runner/gharunners/ai-perf-cache} + data-models: {path: /data/models} + squash: + dir: /data/home/sa-gha-runner/gharunners/squash + visibility: shared + import: compute + lock-timeout-s: 600 + srt-slurm: + extra: + default_gpu_exporter: + container_image: nvcr.io#nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless + port: 9401 + command: "dcgm-exporter --collect-interval=1000 --address :{port} -f /configs/dcgm-counters-noprof.csv" + network-interface: "" h100-dgxc: gpus-per-node: 8 available-cpu-dram-mib: 2_063_837 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 7cbd3d336c..ae84c99347 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9186,3 +9186,12 @@ - "Route the MiniMax-M3 H100 AgentX config to cluster:h100-cw and align the Mooncake DRAM store size with that cluster's host-memory budget." - "将 MiniMax-M3 H100 AgentX 配置切换到 cluster:h100-cw,并根据该集群的主机内存预算调整 Mooncake DRAM 存储段大小。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Route the MiniMax-M3 H100 AgentX config to the DSXE runner group and add its Slurm cluster profile." + - "将 MiniMax-M3 H100 AgentX 配置切换到 DSXE 运行器组,并添加对应的 Slurm 集群配置。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 From f8ef6a7d65d4a2d2a9258e2e667ada239b2e0c85 Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Thu, 1 Oct 2026 00:02:12 -0700 Subject: [PATCH 04/12] fix(config): use staged MiniMax-M3 checkpoint on DSXE MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 让 DSXE 上的 MiniMax-M3 单节点任务读取已暂存的模型权重。 --- inferencex-e2e/configs/runners.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index b4f2773b2b..96da78e36a 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -366,6 +366,7 @@ clusters: port: 9401 command: "dcgm-exporter --collect-interval=1000 --address :{port} -f /configs/dcgm-counters-noprof.csv" network-interface: "" + single-node-models: staged h100-dgxc: gpus-per-node: 8 available-cpu-dram-mib: 2_063_837 From 051c37f18f0a72b8b879d0e268b1ad1bd58db9a4 Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Thu, 1 Oct 2026 05:41:25 -0700 Subject: [PATCH 05/12] fix(config): use TCP for MiniMax-M3 Mooncake DRAM store MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 MiniMax-M3 的 Mooncake DRAM 存储切换为 TCP 传输,避免在单节点上注册大型 RDMA 内存段。 --- .../minimaxm3/vllm/h100-fp8-mtp/agentic.yaml | 4 ++-- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml index 0a947e7c9a..8411f65384 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml @@ -146,7 +146,7 @@ override_tp8_c6_dram: metadata_server: P2PHANDSHAKE global_segment_size: 134GB local_buffer_size: 4GB - protocol: rdma + protocol: tcp device_name: '' enable_offload: false roles: @@ -189,7 +189,7 @@ override_tp8_c8_dram: metadata_server: P2PHANDSHAKE global_segment_size: 134GB local_buffer_size: 4GB - protocol: rdma + protocol: tcp device_name: '' enable_offload: false roles: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index ae84c99347..5c03d88d02 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9195,3 +9195,12 @@ - "Route the MiniMax-M3 H100 AgentX config to the DSXE runner group and add its Slurm cluster profile." - "将 MiniMax-M3 H100 AgentX 配置切换到 DSXE 运行器组,并添加对应的 Slurm 集群配置。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Use TCP transport for the single-node Mooncake DRAM store in the concurrency 6 and 8 variants." + - "将并发数为 6 和 8 的单节点 Mooncake DRAM 存储变体切换为 TCP 传输。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 From 0e36fba0a1e5224b06bf43d7ad4f27ba40af701b Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 1 Oct 2026 10:43:18 -0500 Subject: [PATCH 06/12] fix(ci): use cluster labels and priority leases for H100 DSXE --- .github/workflows/benchmark-tmpl.yml | 2 -- 1 file changed, 2 deletions(-) diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index 75c79fb14e..35548863fe 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -164,10 +164,8 @@ permissions: jobs: benchmark: - # DSXE runners currently advertise h100-dgxc-new without a cluster label. runs-on: >- ${{ fromJSON( - inputs.runner == 'cluster:h100-dsxe' && '["self-hosted","h100-dgxc-new"]' || vars.PRIORITY_SCHEDULER_ENABLED == 'true' && ( vars.NODE_SLOT_SCHEDULER_ENABLED == 'true' && From f1050134ef61888961cb2ddec357fd7419cf5b4a Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Thu, 1 Oct 2026 11:42:53 -0700 Subject: [PATCH 07/12] fix(config): reserve GPU memory for Mooncake TCP MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 Mooncake TCP 传输预留 GPU 内存。 --- .../minimaxm3/vllm/h100-fp8-mtp/agentic.yaml | 4 ++++ inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 2 files changed, 13 insertions(+) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml index 8411f65384..fe826f4336 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml @@ -152,6 +152,8 @@ override_tp8_c6_dram: roles: agg: args: + # Leave room for Mooncake TCP CUDA contexts on GPU 0 during KV transfers. + gpu-memory-utilization: 0.85 max-num-seqs: 12 max-cudagraph-capture-size: 48 kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true}}' @@ -195,6 +197,8 @@ override_tp8_c8_dram: roles: agg: args: + # Leave room for Mooncake TCP CUDA contexts on GPU 0 during KV transfers. + gpu-memory-utilization: 0.85 max-num-seqs: 16 max-cudagraph-capture-size: 64 kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true}}' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index c1cb090e40..1b8e1e50c8 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9216,3 +9216,12 @@ - "Use TCP transport for the single-node Mooncake DRAM store in the concurrency 6 and 8 variants." - "将并发数为 6 和 8 的单节点 Mooncake DRAM 存储变体切换为 TCP 传输。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Reserve GPU memory for Mooncake TCP transfers in the MiniMax-M3 H100 DRAM variants." + - "为 MiniMax-M3 H100 DRAM 变体的 Mooncake TCP 传输预留 GPU 内存。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 From 4467aab4f78b07b23d4097cc457c4ecf23a3bd8c Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Thu, 1 Oct 2026 16:25:07 -0700 Subject: [PATCH 08/12] fix(config): align MiniMax DRAM context limits MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 MiniMax DRAM 变体统一服务与重放上下文上限。 --- .../minimaxm3/vllm/h100-fp8-mtp/agentic.yaml | 4 ++++ inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 2 files changed, 13 insertions(+) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml index fe826f4336..835916b076 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml @@ -154,6 +154,7 @@ override_tp8_c6_dram: args: # Leave room for Mooncake TCP CUDA contexts on GPU 0 during KV transfers. gpu-memory-utilization: 0.85 + max-model-len: 655360 max-num-seqs: 12 max-cudagraph-capture-size: 48 kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true}}' @@ -167,6 +168,7 @@ override_tp8_c6_dram: CONC: '6' KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '1676' + AIPERF_MAX_CONTEXT_LENGTH: '655360' override_tp8_c8_dram: # The worker's Mooncake client and the master run the same pinned release. @@ -199,6 +201,7 @@ override_tp8_c8_dram: args: # Leave room for Mooncake TCP CUDA contexts on GPU 0 during KV transfers. gpu-memory-utilization: 0.85 + max-model-len: 655360 max-num-seqs: 16 max-cudagraph-capture-size: 64 kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true}}' @@ -212,3 +215,4 @@ override_tp8_c8_dram: CONC: '8' KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '1676' + AIPERF_MAX_CONTEXT_LENGTH: '655360' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 1b8e1e50c8..99ca7b9db7 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9225,3 +9225,12 @@ - "Reserve GPU memory for Mooncake TCP transfers in the MiniMax-M3 H100 DRAM variants." - "为 MiniMax-M3 H100 DRAM 变体的 Mooncake TCP 传输预留 GPU 内存。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Set matching 640K service and replay context limits for the MiniMax-M3 H100 Mooncake DRAM variants." + - "为 MiniMax-M3 H100 Mooncake DRAM 变体设置一致的 640K 服务与重放上下文上限。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 From 439616c1bbab5d1c626df5bf5b83dbe7aeaac8b3 Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Thu, 1 Oct 2026 19:25:16 -0700 Subject: [PATCH 09/12] fix(config): reuse MiniMax Mooncake TCP connections MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 在 MiniMax Mooncake DRAM 变体中复用 TCP 连接,避免短连接重复创建。 --- .../minimaxm3/vllm/h100-fp8-mtp/agentic.yaml | 2 ++ inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 2 files changed, 11 insertions(+) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml index 835916b076..0110f2a780 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml @@ -162,6 +162,7 @@ override_tp8_c6_dram: PYTHONHASHSEED: '0' MC_SLICE_SIZE: '1048576' MC_WORKERS_PER_CTX: '4' + MC_TCP_ENABLE_CONNECTION_POOL: '1' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' benchmark: env: @@ -209,6 +210,7 @@ override_tp8_c8_dram: PYTHONHASHSEED: '0' MC_SLICE_SIZE: '1048576' MC_WORKERS_PER_CTX: '4' + MC_TCP_ENABLE_CONNECTION_POOL: '1' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' benchmark: env: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 4c00cefcfa..3b9a4b3112 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9256,3 +9256,12 @@ - "Set matching 640K service and replay context limits for the MiniMax-M3 H100 Mooncake DRAM variants." - "为 MiniMax-M3 H100 Mooncake DRAM 变体设置一致的 640K 服务与重放上下文上限。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Reuse Mooncake TCP connections in the MiniMax-M3 H100 DRAM variants." + - "在 MiniMax-M3 H100 DRAM 变体中复用 Mooncake TCP 连接。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 From 5f09e094577842a6925c578df7785293bab65de8 Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Fri, 2 Oct 2026 01:02:08 -0700 Subject: [PATCH 10/12] fix: increase Mooncake transfer slices for H100 DRAM runs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Set larger TCP transfer slices for the MiniMax-M3 H100 DRAM variants and use a separate live replay error threshold. Keep final result validation unchanged. 中文:增大 MiniMax-M3 H100 DRAM 变体的 Mooncake TCP 传输分片,并单独设置重放过程中的实时错误阈值。最终结果校验保持不变。 --- .../minimaxm3/vllm/h100-fp8-mtp/agentic.yaml | 6 ++++-- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 2 files changed, 13 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml index 0110f2a780..6841db940b 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml @@ -160,7 +160,7 @@ override_tp8_c6_dram: kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true}}' env: PYTHONHASHSEED: '0' - MC_SLICE_SIZE: '1048576' + MC_SLICE_SIZE: '8388608' MC_WORKERS_PER_CTX: '4' MC_TCP_ENABLE_CONNECTION_POOL: '1' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' @@ -170,6 +170,7 @@ override_tp8_c6_dram: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '1676' AIPERF_MAX_CONTEXT_LENGTH: '655360' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' override_tp8_c8_dram: # The worker's Mooncake client and the master run the same pinned release. @@ -208,7 +209,7 @@ override_tp8_c8_dram: kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true}}' env: PYTHONHASHSEED: '0' - MC_SLICE_SIZE: '1048576' + MC_SLICE_SIZE: '8388608' MC_WORKERS_PER_CTX: '4' MC_TCP_ENABLE_CONNECTION_POOL: '1' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' @@ -218,3 +219,4 @@ override_tp8_c8_dram: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '1676' AIPERF_MAX_CONTEXT_LENGTH: '655360' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 64b04b8944..f5a1f42362 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9274,3 +9274,12 @@ - "Reuse Mooncake TCP connections in the MiniMax-M3 H100 DRAM variants." - "在 MiniMax-M3 H100 DRAM 变体中复用 Mooncake TCP 连接。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Use larger Mooncake transfer slices and a separate live replay error threshold for the MiniMax-M3 H100 DRAM variants; retain the final result validation threshold." + - "为 MiniMax-M3 H100 DRAM 变体使用更大的 Mooncake 传输分片,并单独设置重放过程中的实时错误阈值;最终结果校验阈值保持不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 From 1bebab03cf231cab3b1f370c78bcd343934eb0d4 Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Fri, 2 Oct 2026 04:27:16 -0700 Subject: [PATCH 11/12] fix: bound Mooncake TCP lanes for H100 DRAM variants MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 H100 DRAM 变体使用有上限的 Mooncake TCP 连接通道。 --- .../srt-slurm-recipes/configs/vllm-mooncake-0.3.13.sh | 9 +++++++++ .../minimaxm3/vllm/h100-fp8-mtp/agentic.yaml | 10 ++++++---- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 3 files changed, 24 insertions(+), 4 deletions(-) create mode 100755 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/vllm-mooncake-0.3.13.sh diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/vllm-mooncake-0.3.13.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/vllm-mooncake-0.3.13.sh new file mode 100755 index 0000000000..b4301604da --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/vllm-mooncake-0.3.13.sh @@ -0,0 +1,9 @@ +#!/usr/bin/env bash +# Pin the worker's Mooncake store client to the release the master runs. +set -eo pipefail +pip_install=(python3 -m pip install) +if python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages; then + pip_install+=(--break-system-packages) +fi +"${pip_install[@]}" --quiet --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.13.post1 +python3 -c "from mooncake.store import MooncakeDistributedStore" >/dev/null diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml index 6841db940b..8b2eaba115 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml @@ -125,7 +125,7 @@ override_tp8_c5: # 5 GiB CPU weight offload and the 4 GiB local buffer = 134 GB. override_tp8_c6_dram: # The worker's Mooncake client and the master run the same pinned release. - setup_script: vllm-mooncake-0.3.11.sh + setup_script: vllm-mooncake-0.3.13.sh services: - name: mooncake-master type: mooncake-master @@ -136,7 +136,7 @@ override_tp8_c6_dram: preamble: >- pip_install=(python3 -m pip install); python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages && pip_install+=(--break-system-packages); - "${pip_install[@]}" --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.11.post1 + "${pip_install[@]}" --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.13.post1 args: - '--eviction_high_watermark_ratio=0.80' - '--eviction_ratio=0.10' @@ -163,6 +163,7 @@ override_tp8_c6_dram: MC_SLICE_SIZE: '8388608' MC_WORKERS_PER_CTX: '4' MC_TCP_ENABLE_CONNECTION_POOL: '1' + MC_TCP_LANES_PER_PEER: '4' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' benchmark: env: @@ -174,7 +175,7 @@ override_tp8_c6_dram: override_tp8_c8_dram: # The worker's Mooncake client and the master run the same pinned release. - setup_script: vllm-mooncake-0.3.11.sh + setup_script: vllm-mooncake-0.3.13.sh services: - name: mooncake-master type: mooncake-master @@ -185,7 +186,7 @@ override_tp8_c8_dram: preamble: >- pip_install=(python3 -m pip install); python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages && pip_install+=(--break-system-packages); - "${pip_install[@]}" --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.11.post1 + "${pip_install[@]}" --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.13.post1 args: - '--eviction_high_watermark_ratio=0.80' - '--eviction_ratio=0.10' @@ -212,6 +213,7 @@ override_tp8_c8_dram: MC_SLICE_SIZE: '8388608' MC_WORKERS_PER_CTX: '4' MC_TCP_ENABLE_CONNECTION_POOL: '1' + MC_TCP_LANES_PER_PEER: '4' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' benchmark: env: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index f5a1f42362..fd7f462de9 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9283,3 +9283,12 @@ - "Use larger Mooncake transfer slices and a separate live replay error threshold for the MiniMax-M3 H100 DRAM variants; retain the final result validation threshold." - "为 MiniMax-M3 H100 DRAM 变体使用更大的 Mooncake 传输分片,并单独设置重放过程中的实时错误阈值;最终结果校验阈值保持不变。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Pin the H100 DRAM variants to Mooncake 0.3.13.post1 with bounded TCP connection lanes." + - "将 H100 DRAM 变体固定为 Mooncake 0.3.13.post1,并使用有上限的 TCP 连接通道。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 From 0cfb6ac1db447de975df53407a797bc7d703ea17 Mon Sep 17 00:00:00 2001 From: Rohit Pujar Nagraj Date: Fri, 2 Oct 2026 07:54:05 -0700 Subject: [PATCH 12/12] fix: tune Mooncake TCP lane admission for H100 DRAM variants MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 调整 H100 DRAM 变体的 Mooncake TCP 连接通道接纳参数。 --- .../minimaxm3/vllm/h100-fp8-mtp/agentic.yaml | 12 ++++++++++-- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 2 files changed, 19 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml index 8b2eaba115..f4e8d21986 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml @@ -161,9 +161,13 @@ override_tp8_c6_dram: env: PYTHONHASHSEED: '0' MC_SLICE_SIZE: '8388608' + MC_TCP_SLICE_SIZE: '8388608' MC_WORKERS_PER_CTX: '4' MC_TCP_ENABLE_CONNECTION_POOL: '1' - MC_TCP_LANES_PER_PEER: '4' + MC_TCP_LANES_PER_PEER: '16' + MC_TCP_MAX_QUEUED_TRANSFERS_PER_PEER: '65535' + MC_TCP_MAX_PENDING_ADMISSIONS_PER_PEER: '65535' + MC_TCP_ADMISSION_TIMEOUT_MS: '60000' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' benchmark: env: @@ -211,9 +215,13 @@ override_tp8_c8_dram: env: PYTHONHASHSEED: '0' MC_SLICE_SIZE: '8388608' + MC_TCP_SLICE_SIZE: '8388608' MC_WORKERS_PER_CTX: '4' MC_TCP_ENABLE_CONNECTION_POOL: '1' - MC_TCP_LANES_PER_PEER: '4' + MC_TCP_LANES_PER_PEER: '16' + MC_TCP_MAX_QUEUED_TRANSFERS_PER_PEER: '65535' + MC_TCP_MAX_PENDING_ADMISSIONS_PER_PEER: '65535' + MC_TCP_ADMISSION_TIMEOUT_MS: '60000' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' benchmark: env: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index fd7f462de9..21e1581558 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9292,3 +9292,12 @@ - "Pin the H100 DRAM variants to Mooncake 0.3.13.post1 with bounded TCP connection lanes." - "将 H100 DRAM 变体固定为 Mooncake 0.3.13.post1,并使用有上限的 TCP 连接通道。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Set TCP transfer slice size and bounded lane admission settings for the MiniMax-M3 H100 DRAM variants." + - "为 MiniMax-M3 H100 DRAM 变体设置 TCP 传输分片大小和有上限的连接通道接纳参数。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620