Skip to content
Original file line number Diff line number Diff line change
@@ -1,11 +1,11 @@
# MiniMax-M3 MXFP8 AgentX on H100 with vLLM EAGLE3 (the GQA draft head) and
# optional Mooncake DRAM KV offload. 26 GiB of weights per GPU live in host memory.
# optional Mooncake DRAM KV offload and an FP8 MSA indexer cache.
base:
schema: 2
name: minimaxm3-fp8-h100-vllm-agentic
model:
path: hf:MiniMaxAI/MiniMax-M3-MXFP8
container: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4
container: vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89
precision: fp8
resources:
gpu_type: h100
Expand All @@ -20,9 +20,7 @@ base:
engine:
type: vllm
connector: null
# The engine may take the full VLLM_ENGINE_READY_TIMEOUT_S to become ready.
# Lazy loads of the 31 shards from shared NFS ran at ~105 s/shard, past the
# script's 3600 s, so both windows are two hours.
# Allow up to two hours for model loading and engine readiness.
health_check:
interval_seconds: 10
max_attempts: 720
Expand All @@ -35,11 +33,12 @@ base:
served-model-name: MiniMaxAI/MiniMax-M3-MXFP8
tensor-parallel-size: 8
data-parallel-size: 1
gpu-memory-utilization: 0.90
cpu-offload-gb: 26
gpu-memory-utilization: 0.93
cpu-offload-gb: 5
attention-backend: TRITON_ATTN
safetensors-load-strategy: lazy
kv-cache-dtype: fp8
attention-config: '{"indexer_kv_dtype":"fp8"}'
block-size: 128
language-model-only: true
enable-prefix-caching: true
Expand All @@ -55,6 +54,7 @@ base:
env:
PYTHONNOUSERSITE: '1'
VLLM_ENGINE_READY_TIMEOUT_S: '7200'
VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0'
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
Expand Down Expand Up @@ -121,8 +121,8 @@ override_tp8_c5:
KV_OFFLOADING: none

# DRAM points offload KV to an embedded Mooncake store. Per rank: the host budget
# (1731 GB = 1612 GiB) less the 414 GiB checkpoint page cache, over TP8, less the
# 26 GiB CPU weight offload and the 4 GiB local buffer = 119 GB.
# (1676 GB = 1561 GiB) less the 414 GiB checkpoint page cache, over TP8, less the
# 5 GiB CPU weight offload and the 4 GiB local buffer = 134 GB.
override_tp8_c6_dram:
# The worker's Mooncake client and the master run the same pinned release.
setup_script: vllm-mooncake-0.3.11.sh
Expand All @@ -144,9 +144,9 @@ override_tp8_c6_dram:
store_config:
mode: embedded
metadata_server: P2PHANDSHAKE
global_segment_size: 119GB
global_segment_size: 134GB
local_buffer_size: 4GB
protocol: rdma
protocol: tcp
device_name: ''
enable_offload: false
roles:
Expand All @@ -164,7 +164,7 @@ override_tp8_c6_dram:
env:
CONC: '6'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '1731'
TOTAL_CPU_DRAM_GB: '1676'

override_tp8_c8_dram:
# The worker's Mooncake client and the master run the same pinned release.
Expand All @@ -187,9 +187,9 @@ override_tp8_c8_dram:
store_config:
mode: embedded
metadata_server: P2PHANDSHAKE
global_segment_size: 119GB
global_segment_size: 134GB
local_buffer_size: 4GB
protocol: rdma
protocol: tcp
device_name: ''
enable_offload: false
roles:
Expand All @@ -207,4 +207,4 @@ override_tp8_c8_dram:
env:
CONC: '8'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '1731'
TOTAL_CPU_DRAM_GB: '1676'
4 changes: 2 additions & 2 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5367,10 +5367,10 @@ qwen3.5-fp4-b200-trt-mtp:
srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml

minimaxm3-fp8-h100-vllm-agentic-mtp:
image: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4
image: vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89
model: MiniMaxAI/MiniMax-M3-MXFP8
model-prefix: minimaxm3
runner: cluster:h100-dgxc
runner: cluster:h100-dsxe
precision: fp8
framework: vllm
multinode: false
Expand Down
41 changes: 41 additions & 0 deletions inferencex-e2e/configs/runners.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,9 @@ labels:
h100:
- h100-cw_00
- h100-cw_01
- h100-dgxc-new_00
- h100-dgxc-new_01
- h100-dgxc-new_02
- h100-dgxc-slurm_00
- h100-dgxc-slurm_01
- h100-dgxc-slurm_02
Expand Down Expand Up @@ -156,6 +159,14 @@ labels:
cluster:h100-cw:
- h100-cw_00
- h100-cw_01
h100-dgxc-new:
- h100-dgxc-new_00
- h100-dgxc-new_01
- h100-dgxc-new_02
cluster:h100-dsxe:
- h100-dgxc-new_00
- h100-dgxc-new_01
- h100-dgxc-new_02
cluster:h100-dgxc:
- h100-dgxc-slurm_00
- h100-dgxc-slurm_01
Expand Down Expand Up @@ -326,6 +337,36 @@ clusters:
network-interface: ""
gpus-per-node-directive: false
single-node-exclusive: false
h100-dsxe:
gpus-per-node: 8
available-cpu-dram-mib: 1_998_848
arch: x86_64
models:
entries:
MiniMax-M3-MXFP8: {root: data-models, dir: MiniMax-M3-MXFP8}
scheduler: slurm
slurm:
partition: batch_1
account: benchmark
exclusive: false
gres: "gpu:{gpus}"
volumes:
hf-hub-cache: {path: /data/home/sa-gha-runner/gharunners/hf-hub-cache}
aiperf-cache: {path: /data/home/sa-gha-runner/gharunners/ai-perf-cache}
data-models: {path: /data/models}
squash:
dir: /data/home/sa-gha-runner/gharunners/squash
visibility: shared
import: compute
lock-timeout-s: 600
srt-slurm:
extra:
default_gpu_exporter:
container_image: nvcr.io#nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless
port: 9401
command: "dcgm-exporter --collect-interval=1000 --address :{port} -f /configs/dcgm-counters-noprof.csv"
network-interface: ""
single-node-models: staged
h100-dgxc:
gpus-per-node: 8
available-cpu-dram-mib: 2_063_837
Expand Down
36 changes: 36 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9180,3 +9180,39 @@
- "Update the SGLang ROCm image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 on this arm only. The other MI355X arms are left on their current tags."
- "No serving flag outside the HiCache block changes, and no other config key is touched. Each of the 19 matrix points resolves to exactly one recipe override, with no override left unused."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3611

- config-keys:
- minimaxm3-fp8-h100-vllm-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Update the MiniMax-M3 H100 vLLM image, enable the FP8 MSA indexer cache, adjust CPU weight offload and GPU memory settings, and resize the Mooncake store segments."
- "更新 MiniMax-M3 H100 的 vLLM 镜像,启用 FP8 MSA 索引缓存,调整 CPU 权重卸载与 GPU 内存设置,并调整 Mooncake 存储段大小。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620

- config-keys:
- minimaxm3-fp8-h100-vllm-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Route the MiniMax-M3 H100 AgentX config to cluster:h100-cw and align the Mooncake DRAM store size with that cluster's host-memory budget."
- "将 MiniMax-M3 H100 AgentX 配置切换到 cluster:h100-cw,并根据该集群的主机内存预算调整 Mooncake DRAM 存储段大小。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620

- config-keys:
- minimaxm3-fp8-h100-vllm-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Route the MiniMax-M3 H100 AgentX config to the DSXE runner group and add its Slurm cluster profile."
- "将 MiniMax-M3 H100 AgentX 配置切换到 DSXE 运行器组,并添加对应的 Slurm 集群配置。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620

- config-keys:
- minimaxm3-fp8-h100-vllm-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Use TCP transport for the single-node Mooncake DRAM store in the concurrency 6 and 8 variants."
- "将并发数为 6 和 8 的单节点 Mooncake DRAM 存储变体切换为 TCP 传输。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620
Loading