diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/vllm-mooncake-0.3.13.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/vllm-mooncake-0.3.13.sh new file mode 100755 index 0000000000..b4301604da --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/vllm-mooncake-0.3.13.sh @@ -0,0 +1,9 @@ +#!/usr/bin/env bash +# Pin the worker's Mooncake store client to the release the master runs. +set -eo pipefail +pip_install=(python3 -m pip install) +if python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages; then + pip_install+=(--break-system-packages) +fi +"${pip_install[@]}" --quiet --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.13.post1 +python3 -c "from mooncake.store import MooncakeDistributedStore" >/dev/null diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml index dc2aa8f058..f4e8d21986 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/h100-fp8-mtp/agentic.yaml @@ -1,11 +1,11 @@ # MiniMax-M3 MXFP8 AgentX on H100 with vLLM EAGLE3 (the GQA draft head) and -# optional Mooncake DRAM KV offload. 26 GiB of weights per GPU live in host memory. +# optional Mooncake DRAM KV offload and an FP8 MSA indexer cache. base: schema: 2 name: minimaxm3-fp8-h100-vllm-agentic model: path: hf:MiniMaxAI/MiniMax-M3-MXFP8 - container: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4 + container: vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89 precision: fp8 resources: gpu_type: h100 @@ -20,9 +20,7 @@ base: engine: type: vllm connector: null - # The engine may take the full VLLM_ENGINE_READY_TIMEOUT_S to become ready. - # Lazy loads of the 31 shards from shared NFS ran at ~105 s/shard, past the - # script's 3600 s, so both windows are two hours. + # Allow up to two hours for model loading and engine readiness. health_check: interval_seconds: 10 max_attempts: 720 @@ -35,11 +33,12 @@ base: served-model-name: MiniMaxAI/MiniMax-M3-MXFP8 tensor-parallel-size: 8 data-parallel-size: 1 - gpu-memory-utilization: 0.90 - cpu-offload-gb: 26 + gpu-memory-utilization: 0.93 + cpu-offload-gb: 5 attention-backend: TRITON_ATTN safetensors-load-strategy: lazy kv-cache-dtype: fp8 + attention-config: '{"indexer_kv_dtype":"fp8"}' block-size: 128 language-model-only: true enable-prefix-caching: true @@ -55,6 +54,7 @@ base: env: PYTHONNOUSERSITE: '1' VLLM_ENGINE_READY_TIMEOUT_S: '7200' + VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' benchmark: type: custom command: bash /infmax-workspace/benchmarks/srt_agentic.sh @@ -121,11 +121,11 @@ override_tp8_c5: KV_OFFLOADING: none # DRAM points offload KV to an embedded Mooncake store. Per rank: the host budget -# (1731 GB = 1612 GiB) less the 414 GiB checkpoint page cache, over TP8, less the -# 26 GiB CPU weight offload and the 4 GiB local buffer = 119 GB. +# (1676 GB = 1561 GiB) less the 414 GiB checkpoint page cache, over TP8, less the +# 5 GiB CPU weight offload and the 4 GiB local buffer = 134 GB. override_tp8_c6_dram: # The worker's Mooncake client and the master run the same pinned release. - setup_script: vllm-mooncake-0.3.11.sh + setup_script: vllm-mooncake-0.3.13.sh services: - name: mooncake-master type: mooncake-master @@ -136,7 +136,7 @@ override_tp8_c6_dram: preamble: >- pip_install=(python3 -m pip install); python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages && pip_install+=(--break-system-packages); - "${pip_install[@]}" --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.11.post1 + "${pip_install[@]}" --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.13.post1 args: - '--eviction_high_watermark_ratio=0.80' - '--eviction_ratio=0.10' @@ -144,31 +144,42 @@ override_tp8_c6_dram: store_config: mode: embedded metadata_server: P2PHANDSHAKE - global_segment_size: 119GB + global_segment_size: 134GB local_buffer_size: 4GB - protocol: rdma + protocol: tcp device_name: '' enable_offload: false roles: agg: args: + # Leave room for Mooncake TCP CUDA contexts on GPU 0 during KV transfers. + gpu-memory-utilization: 0.85 + max-model-len: 655360 max-num-seqs: 12 max-cudagraph-capture-size: 48 kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true}}' env: PYTHONHASHSEED: '0' - MC_SLICE_SIZE: '1048576' + MC_SLICE_SIZE: '8388608' + MC_TCP_SLICE_SIZE: '8388608' MC_WORKERS_PER_CTX: '4' + MC_TCP_ENABLE_CONNECTION_POOL: '1' + MC_TCP_LANES_PER_PEER: '16' + MC_TCP_MAX_QUEUED_TRANSFERS_PER_PEER: '65535' + MC_TCP_MAX_PENDING_ADMISSIONS_PER_PEER: '65535' + MC_TCP_ADMISSION_TIMEOUT_MS: '60000' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' benchmark: env: CONC: '6' KV_OFFLOADING: dram - TOTAL_CPU_DRAM_GB: '1731' + TOTAL_CPU_DRAM_GB: '1676' + AIPERF_MAX_CONTEXT_LENGTH: '655360' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' override_tp8_c8_dram: # The worker's Mooncake client and the master run the same pinned release. - setup_script: vllm-mooncake-0.3.11.sh + setup_script: vllm-mooncake-0.3.13.sh services: - name: mooncake-master type: mooncake-master @@ -179,7 +190,7 @@ override_tp8_c8_dram: preamble: >- pip_install=(python3 -m pip install); python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages && pip_install+=(--break-system-packages); - "${pip_install[@]}" --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.11.post1 + "${pip_install[@]}" --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.13.post1 args: - '--eviction_high_watermark_ratio=0.80' - '--eviction_ratio=0.10' @@ -187,24 +198,35 @@ override_tp8_c8_dram: store_config: mode: embedded metadata_server: P2PHANDSHAKE - global_segment_size: 119GB + global_segment_size: 134GB local_buffer_size: 4GB - protocol: rdma + protocol: tcp device_name: '' enable_offload: false roles: agg: args: + # Leave room for Mooncake TCP CUDA contexts on GPU 0 during KV transfers. + gpu-memory-utilization: 0.85 + max-model-len: 655360 max-num-seqs: 16 max-cudagraph-capture-size: 64 kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true}}' env: PYTHONHASHSEED: '0' - MC_SLICE_SIZE: '1048576' + MC_SLICE_SIZE: '8388608' + MC_TCP_SLICE_SIZE: '8388608' MC_WORKERS_PER_CTX: '4' + MC_TCP_ENABLE_CONNECTION_POOL: '1' + MC_TCP_LANES_PER_PEER: '16' + MC_TCP_MAX_QUEUED_TRANSFERS_PER_PEER: '65535' + MC_TCP_MAX_PENDING_ADMISSIONS_PER_PEER: '65535' + MC_TCP_ADMISSION_TIMEOUT_MS: '60000' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' benchmark: env: CONC: '8' KV_OFFLOADING: dram - TOTAL_CPU_DRAM_GB: '1731' + TOTAL_CPU_DRAM_GB: '1676' + AIPERF_MAX_CONTEXT_LENGTH: '655360' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25' diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 052eb90726..fdf2e81e66 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -5367,10 +5367,10 @@ qwen3.5-fp4-b200-trt-mtp: srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml minimaxm3-fp8-h100-vllm-agentic-mtp: - image: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4 + image: vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89 model: MiniMaxAI/MiniMax-M3-MXFP8 model-prefix: minimaxm3 - runner: cluster:h100-dgxc + runner: cluster:h100-dsxe precision: fp8 framework: vllm multinode: false diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index 7cd6fa2651..0c5ee3add3 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -2,6 +2,9 @@ labels: h100: - h100-cw_00 - h100-cw_01 + - h100-dgxc-new_00 + - h100-dgxc-new_01 + - h100-dgxc-new_02 - h100-dgxc-slurm_00 - h100-dgxc-slurm_01 - h100-dgxc-slurm_02 @@ -174,6 +177,14 @@ labels: cluster:h100-cw: - h100-cw_00 - h100-cw_01 + h100-dgxc-new: + - h100-dgxc-new_00 + - h100-dgxc-new_01 + - h100-dgxc-new_02 + cluster:h100-dsxe: + - h100-dgxc-new_00 + - h100-dgxc-new_01 + - h100-dgxc-new_02 cluster:h100-dgxc: - h100-dgxc-slurm_00 - h100-dgxc-slurm_01 @@ -399,6 +410,36 @@ clusters: network-interface: "" gpus-per-node-directive: false single-node-exclusive: false + h100-dsxe: + gpus-per-node: 8 + available-cpu-dram-mib: 1_998_848 + arch: x86_64 + models: + entries: + MiniMax-M3-MXFP8: {root: data-models, dir: MiniMax-M3-MXFP8} + scheduler: slurm + slurm: + partition: batch_1 + account: benchmark + exclusive: false + gres: "gpu:{gpus}" + volumes: + hf-hub-cache: {path: /data/home/sa-gha-runner/gharunners/hf-hub-cache} + aiperf-cache: {path: /data/home/sa-gha-runner/gharunners/ai-perf-cache} + data-models: {path: /data/models} + squash: + dir: /data/home/sa-gha-runner/gharunners/squash + visibility: shared + import: compute + lock-timeout-s: 600 + srt-slurm: + extra: + default_gpu_exporter: + container_image: nvcr.io#nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless + port: 9401 + command: "dcgm-exporter --collect-interval=1000 --address :{port} -f /configs/dcgm-counters-noprof.csv" + network-interface: "" + single-node-models: staged h100-dgxc: gpus-per-node: 8 available-cpu-dram-mib: 2_063_837 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index db5eaa6965..21e1581558 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9211,3 +9211,93 @@ - "Add --attention-config '{\"indexer_kv_dtype\":\"mxfp4\",\"indexer_sparse_logits\":true}' and --block-size 128 to enable the vllm-project/vllm#58671 ROCm paged MXFP4 sparse-logits indexer, replacing the dense fp8 indexer path. A live A/B test (TP2 c16, matched 900s window, vllm-project/vllm#58208 reverted via vllm-project/vllm#59125 so the dense fallback doesn't crash) measured +14.7/+14.9% p50/p90 interactivity and -9.5/-10.8% p50/p90 e2e latency over the dense path, with throughput/GPU unchanged." - "Drop c128 from both TP2 and TP4. Neither c128 point was on the Pareto frontier in #3555's run 36528242520: TP2 c64 dominated both (P90 E2EL 58 s against 185 s and 85 s, at 111k against 85k and 79k total tok/s/GPU)." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3571 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Update the MiniMax-M3 H100 vLLM image, enable the FP8 MSA indexer cache, adjust CPU weight offload and GPU memory settings, and resize the Mooncake store segments." + - "更新 MiniMax-M3 H100 的 vLLM 镜像,启用 FP8 MSA 索引缓存,调整 CPU 权重卸载与 GPU 内存设置,并调整 Mooncake 存储段大小。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Route the MiniMax-M3 H100 AgentX config to cluster:h100-cw and align the Mooncake DRAM store size with that cluster's host-memory budget." + - "将 MiniMax-M3 H100 AgentX 配置切换到 cluster:h100-cw,并根据该集群的主机内存预算调整 Mooncake DRAM 存储段大小。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Route the MiniMax-M3 H100 AgentX config to the DSXE runner group and add its Slurm cluster profile." + - "将 MiniMax-M3 H100 AgentX 配置切换到 DSXE 运行器组,并添加对应的 Slurm 集群配置。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Use TCP transport for the single-node Mooncake DRAM store in the concurrency 6 and 8 variants." + - "将并发数为 6 和 8 的单节点 Mooncake DRAM 存储变体切换为 TCP 传输。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Reserve GPU memory for Mooncake TCP transfers in the MiniMax-M3 H100 DRAM variants." + - "为 MiniMax-M3 H100 DRAM 变体的 Mooncake TCP 传输预留 GPU 内存。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Set matching 640K service and replay context limits for the MiniMax-M3 H100 Mooncake DRAM variants." + - "为 MiniMax-M3 H100 Mooncake DRAM 变体设置一致的 640K 服务与重放上下文上限。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Reuse Mooncake TCP connections in the MiniMax-M3 H100 DRAM variants." + - "在 MiniMax-M3 H100 DRAM 变体中复用 Mooncake TCP 连接。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Use larger Mooncake transfer slices and a separate live replay error threshold for the MiniMax-M3 H100 DRAM variants; retain the final result validation threshold." + - "为 MiniMax-M3 H100 DRAM 变体使用更大的 Mooncake 传输分片,并单独设置重放过程中的实时错误阈值;最终结果校验阈值保持不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Pin the H100 DRAM variants to Mooncake 0.3.13.post1 with bounded TCP connection lanes." + - "将 H100 DRAM 变体固定为 Mooncake 0.3.13.post1,并使用有上限的 TCP 连接通道。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620 + +- config-keys: + - minimaxm3-fp8-h100-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Set TCP transfer slice size and bounded lane admission settings for the MiniMax-M3 H100 DRAM variants." + - "为 MiniMax-M3 H100 DRAM 变体设置 TCP 传输分片大小和有上限的连接通道接纳参数。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3620