From 19500ca266239673dfd11835c8318a46d21fa0e1 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 07:41:47 +0000 Subject: [PATCH 1/2] feat(powerx): require measured power for MiniMax-M3 B200 AgentX MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add require-power on MiniMax-M3 B200 vLLM and TRT-LLM AgentX scenario arms, and wire agentic matrix/workflow support so the flag takes effect. 为 MiniMax-M3 B200 AgentX(vLLM + TRT-LLM)启用 require-power,并接通 AgentX 矩阵与 sweep 工作流,使该标记生效。 Co-authored-by: Wenyao Gao --- .github/workflows/run-sweep.yml | 2 ++ inferencex-e2e/configs/nvidia-master.yaml | 3 +++ inferencex-e2e/infx/matrix/generate.py | 5 +++++ inferencex-e2e/infx/matrix/validation.py | 3 +++ inferencex-e2e/perf-changelog.yaml | 10 ++++++++++ 5 files changed, 23 insertions(+) diff --git a/.github/workflows/run-sweep.yml b/.github/workflows/run-sweep.yml index 65e99d5dbc..eaed70d2dc 100644 --- a/.github/workflows/run-sweep.yml +++ b/.github/workflows/run-sweep.yml @@ -580,6 +580,7 @@ jobs: with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run + require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }} runner: ${{ matrix.config.runner }} priority: ${{ matrix.config.priority }} queue-token: ${{ matrix.config['queue-token'] }} @@ -614,6 +615,7 @@ jobs: with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run + require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }} runner: ${{ matrix.config.runner }} node-count: ${{ matrix.config.node-count }} priority: ${{ matrix.config.priority }} diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 63dcf8fff8..63b29050c8 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -5684,10 +5684,12 @@ minimaxm3-fp4-b200-vllm-agentic-mtp: scenarios: agentic-coding: - dram-utilization: 0.683 + require-power: true search-space: - { tp: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [1], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b200-fp4-mtp/agentic.yaml } - { tp: 4, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 5, 10, 15, 20], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b200-fp4-mtp/agentic.yaml } - dram-utilization: 1.0 + require-power: true search-space: - { tp: 4, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [15, 20, 25, 30, 32, 34, 36, 38, 40], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b200-fp4-mtp/agentic.yaml } minimaxm3-fp4-b200-trtllm-agentic-mtp: @@ -5704,6 +5706,7 @@ minimaxm3-fp4-b200-trtllm-agentic-mtp: # every point, so dram-utilization only reports a nominal budget here and # does not size the pool. - dram-utilization: 0.8 + require-power: true search-space: - { tp: 4, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: native }, conc-list: [5, 10, 15, 20, 25, 30, 35, 40, 45], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/trtllm/b200-fp4-mtp/agentic.yaml } - { tp: 8, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: native }, conc-list: [1, 5], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/trtllm/b200-fp4-mtp/agentic.yaml } diff --git a/inferencex-e2e/infx/matrix/generate.py b/inferencex-e2e/infx/matrix/generate.py index 4b64f011df..6080f25035 100644 --- a/inferencex-e2e/infx/matrix/generate.py +++ b/inferencex-e2e/infx/matrix/generate.py @@ -1000,6 +1000,11 @@ def _agentic_entries( if is_multinode: entry[Fields.DISAGG.value] = disagg entry[Fields.SCENARIO_TYPE.value] = "agentic-coding" + require_power = scenario.get( + Fields.REQUIRE_POWER.value, scenario.get("require_power", False) + ) + if require_power: + entry[Fields.REQUIRE_POWER.value] = True if kv_offload_backend is not None: entry[Fields.KV_OFFLOAD_BACKEND.value] = kv_offload_backend entry.update(component_metadata(benchmark, config)) diff --git a/inferencex-e2e/infx/matrix/validation.py b/inferencex-e2e/infx/matrix/validation.py index f44d383ca4..0415349115 100644 --- a/inferencex-e2e/infx/matrix/validation.py +++ b/inferencex-e2e/infx/matrix/validation.py @@ -315,6 +315,7 @@ class SingleNodeAgenticMatrixEntry(BaseModel): precision: str framework: str runner: str + require_power: bool = Field(default=False, alias=Fields.REQUIRE_POWER.value, strict=True) tp: int pp: int = Field(gt=0, strict=True) dcp_size: int = Field(alias=Fields.DCP_SIZE.value, gt=0, strict=True) @@ -367,6 +368,7 @@ class MultiNodeAgenticMatrixEntry(BaseModel): framework: str spec_decoding: Literal["mtp", "draft_model", "none"] = Field(alias=Fields.SPEC_DECODING.value) runner: str + require_power: bool = Field(default=False, alias=Fields.REQUIRE_POWER.value, strict=True) node_count: int = Field(alias=Fields.NODE_COUNT.value, gt=0, strict=True) prefill: WorkerConfig decode: WorkerConfig @@ -717,6 +719,7 @@ class AgenticCodingConfig(BaseModel): model_config = ConfigDict(extra="forbid", populate_by_name=True) + require_power: bool = Field(default=False, alias=Fields.REQUIRE_POWER.value, strict=True) search_space: list[AgenticCodingSearchSpaceEntry] = Field(alias=Fields.SEARCH_SPACE.value) dram_utilization: float | None = Field( default=None, alias=Fields.DRAM_UTILIZATION.value, gt=0, le=1 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9b9c48a0ef..a488aee122 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9168,3 +9168,13 @@ - "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid." - "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605 + +- config-keys: + - minimaxm3-fp4-b200-vllm-agentic-mtp + - minimaxm3-fp4-b200-trtllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Require validated GPU power for MiniMax-M3 B200 AgentX (vLLM + TRT-LLM). Completes PowerX measured-power coverage for this model x SKU while preserving existing topology and concurrency." + - "为 MiniMax-M3 B200 AgentX(vLLM + TRT-LLM)强制校验 GPU 功耗,补齐该模型×SKU 的 PowerX 实测功耗覆盖,保留现有拓扑与并发。" + pr-link: TBD From bb811553e6ed5302653e2de27a62339e1ca99e28 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 07:42:30 +0000 Subject: [PATCH 2/2] chore(powerx): set MiniMax-M3 B200 changelog pr-link to #3628 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 MiniMax-M3 B200 PowerX changelog 的 pr-link 更新为 #3628。 Co-authored-by: Wenyao Gao --- inferencex-e2e/perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index a488aee122..efd9fbd8ab 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9177,4 +9177,4 @@ description: - "Require validated GPU power for MiniMax-M3 B200 AgentX (vLLM + TRT-LLM). Completes PowerX measured-power coverage for this model x SKU while preserving existing topology and concurrency." - "为 MiniMax-M3 B200 AgentX(vLLM + TRT-LLM)强制校验 GPU 功耗,补齐该模型×SKU 的 PowerX 实测功耗覆盖,保留现有拓扑与并发。" - pr-link: TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3628