diff --git a/.github/workflows/run-sweep.yml b/.github/workflows/run-sweep.yml index 65e99d5dbc..eaed70d2dc 100644 --- a/.github/workflows/run-sweep.yml +++ b/.github/workflows/run-sweep.yml @@ -580,6 +580,7 @@ jobs: with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run + require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }} runner: ${{ matrix.config.runner }} priority: ${{ matrix.config.priority }} queue-token: ${{ matrix.config['queue-token'] }} @@ -614,6 +615,7 @@ jobs: with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run + require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }} runner: ${{ matrix.config.runner }} node-count: ${{ matrix.config.node-count }} priority: ${{ matrix.config.priority }} diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 63dcf8fff8..a07aa5e778 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1537,6 +1537,7 @@ kimik3-fp4-b300-vllm-agentic-dspark: # Sizes the Mooncake segment: this resolves to total-cpu-dram-gb, which # the recipe divides across the 8 TP ranks. - dram-utilization: 0.75 + require-power: true search-space: # TP8-only: a ~1.5 TB MXFP4 checkpoint does not fit below 8 GPUs. # One entry for every arm: the recipe drafts at DSpark 7 up to conc 8, diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index d2df756784..0659513403 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -170,7 +170,7 @@ Sources: [`configs/CONFIGS.md`](../configs/CONFIGS.md), [`validation.py`](../inf 6. For srt-slurm, update recipe and master entry together. For llm-d, update the llm-d recipe/orchestration and master entry together. 7. Append the trigger entry, generate only the affected key first, and inspect every emitted point. -Fixed-sequence `8192/1024` scenarios may set `require-power: true` to opt into validated measured power. The matrix passes this flag to standard sweeps and manual E2E throughput jobs; eval-only and AgentX rows do not inherit it. Omit the field to preserve existing behavior. Enable it only alongside the corresponding runtime and result adapter, then qualify the complete selected scope. +Fixed-sequence `8192/1024` scenarios and `agentic-coding` scenario arms may set `require-power: true` to opt into validated measured power. The matrix passes this flag to standard sweeps and AgentX throughput jobs; eval-only rows do not inherit it. Omit the field to preserve existing behavior. Enable it only alongside the corresponding runtime and result adapter, then qualify the complete selected scope. ## Register and set up a runner diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 0d023deb1c..8392d01c76 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -152,7 +152,7 @@ STP(Single Token Prediction,单 Token 预测)是每次前向传播生成 6. srt-slurm 必须同时更新配方和主条目;llm-d 必须同时更新 llm-d 配方/编排和主条目。 7. 追加触发条目,先只生成受影响的 key,并检查每个生成点。 -固定序列 `8192/1024` 场景可设置 `require-power: true`,要求经过验证的实测功耗。矩阵将此标记传递给标准 sweep 和手动 E2E 吞吐作业;eval-only 和 AgentX 行不继承该标记。省略此字段可保留现有行为。仅在对应 runtime 和结果适配器同时交付时启用,然后验证完整选定范围。 +固定序列 `8192/1024` 场景以及 `agentic-coding` 场景臂可设置 `require-power: true`,要求经过验证的实测功耗。矩阵将此标记传递给标准 sweep 和 AgentX 吞吐作业;eval-only 行不继承该标记。省略此字段可保留现有行为。仅在对应 runtime 和结果适配器同时交付时启用,然后验证完整选定范围。 ## 注册并设置 runner diff --git a/inferencex-e2e/infx/matrix/generate.py b/inferencex-e2e/infx/matrix/generate.py index 4b64f011df..6080f25035 100644 --- a/inferencex-e2e/infx/matrix/generate.py +++ b/inferencex-e2e/infx/matrix/generate.py @@ -1000,6 +1000,11 @@ def _agentic_entries( if is_multinode: entry[Fields.DISAGG.value] = disagg entry[Fields.SCENARIO_TYPE.value] = "agentic-coding" + require_power = scenario.get( + Fields.REQUIRE_POWER.value, scenario.get("require_power", False) + ) + if require_power: + entry[Fields.REQUIRE_POWER.value] = True if kv_offload_backend is not None: entry[Fields.KV_OFFLOAD_BACKEND.value] = kv_offload_backend entry.update(component_metadata(benchmark, config)) diff --git a/inferencex-e2e/infx/matrix/validation.py b/inferencex-e2e/infx/matrix/validation.py index f44d383ca4..0415349115 100644 --- a/inferencex-e2e/infx/matrix/validation.py +++ b/inferencex-e2e/infx/matrix/validation.py @@ -315,6 +315,7 @@ class SingleNodeAgenticMatrixEntry(BaseModel): precision: str framework: str runner: str + require_power: bool = Field(default=False, alias=Fields.REQUIRE_POWER.value, strict=True) tp: int pp: int = Field(gt=0, strict=True) dcp_size: int = Field(alias=Fields.DCP_SIZE.value, gt=0, strict=True) @@ -367,6 +368,7 @@ class MultiNodeAgenticMatrixEntry(BaseModel): framework: str spec_decoding: Literal["mtp", "draft_model", "none"] = Field(alias=Fields.SPEC_DECODING.value) runner: str + require_power: bool = Field(default=False, alias=Fields.REQUIRE_POWER.value, strict=True) node_count: int = Field(alias=Fields.NODE_COUNT.value, gt=0, strict=True) prefill: WorkerConfig decode: WorkerConfig @@ -717,6 +719,7 @@ class AgenticCodingConfig(BaseModel): model_config = ConfigDict(extra="forbid", populate_by_name=True) + require_power: bool = Field(default=False, alias=Fields.REQUIRE_POWER.value, strict=True) search_space: list[AgenticCodingSearchSpaceEntry] = Field(alias=Fields.SEARCH_SPACE.value) dram_utilization: float | None = Field( default=None, alias=Fields.DRAM_UTILIZATION.value, gt=0, le=1 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9b9c48a0ef..0a663945c2 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9168,3 +9168,12 @@ - "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid." - "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Enable strict measured power (require-power) for the full Kimi-K3 B300 AgentX curve (conc-list [1,2,4,8,16,24,32,40,48,56,70])." + - "为完整 Kimi-K3 B300 AgentX 曲线(conc-list [1,2,4,8,16,24,32,40,48,56,70])启用严格实测功耗(require-power)。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3632