From 2b6e55c019f2c275c02beeef169096aa663ebe18 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 25 Sep 2026 11:44:39 -0700 Subject: [PATCH 1/2] fix: avoid profiling stalls in H200 Kimi-K3 power sampling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 H200 Kimi-K3 latency 功耗采集选择无 profiling 的 DCGM counters,保持官方 SRT pin、采样时序与严格校验不变。 --- inferencex-e2e/configs/nvidia-master.yaml | 1 + inferencex-e2e/perf-changelog.yaml | 11 +++++++++++ 2 files changed, 12 insertions(+) diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 63dcf8fff8..b4cec74ffb 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1407,6 +1407,7 @@ kimik3-fp4-h200-vllm-agentic-latency: - dram-utilization: 0.80 search-space: # Latency-oriented: TP16 x DP2 with EP32 across four H200 nodes. + # Power telemetry omits DCGM profiling counters to avoid watch-repair stalls. - spec-decoding: mtp kv-offloading: none conc-list: [1, 2, 3, 4, 5, 6, 7, 8, 10, 12] diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9b9c48a0ef..5d79eae649 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9168,3 +9168,14 @@ - "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid." - "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605 + +- config-keys: + - kimik3-fp4-h200-vllm-agentic-latency + scenario-type: + - agentic-coding + description: + - "Use profiling-free DCGM counters for Kimi-K3 H200 latency power telemetry to avoid synchronous profiling-watch recovery stalls. Retain power and GPU utilization; omit profiling metrics." + - "Keep the stock exporter image, 100 ms watch interval, 1 s scrape interval, 2 s timeout, and strict power validation unchanged." + - "Kimi-K3 H200 latency 功耗遥测改用无 profiling 的 DCGM counters,避免同步恢复 profiling watch 导致采样卡顿;保留功耗和 GPU 利用率,省略 profiling 指标。" + - "保持原 exporter 镜像、100 ms watch 间隔、1 s 抓取间隔、2 s 超时与严格功耗校验不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3550 From 9fa1523a1c65186e313eb945dc0a76df137ec6f1 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 03:16:39 -0700 Subject: [PATCH 2/2] fix: drain late AgentX responses for H200 Kimi-K3 coverage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 H200 Kimi-K3 latency AgentX 将 benchmark grace 提到 1800s,避免默认 30s 截断导致 ProfileMetricCoverage 卡在 ~94%。 --- .../vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml | 4 ++++ inferencex-e2e/perf-changelog.yaml | 11 +++++++++++ 2 files changed, 15 insertions(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml index 0d9e535e4c..9d917f6088 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/h200-fp4/agentx/agg-tp16dp2ep32-latency.yaml @@ -106,5 +106,9 @@ benchmark: IS_MULTINODE: "true" AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" + # Long responses admitted near the end of the measurement window can take + # minutes to finish; bound their drain without extending admission. + # Default 30s grace cancels mid-response and fails ProfileMetricCoverage (~94%). + AIPERF_BENCHMARK_GRACE_PERIOD: "1800" HF_HUB_CACHE: "/hf_hub_cache" WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 5d79eae649..1cc64479ee 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9179,3 +9179,14 @@ - "Kimi-K3 H200 latency 功耗遥测改用无 profiling 的 DCGM counters,避免同步恢复 profiling watch 导致采样卡顿;保留功耗和 GPU 利用率,省略 profiling 指标。" - "保持原 exporter 镜像、100 ms watch 间隔、1 s 抓取间隔、2 s 超时与严格功耗校验不变。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3550 + +- config-keys: + - kimik3-fp4-h200-vllm-agentic-latency + scenario-type: + - agentic-coding + description: + - "Raise AgentX benchmark grace period to 1800s on the H200 Kimi-K3 latency recipe so in-flight long responses admitted near the 3600s cutoff can finish and extend TTFT/ITL coverage past the 95% ProfileMetricCoverage gate without changing admission duration, concurrency, or topology." + - "Matches the stock srt_agentic.sh AIPERF_BENCHMARK_GRACE_PERIOD wiring and the MI325X GLM-5.2 agentic precedent; leaves the one-hour profile window and noprof power path unchanged." + - "将 H200 Kimi-K3 latency AgentX 的 benchmark grace 提到 1800s,使接近 3600s 截止时已准入的长响应可完成并延续 TTFT/ITL 覆盖至 ≥95%,不改变 admission 时长、并发或拓扑。" + - "对齐 srt_agentic.sh 的 AIPERF_BENCHMARK_GRACE_PERIOD 接线与 MI325X GLM-5.2 agentic 先例;保持一小时 profile 窗口与 noprof 功耗路径不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3550