From d809d79bef2221e5239632f7b203a3475e84a65d Mon Sep 17 00:00:00 2001 From: "claude[bot]" <41898282+claude[bot]@users.noreply.github.com> Date: Thu, 1 Oct 2026 12:52:47 +0000 Subject: [PATCH 1/2] perf(dsr1): bump B200 FP4 SGLang MTP image to v0.5.20-cu130 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pin lmsysorg/sglang:v0.5.20-cu130 by digest and replace the CLI flags SGLang removed after v0.5.16: --cuda-graph-max-bs becomes --cuda-graph-max-bs-decode and --disable-piecewise-cuda-graph becomes --cuda-graph-backend-prefill=disabled (the exact v0.5.16 alias targets). 将 dsr1-fp4-b200-sglang-mtp 的 SGLang 镜像更新为按 digest 固定的 lmsysorg/sglang:v0.5.20-cu130,并替换 v0.5.16 之后移除的 CLI 参数: --cuda-graph-max-bs 改为 --cuda-graph-max-bs-decode, --disable-piecewise-cuda-graph 改为 --cuda-graph-backend-prefill=disabled (即 v0.5.16 中这两个别名的实际目标)。 Co-Authored-By: Claude Opus 5.5 --- .../srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml | 6 +++--- inferencex-e2e/configs/nvidia-master.yaml | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml index 88ae02628c..3d8245e990 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml @@ -3,7 +3,7 @@ base: name: dsr1-fp4-b200-sglang-mtp-8k1k model: path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 - container: lmsysorg/sglang:v0.5.16-cu130 + container: lmsysorg/sglang:v0.5.20-cu130@sha256:06e4f2ed21afde4ff513cda65070124e727ba23ccaeff7712b8c40e1097d611f precision: fp4 resources: gpu_type: b200 @@ -25,7 +25,7 @@ base: trust-remote-code: true tensor-parallel-size: 4 data-parallel-size: 1 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 max-running-requests: 256 mem-fraction-static: 0.85 kv-cache-dtype: fp8_e4m3 @@ -34,7 +34,7 @@ base: quantization: modelopt_fp4 enable-flashinfer-allreduce-fusion: true scheduler-recv-interval: 10 - disable-piecewise-cuda-graph: true + cuda-graph-backend-prefill: disabled attention-backend: trtllm_mla moe-runner-backend: flashinfer_trtllm stream-interval: 10 diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 63dcf8fff8..7722745c3d 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -893,7 +893,7 @@ dsr1-fp4-b200-sglang: # - { tp: 8, ep: 8, offloading: none, conc-list: [1, 2, 4, 8, 12, 16, 32, 64, 128, 256, 512] } dsr1-fp4-b200-sglang-mtp: - image: lmsysorg/sglang:v0.5.16-cu130 + image: lmsysorg/sglang:v0.5.20-cu130@sha256:06e4f2ed21afde4ff513cda65070124e727ba23ccaeff7712b8c40e1097d611f model: nvidia/DeepSeek-R1-0528-FP4-V2 model-prefix: dsr1 runner: cluster:b200-nscale From caf4afc7dbf1f0bff53c7447ca32c9994d75b938 Mon Sep 17 00:00:00 2001 From: "claude[bot]" <41898282+claude[bot]@users.noreply.github.com> Date: Thu, 1 Oct 2026 13:21:37 +0000 Subject: [PATCH 2/2] chore(changelog): record dsr1-fp4-b200-sglang-mtp SGLang v0.5.20 bump MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 dsr1-fp4-b200-sglang-mtp 的 SGLang v0.5.20 镜像更新追加 perf-changelog 条目。 Co-Authored-By: Claude Opus 5.5 --- inferencex-e2e/perf-changelog.yaml | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9b9c48a0ef..9bbc7a7c80 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9168,3 +9168,9 @@ - "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid." - "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605 + +- config-keys: + - dsr1-fp4-b200-sglang-mtp + description: + - "Update SGLang from v0.5.16-cu130 to v0.5.20-cu130 and rename two CUDA graph flags removed upstream." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3634