diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml index 88ae02628c..3d8245e990 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml @@ -3,7 +3,7 @@ base: name: dsr1-fp4-b200-sglang-mtp-8k1k model: path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 - container: lmsysorg/sglang:v0.5.16-cu130 + container: lmsysorg/sglang:v0.5.20-cu130@sha256:06e4f2ed21afde4ff513cda65070124e727ba23ccaeff7712b8c40e1097d611f precision: fp4 resources: gpu_type: b200 @@ -25,7 +25,7 @@ base: trust-remote-code: true tensor-parallel-size: 4 data-parallel-size: 1 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 max-running-requests: 256 mem-fraction-static: 0.85 kv-cache-dtype: fp8_e4m3 @@ -34,7 +34,7 @@ base: quantization: modelopt_fp4 enable-flashinfer-allreduce-fusion: true scheduler-recv-interval: 10 - disable-piecewise-cuda-graph: true + cuda-graph-backend-prefill: disabled attention-backend: trtllm_mla moe-runner-backend: flashinfer_trtllm stream-interval: 10 diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 63dcf8fff8..7722745c3d 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -893,7 +893,7 @@ dsr1-fp4-b200-sglang: # - { tp: 8, ep: 8, offloading: none, conc-list: [1, 2, 4, 8, 12, 16, 32, 64, 128, 256, 512] } dsr1-fp4-b200-sglang-mtp: - image: lmsysorg/sglang:v0.5.16-cu130 + image: lmsysorg/sglang:v0.5.20-cu130@sha256:06e4f2ed21afde4ff513cda65070124e727ba23ccaeff7712b8c40e1097d611f model: nvidia/DeepSeek-R1-0528-FP4-V2 model-prefix: dsr1 runner: cluster:b200-nscale diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9b9c48a0ef..9bbc7a7c80 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9168,3 +9168,9 @@ - "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid." - "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605 + +- config-keys: + - dsr1-fp4-b200-sglang-mtp + description: + - "Update SGLang from v0.5.16-cu130 to v0.5.20-cu130 and rename two CUDA graph flags removed upstream." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3634