From 9fac19eabcdb9674b77dd942b63cbe662de14989 Mon Sep 17 00:00:00 2001 From: "jacky.cheng" Date: Wed, 30 Sep 2026 16:10:34 +0000 Subject: [PATCH 1/2] [AMD][Qwen3.5] Extend the MI355X FP8 AgentX TP4 HiCache band to concurrency 80 --- .../sglang/mi355x-fp8-mtp/agentic.yaml | 106 +++++++++++++++++- inferencex-e2e/configs/amd-master.yaml | 11 +- inferencex-e2e/perf-changelog.yaml | 12 ++ 3 files changed, 123 insertions(+), 6 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml index 6d0dc2c178..02d94b36de 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml @@ -5,7 +5,7 @@ base: name: qwen3.5-fp8-mi355x-sglang-agentic model: path: hf:Qwen/Qwen3.5-397B-A17B-FP8 - container: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 + container: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 precision: fp8 resources: gpu_type: mi355x @@ -149,6 +149,7 @@ override_tp4_hicache_c16: cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -167,6 +168,7 @@ override_tp4_hicache_c18: cuda-graph-max-bs-decode: 36 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -185,6 +187,7 @@ override_tp4_hicache_c20: cuda-graph-max-bs-decode: 40 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -203,6 +206,7 @@ override_tp4_hicache_c22: cuda-graph-max-bs-decode: 44 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -221,6 +225,7 @@ override_tp4_hicache_c24: cuda-graph-max-bs-decode: 48 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -239,6 +244,7 @@ override_tp4_hicache_c28: cuda-graph-max-bs-decode: 56 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -257,6 +263,7 @@ override_tp4_hicache_c32: cuda-graph-max-bs-decode: 64 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -275,6 +282,7 @@ override_tp4_hicache_c36: cuda-graph-max-bs-decode: 72 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -293,6 +301,7 @@ override_tp4_hicache_c40: cuda-graph-max-bs-decode: 80 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -300,3 +309,98 @@ override_tp4_hicache_c40: env: CONC: '40' KV_OFFLOADING: dram + +override_tp4_hicache_c48: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 96 + cuda-graph-max-bs-decode: 96 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 400 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '48' + KV_OFFLOADING: dram + +override_tp4_hicache_c56: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 112 + cuda-graph-max-bs-decode: 112 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 400 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '56' + KV_OFFLOADING: dram + +override_tp4_hicache_c64: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 400 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '64' + KV_OFFLOADING: dram + +override_tp4_hicache_c72: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 144 + cuda-graph-max-bs-decode: 128 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 400 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '72' + KV_OFFLOADING: dram + +override_tp4_hicache_c80: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 160 + cuda-graph-max-bs-decode: 128 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 400 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '80' + KV_OFFLOADING: dram diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index 62b370ca5d..328c29a3f1 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -222,11 +222,12 @@ qwen3.5-fp8-mi355x-sglang-mtp: # MI355X FP8 AgentX, TP4 only. The grid follows the TP4 band of the B200 FP8 # AgentX arm so the two SKUs are directly comparable point for point, with the -# HiCache band extended to concurrency 36 and 40. TP2 is out of reach: the -# ~400 GB FP8 checkpoint leaves too little KV headroom there for the -# 256k-capped AgentX corpus. +# HiCache band extended to concurrency 80 to reach into the throughput end that +# the host-DRAM tier makes available. TP2 is out of reach: the ~400 GB FP8 +# checkpoint leaves too little KV headroom there for the 256k-capped AgentX +# corpus. qwen3.5-fp8-mi355x-sglang-agentic-mtp: - image: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 + image: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 model: Qwen/Qwen3.5-397B-A17B-FP8 model-prefix: qwen3.5 runner: cluster:mi355x-amds @@ -238,7 +239,7 @@ qwen3.5-fp8-mi355x-sglang-agentic-mtp: - dram-utilization: 0.80 search-space: - { tp: 4, ep: 1, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 14], srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml } - - { tp: 4, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [16, 18, 20, 22, 24, 28, 32, 36, 40], srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml } + - { tp: 4, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [16, 18, 20, 22, 24, 28, 32, 36, 40, 48, 56, 64, 72, 80], srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml } # MI325X official matrix selected from the complete 62-point fast sweep. TP2 # peaks at c4, TP4/TEP4 at c40, and TP8/TEP8 at c64; the next point beyond each diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index d7fc64f0d6..903325ddc8 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9149,3 +9149,15 @@ - "Use AITER attention and allreduce fusion, FP8 KV (fp8_e4m3), and EAGLE MTP (3 steps, topk 1, 4 draft tokens); do not enable ROCm INT4 quick all-reduce. HiCache uses ratio 1.5, write_through_selective, kernel I/O and page_first layout. The matching cookbook recipe is sgl-project/sglang#41849." - "The embedded MTP head runs at its stored precision: block-FP8 expert weights and checkpoint dtype for mtp.fc and gates. No separate draft, draft dtype override or submission-side quantization is used." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3602 + +- config-keys: + - qwen3.5-fp8-mi355x-sglang-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Extend the TP4 HiCache band from concurrency 40 up to 80, adding concurrency 48, 56, 64, 72 and 80. The arm goes from 14 to 19 points; the GPU-resident band [1, 4, 8, 12, 14] and the existing HiCache points [16, 18, 20, 22, 24, 28, 32, 36, 40] are unchanged. The original grid was copied from the TP4 band of qwen3.5-fp8-b200-sglang-agentic-mtp, which stops at 32 because a B200 node has far less HBM per GPU; MI355X at 288 GB per GPU has the device KV headroom to keep feeding the host-DRAM tier past that, and these five points sample the throughput end the tier makes available." + - "Pin the HiCache host pool with hicache-size 400 on every HiCache point, which overrides hicache-ratio. The band previously sized the host tier only by ratio, so the pool grew and shrank with the device KV pool and was never stated in bytes; a fixed 400 GB per rank gives every point in the band the same host tier and makes the high-concurrency points reproducible. At TP4 that is 1600 GB across the four ranks, inside the node budget implied by dram-utilization 0.80. This changes the nine previously merged HiCache points as well as the five new ones, so their numbers are expected to move." + - "The five new points reuse the existing HiCache override shape: enable-hierarchical-cache, hicache-ratio 1.5, hicache-size 400, hicache-write-policy write_through_selective, hicache-io-backend kernel and hicache-mem-layout page_first. Admission continues to follow 2x CONC, and the decode graph batch follows it capped at 128, so concurrency 72 and 80 both clamp cuda-graph-max-bs-decode to 128 while max-running-requests reaches 144 and 160." + - "Update the SGLang ROCm image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 on this arm only. The other MI355X arms are left on their current tags." + - "No serving flag outside the HiCache block changes, and no other config key is touched. Each of the 19 matrix points resolves to exactly one recipe override, with no override left unused." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3611 From 4a673e787fe3447760d6e1bc834f4c1a8a1dd31e Mon Sep 17 00:00:00 2001 From: "jacky.cheng" Date: Wed, 30 Sep 2026 16:21:38 +0000 Subject: [PATCH 2/2] [AMD][Qwen3.5] Tier the MI355X FP8 AgentX HiCache host pool by concurrency --- .../sglang/mi355x-fp8-mtp/agentic.yaml | 22 +++++++------------ inferencex-e2e/perf-changelog.yaml | 2 +- 2 files changed, 9 insertions(+), 15 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml index 02d94b36de..8c86408daf 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml @@ -70,9 +70,9 @@ base: AIPERF_APPLY_CHAT_TEMPLATE: 'true' # One variant per point. Admission is 2x CONC and the decode graph batch follows -# it up to 128. HiCache holds 1.5x the device KV pool and skips non-reusable -# blocks; the ratio keeps the host pool proportional to TP without pinning a -# byte count. +# it up to 128. HiCache skips non-reusable blocks and is sized in tiers: CONC 16 +# to 28 keep the 1.5x ratio, CONC 32 to 56 pin a 253 GB host pool, and CONC 64 to +# 80 pin 400 GB. hicache-size overrides hicache-ratio where it is set. override_tp4_c1: roles: @@ -149,7 +149,6 @@ override_tp4_hicache_c16: cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -168,7 +167,6 @@ override_tp4_hicache_c18: cuda-graph-max-bs-decode: 36 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -187,7 +185,6 @@ override_tp4_hicache_c20: cuda-graph-max-bs-decode: 40 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -206,7 +203,6 @@ override_tp4_hicache_c22: cuda-graph-max-bs-decode: 44 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -225,7 +221,6 @@ override_tp4_hicache_c24: cuda-graph-max-bs-decode: 48 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -244,7 +239,6 @@ override_tp4_hicache_c28: cuda-graph-max-bs-decode: 56 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -263,7 +257,7 @@ override_tp4_hicache_c32: cuda-graph-max-bs-decode: 64 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 + hicache-size: 253 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -282,7 +276,7 @@ override_tp4_hicache_c36: cuda-graph-max-bs-decode: 72 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 + hicache-size: 253 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -301,7 +295,7 @@ override_tp4_hicache_c40: cuda-graph-max-bs-decode: 80 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 + hicache-size: 253 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -320,7 +314,7 @@ override_tp4_hicache_c48: cuda-graph-max-bs-decode: 96 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 + hicache-size: 253 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -339,7 +333,7 @@ override_tp4_hicache_c56: cuda-graph-max-bs-decode: 112 enable-hierarchical-cache: true hicache-ratio: 1.5 - hicache-size: 400 + hicache-size: 253 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 903325ddc8..b856ff67b2 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9156,7 +9156,7 @@ - agentic-coding description: - "Extend the TP4 HiCache band from concurrency 40 up to 80, adding concurrency 48, 56, 64, 72 and 80. The arm goes from 14 to 19 points; the GPU-resident band [1, 4, 8, 12, 14] and the existing HiCache points [16, 18, 20, 22, 24, 28, 32, 36, 40] are unchanged. The original grid was copied from the TP4 band of qwen3.5-fp8-b200-sglang-agentic-mtp, which stops at 32 because a B200 node has far less HBM per GPU; MI355X at 288 GB per GPU has the device KV headroom to keep feeding the host-DRAM tier past that, and these five points sample the throughput end the tier makes available." - - "Pin the HiCache host pool with hicache-size 400 on every HiCache point, which overrides hicache-ratio. The band previously sized the host tier only by ratio, so the pool grew and shrank with the device KV pool and was never stated in bytes; a fixed 400 GB per rank gives every point in the band the same host tier and makes the high-concurrency points reproducible. At TP4 that is 1600 GB across the four ranks, inside the node budget implied by dram-utilization 0.80. This changes the nine previously merged HiCache points as well as the five new ones, so their numbers are expected to move." + - "Size the HiCache host pool in tiers by concurrency instead of by ratio alone. Concurrency 16 through 28 keep ratio-only sizing and are unchanged from what merged in #3602. Concurrency 32 through 56 pin hicache-size 253, the value already used by the MI355X MXFP4 AgentX arm. Concurrency 64 through 80 pin hicache-size 400, since the largest points need the deepest host tier to hold prefix across the 256k-capped corpus. hicache-size overrides hicache-ratio where it is set. Of the points that merged in #3602 only concurrency 32, 36 and 40 change behaviour; their numbers are expected to move." - "The five new points reuse the existing HiCache override shape: enable-hierarchical-cache, hicache-ratio 1.5, hicache-size 400, hicache-write-policy write_through_selective, hicache-io-backend kernel and hicache-mem-layout page_first. Admission continues to follow 2x CONC, and the decode graph batch follows it capped at 128, so concurrency 72 and 80 both clamp cuda-graph-max-bs-decode to 128 while max-running-requests reaches 144 and 160." - "Update the SGLang ROCm image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 on this arm only. The other MI355X arms are left on their current tags." - "No serving flag outside the HiCache block changes, and no other config key is touched. Each of the 19 matrix points resolves to exactly one recipe override, with no override left unused."