diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml index 6d0dc2c178..8c86408daf 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml @@ -5,7 +5,7 @@ base: name: qwen3.5-fp8-mi355x-sglang-agentic model: path: hf:Qwen/Qwen3.5-397B-A17B-FP8 - container: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 + container: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 precision: fp8 resources: gpu_type: mi355x @@ -70,9 +70,9 @@ base: AIPERF_APPLY_CHAT_TEMPLATE: 'true' # One variant per point. Admission is 2x CONC and the decode graph batch follows -# it up to 128. HiCache holds 1.5x the device KV pool and skips non-reusable -# blocks; the ratio keeps the host pool proportional to TP without pinning a -# byte count. +# it up to 128. HiCache skips non-reusable blocks and is sized in tiers: CONC 16 +# to 28 keep the 1.5x ratio, CONC 32 to 56 pin a 253 GB host pool, and CONC 64 to +# 80 pin 400 GB. hicache-size overrides hicache-ratio where it is set. override_tp4_c1: roles: @@ -257,6 +257,7 @@ override_tp4_hicache_c32: cuda-graph-max-bs-decode: 64 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 253 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -275,6 +276,7 @@ override_tp4_hicache_c36: cuda-graph-max-bs-decode: 72 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 253 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -293,6 +295,7 @@ override_tp4_hicache_c40: cuda-graph-max-bs-decode: 80 enable-hierarchical-cache: true hicache-ratio: 1.5 + hicache-size: 253 hicache-write-policy: write_through_selective hicache-io-backend: kernel hicache-mem-layout: page_first @@ -300,3 +303,98 @@ override_tp4_hicache_c40: env: CONC: '40' KV_OFFLOADING: dram + +override_tp4_hicache_c48: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 96 + cuda-graph-max-bs-decode: 96 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 253 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '48' + KV_OFFLOADING: dram + +override_tp4_hicache_c56: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 112 + cuda-graph-max-bs-decode: 112 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 253 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '56' + KV_OFFLOADING: dram + +override_tp4_hicache_c64: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 400 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '64' + KV_OFFLOADING: dram + +override_tp4_hicache_c72: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 144 + cuda-graph-max-bs-decode: 128 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 400 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '72' + KV_OFFLOADING: dram + +override_tp4_hicache_c80: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-running-requests: 160 + cuda-graph-max-bs-decode: 128 + enable-hierarchical-cache: true + hicache-ratio: 1.5 + hicache-size: 400 + hicache-write-policy: write_through_selective + hicache-io-backend: kernel + hicache-mem-layout: page_first + benchmark: + env: + CONC: '80' + KV_OFFLOADING: dram diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index a852be37b6..5df8735385 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -222,11 +222,12 @@ qwen3.5-fp8-mi355x-sglang-mtp: # MI355X FP8 AgentX, TP4 only. The grid follows the TP4 band of the B200 FP8 # AgentX arm so the two SKUs are directly comparable point for point, with the -# HiCache band extended to concurrency 36 and 40. TP2 is out of reach: the -# ~400 GB FP8 checkpoint leaves too little KV headroom there for the -# 256k-capped AgentX corpus. +# HiCache band extended to concurrency 80 to reach into the throughput end that +# the host-DRAM tier makes available. TP2 is out of reach: the ~400 GB FP8 +# checkpoint leaves too little KV headroom there for the 256k-capped AgentX +# corpus. qwen3.5-fp8-mi355x-sglang-agentic-mtp: - image: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 + image: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 model: Qwen/Qwen3.5-397B-A17B-FP8 model-prefix: qwen3.5 runner: cluster:mi355x-amds @@ -238,7 +239,7 @@ qwen3.5-fp8-mi355x-sglang-agentic-mtp: - dram-utilization: 0.80 search-space: - { tp: 4, ep: 1, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 14], srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml } - - { tp: 4, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [16, 18, 20, 22, 24, 28, 32, 36, 40], srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml } + - { tp: 4, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [16, 18, 20, 22, 24, 28, 32, 36, 40, 48, 56, 64, 72, 80], srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/agentic.yaml } # MI325X official matrix selected from the complete 62-point fast sweep. TP2 # peaks at c4, TP4/TEP4 at c40, and TP8/TEP8 at c64; the next point beyond each diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9b9c48a0ef..b8dfc3cf5d 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9168,3 +9168,15 @@ - "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid." - "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605 + +- config-keys: + - qwen3.5-fp8-mi355x-sglang-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Extend the TP4 HiCache band from concurrency 40 up to 80, adding concurrency 48, 56, 64, 72 and 80. The arm goes from 14 to 19 points; the GPU-resident band [1, 4, 8, 12, 14] and the existing HiCache points [16, 18, 20, 22, 24, 28, 32, 36, 40] are unchanged. The original grid was copied from the TP4 band of qwen3.5-fp8-b200-sglang-agentic-mtp, which stops at 32 because a B200 node has far less HBM per GPU; MI355X at 288 GB per GPU has the device KV headroom to keep feeding the host-DRAM tier past that, and these five points sample the throughput end the tier makes available." + - "Size the HiCache host pool in tiers by concurrency instead of by ratio alone. Concurrency 16 through 28 keep ratio-only sizing and are unchanged from what merged in #3602. Concurrency 32 through 56 pin hicache-size 253, the value already used by the MI355X MXFP4 AgentX arm. Concurrency 64 through 80 pin hicache-size 400, since the largest points need the deepest host tier to hold prefix across the 256k-capped corpus. hicache-size overrides hicache-ratio where it is set. Of the points that merged in #3602 only concurrency 32, 36 and 40 change behaviour; their numbers are expected to move." + - "The five new points reuse the existing HiCache override shape: enable-hierarchical-cache, hicache-ratio 1.5, hicache-size 400, hicache-write-policy write_through_selective, hicache-io-backend kernel and hicache-mem-layout page_first. Admission continues to follow 2x CONC, and the decode graph batch follows it capped at 128, so concurrency 72 and 80 both clamp cuda-graph-max-bs-decode to 128 while max-running-requests reaches 144 and 160." + - "Update the SGLang ROCm image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 on this arm only. The other MI355X arms are left on their current tags." + - "No serving flag outside the HiCache block changes, and no other config key is touched. Each of the 19 matrix points resolves to exactly one recipe override, with no override left unused." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3611