diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml index 2ef179373b..f76bb3c4ab 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -13,11 +13,11 @@ base: model: repo: "deepseek-ai/DeepSeek-V4-Pro-0813" container: - image: "lmsysorg/sglang:v0.5.19-cu130@sha256:d6e7288627be8b02be88e4bba38e73f6d50e2826869f753c13a4c4385ab3eda9" + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" dynamo: install: true source: - wheel: "1.5.0.dev20260910" + wheel: "1.5.0.dev20260914" slurm: time_limit: "4:00:00" health_check: @@ -77,7 +77,7 @@ base: weight-loader-prefetch-checkpoints: true stream-interval: 10 watchdog-timeout: 1000000 - mem-fraction-static: 0.94 + mem-fraction-static: 0.90 page-size: 256 chunked-prefill-size: 8192 max-prefill-tokens: 8192 @@ -107,6 +107,8 @@ base: RESULT_DIR: "/logs/agentic" PORT: "8000" IS_MULTINODE: "false" + PP_SIZE: "1" + PCP_SIZE: "1" AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" @@ -115,15 +117,15 @@ base: # Low-latency AgentX aggregate topology: one TP4 worker occupies one # four-GPU GB300 node and serves both prefill and decode with DSpark K=6. -override_tp4: - name: "agg-gb300-tp4-mtp-lowlatency" +override_tp4_c4: + name: "agg-gb300-tp4-c4-mtp-lowlatency" roles: agg: nodes: 1 gpus: 4 args: - max-running-requests: 32 - cuda-graph-max-bs-decode: 32 + max-running-requests: 8 + cuda-graph-max-bs-decode: 8 tp-size: 4 benchmark: env: @@ -131,15 +133,15 @@ override_tp4: # Low-latency AgentX aggregate topology: one TP8 worker spans two # four-GPU GB300 nodes and serves both prefill and decode with DSpark K=6. -override_tp8: - name: "agg-gb300-tp8-mtp-lowlatency" +override_tp8_c1: + name: "agg-gb300-tp8-c1-mtp-lowlatency" roles: agg: nodes: 2 gpus: 8 args: - max-running-requests: 4 - cuda-graph-max-bs-decode: 4 + max-running-requests: 2 + cuda-graph-max-bs-decode: 2 tp-size: 8 benchmark: env: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index ad2af4be37..74fb50da39 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -13,11 +13,11 @@ base: model: repo: "deepseek-ai/DeepSeek-V4-Pro-0813" container: - image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21" + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" dynamo: install: true source: - wheel: "1.5.0.dev20260902" + wheel: "1.5.0.dev20260914" slurm: time_limit: "8:00:00" health_check: @@ -103,7 +103,6 @@ base: moe-dense-tp-size: 1 moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true disaggregation-transfer-backend: mooncake disaggregation-mode: prefill load-balance-method: total_tokens @@ -172,7 +171,6 @@ base: moe-dense-tp-size: 1 moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true disaggregation-transfer-backend: mooncake disaggregation-mode: decode load-balance-method: total_tokens @@ -208,15 +206,9 @@ base: AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" HF_HUB_CACHE: "/hf_hub_cache" -# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 -# (2P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 480. -# -# Uses the flat single-variant srtctl schema the agentic CI flow expects; -# resources + backend (prefill/decode env + sglang_config) are normalized -# from the Pareto run. -# Concurrency is exported into srt_agentic.sh from the master-config conc-list. -override_1p1d_c480: - name: "disagg-gb300-2p4d-dep8-dep16-c480-mtp-kvoffload" +# One DEP8 prefill worker and one DEP16 decode worker at concurrency 64. +override_1p1d_c64: + name: "disagg-gb300-1p1d-dep8-dep16-c64-mtp-kvoffload" roles: prefill: nodes: 2 @@ -224,11 +216,47 @@ override_1p1d_c480: gpus: 8 args: max-running-requests: 256 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 + hicache-mem-layout: page_first_direct + decode: + gpus: 16 + args: + max-running-requests: 128 + cuda-graph-max-bs-decode: 256 + +# One DEP8 prefill worker and one DEP16 decode worker at concurrency 240. +override_1p1d_c240: + name: "disagg-gb300-1p1d-dep8-dep16-c240-mtp-kvoffload" + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + args: + max-running-requests: 512 + cuda-graph-max-bs-decode: 256 decode: gpus: 16 args: - cuda-graph-max-bs: 256 + max-running-requests: 512 + cuda-graph-max-bs-decode: 256 + +# Two DEP8 prefill workers and one DEP16 decode worker at concurrency 480. +override_2p1d_c480: + name: "disagg-gb300-2p1d-dep8-dep16-c480-mtp-kvoffload" + roles: + prefill: + nodes: 4 + workers: 2 + gpus: 8 + args: + max-running-requests: 512 + cuda-graph-max-bs-decode: 256 + decode: + gpus: 16 + args: + max-running-requests: 1024 + cuda-graph-max-bs-decode: 256 # Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300 # (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960. @@ -247,13 +275,13 @@ override_2p1d_c960: OMP_NUM_THREADS: '1' args: max-running-requests: 256 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 decode: env: OMP_NUM_THREADS: '1' SGLANG_DSV4_MHC_PREWARM: '1' args: - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 # Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 # (6P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1440. @@ -271,11 +299,11 @@ override_3p1d_c1440: gpus: 8 args: max-running-requests: 512 - cuda-graph-max-bs: 512 + cuda-graph-max-bs-decode: 512 decode: gpus: 16 args: - cuda-graph-max-bs: 512 + cuda-graph-max-bs-decode: 512 # Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 # (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920. @@ -301,13 +329,13 @@ override_4p1d_c1920: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" args: max-running-requests: 1024 - cuda-graph-max-bs: 1024 + cuda-graph-max-bs-decode: 1024 decode: gpus: 16 env: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" args: - cuda-graph-max-bs: 192 + cuda-graph-max-bs-decode: 192 benchmark: env: AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 052eb90726..0868ddf704 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -6639,20 +6639,20 @@ kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2: additional-settings: - "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml" dsv4-fp4-gb300-dynamo-sglang-agentic-agg: - image: lmsysorg/sglang:v0.5.19-cu130@sha256:d6e7288627be8b02be88e4bba38e73f6d50e2826869f753c13a4c4385ab3eda9 + image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 model: deepseek-ai/DeepSeek-V4-Pro-0813 model-prefix: dsv4 runner: cluster:gb300-nv precision: fp4 framework: dynamo-sglang - router: { name: dynamo-router, version: "1.5.0.dev20260910" } + router: { name: dynamo-router, version: "1.5.0.dev20260914" } multinode: true disagg: false scenarios: agentic-coding: - search-space: - spec-decoding: draft_model - conc-list: [1, 4] + conc-list: [1] num-nodes: 2 worker: num-worker: 1 @@ -6660,10 +6660,10 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp8" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp8_c1" - search-space: - spec-decoding: draft_model - conc-list: [8] + conc-list: [4] num-nodes: 1 worker: num-worker: 1 @@ -6671,15 +6671,15 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg: ep: 1 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4_c4" dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21 + image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 model: deepseek-ai/DeepSeek-V4-Pro-0813 model-prefix: dsv4 runner: cluster:gb300-nv precision: fp4 framework: dynamo-sglang - router: { name: dynamo-router, version: "1.5.0.dev20260902" } + router: { name: dynamo-router, version: "1.5.0.dev20260914" } kv-p2p-transfer: mooncake multinode: true disagg: true @@ -6688,7 +6688,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - dram-utilization: 0.80 search-space: - spec-decoding: draft_model - conc-list: [480] + conc-list: [64] kv-offloading: dram kv-offload-backend: { name: hicache } prefill: @@ -6697,7 +6697,39 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c480" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c64" + decode: + num-worker: 1 + tp: 16 + ep: 16 + dp-attn: true + - spec-decoding: draft_model + conc-list: [240] + kv-offloading: dram + kv-offload-backend: { name: hicache } + prefill: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c240" + decode: + num-worker: 1 + tp: 16 + ep: 16 + dp-attn: true + - spec-decoding: draft_model + conc-list: [480] + kv-offloading: dram + kv-offload-backend: { name: hicache } + prefill: + num-worker: 2 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_2p1d_c480" decode: num-worker: 1 tp: 16 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index db5eaa6965..5b1bdefe39 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9211,3 +9211,17 @@ - "Add --attention-config '{\"indexer_kv_dtype\":\"mxfp4\",\"indexer_sparse_logits\":true}' and --block-size 128 to enable the vllm-project/vllm#58671 ROCm paged MXFP4 sparse-logits indexer, replacing the dense fp8 indexer path. A live A/B test (TP2 c16, matched 900s window, vllm-project/vllm#58208 reverted via vllm-project/vllm#59125 so the dense fallback doesn't crash) measured +14.7/+14.9% p50/p90 interactivity and -9.5/-10.8% p50/p90 e2e latency over the dense path, with throughput/GPU unchanged." - "Drop c128 from both TP2 and TP4. Neither c128 point was on the Pareto frontier in #3555's run 36528242520: TP2 c64 dominated both (P90 E2EL 58 s against 185 s and 85 s, at 111k against 85k and 79k total tok/s/GPU)." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3571 + +- config-keys: + - dsv4-fp4-gb300-dynamo-sglang-agentic-agg + - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Refresh DeepSeek-V4-Pro-0813 GB300 Dynamo+SGLang AgentX on nightly-dev-20260916-c9a8fba9 and Dynamo 1.5.0.dev20260914. Keep TP8 c1 and TP4 c4 aggregate points, and DEP8 prefill / DEP16 decode points at c64, c240, c480, c960, c1440 and c1920." + - "Replace the long-TTFT 1P1D c480 point with 1P1D c240 and 2P1D c480, retain DSpark K=6 on both serving stages, and use the current SGLang CUDA graph flags. Adapt the recipes and registry to the shared variant layout under inferencex-e2e/." + - "Drop enable-w4a4-mxfp4-megamoe from the disaggregated recipes, and set PP_SIZE and PCP_SIZE in the aggregate benchmark environment." + - "将 DeepSeek-V4-Pro-0813 GB300 Dynamo+SGLang AgentX 更新至 nightly-dev-20260916-c9a8fba9 和 Dynamo 1.5.0.dev20260914。聚合配置保留 TP8 c1 与 TP4 c4;DEP8 预填充、DEP16 解码配置覆盖 c64、c240、c480、c960、c1440 和 c1920。" + - "将首 token 延迟较长的 1P1D c480 替换为 1P1D c240 与 2P1D c480,在预填充和解码阶段均使用 DSpark K=6,并采用当前的 SGLang CUDA graph 参数。配方和注册项适配 inferencex-e2e/ 下的共享变体布局。" + - "分离式配方移除 enable-w4a4-mxfp4-megamoe;聚合配置的基准测试环境设置 PP_SIZE 与 PCP_SIZE。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3630