Skip to content
Open
Original file line number Diff line number Diff line change
Expand Up @@ -13,11 +13,11 @@ base:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro-0813"
container:
image: "lmsysorg/sglang:v0.5.19-cu130@sha256:d6e7288627be8b02be88e4bba38e73f6d50e2826869f753c13a4c4385ab3eda9"
image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9"
dynamo:
install: true
source:
wheel: "1.5.0.dev20260910"
wheel: "1.5.0.dev20260914"
slurm:
time_limit: "4:00:00"
health_check:
Expand Down Expand Up @@ -77,7 +77,7 @@ base:
weight-loader-prefetch-checkpoints: true
stream-interval: 10
watchdog-timeout: 1000000
mem-fraction-static: 0.94
mem-fraction-static: 0.90
page-size: 256
chunked-prefill-size: 8192
max-prefill-tokens: 8192
Expand Down Expand Up @@ -107,6 +107,8 @@ base:
RESULT_DIR: "/logs/agentic"
PORT: "8000"
IS_MULTINODE: "false"
PP_SIZE: "1"
PCP_SIZE: "1"
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:"
Expand All @@ -115,31 +117,31 @@ base:

# Low-latency AgentX aggregate topology: one TP4 worker occupies one
# four-GPU GB300 node and serves both prefill and decode with DSpark K=6.
override_tp4:
name: "agg-gb300-tp4-mtp-lowlatency"
override_tp4_c4:
name: "agg-gb300-tp4-c4-mtp-lowlatency"
roles:
agg:
nodes: 1
gpus: 4
args:
max-running-requests: 32
cuda-graph-max-bs-decode: 32
max-running-requests: 8
cuda-graph-max-bs-decode: 8
tp-size: 4
benchmark:
env:
TP: "4"

# Low-latency AgentX aggregate topology: one TP8 worker spans two
# four-GPU GB300 nodes and serves both prefill and decode with DSpark K=6.
override_tp8:
name: "agg-gb300-tp8-mtp-lowlatency"
override_tp8_c1:
name: "agg-gb300-tp8-c1-mtp-lowlatency"
roles:
agg:
nodes: 2
gpus: 8
args:
max-running-requests: 4
cuda-graph-max-bs-decode: 4
max-running-requests: 2
cuda-graph-max-bs-decode: 2
tp-size: 8
benchmark:
env:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -13,11 +13,11 @@ base:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro-0813"
container:
image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21"
image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9"
dynamo:
install: true
source:
wheel: "1.5.0.dev20260902"
wheel: "1.5.0.dev20260914"
slurm:
time_limit: "8:00:00"
health_check:
Expand Down Expand Up @@ -103,7 +103,6 @@ base:
moe-dense-tp-size: 1
moe-a2a-backend: megamoe
enable-deepseek-v4-fp4-indexer: true
enable-w4a4-mxfp4-megamoe: true
disaggregation-transfer-backend: mooncake
disaggregation-mode: prefill
load-balance-method: total_tokens
Expand Down Expand Up @@ -172,7 +171,6 @@ base:
moe-dense-tp-size: 1
moe-a2a-backend: megamoe
enable-deepseek-v4-fp4-indexer: true
enable-w4a4-mxfp4-megamoe: true
disaggregation-transfer-backend: mooncake
disaggregation-mode: decode
load-balance-method: total_tokens
Expand Down Expand Up @@ -208,27 +206,57 @@ base:
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"

# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300
# (2P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 480.
#
# Uses the flat single-variant srtctl schema the agentic CI flow expects;
# resources + backend (prefill/decode env + sglang_config) are normalized
# from the Pareto run.
# Concurrency is exported into srt_agentic.sh from the master-config conc-list.
override_1p1d_c480:
name: "disagg-gb300-2p4d-dep8-dep16-c480-mtp-kvoffload"
# One DEP8 prefill worker and one DEP16 decode worker at concurrency 64.
override_1p1d_c64:
name: "disagg-gb300-1p1d-dep8-dep16-c64-mtp-kvoffload"
roles:
prefill:
nodes: 2
workers: 1
gpus: 8
args:
max-running-requests: 256
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256
hicache-mem-layout: page_first_direct
decode:
gpus: 16
args:
max-running-requests: 128
cuda-graph-max-bs-decode: 256

# One DEP8 prefill worker and one DEP16 decode worker at concurrency 240.
override_1p1d_c240:
name: "disagg-gb300-1p1d-dep8-dep16-c240-mtp-kvoffload"
roles:
prefill:
nodes: 2
workers: 1
gpus: 8
args:
max-running-requests: 512
cuda-graph-max-bs-decode: 256
decode:
gpus: 16
args:
cuda-graph-max-bs: 256
max-running-requests: 512
cuda-graph-max-bs-decode: 256

# Two DEP8 prefill workers and one DEP16 decode worker at concurrency 480.
override_2p1d_c480:
name: "disagg-gb300-2p1d-dep8-dep16-c480-mtp-kvoffload"
roles:
prefill:
nodes: 4
workers: 2
gpus: 8
args:
max-running-requests: 512
cuda-graph-max-bs-decode: 256
decode:
gpus: 16
args:
max-running-requests: 1024
cuda-graph-max-bs-decode: 256

# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300
# (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960.
Expand All @@ -247,13 +275,13 @@ override_2p1d_c960:
OMP_NUM_THREADS: '1'
args:
max-running-requests: 256
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256
decode:
env:
OMP_NUM_THREADS: '1'
SGLANG_DSV4_MHC_PREWARM: '1'
args:
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256

# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300
# (6P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1440.
Expand All @@ -271,11 +299,11 @@ override_3p1d_c1440:
gpus: 8
args:
max-running-requests: 512
cuda-graph-max-bs: 512
cuda-graph-max-bs-decode: 512
decode:
gpus: 16
args:
cuda-graph-max-bs: 512
cuda-graph-max-bs-decode: 512

# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300
# (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920.
Expand All @@ -301,13 +329,13 @@ override_4p1d_c1920:
SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600"
args:
max-running-requests: 1024
cuda-graph-max-bs: 1024
cuda-graph-max-bs-decode: 1024
decode:
gpus: 16
env:
SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600"
args:
cuda-graph-max-bs: 192
cuda-graph-max-bs-decode: 192
benchmark:
env:
AIPERF_HTTP_TCP_USER_TIMEOUT: "900000"
52 changes: 42 additions & 10 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6639,47 +6639,47 @@ kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2:
additional-settings:
- "CONFIG_FILE=recipes/kimik3/vllm/gb200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml"
dsv4-fp4-gb300-dynamo-sglang-agentic-agg:
image: lmsysorg/sglang:v0.5.19-cu130@sha256:d6e7288627be8b02be88e4bba38e73f6d50e2826869f753c13a4c4385ab3eda9
image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9
model: deepseek-ai/DeepSeek-V4-Pro-0813
model-prefix: dsv4
runner: cluster:gb300-nv
precision: fp4
framework: dynamo-sglang
router: { name: dynamo-router, version: "1.5.0.dev20260910" }
router: { name: dynamo-router, version: "1.5.0.dev20260914" }
multinode: true
disagg: false
scenarios:
agentic-coding:
- search-space:
- spec-decoding: draft_model
conc-list: [1, 4]
conc-list: [1]
num-nodes: 2
worker:
num-worker: 1
tp: 8
ep: 1
dp-attn: false
additional-settings:
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp8"
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp8_c1"
- search-space:
- spec-decoding: draft_model
conc-list: [8]
conc-list: [4]
num-nodes: 1
worker:
num-worker: 1
tp: 4
ep: 1
dp-attn: false
additional-settings:
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4"
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4_c4"
dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21
image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9
model: deepseek-ai/DeepSeek-V4-Pro-0813
model-prefix: dsv4
runner: cluster:gb300-nv
precision: fp4
framework: dynamo-sglang
router: { name: dynamo-router, version: "1.5.0.dev20260902" }
router: { name: dynamo-router, version: "1.5.0.dev20260914" }
kv-p2p-transfer: mooncake
multinode: true
disagg: true
Expand All @@ -6688,7 +6688,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
- dram-utilization: 0.80
search-space:
- spec-decoding: draft_model
conc-list: [480]
conc-list: [64]
kv-offloading: dram
kv-offload-backend: { name: hicache }
prefill:
Expand All @@ -6697,7 +6697,39 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
ep: 8
dp-attn: true
additional-settings:
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c480"
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c64"
decode:
num-worker: 1
tp: 16
ep: 16
dp-attn: true
- spec-decoding: draft_model
conc-list: [240]
kv-offloading: dram
kv-offload-backend: { name: hicache }
prefill:
num-worker: 1
tp: 8
ep: 8
dp-attn: true
additional-settings:
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c240"
decode:
num-worker: 1
tp: 16
ep: 16
dp-attn: true
- spec-decoding: draft_model
conc-list: [480]
kv-offloading: dram
kv-offload-backend: { name: hicache }
prefill:
num-worker: 2
tp: 8
ep: 8
dp-attn: true
additional-settings:
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_2p1d_c480"
decode:
num-worker: 1
tp: 16
Expand Down
14 changes: 14 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9211,3 +9211,17 @@
- "Add --attention-config '{\"indexer_kv_dtype\":\"mxfp4\",\"indexer_sparse_logits\":true}' and --block-size 128 to enable the vllm-project/vllm#58671 ROCm paged MXFP4 sparse-logits indexer, replacing the dense fp8 indexer path. A live A/B test (TP2 c16, matched 900s window, vllm-project/vllm#58208 reverted via vllm-project/vllm#59125 so the dense fallback doesn't crash) measured +14.7/+14.9% p50/p90 interactivity and -9.5/-10.8% p50/p90 e2e latency over the dense path, with throughput/GPU unchanged."
- "Drop c128 from both TP2 and TP4. Neither c128 point was on the Pareto frontier in #3555's run 36528242520: TP2 c64 dominated both (P90 E2EL 58 s against 185 s and 85 s, at 111k against 85k and 79k total tok/s/GPU)."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3571

- config-keys:
- dsv4-fp4-gb300-dynamo-sglang-agentic-agg
- dsv4-fp4-gb300-dynamo-sglang-agentic-disagg
scenario-type:
- agentic-coding
description:
- "Refresh DeepSeek-V4-Pro-0813 GB300 Dynamo+SGLang AgentX on nightly-dev-20260916-c9a8fba9 and Dynamo 1.5.0.dev20260914. Keep TP8 c1 and TP4 c4 aggregate points, and DEP8 prefill / DEP16 decode points at c64, c240, c480, c960, c1440 and c1920."
- "Replace the long-TTFT 1P1D c480 point with 1P1D c240 and 2P1D c480, retain DSpark K=6 on both serving stages, and use the current SGLang CUDA graph flags. Adapt the recipes and registry to the shared variant layout under inferencex-e2e/."
- "Drop enable-w4a4-mxfp4-megamoe from the disaggregated recipes, and set PP_SIZE and PCP_SIZE in the aggregate benchmark environment."
- "将 DeepSeek-V4-Pro-0813 GB300 Dynamo+SGLang AgentX 更新至 nightly-dev-20260916-c9a8fba9 和 Dynamo 1.5.0.dev20260914。聚合配置保留 TP8 c1 与 TP4 c4;DEP8 预填充、DEP16 解码配置覆盖 c64、c240、c480、c960、c1440 和 c1920。"
- "将首 token 延迟较长的 1P1D c480 替换为 1P1D c240 与 2P1D c480,在预填充和解码阶段均使用 DSpark K=6,并采用当前的 SGLang CUDA graph 参数。配方和注册项适配 inferencex-e2e/ 下的共享变体布局。"
- "分离式配方移除 enable-w4a4-mxfp4-megamoe;聚合配置的基准测试环境设置 PP_SIZE 与 PCP_SIZE。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3630
Loading