Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -107,28 +107,15 @@ base:
RESULT_DIR: "/logs/agentic"
PORT: "8000"
IS_MULTINODE: "false"
# The AgentX power path checks the GPU topology (TP from each variant) before replay.
PP_SIZE: "1"
PCP_SIZE: "1"
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"

# Low-latency AgentX aggregate topology: one TP4 worker occupies one
# four-GPU GB300 node and serves both prefill and decode with DSpark K=6.
override_tp4:
name: "agg-gb300-tp4-mtp-lowlatency"
roles:
agg:
nodes: 1
gpus: 4
args:
max-running-requests: 32
cuda-graph-max-bs-decode: 32
tp-size: 4
benchmark:
env:
TP: "4"

# Low-latency AgentX aggregate topology: one TP8 worker spans two
# four-GPU GB300 nodes and serves both prefill and decode with DSpark K=6.
override_tp8:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -13,16 +13,22 @@ base:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro-0813"
container:
image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21"
image: "lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9"
dynamo:
install: true
source:
wheel: "1.5.0.dev20260902"
wheel: "1.5.0.dev20260914"
slurm:
time_limit: "8:00:00"
health_check:
max_attempts: 1440
interval_seconds: 10

# Tachometer's per-node exporters slow decode steps by a few percent and add
# host memory on the head node; this ladder is a throughput measurement.
observability:
tachometer:
enabled: false
resources:
gpu_type: gb300
gpus_per_node: 4
Expand All @@ -37,6 +43,51 @@ base:
node: dedicated
options:
max_payload_mb: 32
- name: mooncake-master
type: mooncake-master
args:
- --eviction_high_watermark_ratio=0.90
# Decode ranks use a chunk cache, so their host DRAM only joins the store
# through standalone stores: four per decode node, so segments stay uniform
# with prefill's per-rank ones instead of one oversized segment per node.
- &decode-store
name: store-decode-0
type: mooncake-store
placement:
node: decode
args: ["--port", "8800"]
env:
MOONCAKE_PROTOCOL: rdma
MOONCAKE_DEVICE: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
MOONCAKE_GLOBAL_SEGMENT_SIZE: 180gb
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
preamble: |
ulimit -n 1048576
ulimit -l unlimited
# Cold container starts on a decode node have exceeded srtctl's 120 s default;
# the <<: merge is shallow, so every store repeats the timeout.
readiness:
port: 8800
timeout_seconds: 600
- <<: *decode-store
name: store-decode-1
args: ["--port", "8801"]
readiness:
port: 8801
timeout_seconds: 600
- <<: *decode-store
name: store-decode-2
args: ["--port", "8802"]
readiness:
port: 8802
timeout_seconds: 600
- <<: *decode-store
name: store-decode-3
args: ["--port", "8803"]
readiness:
port: 8803
timeout_seconds: 600
frontend:
type: dynamo
nginx_session_affinity: true
Expand Down Expand Up @@ -71,7 +122,7 @@ base:
SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1"
SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1'
SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1'
SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216'
SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '17408'
SGLANG_OPT_USE_ONLINE_COMPRESS: '0'
NCCL_MNNVL_ENABLE: '1'
NCCL_CUMEM_ENABLE: '1'
Expand All @@ -86,6 +137,15 @@ base:
SGLANG_LOG_MS: '1'
SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60'
SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1'
MOONCAKE_PROTOCOL: rdma
# The store only registers host buffers on RDMA; without an explicit list it
# falls back to NVLink for same-domain peers and aborts at the warmup put.
MOONCAKE_DEVICE: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb
MOONCAKE_STANDALONE_STORAGE: '0'
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
args:
host: 0.0.0.0
served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813
Expand All @@ -107,10 +167,11 @@ base:
disaggregation-transfer-backend: mooncake
disaggregation-mode: prefill
load-balance-method: total_tokens
mem-fraction-static: 0.85
# 16k-token chunks per DP rank need ~12 GiB more for the DSV4 indexer.
mem-fraction-static: 0.8
page-size: 256
swa-full-tokens-ratio: 0.02
chunked-prefill-size: 65536
chunked-prefill-size: 131072
disable-flashinfer-autotune: true
model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}'
kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}'
Expand All @@ -119,10 +180,9 @@ base:
speculative-num-steps: 1
speculative-eagle-topk: 1
speculative-num-draft-tokens: 7
enable-hierarchical-cache: true
hicache-write-policy: write_back
hicache-ratio: 1
hicache-io-backend: direct
enable-unified-cache-external-linker: true
unified-cache-external-linker-backend: mooncake
hicache-storage-backend-extra-config: '{"enable_group_semantics":true}'
decode:
nodes: 4
workers: 1
Expand Down Expand Up @@ -155,6 +215,8 @@ base:
SGLANG_LOG_FORWARD_ITERS: '1'
SGLANG_LOG_MS: '1'
SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60'
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
args:
host: 0.0.0.0
served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813
Expand Down Expand Up @@ -197,6 +259,10 @@ base:
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
# The replay client holds ~300 GB of host memory; on the default head node it
# shares prefill_0's node and OOMs it. Put it on the etcd/nats node instead.
placement:
node: dedicated
env:
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
Expand Down Expand Up @@ -224,11 +290,11 @@ override_1p1d_c480:
gpus: 8
args:
max-running-requests: 256
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256
decode:
gpus: 16
args:
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256

# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300
# (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960.
Expand All @@ -247,13 +313,13 @@ override_2p1d_c960:
OMP_NUM_THREADS: '1'
args:
max-running-requests: 256
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256
decode:
env:
OMP_NUM_THREADS: '1'
SGLANG_DSV4_MHC_PREWARM: '1'
args:
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256

# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300
# (6P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1440.
Expand All @@ -271,21 +337,21 @@ override_3p1d_c1440:
gpus: 8
args:
max-running-requests: 512
cuda-graph-max-bs: 512
cuda-graph-max-bs-decode: 512
decode:
gpus: 16
args:
cuda-graph-max-bs: 512
cuda-graph-max-bs-decode: 512

# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300
# (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920.
# (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 2400.
#
# Uses the flat single-variant srtctl schema the agentic CI flow expects;
# resources + backend (prefill/decode env + sglang_config) are normalized
# from the Pareto run.
# Concurrency is exported into srt_agentic.sh from the master-config conc-list.
override_4p1d_c1920:
name: "disagg-gb300-8p4d-dep8-dep16-c1920-mtp-kvoffload"
override_4p1d_c2400:
name: "disagg-gb300-8p4d-dep8-dep16-c2400-mtp-kvoffload"
frontend:
nginx_keepalive_timeout: "900s"
env:
Expand All @@ -301,13 +367,13 @@ override_4p1d_c1920:
SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600"
args:
max-running-requests: 1024
cuda-graph-max-bs: 1024
cuda-graph-max-bs-decode: 1024
decode:
gpus: 16
env:
SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600"
args:
cuda-graph-max-bs: 192
cuda-graph-max-bs-decode: 192
benchmark:
env:
AIPERF_HTTP_TCP_USER_TIMEOUT: "900000"
27 changes: 8 additions & 19 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6661,19 +6661,8 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg:
dp-attn: false
additional-settings:
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp8"
- search-space:
- spec-decoding: draft_model
conc-list: [8]
num-nodes: 1
worker:
num-worker: 1
tp: 4
ep: 1
dp-attn: false
additional-settings:
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4"
dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21
image: lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9
model: deepseek-ai/DeepSeek-V4-Pro-0813
model-prefix: dsv4
runner: cluster:gb300-nv
Expand All @@ -6688,9 +6677,9 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
- dram-utilization: 0.80
search-space:
- spec-decoding: draft_model
conc-list: [480]
conc-list: [120, 480]
kv-offloading: dram
kv-offload-backend: { name: hicache }
kv-offload-backend: { name: mooncake, version: "0.3.13" }
prefill:
num-worker: 1
tp: 8
Expand All @@ -6706,7 +6695,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
- spec-decoding: draft_model
conc-list: [960]
kv-offloading: dram
kv-offload-backend: { name: hicache }
kv-offload-backend: { name: mooncake, version: "0.3.13" }
prefill:
num-worker: 2
tp: 8
Expand All @@ -6722,7 +6711,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
- spec-decoding: draft_model
conc-list: [1440]
kv-offloading: dram
kv-offload-backend: { name: hicache }
kv-offload-backend: { name: mooncake, version: "0.3.13" }
prefill:
num-worker: 3
tp: 8
Expand All @@ -6736,16 +6725,16 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
ep: 16
dp-attn: true
- spec-decoding: draft_model
conc-list: [1920]
conc-list: [2400]
kv-offloading: dram
kv-offload-backend: { name: hicache }
kv-offload-backend: { name: mooncake, version: "0.3.13" }
prefill:
num-worker: 4
tp: 8
ep: 8
dp-attn: true
additional-settings:
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_4p1d_c1920"
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_4p1d_c2400"
decode:
num-worker: 1
tp: 16
Expand Down
22 changes: 22 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9168,3 +9168,25 @@
- "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid."
- "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605

- config-keys:
- dsv4-fp4-gb300-dynamo-sglang-agentic-disagg
- dsv4-fp4-gb300-dynamo-sglang-agentic-agg
scenario-type:
- agentic-coding
description:
- "Move the GB300 DeepSeek-V4-Pro AgentX disaggregated ladder from hierarchical cache to the Mooncake unified-cache external linker on the image's own SGLang, bump the image to lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9, and bump the Dynamo wheel to 1.5.0.dev20260914, whose SGLang worker no longer needs ServerArgs.get_model_config."
- "Decode nodes lend host DRAM to the store through srt-slurm mooncake-store services: four 180GB standalone stores per decode node, started before the workers against the managed master and its HTTP metadata server and gated on their readiness ports with a 600-second timeout, because cold container starts on decode nodes exceeded srtctl's 120-second default. Prefill keeps a 140GB segment per rank."
- "Prefill runs 16k-token chunks per DP rank (chunked-prefill-size 131072, SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408) with mem-fraction-static 0.80, which the DSV4 indexer needs at that chunk size."
- "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent."
- "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the prefill role and the decode stores, the processes that act as store clients, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put."
- "The ladder's widest point moves from concurrency 1920 to 2400 on the same 4P1D topology, where prefill still had headroom, and the 1P1D point also runs at concurrency 120. At equal P90 interactivity it delivers about twice the throughput per GPU of the aggregate TP4 concurrency-8 point, which is removed from dsv4-fp4-gb300-dynamo-sglang-agentic-agg."
- "The aggregate recipe sets PP_SIZE and PCP_SIZE to 1 in its benchmark environment: the AgentX power path now checks TP, PP_SIZE and PCP_SIZE before replay, and multi-node aggregate jobs do not receive them from the workflow."
- "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。"
- "decode 节点通过 srt-slurm 的 mooncake-store 服务向 store 借出主机 DRAM:每个 decode 节点 4 个 180GB 的独立 store,在 worker 之前启动,连接托管的 master 及其 HTTP metadata server,并以就绪端口作为启动检查,超时设为 600 秒,因为 decode 节点上冷启动容器曾超过 srtctl 默认的 120 秒。prefill 每个 rank 保留 140GB segment。"
- "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。"
- "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。"
- "配方在 prefill 角色与 decode 节点的 store 上(即作为 store 客户端的进程)显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。"
- "阶梯中吞吐最高的点在相同的 4P1D 拓扑下由并发 1920 调整为 2400(该并发下 prefill 仍有余量),1P1D 点同时增加并发 120。在相同的 P90 交互性下,其单卡吞吐约为聚合 TP4 并发 8 点的两倍,因此从 dsv4-fp4-gb300-dynamo-sglang-agentic-agg 中移除该聚合点。"
- "聚合式配方在 benchmark 环境中设置 PP_SIZE 与 PCP_SIZE 为 1:AgentX 功耗路径现在会在请求回放前检查 TP、PP_SIZE 与 PCP_SIZE,而多节点聚合任务不会从工作流获得这两个变量。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187
Loading