diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml index 2ef179373b..78f8c99a4e 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -107,28 +107,15 @@ base: RESULT_DIR: "/logs/agentic" PORT: "8000" IS_MULTINODE: "false" + # The AgentX power path checks the GPU topology (TP from each variant) before replay. + PP_SIZE: "1" + PCP_SIZE: "1" AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" HF_HUB_CACHE: "/hf_hub_cache" -# Low-latency AgentX aggregate topology: one TP4 worker occupies one -# four-GPU GB300 node and serves both prefill and decode with DSpark K=6. -override_tp4: - name: "agg-gb300-tp4-mtp-lowlatency" - roles: - agg: - nodes: 1 - gpus: 4 - args: - max-running-requests: 32 - cuda-graph-max-bs-decode: 32 - tp-size: 4 - benchmark: - env: - TP: "4" - # Low-latency AgentX aggregate topology: one TP8 worker spans two # four-GPU GB300 nodes and serves both prefill and decode with DSpark K=6. override_tp8: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index ad2af4be37..4ab2c1734d 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -13,16 +13,22 @@ base: model: repo: "deepseek-ai/DeepSeek-V4-Pro-0813" container: - image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21" + image: "lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9" dynamo: install: true source: - wheel: "1.5.0.dev20260902" + wheel: "1.5.0.dev20260914" slurm: time_limit: "8:00:00" health_check: max_attempts: 1440 interval_seconds: 10 + + # Tachometer's per-node exporters slow decode steps by a few percent and add + # host memory on the head node; this ladder is a throughput measurement. + observability: + tachometer: + enabled: false resources: gpu_type: gb300 gpus_per_node: 4 @@ -37,6 +43,55 @@ base: node: dedicated options: max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + args: + - --eviction_high_watermark_ratio=0.90 + # Decode ranks use a chunk cache, so their host DRAM only joins the store + # through standalone stores: four per decode node, so segments stay uniform + # with prefill's per-rank ones instead of one oversized segment per node. + - &decode-store + name: store-decode-0 + type: mooncake-store + placement: + node: decode + args: ["--port", "8800"] + env: + MOONCAKE_PROTOCOL: rdma + MOONCAKE_DEVICE: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" + MOONCAKE_GLOBAL_SEGMENT_SIZE: 180gb + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + # Advertise the store transfer engine on the node address instead of the first active + # interface: some GB300 nodes expose a BMC Redfish interface (bmc_redfish0, the same + # 10.0.1.2 on every node) ahead of the fabric interface. + MC_TCP_BIND_ADDRESS: "{node_ip}" + preamble: | + ulimit -n 1048576 + ulimit -l unlimited + # Cold container starts on a decode node have exceeded srtctl's 120 s default; + # the <<: merge is shallow, so every store repeats the timeout. + readiness: + port: 8800 + timeout_seconds: 600 + - <<: *decode-store + name: store-decode-1 + args: ["--port", "8801"] + readiness: + port: 8801 + timeout_seconds: 600 + - <<: *decode-store + name: store-decode-2 + args: ["--port", "8802"] + readiness: + port: 8802 + timeout_seconds: 600 + - <<: *decode-store + name: store-decode-3 + args: ["--port", "8803"] + readiness: + port: 8803 + timeout_seconds: 600 frontend: type: dynamo nginx_session_affinity: true @@ -71,7 +126,7 @@ base: SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '17408' SGLANG_OPT_USE_ONLINE_COMPRESS: '0' NCCL_MNNVL_ENABLE: '1' NCCL_CUMEM_ENABLE: '1' @@ -86,6 +141,19 @@ base: SGLANG_LOG_MS: '1' SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + MOONCAKE_PROTOCOL: rdma + # The store only registers host buffers on RDMA; without an explicit list it + # falls back to NVLink for same-domain peers and aborts at the warmup put. + MOONCAKE_DEVICE: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" + MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb + MOONCAKE_STANDALONE_STORAGE: '0' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + # Advertise the store transfer engine on the node address instead of the first active + # interface: some GB300 nodes expose a BMC Redfish interface (bmc_redfish0, the same + # 10.0.1.2 on every node) ahead of the fabric interface. + MC_TCP_BIND_ADDRESS: "{node}" + WITH_NVIDIA_PEERMEM: '0' args: host: 0.0.0.0 served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 @@ -107,10 +175,11 @@ base: disaggregation-transfer-backend: mooncake disaggregation-mode: prefill load-balance-method: total_tokens - mem-fraction-static: 0.85 + # 16k-token chunks per DP rank need ~12 GiB more for the DSV4 indexer. + mem-fraction-static: 0.8 page-size: 256 swa-full-tokens-ratio: 0.02 - chunked-prefill-size: 65536 + chunked-prefill-size: 131072 disable-flashinfer-autotune: true model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' @@ -119,10 +188,9 @@ base: speculative-num-steps: 1 speculative-eagle-topk: 1 speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 1 - hicache-io-backend: direct + enable-unified-cache-external-linker: true + unified-cache-external-linker-backend: mooncake + hicache-storage-backend-extra-config: '{"enable_group_semantics":true}' decode: nodes: 4 workers: 1 @@ -155,6 +223,8 @@ base: SGLANG_LOG_FORWARD_ITERS: '1' SGLANG_LOG_MS: '1' SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' args: host: 0.0.0.0 served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 @@ -197,6 +267,10 @@ base: benchmark: type: custom command: bash /infmax-workspace/benchmarks/srt_agentic.sh + # The replay client holds ~300 GB of host memory; on the default head node it + # shares prefill_0's node and OOMs it. Put it on the etcd/nats node instead. + placement: + node: dedicated env: INFMAX_CONTAINER_WORKSPACE: /infmax-workspace RESULT_DIR: /logs/agentic @@ -224,11 +298,11 @@ override_1p1d_c480: gpus: 8 args: max-running-requests: 256 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 decode: gpus: 16 args: - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 # Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300 # (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960. @@ -247,13 +321,13 @@ override_2p1d_c960: OMP_NUM_THREADS: '1' args: max-running-requests: 256 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 decode: env: OMP_NUM_THREADS: '1' SGLANG_DSV4_MHC_PREWARM: '1' args: - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 # Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 # (6P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1440. @@ -271,21 +345,21 @@ override_3p1d_c1440: gpus: 8 args: max-running-requests: 512 - cuda-graph-max-bs: 512 + cuda-graph-max-bs-decode: 512 decode: gpus: 16 args: - cuda-graph-max-bs: 512 + cuda-graph-max-bs-decode: 512 # Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 -# (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920. +# (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 2400. # # Uses the flat single-variant srtctl schema the agentic CI flow expects; # resources + backend (prefill/decode env + sglang_config) are normalized # from the Pareto run. # Concurrency is exported into srt_agentic.sh from the master-config conc-list. -override_4p1d_c1920: - name: "disagg-gb300-8p4d-dep8-dep16-c1920-mtp-kvoffload" +override_4p1d_c2400: + name: "disagg-gb300-8p4d-dep8-dep16-c2400-mtp-kvoffload" frontend: nginx_keepalive_timeout: "900s" env: @@ -301,13 +375,13 @@ override_4p1d_c1920: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" args: max-running-requests: 1024 - cuda-graph-max-bs: 1024 + cuda-graph-max-bs-decode: 1024 decode: gpus: 16 env: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" args: - cuda-graph-max-bs: 192 + cuda-graph-max-bs-decode: 192 benchmark: env: AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 052eb90726..d314c233f0 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -6661,19 +6661,8 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg: dp-attn: false additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp8" - - search-space: - - spec-decoding: draft_model - conc-list: [8] - num-nodes: 1 - worker: - num-worker: 1 - tp: 4 - ep: 1 - dp-attn: false - additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4" dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21 + image: lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9 model: deepseek-ai/DeepSeek-V4-Pro-0813 model-prefix: dsv4 runner: cluster:gb300-nv @@ -6688,9 +6677,9 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - dram-utilization: 0.80 search-space: - spec-decoding: draft_model - conc-list: [480] + conc-list: [120, 480] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: num-worker: 1 tp: 8 @@ -6706,7 +6695,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [960] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: num-worker: 2 tp: 8 @@ -6722,7 +6711,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [1440] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: num-worker: 3 tp: 8 @@ -6736,16 +6725,16 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 16 dp-attn: true - spec-decoding: draft_model - conc-list: [1920] + conc-list: [2400] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: num-worker: 4 tp: 8 ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_4p1d_c1920" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_4p1d_c2400" decode: num-worker: 1 tp: 16 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index db5eaa6965..8488a9a234 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9211,3 +9211,27 @@ - "Add --attention-config '{\"indexer_kv_dtype\":\"mxfp4\",\"indexer_sparse_logits\":true}' and --block-size 128 to enable the vllm-project/vllm#58671 ROCm paged MXFP4 sparse-logits indexer, replacing the dense fp8 indexer path. A live A/B test (TP2 c16, matched 900s window, vllm-project/vllm#58208 reverted via vllm-project/vllm#59125 so the dense fallback doesn't crash) measured +14.7/+14.9% p50/p90 interactivity and -9.5/-10.8% p50/p90 e2e latency over the dense path, with throughput/GPU unchanged." - "Drop c128 from both TP2 and TP4. Neither c128 point was on the Pareto frontier in #3555's run 36528242520: TP2 c64 dominated both (P90 E2EL 58 s against 185 s and 85 s, at 111k against 85k and 79k total tok/s/GPU)." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3571 + +- config-keys: + - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg + - dsv4-fp4-gb300-dynamo-sglang-agentic-agg + scenario-type: + - agentic-coding + description: + - "Move the GB300 DeepSeek-V4-Pro AgentX disaggregated ladder from hierarchical cache to the Mooncake unified-cache external linker on the image's own SGLang, bump the image to lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9, and bump the Dynamo wheel to 1.5.0.dev20260914, whose SGLang worker no longer needs ServerArgs.get_model_config." + - "Decode nodes lend host DRAM to the store through srt-slurm mooncake-store services: four 180GB standalone stores per decode node, started before the workers against the managed master and its HTTP metadata server and gated on their readiness ports with a 600-second timeout, because cold container starts on decode nodes exceeded srtctl's 120-second default. Prefill keeps a 140GB segment per rank." + - "Prefill runs 16k-token chunks per DP rank (chunked-prefill-size 131072, SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408) with mem-fraction-static 0.80, which the DSV4 indexer needs at that chunk size." + - "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent." + - "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the prefill role and the decode stores, the processes that act as store clients, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." + - "The ladder's widest point moves from concurrency 1920 to 2400 on the same 4P1D topology, where prefill still had headroom, and the 1P1D point also runs at concurrency 120. At equal P90 interactivity it delivers about twice the throughput per GPU of the aggregate TP4 concurrency-8 point, which is removed from dsv4-fp4-gb300-dynamo-sglang-agentic-agg." + - "The aggregate recipe sets PP_SIZE and PCP_SIZE to 1 in its benchmark environment: the AgentX power path now checks TP, PP_SIZE and PCP_SIZE before replay, and multi-node aggregate jobs do not receive them from the workflow." + - "The prefill role and the decode store services set MC_TCP_BIND_ADDRESS to their node so each store transfer engine advertises the node address instead of the first active interface; some GB300 nodes expose a BMC Redfish interface with the same 10.0.1.2 address on every node, which peers cannot reach." + - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。" + - "decode 节点通过 srt-slurm 的 mooncake-store 服务向 store 借出主机 DRAM:每个 decode 节点 4 个 180GB 的独立 store,在 worker 之前启动,连接托管的 master 及其 HTTP metadata server,并以就绪端口作为启动检查,超时设为 600 秒,因为 decode 节点上冷启动容器曾超过 srtctl 默认的 120 秒。prefill 每个 rank 保留 140GB segment。" + - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" + - "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。" + - "配方在 prefill 角色与 decode 节点的 store 上(即作为 store 客户端的进程)显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" + - "阶梯中吞吐最高的点在相同的 4P1D 拓扑下由并发 1920 调整为 2400(该并发下 prefill 仍有余量),1P1D 点同时增加并发 120。在相同的 P90 交互性下,其单卡吞吐约为聚合 TP4 并发 8 点的两倍,因此从 dsv4-fp4-gb300-dynamo-sglang-agentic-agg 中移除该聚合点。" + - "聚合式配方在 benchmark 环境中设置 PP_SIZE 与 PCP_SIZE 为 1:AgentX 功耗路径现在会在请求回放前检查 TP、PP_SIZE 与 PCP_SIZE,而多节点聚合任务不会从工作流获得这两个变量。" + - "prefill 角色与 decode 的 store 服务将 MC_TCP_BIND_ADDRESS 设为所在节点,使每个 store 传输引擎对外公布节点自身地址,而不是第一个活动网卡;部分 GB300 节点带有 BMC Redfish 网卡,其地址在每个节点上都是 10.0.1.2,其他节点无法访问。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187