From e8afb8965ed23de4bd99c17ef6ec0ac53600b7ae Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Wed, 30 Sep 2026 09:46:31 +0800 Subject: [PATCH 01/12] feat(agentx): move GB300 DSV4 AgentX disagg to the Mooncake external linker Replace hierarchical cache with the Mooncake unified-cache external linker on the image's own SGLang and bump the image to nightly-dev-cu13-20260916-c9a8fba9. Decode nodes lend host DRAM to the store through four 180GB standalone mooncake_client daemons each, started by the recipe's setup script and registered with the HTTP metadata server; the job fails unless every daemon serves and resolves. Prefill runs 16k-token chunks per DP rank with MegaMoE sized to match and mem-fraction-static 0.80. The benchmark client moves to the dedicated etcd/nats node, where it no longer OOM-kills prefill_0, and Tachometer is off for this ladder. The gb300-nv srt lane gains role_env, which sets this cluster's RDMA devices on the recipe's prefill and decode roles in place of the removed bash launcher block. --- .../configs/dsv4-gb300-mooncake-sidecar.sh | 77 +++++++++++++++++++ .../sitecustomize.py | 26 +++++++ .../gb300-fp4/agentx/disagg-variants.yaml | 73 ++++++++++++++---- inferencex-e2e/configs/nvidia-master.yaml | 10 +-- .../infx/launch/drivers/srt/__init__.py | 1 + .../infx/launch/drivers/srt/lanes.py | 35 +++++++++ .../infx/tests/launch/test_srt_policy.py | 30 +++++++- inferencex-e2e/perf-changelog.yaml | 17 ++++ 8 files changed, 247 insertions(+), 22 deletions(-) create mode 100755 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-mooncake-sidecar.sh create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-mooncake-sidecar.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-mooncake-sidecar.sh new file mode 100755 index 0000000000..a4a8076204 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-mooncake-sidecar.sh @@ -0,0 +1,77 @@ +#!/usr/bin/env bash +# On decode nodes, start standalone Mooncake store daemons that lend host DRAM +# to the store. No-op on roles that do not set MC_SIDECAR_SEGMENT_SIZE. +set -euo pipefail + +# Decode ranks use a chunk cache, so their host memory never joins the store on +# its own. Standalone daemons on the decode nodes mount it instead; SGLang never +# talks to them, only the master sees their segments. Only roles that set +# MC_SIDECAR_SEGMENT_SIZE start them. +if [ -z "${MC_SIDECAR_SEGMENT_SIZE:-}" ]; then + exit 0 +fi +for var in MOONCAKE_MASTER MOONCAKE_TE_META_DATA_SERVER; do + if [ -z "${!var:-}" ]; then + echo "ERROR: ${var} unset; cannot start Mooncake store daemons" >&2 + exit 1 + fi +done + +MY_IP=$(hostname -i | awk '{print $1}') +HOST=$(hostname) +COUNT="${MC_SIDECAR_COUNT:-1}" +echo "[mc-sidecar] host=${MY_IP} master=${MOONCAKE_MASTER} count=${COUNT} size=${MC_SIDECAR_SEGMENT_SIZE}" + +# One daemon per rank-equivalent keeps segment sizes uniform across the pool; a +# single oversized segment would concentrate every put on this node's NICs. +PIDS=() +for i in $(seq 0 $((COUNT - 1))); do + PORT=$((50052 + i)) + # Writers resolve a segment's transfer-engine descriptor through the HTTP + # metadata server. A daemon on P2PHANDSHAKE mounts fine, but every remote + # open of its segment 404s and the puts placed there are revoked. + nohup mooncake_client \ + --host="${MY_IP}" \ + --port="${PORT}" \ + --master_server_address="${MOONCAKE_MASTER}" \ + --metadata_server="${MOONCAKE_TE_META_DATA_SERVER}" \ + --protocol="${MOONCAKE_PROTOCOL:-rdma}" \ + --device_names="${MOONCAKE_DEVICE:-}" \ + --global_segment_size="${MC_SIDECAR_SEGMENT_SIZE}" \ + --local_buffer_size=1GB \ + > "/logs/mc_sidecar_${HOST}_${PORT}.out" 2>&1 & + PIDS+=($!) +done + +mounted=0 +for _ in $(seq 1 180); do + mounted=$( (grep -l "Starting real client service" /logs/mc_sidecar_"${HOST}"_*.out 2>/dev/null || true) | wc -l) + [ "${mounted}" -ge "${COUNT}" ] && break + sleep 1 +done +if [ "${mounted}" -lt "${COUNT}" ]; then + echo "ERROR: only ${mounted}/${COUNT} Mooncake store daemons serving within 180s" >&2 + exit 1 +fi + +# A daemon has exited shortly after startup before; require all to survive. +sleep 20 +for pid in "${PIDS[@]}"; do + if ! kill -0 "${pid}" 2>/dev/null; then + echo "ERROR: Mooncake store daemon pid ${pid} exited after startup" >&2 + exit 1 + fi +done + +# The descriptor must resolve exactly the way remote writers look it up. The +# segment name is the transfer-engine host:port, not the RPC listen port. +for f in /logs/mc_sidecar_"${HOST}"_*.out; do + ep=$(grep -aoE "parseHostNameWithPort\. server_name: [0-9.]+ port: [0-9]+" "${f}" | head -1 | awk '{print $3 ":" $5}' || true) + key="mooncake%2Fram%2F${ep%:*}%3A${ep#*:}" + code=$(curl -s -o /dev/null -w '%{http_code}' "${MOONCAKE_TE_META_DATA_SERVER}?key=${key}" || true) + if [ -z "${ep}" ] || [ "${code}" != "200" ]; then + echo "ERROR: descriptor for '${ep}' not on metadata server (http=${code})" >&2 + exit 1 + fi +done +echo "[mc-sidecar] ${COUNT} daemons serving and resolvable" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py new file mode 100644 index 0000000000..a9c2086b56 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py @@ -0,0 +1,26 @@ +"""TEMPORARY: restore ServerArgs.get_model_config for the pinned Dynamo wheel. + +The image's SGLang moved model-config resolution out of ServerArgs, but the +pinned Dynamo wheel still calls ``server_args.get_model_config()`` while +parsing worker arguments. Re-attach the accessor so the wheel keeps working. + +This file is picked up because its directory is on PYTHONPATH, so it is +imported by every interpreter in the job, including ones that never import +SGLang. Failing to import SGLang there is expected and must stay silent. + +Remove this directory and its PYTHONPATH entry once the pinned Dynamo wheel +stops calling the accessor. +""" + +try: + from sglang.srt.arg_groups.model_override_base import model_config_of + from sglang.srt.server_args import ServerArgs +except Exception: # noqa: BLE001 - non-SGLang interpreters legitimately fail here + pass +else: + if not hasattr(ServerArgs, "get_model_config"): + + def get_model_config(self): + return model_config_of(self) + + ServerArgs.get_model_config = get_model_config diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index ad2af4be37..82dc2c6453 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -13,7 +13,7 @@ base: model: repo: "deepseek-ai/DeepSeek-V4-Pro-0813" container: - image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21" + image: "lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9" dynamo: install: true source: @@ -23,6 +23,21 @@ base: health_check: max_attempts: 1440 interval_seconds: 10 + + # Starts the standalone Mooncake store daemons that lend decode-node host + # DRAM to the store (see MC_SIDECAR_* on the decode role). + setup_script: dsv4-gb300-mooncake-sidecar.sh + + # TEMPORARY: the pinned Dynamo wheel still calls ServerArgs.get_model_config, + # which this image's SGLang no longer has. Remove once the wheel catches up. + environment: + PYTHONPATH: /configs/sglang-server-args-compat + + # Tachometer's per-node exporters slow decode steps by a few percent and add + # host memory on the head node; this ladder is a throughput measurement. + observability: + tachometer: + enabled: false resources: gpu_type: gb300 gpus_per_node: 4 @@ -37,6 +52,10 @@ base: node: dedicated options: max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + args: + - --eviction_high_watermark_ratio=0.90 frontend: type: dynamo nginx_session_affinity: true @@ -71,7 +90,7 @@ base: SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '17408' SGLANG_OPT_USE_ONLINE_COMPRESS: '0' NCCL_MNNVL_ENABLE: '1' NCCL_CUMEM_ENABLE: '1' @@ -86,6 +105,13 @@ base: SGLANG_LOG_MS: '1' SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + MOONCAKE_PROTOCOL: rdma + MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb + MOONCAKE_STANDALONE_STORAGE: '0' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' args: host: 0.0.0.0 served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 @@ -107,10 +133,11 @@ base: disaggregation-transfer-backend: mooncake disaggregation-mode: prefill load-balance-method: total_tokens - mem-fraction-static: 0.85 + # 16k-token chunks per DP rank need ~12 GiB more for the DSV4 indexer. + mem-fraction-static: 0.8 page-size: 256 swa-full-tokens-ratio: 0.02 - chunked-prefill-size: 65536 + chunked-prefill-size: 131072 disable-flashinfer-autotune: true model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' @@ -119,10 +146,9 @@ base: speculative-num-steps: 1 speculative-eagle-topk: 1 speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 1 - hicache-io-backend: direct + enable-unified-cache-external-linker: true + unified-cache-external-linker-backend: mooncake + hicache-storage-backend-extra-config: '{"enable_group_semantics":true}' decode: nodes: 4 workers: 1 @@ -155,6 +181,17 @@ base: SGLANG_LOG_FORWARD_ITERS: '1' SGLANG_LOG_MS: '1' SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + MOONCAKE_PROTOCOL: rdma + MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb + MOONCAKE_STANDALONE_STORAGE: '0' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + # Consumed by the setup script: each decode node starts this many + # standalone Mooncake store daemons of this size to lend its host DRAM. + MC_SIDECAR_SEGMENT_SIZE: 180GB + MC_SIDECAR_COUNT: '4' args: host: 0.0.0.0 served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 @@ -197,6 +234,10 @@ base: benchmark: type: custom command: bash /infmax-workspace/benchmarks/srt_agentic.sh + # The replay client holds ~300 GB of host memory; on the default head node it + # shares prefill_0's node and OOMs it. Put it on the etcd/nats node instead. + placement: + node: dedicated env: INFMAX_CONTAINER_WORKSPACE: /infmax-workspace RESULT_DIR: /logs/agentic @@ -224,11 +265,11 @@ override_1p1d_c480: gpus: 8 args: max-running-requests: 256 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 decode: gpus: 16 args: - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 # Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300 # (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960. @@ -247,13 +288,13 @@ override_2p1d_c960: OMP_NUM_THREADS: '1' args: max-running-requests: 256 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 decode: env: OMP_NUM_THREADS: '1' SGLANG_DSV4_MHC_PREWARM: '1' args: - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 # Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 # (6P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1440. @@ -271,11 +312,11 @@ override_3p1d_c1440: gpus: 8 args: max-running-requests: 512 - cuda-graph-max-bs: 512 + cuda-graph-max-bs-decode: 512 decode: gpus: 16 args: - cuda-graph-max-bs: 512 + cuda-graph-max-bs-decode: 512 # Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 # (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920. @@ -301,13 +342,13 @@ override_4p1d_c1920: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" args: max-running-requests: 1024 - cuda-graph-max-bs: 1024 + cuda-graph-max-bs-decode: 1024 decode: gpus: 16 env: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" args: - cuda-graph-max-bs: 192 + cuda-graph-max-bs-decode: 192 benchmark: env: AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 07c47e6e40..466776f0b3 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -6673,7 +6673,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg: additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4" dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21 + image: lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9 model: deepseek-ai/DeepSeek-V4-Pro-0813 model-prefix: dsv4 runner: cluster:gb300-nv @@ -6690,7 +6690,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [480] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake } prefill: num-worker: 1 tp: 8 @@ -6706,7 +6706,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [960] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake } prefill: num-worker: 2 tp: 8 @@ -6722,7 +6722,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [1440] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake } prefill: num-worker: 3 tp: 8 @@ -6738,7 +6738,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [1920] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake } prefill: num-worker: 4 tp: 8 diff --git a/inferencex-e2e/infx/launch/drivers/srt/__init__.py b/inferencex-e2e/infx/launch/drivers/srt/__init__.py index 4a3aa0cd3d..0951721dc0 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/__init__.py +++ b/inferencex-e2e/infx/launch/drivers/srt/__init__.py @@ -138,6 +138,7 @@ def run_multinode(launch: Launch) -> int: checkout = prepare_checkout(run, checkout_dir(run, shared=shared), power=decision.dcgm) overrides = eval_overrides(checkout.root / "recipes", lane, request) + overrides += lanes.role_env_overrides(lane, config_file) system_python = ( "/usr/bin/python3" if shared and os.access("/usr/bin/python3", os.X_OK) else None ) diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index e8217eb607..cc5defad3a 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -2,6 +2,7 @@ from __future__ import annotations +import fnmatch from collections.abc import Mapping from dataclasses import dataclass, field from typing import TYPE_CHECKING @@ -25,6 +26,15 @@ class LaneMount: world_writable: bool = False +@dataclass(frozen=True) +class RoleEnv: + """Environment the lane sets on the worker roles of the recipes ``recipe`` globs.""" + + recipe: str + roles: tuple[str, ...] + env: Mapping[str, str] + + @dataclass(frozen=True) class SrtLane: """How one cluster's multi-node srt-slurm lane differs from the others.""" @@ -41,6 +51,7 @@ class SrtLane: time_limit: str | None = None long_time_limit: str | None = None long_time: Match | None = None + role_env: tuple[RoleEnv, ...] = () _DYNAMO = any_of("dynamo-sglang", "dynamo-trt", "dynamo-vllm") @@ -91,6 +102,17 @@ class SrtLane: long_time=Match( any_of("dsv4"), frameworks=any_of("dynamo-sglang", "dynamo-trt"), agentic=True ), + # The Mooncake store transfers from host buffers that only the RDMA transport + # registers on the fly. Without an explicit device list the client falls back to + # NVLink for same-domain peers, whose address lookup fails and aborts every worker + # during the store warmup put. These are this cluster's RDMA devices. + role_env=( + RoleEnv( + "recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml", + ("prefill", "decode"), + {"MOONCAKE_DEVICE": "mlx5_0,mlx5_1,mlx5_2,mlx5_3"}, + ), + ), ), ("h100-dgxc", LaunchPath.SRT_MULTI): SrtLane(frameworks=any_of("dynamo-sglang", "dynamo-trt")), ("h200-dgxc", LaunchPath.SRT_MULTI): SrtLane( @@ -147,6 +169,19 @@ def config_file(request: SrtRequest) -> str: return request.config_file +def role_env_overrides(lane: SrtLane, config_file: str) -> list[str]: + """``--set`` arguments for the lane's role environment on the recipe ``config_file`` names.""" + recipe = config_file.partition(":")[0] + overrides: list[str] = [] + for rule in lane.role_env: + if not fnmatch.fnmatchcase(recipe, rule.recipe): + continue + for role in rule.roles: + for name, value in rule.env.items(): + overrides += ["--set", f"roles.{role}.env.{name}={value}"] + return overrides + + def srt_time_limit( cluster_id: str, request: LaunchRequest, lane: SrtLane | None, srt: SrtSlurmSettings ) -> str: diff --git a/inferencex-e2e/infx/tests/launch/test_srt_policy.py b/inferencex-e2e/infx/tests/launch/test_srt_policy.py index ce4b3d00d5..9169668e4e 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_policy.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_policy.py @@ -7,7 +7,7 @@ from infx.launch import policy from infx.launch.context import LaunchError from infx.launch.drivers.srt import models -from infx.launch.drivers.srt.lanes import SrtLane, srt_lane, srt_time_limit +from infx.launch.drivers.srt.lanes import RoleEnv, SrtLane, role_env_overrides, srt_lane, srt_time_limit from infx.launch.drivers.srt.models import ( Override, checkpoint, @@ -183,3 +183,31 @@ def test_srt_time_limits(monkeypatch, profile, lane, env, limit): srt = SrtSlurmSettings.model_validate({"network-interface": "", **profile}) point = request(SALLOC_TIME_LIMIT="480", EVAL_ONLY="false", **env) assert srt_time_limit("c", point, lane, srt) == limit + + +DISAGG = "recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml" + + +@pytest.mark.parametrize(("config_file", "expected"), [ + (f"{DISAGG}:override_4p1d_c1920", [ + "--set", "roles.prefill.env.MOONCAKE_DEVICE=mlx5_0,mlx5_1,mlx5_2,mlx5_3", + "--set", "roles.decode.env.MOONCAKE_DEVICE=mlx5_0,mlx5_1,mlx5_2,mlx5_3", + ]), + (DISAGG, [ + "--set", "roles.prefill.env.MOONCAKE_DEVICE=mlx5_0,mlx5_1,mlx5_2,mlx5_3", + "--set", "roles.decode.env.MOONCAKE_DEVICE=mlx5_0,mlx5_1,mlx5_2,mlx5_3", + ]), + ("recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp8", []), +], ids=["disagg-variant", "disagg-file", "agg-recipe-untouched"]) # fmt: skip +def test_gb300_sets_the_mooncake_devices_only_on_the_disagg_recipe_roles(config_file, expected): + lane = srt_lane("gb300-nv", LaunchPath.SRT_MULTI) + assert role_env_overrides(lane, config_file) == expected + + +def test_role_env_applies_every_rule_whose_glob_matches(): + lane = SrtLane(role_env=( + RoleEnv("recipes/a/*.yaml", ("decode",), {"X": "1"}), + RoleEnv("recipes/b/*.yaml", ("decode",), {"Y": "2"}), + )) # fmt: skip + assert role_env_overrides(lane, "recipes/a/r.yaml:override_x") == ["--set", "roles.decode.env.X=1"] + assert role_env_overrides(SrtLane(), "recipes/a/r.yaml") == [] diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index f4d44204ea..5fc77a69ff 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9120,3 +9120,20 @@ - "Restore the DCP8 LMCache bands on the native srt-slurm recipe: concurrency 14 and 16 (DSpark 3, ReplaySSM) and 48, 56 and 72 (no draft) run ATOM's in-process lmcache_offload connector through roles.agg.args.extra-kv-connectors (srt-slurm patch 507), with 128 GB/rank up to 48 and 192 GB/rank at 56 and 72. Concurrency 1 and 4 stay GPU-resident." - "No change to the Inferact/Kimi-K3-DSpark draft's precision: online_quant_config still excludes every draft linear (layers.*, context_proj), so its weights and activations stay BF16, and it keeps the target's FP8 KV cache (kv_cache_dtype fp8). FlyDSL FP8 prefill attention applies only to the target, since the draft runs its block pass as decode attention." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3407 + +- config-keys: + - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Move the GB300 DeepSeek-V4-Pro AgentX disaggregated ladder from hierarchical cache to the Mooncake unified-cache external linker on the image's own SGLang, and bump the image to lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9." + - "Decode nodes lend host DRAM to the store through standalone mooncake_client daemons: the setup script starts four 180GB daemons per decode node, registered with the HTTP metadata server prefill writers resolve segments through, and fails the job unless every daemon is serving and its descriptor resolves. Prefill keeps a 140GB segment per rank." + - "Prefill runs 16k-token chunks per DP rank (chunked-prefill-size 131072, SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408) with mem-fraction-static 0.80, which the DSV4 indexer needs at that chunk size." + - "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent." + - "The gb300-nv srt lane sets this cluster's RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the recipe's prefill and decode roles: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." + - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,并将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9。" + - "decode 节点通过独立的 mooncake_client 守护进程向 store 借出主机 DRAM:setup 脚本在每个 decode 节点启动 4 个 180GB 守护进程,并注册到 prefill 写入方解析 segment 所用的 HTTP metadata server;任一守护进程未就绪或其描述符无法解析时作业直接失败。prefill 每个 rank 保留 140GB segment。" + - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" + - "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。" + - "gb300-nv 的 srt lane 为该配方的 prefill 与 decode 角色设置本集群的 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3):未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187 From 457e16eb835525dac82911180ef0c3dffe174d13 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Wed, 30 Sep 2026 10:04:41 +0800 Subject: [PATCH 02/12] chore(agentx): bump the Dynamo wheel and drop the get_model_config shim Dynamo 1.5.0.dev20260914 falls back when ServerArgs.get_model_config is missing (ai-dynamo/dynamo#14234), matching the B200 recipes on the same image, so the sitecustomize shim and its PYTHONPATH entry go away. --- .../sitecustomize.py | 26 ------------------- .../gb300-fp4/agentx/disagg-variants.yaml | 7 +---- inferencex-e2e/perf-changelog.yaml | 4 +-- 3 files changed, 3 insertions(+), 34 deletions(-) delete mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py deleted file mode 100644 index a9c2086b56..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py +++ /dev/null @@ -1,26 +0,0 @@ -"""TEMPORARY: restore ServerArgs.get_model_config for the pinned Dynamo wheel. - -The image's SGLang moved model-config resolution out of ServerArgs, but the -pinned Dynamo wheel still calls ``server_args.get_model_config()`` while -parsing worker arguments. Re-attach the accessor so the wheel keeps working. - -This file is picked up because its directory is on PYTHONPATH, so it is -imported by every interpreter in the job, including ones that never import -SGLang. Failing to import SGLang there is expected and must stay silent. - -Remove this directory and its PYTHONPATH entry once the pinned Dynamo wheel -stops calling the accessor. -""" - -try: - from sglang.srt.arg_groups.model_override_base import model_config_of - from sglang.srt.server_args import ServerArgs -except Exception: # noqa: BLE001 - non-SGLang interpreters legitimately fail here - pass -else: - if not hasattr(ServerArgs, "get_model_config"): - - def get_model_config(self): - return model_config_of(self) - - ServerArgs.get_model_config = get_model_config diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index 82dc2c6453..1119adf524 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -17,7 +17,7 @@ base: dynamo: install: true source: - wheel: "1.5.0.dev20260902" + wheel: "1.5.0.dev20260914" slurm: time_limit: "8:00:00" health_check: @@ -28,11 +28,6 @@ base: # DRAM to the store (see MC_SIDECAR_* on the decode role). setup_script: dsv4-gb300-mooncake-sidecar.sh - # TEMPORARY: the pinned Dynamo wheel still calls ServerArgs.get_model_config, - # which this image's SGLang no longer has. Remove once the wheel catches up. - environment: - PYTHONPATH: /configs/sglang-server-args-compat - # Tachometer's per-node exporters slow decode steps by a few percent and add # host memory on the head node; this ladder is a throughput measurement. observability: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 5fc77a69ff..318ca8269f 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9126,12 +9126,12 @@ scenario-type: - agentic-coding description: - - "Move the GB300 DeepSeek-V4-Pro AgentX disaggregated ladder from hierarchical cache to the Mooncake unified-cache external linker on the image's own SGLang, and bump the image to lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9." + - "Move the GB300 DeepSeek-V4-Pro AgentX disaggregated ladder from hierarchical cache to the Mooncake unified-cache external linker on the image's own SGLang, bump the image to lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9, and bump the Dynamo wheel to 1.5.0.dev20260914, whose SGLang worker no longer needs ServerArgs.get_model_config." - "Decode nodes lend host DRAM to the store through standalone mooncake_client daemons: the setup script starts four 180GB daemons per decode node, registered with the HTTP metadata server prefill writers resolve segments through, and fails the job unless every daemon is serving and its descriptor resolves. Prefill keeps a 140GB segment per rank." - "Prefill runs 16k-token chunks per DP rank (chunked-prefill-size 131072, SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408) with mem-fraction-static 0.80, which the DSV4 indexer needs at that chunk size." - "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent." - "The gb300-nv srt lane sets this cluster's RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the recipe's prefill and decode roles: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." - - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,并将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9。" + - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。" - "decode 节点通过独立的 mooncake_client 守护进程向 store 借出主机 DRAM:setup 脚本在每个 decode 节点启动 4 个 180GB 守护进程,并注册到 prefill 写入方解析 segment 所用的 HTTP metadata server;任一守护进程未就绪或其描述符无法解析时作业直接失败。prefill 每个 rank 保留 140GB segment。" - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" - "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。" From 226e686d059f666bf30fc455a37c48718f7c0a07 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Wed, 30 Sep 2026 10:13:51 +0800 Subject: [PATCH 03/12] chore(agentx): drop MC_STORE_CLIENT_METRIC, which only restated Mooncake's default --- .../dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml | 2 -- 1 file changed, 2 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index 1119adf524..40ebcfd380 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -103,7 +103,6 @@ base: MOONCAKE_PROTOCOL: rdma MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb MOONCAKE_STANDALONE_STORAGE: '0' - MC_STORE_CLIENT_METRIC: '1' MC_STORE_CLIENT_METRIC_INTERVAL: '5' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' WITH_NVIDIA_PEERMEM: '0' @@ -179,7 +178,6 @@ base: MOONCAKE_PROTOCOL: rdma MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb MOONCAKE_STANDALONE_STORAGE: '0' - MC_STORE_CLIENT_METRIC: '1' MC_STORE_CLIENT_METRIC_INTERVAL: '5' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' WITH_NVIDIA_PEERMEM: '0' From 3f36d39c38fe2d8b934e7d290fd3afbb5ddd8ee3 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Wed, 30 Sep 2026 10:14:52 +0800 Subject: [PATCH 04/12] chore(agentx): drop decode store-client env the sidecar setup no longer reads Decode ranks no longer mount a store segment, so MOONCAKE_GLOBAL_SEGMENT_SIZE and MOONCAKE_STANDALONE_STORAGE had no effect there; the daemons take their size from MC_SIDECAR_SEGMENT_SIZE. --- .../dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml | 2 -- 1 file changed, 2 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index 40ebcfd380..b679026ef1 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -176,8 +176,6 @@ base: SGLANG_LOG_MS: '1' SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' MOONCAKE_PROTOCOL: rdma - MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb - MOONCAKE_STANDALONE_STORAGE: '0' MC_STORE_CLIENT_METRIC_INTERVAL: '5' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' WITH_NVIDIA_PEERMEM: '0' From ac708632dd96d2f7b51f62410d3be68e0be21b5e Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Wed, 30 Sep 2026 10:16:15 +0800 Subject: [PATCH 05/12] refactor(agentx): name the Mooncake RDMA devices in the recipe, not the launcher Set MOONCAKE_DEVICE on the prefill and decode roles directly, as the GB300 vLLM Mooncake recipes already do, and drop the srt lane role_env hook. --- .../gb300-fp4/agentx/disagg-variants.yaml | 6 ++++ .../infx/launch/drivers/srt/__init__.py | 1 - .../infx/launch/drivers/srt/lanes.py | 35 ------------------- .../infx/tests/launch/test_srt_policy.py | 30 +--------------- inferencex-e2e/perf-changelog.yaml | 4 +-- 5 files changed, 9 insertions(+), 67 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index b679026ef1..d265746afa 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -101,6 +101,9 @@ base: SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' MOONCAKE_PROTOCOL: rdma + # The store only registers host buffers on RDMA; without an explicit list it + # falls back to NVLink for same-domain peers and aborts at the warmup put. + MOONCAKE_DEVICE: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb MOONCAKE_STANDALONE_STORAGE: '0' MC_STORE_CLIENT_METRIC_INTERVAL: '5' @@ -176,6 +179,9 @@ base: SGLANG_LOG_MS: '1' SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' MOONCAKE_PROTOCOL: rdma + # The store only registers host buffers on RDMA; without an explicit list it + # falls back to NVLink for same-domain peers and aborts at the warmup put. + MOONCAKE_DEVICE: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" MC_STORE_CLIENT_METRIC_INTERVAL: '5' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' WITH_NVIDIA_PEERMEM: '0' diff --git a/inferencex-e2e/infx/launch/drivers/srt/__init__.py b/inferencex-e2e/infx/launch/drivers/srt/__init__.py index 0951721dc0..4a3aa0cd3d 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/__init__.py +++ b/inferencex-e2e/infx/launch/drivers/srt/__init__.py @@ -138,7 +138,6 @@ def run_multinode(launch: Launch) -> int: checkout = prepare_checkout(run, checkout_dir(run, shared=shared), power=decision.dcgm) overrides = eval_overrides(checkout.root / "recipes", lane, request) - overrides += lanes.role_env_overrides(lane, config_file) system_python = ( "/usr/bin/python3" if shared and os.access("/usr/bin/python3", os.X_OK) else None ) diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index cc5defad3a..e8217eb607 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -2,7 +2,6 @@ from __future__ import annotations -import fnmatch from collections.abc import Mapping from dataclasses import dataclass, field from typing import TYPE_CHECKING @@ -26,15 +25,6 @@ class LaneMount: world_writable: bool = False -@dataclass(frozen=True) -class RoleEnv: - """Environment the lane sets on the worker roles of the recipes ``recipe`` globs.""" - - recipe: str - roles: tuple[str, ...] - env: Mapping[str, str] - - @dataclass(frozen=True) class SrtLane: """How one cluster's multi-node srt-slurm lane differs from the others.""" @@ -51,7 +41,6 @@ class SrtLane: time_limit: str | None = None long_time_limit: str | None = None long_time: Match | None = None - role_env: tuple[RoleEnv, ...] = () _DYNAMO = any_of("dynamo-sglang", "dynamo-trt", "dynamo-vllm") @@ -102,17 +91,6 @@ class SrtLane: long_time=Match( any_of("dsv4"), frameworks=any_of("dynamo-sglang", "dynamo-trt"), agentic=True ), - # The Mooncake store transfers from host buffers that only the RDMA transport - # registers on the fly. Without an explicit device list the client falls back to - # NVLink for same-domain peers, whose address lookup fails and aborts every worker - # during the store warmup put. These are this cluster's RDMA devices. - role_env=( - RoleEnv( - "recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml", - ("prefill", "decode"), - {"MOONCAKE_DEVICE": "mlx5_0,mlx5_1,mlx5_2,mlx5_3"}, - ), - ), ), ("h100-dgxc", LaunchPath.SRT_MULTI): SrtLane(frameworks=any_of("dynamo-sglang", "dynamo-trt")), ("h200-dgxc", LaunchPath.SRT_MULTI): SrtLane( @@ -169,19 +147,6 @@ def config_file(request: SrtRequest) -> str: return request.config_file -def role_env_overrides(lane: SrtLane, config_file: str) -> list[str]: - """``--set`` arguments for the lane's role environment on the recipe ``config_file`` names.""" - recipe = config_file.partition(":")[0] - overrides: list[str] = [] - for rule in lane.role_env: - if not fnmatch.fnmatchcase(recipe, rule.recipe): - continue - for role in rule.roles: - for name, value in rule.env.items(): - overrides += ["--set", f"roles.{role}.env.{name}={value}"] - return overrides - - def srt_time_limit( cluster_id: str, request: LaunchRequest, lane: SrtLane | None, srt: SrtSlurmSettings ) -> str: diff --git a/inferencex-e2e/infx/tests/launch/test_srt_policy.py b/inferencex-e2e/infx/tests/launch/test_srt_policy.py index 9169668e4e..ce4b3d00d5 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_policy.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_policy.py @@ -7,7 +7,7 @@ from infx.launch import policy from infx.launch.context import LaunchError from infx.launch.drivers.srt import models -from infx.launch.drivers.srt.lanes import RoleEnv, SrtLane, role_env_overrides, srt_lane, srt_time_limit +from infx.launch.drivers.srt.lanes import SrtLane, srt_lane, srt_time_limit from infx.launch.drivers.srt.models import ( Override, checkpoint, @@ -183,31 +183,3 @@ def test_srt_time_limits(monkeypatch, profile, lane, env, limit): srt = SrtSlurmSettings.model_validate({"network-interface": "", **profile}) point = request(SALLOC_TIME_LIMIT="480", EVAL_ONLY="false", **env) assert srt_time_limit("c", point, lane, srt) == limit - - -DISAGG = "recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml" - - -@pytest.mark.parametrize(("config_file", "expected"), [ - (f"{DISAGG}:override_4p1d_c1920", [ - "--set", "roles.prefill.env.MOONCAKE_DEVICE=mlx5_0,mlx5_1,mlx5_2,mlx5_3", - "--set", "roles.decode.env.MOONCAKE_DEVICE=mlx5_0,mlx5_1,mlx5_2,mlx5_3", - ]), - (DISAGG, [ - "--set", "roles.prefill.env.MOONCAKE_DEVICE=mlx5_0,mlx5_1,mlx5_2,mlx5_3", - "--set", "roles.decode.env.MOONCAKE_DEVICE=mlx5_0,mlx5_1,mlx5_2,mlx5_3", - ]), - ("recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp8", []), -], ids=["disagg-variant", "disagg-file", "agg-recipe-untouched"]) # fmt: skip -def test_gb300_sets_the_mooncake_devices_only_on_the_disagg_recipe_roles(config_file, expected): - lane = srt_lane("gb300-nv", LaunchPath.SRT_MULTI) - assert role_env_overrides(lane, config_file) == expected - - -def test_role_env_applies_every_rule_whose_glob_matches(): - lane = SrtLane(role_env=( - RoleEnv("recipes/a/*.yaml", ("decode",), {"X": "1"}), - RoleEnv("recipes/b/*.yaml", ("decode",), {"Y": "2"}), - )) # fmt: skip - assert role_env_overrides(lane, "recipes/a/r.yaml:override_x") == ["--set", "roles.decode.env.X=1"] - assert role_env_overrides(SrtLane(), "recipes/a/r.yaml") == [] diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 318ca8269f..b6a2f4e44c 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9130,10 +9130,10 @@ - "Decode nodes lend host DRAM to the store through standalone mooncake_client daemons: the setup script starts four 180GB daemons per decode node, registered with the HTTP metadata server prefill writers resolve segments through, and fails the job unless every daemon is serving and its descriptor resolves. Prefill keeps a 140GB segment per rank." - "Prefill runs 16k-token chunks per DP rank (chunked-prefill-size 131072, SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408) with mem-fraction-static 0.80, which the DSV4 indexer needs at that chunk size." - "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent." - - "The gb300-nv srt lane sets this cluster's RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the recipe's prefill and decode roles: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." + - "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on its prefill and decode roles, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。" - "decode 节点通过独立的 mooncake_client 守护进程向 store 借出主机 DRAM:setup 脚本在每个 decode 节点启动 4 个 180GB 守护进程,并注册到 prefill 写入方解析 segment 所用的 HTTP metadata server;任一守护进程未就绪或其描述符无法解析时作业直接失败。prefill 每个 rank 保留 140GB segment。" - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" - "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。" - - "gb300-nv 的 srt lane 为该配方的 prefill 与 decode 角色设置本集群的 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3):未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" + - "配方在 prefill 与 decode 角色上显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187 From 8933a63ce616d36e6a78a8ab692ae3b19620f437 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Wed, 30 Sep 2026 10:19:54 +0800 Subject: [PATCH 06/12] chore(agentx): record the Mooncake version the GB300 DSV4 disagg ladder runs The image ships mooncake-transfer-engine-cuda13 0.3.13; CONFIGS.md asks independently versioned offload backends to state their version. --- inferencex-e2e/configs/nvidia-master.yaml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 466776f0b3..fa38449d3c 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -6690,7 +6690,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [480] kv-offloading: dram - kv-offload-backend: { name: mooncake } + kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: num-worker: 1 tp: 8 @@ -6706,7 +6706,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [960] kv-offloading: dram - kv-offload-backend: { name: mooncake } + kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: num-worker: 2 tp: 8 @@ -6722,7 +6722,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [1440] kv-offloading: dram - kv-offload-backend: { name: mooncake } + kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: num-worker: 3 tp: 8 @@ -6738,7 +6738,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [1920] kv-offloading: dram - kv-offload-backend: { name: mooncake } + kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: num-worker: 4 tp: 8 From a115decc375ed02c9c5421a0144b9cf64324d231 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Wed, 30 Sep 2026 11:09:17 +0800 Subject: [PATCH 07/12] refactor(agentx): run the decode Mooncake stores as srt-slurm services Replace the recipe setup script with four mooncake-store services on the decode nodes (180gb each, ports 8800-8803). srt-slurm injects the master and HTTP metadata server, starts them before the workers, and gates on their readiness ports, so the custom daemon script goes away. --- .../configs/dsv4-gb300-mooncake-sidecar.sh | 77 ------------------- .../gb300-fp4/agentx/disagg-variants.yaml | 43 +++++++++-- inferencex-e2e/perf-changelog.yaml | 4 +- 3 files changed, 37 insertions(+), 87 deletions(-) delete mode 100755 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-mooncake-sidecar.sh diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-mooncake-sidecar.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-mooncake-sidecar.sh deleted file mode 100755 index a4a8076204..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-mooncake-sidecar.sh +++ /dev/null @@ -1,77 +0,0 @@ -#!/usr/bin/env bash -# On decode nodes, start standalone Mooncake store daemons that lend host DRAM -# to the store. No-op on roles that do not set MC_SIDECAR_SEGMENT_SIZE. -set -euo pipefail - -# Decode ranks use a chunk cache, so their host memory never joins the store on -# its own. Standalone daemons on the decode nodes mount it instead; SGLang never -# talks to them, only the master sees their segments. Only roles that set -# MC_SIDECAR_SEGMENT_SIZE start them. -if [ -z "${MC_SIDECAR_SEGMENT_SIZE:-}" ]; then - exit 0 -fi -for var in MOONCAKE_MASTER MOONCAKE_TE_META_DATA_SERVER; do - if [ -z "${!var:-}" ]; then - echo "ERROR: ${var} unset; cannot start Mooncake store daemons" >&2 - exit 1 - fi -done - -MY_IP=$(hostname -i | awk '{print $1}') -HOST=$(hostname) -COUNT="${MC_SIDECAR_COUNT:-1}" -echo "[mc-sidecar] host=${MY_IP} master=${MOONCAKE_MASTER} count=${COUNT} size=${MC_SIDECAR_SEGMENT_SIZE}" - -# One daemon per rank-equivalent keeps segment sizes uniform across the pool; a -# single oversized segment would concentrate every put on this node's NICs. -PIDS=() -for i in $(seq 0 $((COUNT - 1))); do - PORT=$((50052 + i)) - # Writers resolve a segment's transfer-engine descriptor through the HTTP - # metadata server. A daemon on P2PHANDSHAKE mounts fine, but every remote - # open of its segment 404s and the puts placed there are revoked. - nohup mooncake_client \ - --host="${MY_IP}" \ - --port="${PORT}" \ - --master_server_address="${MOONCAKE_MASTER}" \ - --metadata_server="${MOONCAKE_TE_META_DATA_SERVER}" \ - --protocol="${MOONCAKE_PROTOCOL:-rdma}" \ - --device_names="${MOONCAKE_DEVICE:-}" \ - --global_segment_size="${MC_SIDECAR_SEGMENT_SIZE}" \ - --local_buffer_size=1GB \ - > "/logs/mc_sidecar_${HOST}_${PORT}.out" 2>&1 & - PIDS+=($!) -done - -mounted=0 -for _ in $(seq 1 180); do - mounted=$( (grep -l "Starting real client service" /logs/mc_sidecar_"${HOST}"_*.out 2>/dev/null || true) | wc -l) - [ "${mounted}" -ge "${COUNT}" ] && break - sleep 1 -done -if [ "${mounted}" -lt "${COUNT}" ]; then - echo "ERROR: only ${mounted}/${COUNT} Mooncake store daemons serving within 180s" >&2 - exit 1 -fi - -# A daemon has exited shortly after startup before; require all to survive. -sleep 20 -for pid in "${PIDS[@]}"; do - if ! kill -0 "${pid}" 2>/dev/null; then - echo "ERROR: Mooncake store daemon pid ${pid} exited after startup" >&2 - exit 1 - fi -done - -# The descriptor must resolve exactly the way remote writers look it up. The -# segment name is the transfer-engine host:port, not the RPC listen port. -for f in /logs/mc_sidecar_"${HOST}"_*.out; do - ep=$(grep -aoE "parseHostNameWithPort\. server_name: [0-9.]+ port: [0-9]+" "${f}" | head -1 | awk '{print $3 ":" $5}' || true) - key="mooncake%2Fram%2F${ep%:*}%3A${ep#*:}" - code=$(curl -s -o /dev/null -w '%{http_code}' "${MOONCAKE_TE_META_DATA_SERVER}?key=${key}" || true) - if [ -z "${ep}" ] || [ "${code}" != "200" ]; then - echo "ERROR: descriptor for '${ep}' not on metadata server (http=${code})" >&2 - exit 1 - fi -done -echo "[mc-sidecar] ${COUNT} daemons serving and resolvable" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index d265746afa..cae0b4dda8 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -24,10 +24,6 @@ base: max_attempts: 1440 interval_seconds: 10 - # Starts the standalone Mooncake store daemons that lend decode-node host - # DRAM to the store (see MC_SIDECAR_* on the decode role). - setup_script: dsv4-gb300-mooncake-sidecar.sh - # Tachometer's per-node exporters slow decode steps by a few percent and add # host memory on the head node; this ladder is a throughput measurement. observability: @@ -51,6 +47,41 @@ base: type: mooncake-master args: - --eviction_high_watermark_ratio=0.90 + # Decode ranks use a chunk cache, so their host DRAM only joins the store + # through standalone stores: four per decode node, so segments stay uniform + # with prefill's per-rank ones instead of one oversized segment per node. + - &decode-store + name: store-decode-0 + type: mooncake-store + placement: + node: decode + args: ["--port", "8800"] + env: + MOONCAKE_PROTOCOL: rdma + MOONCAKE_DEVICE: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" + MOONCAKE_GLOBAL_SEGMENT_SIZE: 180gb + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + preamble: | + ulimit -n 1048576 + ulimit -l unlimited + readiness: + port: 8800 + - <<: *decode-store + name: store-decode-1 + args: ["--port", "8801"] + readiness: + port: 8801 + - <<: *decode-store + name: store-decode-2 + args: ["--port", "8802"] + readiness: + port: 8802 + - <<: *decode-store + name: store-decode-3 + args: ["--port", "8803"] + readiness: + port: 8803 frontend: type: dynamo nginx_session_affinity: true @@ -185,10 +216,6 @@ base: MC_STORE_CLIENT_METRIC_INTERVAL: '5' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' WITH_NVIDIA_PEERMEM: '0' - # Consumed by the setup script: each decode node starts this many - # standalone Mooncake store daemons of this size to lend its host DRAM. - MC_SIDECAR_SEGMENT_SIZE: 180GB - MC_SIDECAR_COUNT: '4' args: host: 0.0.0.0 served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index b6a2f4e44c..5e7e400db2 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9127,12 +9127,12 @@ - agentic-coding description: - "Move the GB300 DeepSeek-V4-Pro AgentX disaggregated ladder from hierarchical cache to the Mooncake unified-cache external linker on the image's own SGLang, bump the image to lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9, and bump the Dynamo wheel to 1.5.0.dev20260914, whose SGLang worker no longer needs ServerArgs.get_model_config." - - "Decode nodes lend host DRAM to the store through standalone mooncake_client daemons: the setup script starts four 180GB daemons per decode node, registered with the HTTP metadata server prefill writers resolve segments through, and fails the job unless every daemon is serving and its descriptor resolves. Prefill keeps a 140GB segment per rank." + - "Decode nodes lend host DRAM to the store through srt-slurm mooncake-store services: four 180GB standalone stores per decode node, started before the workers against the managed master and its HTTP metadata server and gated on their readiness ports. Prefill keeps a 140GB segment per rank." - "Prefill runs 16k-token chunks per DP rank (chunked-prefill-size 131072, SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408) with mem-fraction-static 0.80, which the DSV4 indexer needs at that chunk size." - "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent." - "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on its prefill and decode roles, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。" - - "decode 节点通过独立的 mooncake_client 守护进程向 store 借出主机 DRAM:setup 脚本在每个 decode 节点启动 4 个 180GB 守护进程,并注册到 prefill 写入方解析 segment 所用的 HTTP metadata server;任一守护进程未就绪或其描述符无法解析时作业直接失败。prefill 每个 rank 保留 140GB segment。" + - "decode 节点通过 srt-slurm 的 mooncake-store 服务向 store 借出主机 DRAM:每个 decode 节点 4 个 180GB 的独立 store,在 worker 之前启动,连接托管的 master 及其 HTTP metadata server,并以就绪端口作为启动检查。prefill 每个 rank 保留 140GB segment。" - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" - "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。" - "配方在 prefill 与 decode 角色上显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" From 0d9e9f11cbe70128b7c981b5b58bfa92c2cfb2d4 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Wed, 30 Sep 2026 12:06:48 +0800 Subject: [PATCH 08/12] chore(agentx): drop store-client env from the decode role Decode workers no longer act as Mooncake store clients; the decode stores carry their own MOONCAKE_DEVICE/PROTOCOL, and P/D KV transfer does not read them. --- .../dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml | 5 ----- inferencex-e2e/perf-changelog.yaml | 4 ++-- 2 files changed, 2 insertions(+), 7 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index cae0b4dda8..f9dff815a6 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -209,11 +209,6 @@ base: SGLANG_LOG_FORWARD_ITERS: '1' SGLANG_LOG_MS: '1' SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - MOONCAKE_PROTOCOL: rdma - # The store only registers host buffers on RDMA; without an explicit list it - # falls back to NVLink for same-domain peers and aborts at the warmup put. - MOONCAKE_DEVICE: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" - MC_STORE_CLIENT_METRIC_INTERVAL: '5' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' WITH_NVIDIA_PEERMEM: '0' args: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 5e7e400db2..2eaebe5611 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9130,10 +9130,10 @@ - "Decode nodes lend host DRAM to the store through srt-slurm mooncake-store services: four 180GB standalone stores per decode node, started before the workers against the managed master and its HTTP metadata server and gated on their readiness ports. Prefill keeps a 140GB segment per rank." - "Prefill runs 16k-token chunks per DP rank (chunked-prefill-size 131072, SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408) with mem-fraction-static 0.80, which the DSV4 indexer needs at that chunk size." - "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent." - - "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on its prefill and decode roles, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." + - "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the prefill role and the decode stores, the processes that act as store clients, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。" - "decode 节点通过 srt-slurm 的 mooncake-store 服务向 store 借出主机 DRAM:每个 decode 节点 4 个 180GB 的独立 store,在 worker 之前启动,连接托管的 master 及其 HTTP metadata server,并以就绪端口作为启动检查。prefill 每个 rank 保留 140GB segment。" - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" - "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。" - - "配方在 prefill 与 decode 角色上显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" + - "配方在 prefill 角色与 decode 节点的 store 上(即作为 store 客户端的进程)显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187 From 1122e1036c3649c34f0bbde74e1f5954ee22cb22 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Thu, 1 Oct 2026 10:24:13 +0800 Subject: [PATCH 09/12] feat(agentx): move the GB300 DSV4 4P1D point to c2400, add a DEP32 c480 point, and give decode stores 600 s to start --- .../gb300-fp4/agentx/disagg-variants.yaml | 34 +++++++++++++++++-- inferencex-e2e/configs/nvidia-master.yaml | 20 +++++++++-- inferencex-e2e/perf-changelog.yaml | 6 ++-- 3 files changed, 53 insertions(+), 7 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index f9dff815a6..1378fe1692 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -65,23 +65,29 @@ base: preamble: | ulimit -n 1048576 ulimit -l unlimited + # Cold container starts on a decode node have exceeded srtctl's 120 s default; + # the <<: merge is shallow, so every store repeats the timeout. readiness: port: 8800 + timeout_seconds: 600 - <<: *decode-store name: store-decode-1 args: ["--port", "8801"] readiness: port: 8801 + timeout_seconds: 600 - <<: *decode-store name: store-decode-2 args: ["--port", "8802"] readiness: port: 8802 + timeout_seconds: 600 - <<: *decode-store name: store-decode-3 args: ["--port", "8803"] readiness: port: 8803 + timeout_seconds: 600 frontend: type: dynamo nginx_session_affinity: true @@ -290,6 +296,28 @@ override_1p1d_c480: args: cuda-graph-max-bs-decode: 256 +# Low-latency point: 1P x DEP8 prefill / 1D x DEP32 decode (8 nodes; 384 experts / 32 = 12 per +# rank) at concurrency 480. Wider expert parallelism shortens the decode step, filling the +# interactivity range between the DEP16 c480 point and the aggregate recipes. +override_1p1d_dep8_dep32_c480: + name: "disagg-gb300-2p8d-dep8-dep32-c480-mtp-kvoffload" + roles: + prefill: + nodes: 2 + workers: 1 + gpus: 8 + args: + max-running-requests: 256 + cuda-graph-max-bs-decode: 256 + decode: + nodes: 8 + gpus: 32 + args: + tp-size: 32 + dp-size: 32 + ep-size: 32 + cuda-graph-max-bs-decode: 256 + # Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300 # (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960. # @@ -338,14 +366,14 @@ override_3p1d_c1440: cuda-graph-max-bs-decode: 512 # Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 -# (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920. +# (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 2400. # # Uses the flat single-variant srtctl schema the agentic CI flow expects; # resources + backend (prefill/decode env + sglang_config) are normalized # from the Pareto run. # Concurrency is exported into srt_agentic.sh from the master-config conc-list. -override_4p1d_c1920: - name: "disagg-gb300-8p4d-dep8-dep16-c1920-mtp-kvoffload" +override_4p1d_c2400: + name: "disagg-gb300-8p4d-dep8-dep16-c2400-mtp-kvoffload" frontend: nginx_keepalive_timeout: "900s" env: diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index fa38449d3c..c7b4afdcca 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -6703,6 +6703,22 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: tp: 16 ep: 16 dp-attn: true + - spec-decoding: draft_model + conc-list: [480] + kv-offloading: dram + kv-offload-backend: { name: mooncake, version: "0.3.13" } + prefill: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_dep8_dep32_c480" + decode: + num-worker: 1 + tp: 32 + ep: 32 + dp-attn: true - spec-decoding: draft_model conc-list: [960] kv-offloading: dram @@ -6736,7 +6752,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 16 dp-attn: true - spec-decoding: draft_model - conc-list: [1920] + conc-list: [2400] kv-offloading: dram kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: @@ -6745,7 +6761,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 8 dp-attn: true additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_4p1d_c1920" + - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_4p1d_c2400" decode: num-worker: 1 tp: 16 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 2eaebe5611..f752d93afc 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9127,13 +9127,15 @@ - agentic-coding description: - "Move the GB300 DeepSeek-V4-Pro AgentX disaggregated ladder from hierarchical cache to the Mooncake unified-cache external linker on the image's own SGLang, bump the image to lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9, and bump the Dynamo wheel to 1.5.0.dev20260914, whose SGLang worker no longer needs ServerArgs.get_model_config." - - "Decode nodes lend host DRAM to the store through srt-slurm mooncake-store services: four 180GB standalone stores per decode node, started before the workers against the managed master and its HTTP metadata server and gated on their readiness ports. Prefill keeps a 140GB segment per rank." + - "Decode nodes lend host DRAM to the store through srt-slurm mooncake-store services: four 180GB standalone stores per decode node, started before the workers against the managed master and its HTTP metadata server and gated on their readiness ports with a 600-second timeout, because cold container starts on decode nodes exceeded srtctl's 120-second default. Prefill keeps a 140GB segment per rank." - "Prefill runs 16k-token chunks per DP rank (chunked-prefill-size 131072, SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408) with mem-fraction-static 0.80, which the DSV4 indexer needs at that chunk size." - "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent." - "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the prefill role and the decode stores, the processes that act as store clients, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." + - "The ladder's widest point moves from concurrency 1920 to 2400 on the same 4P1D topology, where prefill still had headroom, and a 1P x DEP8 / 1D x DEP32 point at concurrency 480 is added at the low-latency end; the wider expert parallelism shortens the decode step." - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。" - - "decode 节点通过 srt-slurm 的 mooncake-store 服务向 store 借出主机 DRAM:每个 decode 节点 4 个 180GB 的独立 store,在 worker 之前启动,连接托管的 master 及其 HTTP metadata server,并以就绪端口作为启动检查。prefill 每个 rank 保留 140GB segment。" + - "decode 节点通过 srt-slurm 的 mooncake-store 服务向 store 借出主机 DRAM:每个 decode 节点 4 个 180GB 的独立 store,在 worker 之前启动,连接托管的 master 及其 HTTP metadata server,并以就绪端口作为启动检查,超时设为 600 秒,因为 decode 节点上冷启动容器曾超过 srtctl 默认的 120 秒。prefill 每个 rank 保留 140GB segment。" - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" - "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。" - "配方在 prefill 角色与 decode 节点的 store 上(即作为 store 客户端的进程)显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" + - "阶梯中吞吐最高的点在相同的 4P1D 拓扑下由并发 1920 调整为 2400(该并发下 prefill 仍有余量),并在低延迟端新增 1P x DEP8 / 1D x DEP32、并发 480 的点;更宽的专家并行缩短了 decode 单步时间。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187 From db9f30e6354c56d739cc23b5e1279e421b763425 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Thu, 1 Oct 2026 11:13:18 +0800 Subject: [PATCH 10/12] feat(agentx): replace the GB300 DSV4 agg TP4 c8 point with a 1P1D c120 disagg point and drop the DEP32 point --- .../sglang/gb300-fp4/agentx/agg-variants.yaml | 16 ---------- .../gb300-fp4/agentx/disagg-variants.yaml | 22 -------------- inferencex-e2e/configs/nvidia-master.yaml | 29 +------------------ inferencex-e2e/perf-changelog.yaml | 5 ++-- 4 files changed, 4 insertions(+), 68 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml index 2ef179373b..37f67b0db9 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -113,22 +113,6 @@ base: AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" HF_HUB_CACHE: "/hf_hub_cache" -# Low-latency AgentX aggregate topology: one TP4 worker occupies one -# four-GPU GB300 node and serves both prefill and decode with DSpark K=6. -override_tp4: - name: "agg-gb300-tp4-mtp-lowlatency" - roles: - agg: - nodes: 1 - gpus: 4 - args: - max-running-requests: 32 - cuda-graph-max-bs-decode: 32 - tp-size: 4 - benchmark: - env: - TP: "4" - # Low-latency AgentX aggregate topology: one TP8 worker spans two # four-GPU GB300 nodes and serves both prefill and decode with DSpark K=6. override_tp8: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index 1378fe1692..4b58b69561 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -296,28 +296,6 @@ override_1p1d_c480: args: cuda-graph-max-bs-decode: 256 -# Low-latency point: 1P x DEP8 prefill / 1D x DEP32 decode (8 nodes; 384 experts / 32 = 12 per -# rank) at concurrency 480. Wider expert parallelism shortens the decode step, filling the -# interactivity range between the DEP16 c480 point and the aggregate recipes. -override_1p1d_dep8_dep32_c480: - name: "disagg-gb300-2p8d-dep8-dep32-c480-mtp-kvoffload" - roles: - prefill: - nodes: 2 - workers: 1 - gpus: 8 - args: - max-running-requests: 256 - cuda-graph-max-bs-decode: 256 - decode: - nodes: 8 - gpus: 32 - args: - tp-size: 32 - dp-size: 32 - ep-size: 32 - cuda-graph-max-bs-decode: 256 - # Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300 # (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960. # diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index c7b4afdcca..87cc9ca9b2 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -6661,17 +6661,6 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg: dp-attn: false additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp8" - - search-space: - - spec-decoding: draft_model - conc-list: [8] - num-nodes: 1 - worker: - num-worker: 1 - tp: 4 - ep: 1 - dp-attn: false - additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4" dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9 model: deepseek-ai/DeepSeek-V4-Pro-0813 @@ -6688,7 +6677,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - dram-utilization: 0.80 search-space: - spec-decoding: draft_model - conc-list: [480] + conc-list: [120, 480] kv-offloading: dram kv-offload-backend: { name: mooncake, version: "0.3.13" } prefill: @@ -6703,22 +6692,6 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: tp: 16 ep: 16 dp-attn: true - - spec-decoding: draft_model - conc-list: [480] - kv-offloading: dram - kv-offload-backend: { name: mooncake, version: "0.3.13" } - prefill: - num-worker: 1 - tp: 8 - ep: 8 - dp-attn: true - additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_dep8_dep32_c480" - decode: - num-worker: 1 - tp: 32 - ep: 32 - dp-attn: true - spec-decoding: draft_model conc-list: [960] kv-offloading: dram diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index f752d93afc..f5c0ec4b2e 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9123,6 +9123,7 @@ - config-keys: - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg + - dsv4-fp4-gb300-dynamo-sglang-agentic-agg scenario-type: - agentic-coding description: @@ -9131,11 +9132,11 @@ - "Prefill runs 16k-token chunks per DP rank (chunked-prefill-size 131072, SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408) with mem-fraction-static 0.80, which the DSV4 indexer needs at that chunk size." - "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent." - "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the prefill role and the decode stores, the processes that act as store clients, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." - - "The ladder's widest point moves from concurrency 1920 to 2400 on the same 4P1D topology, where prefill still had headroom, and a 1P x DEP8 / 1D x DEP32 point at concurrency 480 is added at the low-latency end; the wider expert parallelism shortens the decode step." + - "The ladder's widest point moves from concurrency 1920 to 2400 on the same 4P1D topology, where prefill still had headroom, and the 1P1D point also runs at concurrency 120. At equal P90 interactivity it delivers about twice the throughput per GPU of the aggregate TP4 concurrency-8 point, which is removed from dsv4-fp4-gb300-dynamo-sglang-agentic-agg." - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。" - "decode 节点通过 srt-slurm 的 mooncake-store 服务向 store 借出主机 DRAM:每个 decode 节点 4 个 180GB 的独立 store,在 worker 之前启动,连接托管的 master 及其 HTTP metadata server,并以就绪端口作为启动检查,超时设为 600 秒,因为 decode 节点上冷启动容器曾超过 srtctl 默认的 120 秒。prefill 每个 rank 保留 140GB segment。" - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" - "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。" - "配方在 prefill 角色与 decode 节点的 store 上(即作为 store 客户端的进程)显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" - - "阶梯中吞吐最高的点在相同的 4P1D 拓扑下由并发 1920 调整为 2400(该并发下 prefill 仍有余量),并在低延迟端新增 1P x DEP8 / 1D x DEP32、并发 480 的点;更宽的专家并行缩短了 decode 单步时间。" + - "阶梯中吞吐最高的点在相同的 4P1D 拓扑下由并发 1920 调整为 2400(该并发下 prefill 仍有余量),1P1D 点同时增加并发 120。在相同的 P90 交互性下,其单卡吞吐约为聚合 TP4 并发 8 点的两倍,因此从 dsv4-fp4-gb300-dynamo-sglang-agentic-agg 中移除该聚合点。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187 From fefe3fb01069004b3ad1700e6d74dd063a78a26d Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Thu, 1 Oct 2026 17:08:05 +0800 Subject: [PATCH 11/12] fix(agentx): give the GB300 DSV4 agg recipe PP_SIZE and PCP_SIZE for the AgentX power topology check --- .../dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml | 3 +++ inferencex-e2e/perf-changelog.yaml | 2 ++ 2 files changed, 5 insertions(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml index 37f67b0db9..78f8c99a4e 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -107,6 +107,9 @@ base: RESULT_DIR: "/logs/agentic" PORT: "8000" IS_MULTINODE: "false" + # The AgentX power path checks the GPU topology (TP from each variant) before replay. + PP_SIZE: "1" + PCP_SIZE: "1" AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 5e50bd2c53..bc4c189d43 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9181,10 +9181,12 @@ - "The benchmark client runs on the dedicated etcd/nats node instead of the head prefill node, where its ~300GB of host memory OOM-killed prefill_0, and Tachometer's per-node exporters are off for this throughput ladder because they slowed decode steps by a few percent." - "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the prefill role and the decode stores, the processes that act as store clients, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." - "The ladder's widest point moves from concurrency 1920 to 2400 on the same 4P1D topology, where prefill still had headroom, and the 1P1D point also runs at concurrency 120. At equal P90 interactivity it delivers about twice the throughput per GPU of the aggregate TP4 concurrency-8 point, which is removed from dsv4-fp4-gb300-dynamo-sglang-agentic-agg." + - "The aggregate recipe sets PP_SIZE and PCP_SIZE to 1 in its benchmark environment: the AgentX power path now checks TP, PP_SIZE and PCP_SIZE before replay, and multi-node aggregate jobs do not receive them from the workflow." - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。" - "decode 节点通过 srt-slurm 的 mooncake-store 服务向 store 借出主机 DRAM:每个 decode 节点 4 个 180GB 的独立 store,在 worker 之前启动,连接托管的 master 及其 HTTP metadata server,并以就绪端口作为启动检查,超时设为 600 秒,因为 decode 节点上冷启动容器曾超过 srtctl 默认的 120 秒。prefill 每个 rank 保留 140GB segment。" - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" - "压测客户端改为运行在独立的 etcd/nats 节点上,而不是 head prefill 节点(其约 300GB 主机内存曾导致 prefill_0 被 OOM kill);该吞吐阶梯关闭 Tachometer 的逐节点 exporter,因为它们会使 decode 单步变慢数个百分点。" - "配方在 prefill 角色与 decode 节点的 store 上(即作为 store 客户端的进程)显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" - "阶梯中吞吐最高的点在相同的 4P1D 拓扑下由并发 1920 调整为 2400(该并发下 prefill 仍有余量),1P1D 点同时增加并发 120。在相同的 P90 交互性下,其单卡吞吐约为聚合 TP4 并发 8 点的两倍,因此从 dsv4-fp4-gb300-dynamo-sglang-agentic-agg 中移除该聚合点。" + - "聚合式配方在 benchmark 环境中设置 PP_SIZE 与 PCP_SIZE 为 1:AgentX 功耗路径现在会在请求回放前检查 TP、PP_SIZE 与 PCP_SIZE,而多节点聚合任务不会从工作流获得这两个变量。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187 From 07543073bb615836dfa4bc70ec3477adc9a3fd65 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Fri, 2 Oct 2026 15:03:55 +0800 Subject: [PATCH 12/12] fix(agentx): advertise the GB300 DSV4 Mooncake store on the node address, not a BMC Redfish interface --- .../dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml | 8 ++++++++ inferencex-e2e/perf-changelog.yaml | 2 ++ 2 files changed, 10 insertions(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index 4b58b69561..4ab2c1734d 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -62,6 +62,10 @@ base: MOONCAKE_GLOBAL_SEGMENT_SIZE: 180gb MC_STORE_CLIENT_METRIC_INTERVAL: '5' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + # Advertise the store transfer engine on the node address instead of the first active + # interface: some GB300 nodes expose a BMC Redfish interface (bmc_redfish0, the same + # 10.0.1.2 on every node) ahead of the fabric interface. + MC_TCP_BIND_ADDRESS: "{node_ip}" preamble: | ulimit -n 1048576 ulimit -l unlimited @@ -145,6 +149,10 @@ base: MOONCAKE_STANDALONE_STORAGE: '0' MC_STORE_CLIENT_METRIC_INTERVAL: '5' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + # Advertise the store transfer engine on the node address instead of the first active + # interface: some GB300 nodes expose a BMC Redfish interface (bmc_redfish0, the same + # 10.0.1.2 on every node) ahead of the fabric interface. + MC_TCP_BIND_ADDRESS: "{node}" WITH_NVIDIA_PEERMEM: '0' args: host: 0.0.0.0 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index d1361f1365..8488a9a234 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9225,6 +9225,7 @@ - "The recipe names the RDMA devices (MOONCAKE_DEVICE=mlx5_0..3) on the prefill role and the decode stores, the processes that act as store clients, like the GB300 vLLM Mooncake recipes: without an explicit list the store client falls back to NVLink for same-domain peers and aborts every worker at the warmup put." - "The ladder's widest point moves from concurrency 1920 to 2400 on the same 4P1D topology, where prefill still had headroom, and the 1P1D point also runs at concurrency 120. At equal P90 interactivity it delivers about twice the throughput per GPU of the aggregate TP4 concurrency-8 point, which is removed from dsv4-fp4-gb300-dynamo-sglang-agentic-agg." - "The aggregate recipe sets PP_SIZE and PCP_SIZE to 1 in its benchmark environment: the AgentX power path now checks TP, PP_SIZE and PCP_SIZE before replay, and multi-node aggregate jobs do not receive them from the workflow." + - "The prefill role and the decode store services set MC_TCP_BIND_ADDRESS to their node so each store transfer engine advertises the node address instead of the first active interface; some GB300 nodes expose a BMC Redfish interface with the same 10.0.1.2 address on every node, which peers cannot reach." - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到镜像自带 SGLang 的 Mooncake unified-cache external linker,将镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,并将 Dynamo wheel 升级到 1.5.0.dev20260914,其 SGLang worker 不再依赖 ServerArgs.get_model_config。" - "decode 节点通过 srt-slurm 的 mooncake-store 服务向 store 借出主机 DRAM:每个 decode 节点 4 个 180GB 的独立 store,在 worker 之前启动,连接托管的 master 及其 HTTP metadata server,并以就绪端口作为启动检查,超时设为 600 秒,因为 decode 节点上冷启动容器曾超过 srtctl 默认的 120 秒。prefill 每个 rank 保留 140GB segment。" - "prefill 每个 DP rank 按 16k token 切分(chunked-prefill-size 131072,SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK 17408),并将 mem-fraction-static 降至 0.80,这是该切分大小下 DSV4 indexer 所需。" @@ -9232,4 +9233,5 @@ - "配方在 prefill 角色与 decode 节点的 store 上(即作为 store 客户端的进程)显式指定 RDMA 设备(MOONCAKE_DEVICE=mlx5_0..3),与 GB300 vLLM Mooncake 配方的做法一致:未显式指定时,store 客户端会对同 NVLink 域的对端退回 NVLink,并在 warmup put 阶段使所有 worker 退出。" - "阶梯中吞吐最高的点在相同的 4P1D 拓扑下由并发 1920 调整为 2400(该并发下 prefill 仍有余量),1P1D 点同时增加并发 120。在相同的 P90 交互性下,其单卡吞吐约为聚合 TP4 并发 8 点的两倍,因此从 dsv4-fp4-gb300-dynamo-sglang-agentic-agg 中移除该聚合点。" - "聚合式配方在 benchmark 环境中设置 PP_SIZE 与 PCP_SIZE 为 1:AgentX 功耗路径现在会在请求回放前检查 TP、PP_SIZE 与 PCP_SIZE,而多节点聚合任务不会从工作流获得这两个变量。" + - "prefill 角色与 decode 的 store 服务将 MC_TCP_BIND_ADDRESS 设为所在节点,使每个 store 传输引擎对外公布节点自身地址,而不是第一个活动网卡;部分 GB300 节点带有 BMC Redfish 网卡,其地址在每个节点上都是 10.0.1.2,其他节点无法访问。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187