From e21ee13a4775e66dbfb493f48f04077f033383af Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Mon, 28 Sep 2026 03:11:51 -0700 Subject: [PATCH 01/14] fix(dsxe): migrate B300 AgentX recipes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 B300 AgentX 配方迁移到重构后的 inferencex-e2e 布局,并采用 EFA 1.50.0 安装流程。 --- .../configs/install-mooncake-efa-cu13.sh | 15 ++ .../b300-fp4/agentx/agg-tp4-c4-mtp.yaml | 128 ++++++++++ .../b300-fp4/agentx/agg-tp4-c8-mtp.yaml | 128 ++++++++++ .../b300-fp4/agentx/agg-tp8-c1-mtp.yaml | 128 ++++++++++ ...agg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 229 +++++++++++++++++ ...agg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 231 ++++++++++++++++++ inferencex-e2e/configs/nvidia-master.yaml | 92 +++++++ inferencex-e2e/perf-changelog.yaml | 13 + 8 files changed, 964 insertions(+) create mode 100755 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh new file mode 100755 index 0000000000..9c0f1efbe0 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh @@ -0,0 +1,15 @@ +#!/usr/bin/env bash +set -eo pipefail + +export EFA_VERSION=1.50.0 +curl --retry 3 --retry-delay 2 -fsSL -o aws-efa-installer-${EFA_VERSION}.tar.gz \ + https://efa-installer.amazonaws.com/aws-efa-installer-${EFA_VERSION}.tar.gz && \ +tar -xf aws-efa-installer-${EFA_VERSION}.tar.gz && \ +cd aws-efa-installer && \ +apt-get update && \ +./efa_installer.sh -y --skip-kmod --skip-limit-conf --no-verify && \ +cd .. && rm -rf aws-efa-installer* && \ +ldconfig + +python3 -m pip uninstall --break-system-packages -y mooncake-transfer-engine-cuda13 mooncake-transfer-engine-efa-cuda13 +python3 -m pip install --break-system-packages --no-deps mooncake-transfer-engine-efa-cuda13==0.3.13.post1 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml new file mode 100644 index 0000000000..5b88820d64 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml @@ -0,0 +1,128 @@ +schema: 2 +name: "agg-b300-tp4-c4-mtp" + +# AgentX aggregate topology: one TP4 worker occupies half of one eight-GPU +# B300 node and serves both prefill and decode with DSpark K=6. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: "b300" + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: "1" + SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" + SGLANG_OPT_USE_ONLINE_COMPRESS: "0" + SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" + SGLANG_OPT_USE_JIT_NORM: "1" + SGLANG_OPT_USE_TOPK_V2: "True" + + args: + served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.90 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: "flashinfer_mxfp4" + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 8 + cuda-graph-max-bs-decode: 8 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 4 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + IS_MULTINODE: "false" + TP: "4" + PP_SIZE: "1" + PCP_SIZE: "1" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml new file mode 100644 index 0000000000..841c5591c0 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml @@ -0,0 +1,128 @@ +schema: 2 +name: "agg-b300-tp4-c8-mtp" + +# AgentX aggregate topology: one TP4 worker occupies half of one eight-GPU +# B300 node and serves both prefill and decode with DSpark K=6. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: "b300" + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: "1" + SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" + SGLANG_OPT_USE_ONLINE_COMPRESS: "0" + SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" + SGLANG_OPT_USE_JIT_NORM: "1" + SGLANG_OPT_USE_TOPK_V2: "True" + + args: + served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.90 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: "flashinfer_mxfp4" + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 16 + cuda-graph-max-bs-decode: 16 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 4 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + IS_MULTINODE: "false" + TP: "4" + PP_SIZE: "1" + PCP_SIZE: "1" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml new file mode 100644 index 0000000000..880042476d --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml @@ -0,0 +1,128 @@ +schema: 2 +name: "agg-b300-tp8-c1-mtp" + +# AgentX aggregate topology: one TP8 worker occupies one eight-GPU B300 node +# and serves both prefill and decode with DSpark K=6. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: "b300" + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: "1" + SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" + SGLANG_OPT_USE_ONLINE_COMPRESS: "0" + SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" + SGLANG_OPT_USE_JIT_NORM: "1" + SGLANG_OPT_USE_TOPK_V2: "True" + + args: + served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.90 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: "flashinfer_mxfp4" + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 2 + cuda-graph-max-bs-decode: 2 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 8 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + IS_MULTINODE: "false" + TP: "8" + PP_SIZE: "1" + PCP_SIZE: "1" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml new file mode 100644 index 0000000000..edaae2c0a2 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -0,0 +1,229 @@ +schema: 2 +name: "disagg-b300-1p1d-dep8-dep8-c240-mtp-kvoffload" + +# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 on B300 +# (1P x DEP8 / 1D x DEP8, DSpark K=6 + hierarchical-cache KV offload), +# tuned for concurrency 240. Concurrency is exported by the master config. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +setup_script: install-mooncake-efa-cu13.sh + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: b300 + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_TCP_REQUEST_TIMEOUT: "60" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + LD_PRELOAD: /opt/amazon/efa/lib/libfabric.so.1 + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs-prefill: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 3 + hicache-io-backend: direct + + decode: + nodes: 1 + workers: 1 + gpus: 8 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + LD_PRELOAD: /opt/amazon/efa/lib/libfabric.so.1 + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 480 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "true" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml new file mode 100644 index 0000000000..8080eea823 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -0,0 +1,231 @@ +schema: 2 +name: "disagg-b300-2p1d-dep8-dep8-c480-mtp-kvoffload" + +# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 on B300 +# (2P x DEP8 / 1D x DEP8, DSpark K=6 + hierarchical-cache KV offload), +# tuned for concurrency 480. Concurrency is exported by the master config. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +setup_script: install-mooncake-efa-cu13.sh + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: b300 + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_TCP_REQUEST_TIMEOUT: "60" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + prefill: + nodes: 2 + workers: 2 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + OMP_NUM_THREADS: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + LD_PRELOAD: /opt/amazon/efa/lib/libfabric.so.1 + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_DSV4_MHC_PREWARM: '1' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs-prefill: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 3 + hicache-io-backend: direct + + decode: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + OMP_NUM_THREADS: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + LD_PRELOAD: /opt/amazon/efa/lib/libfabric.so.1 + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_DSV4_MHC_PREWARM: '1' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 960 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "true" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 07c47e6e40..5e970801a3 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1161,6 +1161,98 @@ dsv4-fp4-b300-sglang-agentic-hicache-eagle: - { tp: 8, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml } - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [64, 128, 256, 384, 512], router: { name: sglang-router, version: "0.3.2" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml } +dsv4-fp4-b300-dynamo-sglang-agentic-agg: + image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 + model: deepseek-ai/DeepSeek-V4-Pro-0813 + model-prefix: dsv4 + runner: cluster:b300-dsxe + precision: fp4 + framework: dynamo-sglang + router: { name: dynamo-router, version: "1.5.0.dev20260914" } + multinode: true + disagg: false + scenarios: + agentic-coding: + - search-space: + - spec-decoding: draft_model + conc-list: [1] + num-nodes: 1 + worker: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml" + - spec-decoding: draft_model + conc-list: [4] + num-nodes: 1 + worker: + num-worker: 1 + tp: 4 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml" + - spec-decoding: draft_model + conc-list: [8] + num-nodes: 1 + worker: + num-worker: 1 + tp: 4 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml" + +dsv4-fp4-b300-dynamo-sglang-agentic-disagg: + image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 + model: deepseek-ai/DeepSeek-V4-Pro-0813 + model-prefix: dsv4 + runner: cluster:b300-dsxe + precision: fp4 + framework: dynamo-sglang + router: { name: dynamo-router, version: "1.5.0.dev20260914" } + kv-p2p-transfer: mooncake + multinode: true + disagg: true + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - spec-decoding: draft_model + conc-list: [240] + kv-offloading: dram + kv-offload-backend: { name: hicache } + prefill: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml" + decode: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + - spec-decoding: draft_model + conc-list: [480] + kv-offloading: dram + kv-offload-backend: { name: hicache } + prefill: + num-worker: 2 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml" + decode: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + qwen3.5-fp8-b200-sglang: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 model: Qwen/Qwen3.5-397B-A17B-FP8 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index d7fc64f0d6..fe67a2b48f 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9149,3 +9149,16 @@ - "Use AITER attention and allreduce fusion, FP8 KV (fp8_e4m3), and EAGLE MTP (3 steps, topk 1, 4 draft tokens); do not enable ROCm INT4 quick all-reduce. HiCache uses ratio 1.5, write_through_selective, kernel I/O and page_first layout. The matching cookbook recipe is sgl-project/sglang#41849." - "The embedded MTP head runs at its stored precision: block-FP8 expert weights and checkpoint dtype for mtp.fc and gates. No separate draft, draft dtype override or submission-side quantization is used." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3602 +- config-keys: + - dsv4-fp4-b300-dynamo-sglang-agentic-agg + - dsv4-fp4-b300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4/c8 recipes, plus disaggregated 1P1D DEP8 c240 and 2P1D DEP8 c480 recipes, under the refactored inferencex-e2e project layout." + - "Use EFA 1.50.0 with mooncake-transfer-engine-efa-cuda13 0.3.13.post1 on cluster:b300-dsxe; rely on the matching Pyxis-provided libfabric ABI without FI_* or LD_LIBRARY_PATH overrides." + - "Use 192 logical CPUs per task, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE nodes." + - "在重构后的 inferencex-e2e 项目布局中新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4/c8,以及分离式 1P1D DEP8 c240 和 2P1D DEP8 c480。" + - "在 cluster:b300-dsxe 上使用 EFA 1.50.0 与 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 Pyxis 提供且 ABI 匹配的 libfabric,不设置 FI_* 或 LD_LIBRARY_PATH 覆盖。" + - "每任务使用 192 个逻辑 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 From 916fae649b4085c5ff2e9cd432ffb388bcd5330b Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Mon, 28 Sep 2026 08:21:09 -0700 Subject: [PATCH 02/14] fix(dsxe): avoid EFA preload during setup MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 移除 B300 分离式配方中的 LD_PRELOAD,避免 EFA 安装时向 dpkg 等初始化进程注入不兼容的 libfabric/libefa。 --- .../agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 2 -- .../agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 2 -- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 3 files changed, 9 insertions(+), 4 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml index edaae2c0a2..a722871c0e 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -84,7 +84,6 @@ roles: MOONCAKE_PROTOCOL: efa MC_MS_FILTERS: rdmap191s0 WITH_NVIDIA_PEERMEM: '0' - LD_PRELOAD: /opt/amazon/efa/lib/libfabric.so.1 NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' DYN_SKIP_SGLANG_LOG_FORMATTING: '1' @@ -159,7 +158,6 @@ roles: MOONCAKE_PROTOCOL: efa MC_MS_FILTERS: rdmap191s0 WITH_NVIDIA_PEERMEM: '0' - LD_PRELOAD: /opt/amazon/efa/lib/libfabric.so.1 NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml index 8080eea823..32dc10b8dc 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -84,7 +84,6 @@ roles: MOONCAKE_PROTOCOL: efa MC_MS_FILTERS: rdmap191s0 WITH_NVIDIA_PEERMEM: '0' - LD_PRELOAD: /opt/amazon/efa/lib/libfabric.so.1 NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' DYN_SKIP_SGLANG_LOG_FORMATTING: '1' @@ -160,7 +159,6 @@ roles: MOONCAKE_PROTOCOL: efa MC_MS_FILTERS: rdmap191s0 WITH_NVIDIA_PEERMEM: '0' - LD_PRELOAD: /opt/amazon/efa/lib/libfabric.so.1 NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index fe67a2b48f..256dbcb4f4 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9162,3 +9162,12 @@ - "在 cluster:b300-dsxe 上使用 EFA 1.50.0 与 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 Pyxis 提供且 ABI 匹配的 libfabric,不设置 FI_* 或 LD_LIBRARY_PATH 覆盖。" - "每任务使用 192 个逻辑 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 + +- config-keys: + - dsv4-fp4-b300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Remove LD_PRELOAD from both B300 disaggregated recipes so the EFA 1.50.0 installer does not inject libfabric/libefa into dpkg and other setup subprocesses. Keep the EFA installer and Mooncake package pins unchanged." + - "从两个 B300 分离式配方中移除 LD_PRELOAD,避免 EFA 1.50.0 安装时向 dpkg 等初始化子进程注入 libfabric/libefa;EFA 安装命令和 Mooncake 包版本保持不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 From b7cebd16a15ffaae1b08be56bcee660362564dc0 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Mon, 28 Sep 2026 11:33:01 -0700 Subject: [PATCH 03/14] fix(dsxe): keep SGLang metrics directories stable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 B300 分离式 prefill 和 decode 角色预建稳定的 Prometheus 多进程目录,避免调度器启动时临时目录被清理。 --- .../configs/install-mooncake-efa-cu13.sh | 1 + .../agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 2 ++ .../agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 2 ++ inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 14 insertions(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh index 9c0f1efbe0..c4b1381f7f 100755 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh @@ -13,3 +13,4 @@ ldconfig python3 -m pip uninstall --break-system-packages -y mooncake-transfer-engine-cuda13 mooncake-transfer-engine-efa-cuda13 python3 -m pip install --break-system-packages --no-deps mooncake-transfer-engine-efa-cuda13==0.3.13.post1 +mkdir -p /tmp/sglang-prometheus-prefill /tmp/sglang-prometheus-decode diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml index a722871c0e..80a51cac05 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -68,6 +68,7 @@ roles: SGLANG_DSV4_MHC_PREWARM: '1' PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' + PROMETHEUS_MULTIPROC_DIR: /tmp/sglang-prometheus-prefill SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' @@ -143,6 +144,7 @@ roles: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' + PROMETHEUS_MULTIPROC_DIR: /tmp/sglang-prometheus-decode SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml index 32dc10b8dc..41bd71dc74 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -67,6 +67,7 @@ roles: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' + PROMETHEUS_MULTIPROC_DIR: /tmp/sglang-prometheus-prefill OMP_NUM_THREADS: '1' SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache @@ -143,6 +144,7 @@ roles: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' + PROMETHEUS_MULTIPROC_DIR: /tmp/sglang-prometheus-decode OMP_NUM_THREADS: '1' SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 256dbcb4f4..cb9315de34 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9171,3 +9171,12 @@ - "Remove LD_PRELOAD from both B300 disaggregated recipes so the EFA 1.50.0 installer does not inject libfabric/libefa into dpkg and other setup subprocesses. Keep the EFA installer and Mooncake package pins unchanged." - "从两个 B300 分离式配方中移除 LD_PRELOAD,避免 EFA 1.50.0 安装时向 dpkg 等初始化子进程注入 libfabric/libefa;EFA 安装命令和 Mooncake 包版本保持不变。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 + +- config-keys: + - dsv4-fp4-b300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Set stable, separate Prometheus multiprocess directories for the B300 prefill and decode roles, creating them in container setup before SGLang starts. This avoids the pinned SGLang build's temporary-directory lifetime race during scheduler metrics initialization." + - "为 B300 prefill 和 decode 角色设置稳定且分离的 Prometheus 多进程目录,并在容器初始化时预先创建,避免所用 SGLang 版本在调度器指标初始化期间出现临时目录生命周期竞争。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 From cee5a8ed20f397bbd64c9e134f004c2354444545 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Mon, 28 Sep 2026 15:55:40 -0700 Subject: [PATCH 04/14] fix(dsxe): preserve Pyxis RDMA libraries for B300 AgentX MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 跳过 EFA 安装器的 rdma-core 升级,避免覆盖 Pyxis 挂载的 libibverbs,并确保安装失败时立即终止初始化。排除 TP8 c1 性能异常的 gpu-11 节点,追加性能变更记录。 --- .../configs/install-mooncake-efa-cu13.sh | 13 +++++++------ .../b300-fp4/agentx/agg-tp4-c4-mtp.yaml | 2 +- .../b300-fp4/agentx/agg-tp4-c8-mtp.yaml | 2 +- .../b300-fp4/agentx/agg-tp8-c1-mtp.yaml | 2 +- ...agg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 2 +- ...agg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 2 +- inferencex-e2e/perf-changelog.yaml | 19 +++++++++++++++++++ 7 files changed, 31 insertions(+), 11 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh index c4b1381f7f..23d7bb7bc2 100755 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh @@ -3,12 +3,13 @@ set -eo pipefail export EFA_VERSION=1.50.0 curl --retry 3 --retry-delay 2 -fsSL -o aws-efa-installer-${EFA_VERSION}.tar.gz \ - https://efa-installer.amazonaws.com/aws-efa-installer-${EFA_VERSION}.tar.gz && \ -tar -xf aws-efa-installer-${EFA_VERSION}.tar.gz && \ -cd aws-efa-installer && \ -apt-get update && \ -./efa_installer.sh -y --skip-kmod --skip-limit-conf --no-verify && \ -cd .. && rm -rf aws-efa-installer* && \ + https://efa-installer.amazonaws.com/aws-efa-installer-${EFA_VERSION}.tar.gz +tar -xf aws-efa-installer-${EFA_VERSION}.tar.gz +cd aws-efa-installer +apt-get update +./efa_installer.sh -y --skip-kmod --skip-limit-conf --no-verify --skip-rdma-core +cd .. +rm -rf aws-efa-installer* ldconfig python3 -m pip uninstall --break-system-packages -y mooncake-transfer-engine-cuda13 mooncake-transfer-engine-efa-cuda13 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml index 5b88820d64..4a0de7e8b3 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml @@ -104,7 +104,7 @@ roles: sbatch_directives: mem: "0" cpus-per-task: "192" - exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: mem: "0" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml index 841c5591c0..455ac340f1 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml @@ -104,7 +104,7 @@ roles: sbatch_directives: mem: "0" cpus-per-task: "192" - exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: mem: "0" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml index 880042476d..a7383fcac7 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml @@ -104,7 +104,7 @@ roles: sbatch_directives: mem: "0" cpus-per-task: "192" - exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: mem: "0" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml index 80a51cac05..0b5ccf6f42 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -208,7 +208,7 @@ roles: sbatch_directives: mem: "0" cpus-per-task: "192" - exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: mem: "0" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml index 41bd71dc74..20ef349147 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -210,7 +210,7 @@ roles: sbatch_directives: mem: "0" cpus-per-task: "192" - exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-16" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: mem: "0" diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index cb9315de34..8f9f2021ab 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9180,3 +9180,22 @@ - "Set stable, separate Prometheus multiprocess directories for the B300 prefill and decode roles, creating them in container setup before SGLang starts. This avoids the pinned SGLang build's temporary-directory lifetime race during scheduler metrics initialization." - "为 B300 prefill 和 decode 角色设置稳定且分离的 Prometheus 多进程目录,并在容器初始化时预先创建,避免所用 SGLang 版本在调度器指标初始化期间出现临时目录生命周期竞争。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 + +- config-keys: + - dsv4-fp4-b300-dynamo-sglang-agentic-agg + - dsv4-fp4-b300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Exclude DSXE gpu-11 from all B300 Dynamo+SGLang AgentX recipes after an otherwise identical TP8 c1 canary measured 175.69 P90 tokens/s/user there versus 291.12 on gpu-17." + - "在其他条件相同的 TP8 c1 预检中,DSXE gpu-11 的 P90 速度为每用户 175.69 token/s,而 gpu-17 为 291.12;因此在所有 B300 Dynamo+SGLang AgentX 配方中排除 gpu-11。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 + +- config-keys: + - dsv4-fp4-b300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Skip the EFA installer's rdma-core upgrade on DSXE so the Pyxis-provided libibverbs and provider files remain intact; fail setup immediately if EFA installation fails." + - "在 DSXE 上跳过 EFA 安装器对 rdma-core 的升级,保留 Pyxis 提供的 libibverbs 和 provider 文件;若 EFA 安装失败则立即终止初始化。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 From 86daade7fe7cf2130f9322b6f1f04bd845de4328 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Mon, 28 Sep 2026 18:24:10 -0700 Subject: [PATCH 05/14] fix(dsxe): use cluster RDMA stack for Mooncake EFA MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit EFA 1.50 安装器与 DSXE Pyxis 挂载的 verbs 软件栈冲突;仅安装 Mooncake EFA wheel,并沿用集群提供的 RDMA 库。追加性能变更记录。 --- .../configs/install-mooncake-efa-cu13.sh | 11 ----------- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 2 files changed, 9 insertions(+), 11 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh index 23d7bb7bc2..09f5350fe2 100755 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh @@ -1,17 +1,6 @@ #!/usr/bin/env bash set -eo pipefail -export EFA_VERSION=1.50.0 -curl --retry 3 --retry-delay 2 -fsSL -o aws-efa-installer-${EFA_VERSION}.tar.gz \ - https://efa-installer.amazonaws.com/aws-efa-installer-${EFA_VERSION}.tar.gz -tar -xf aws-efa-installer-${EFA_VERSION}.tar.gz -cd aws-efa-installer -apt-get update -./efa_installer.sh -y --skip-kmod --skip-limit-conf --no-verify --skip-rdma-core -cd .. -rm -rf aws-efa-installer* -ldconfig - python3 -m pip uninstall --break-system-packages -y mooncake-transfer-engine-cuda13 mooncake-transfer-engine-efa-cuda13 python3 -m pip install --break-system-packages --no-deps mooncake-transfer-engine-efa-cuda13==0.3.13.post1 mkdir -p /tmp/sglang-prometheus-prefill /tmp/sglang-prometheus-decode diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 8f9f2021ab..23d7d5891b 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9199,3 +9199,12 @@ - "Skip the EFA installer's rdma-core upgrade on DSXE so the Pyxis-provided libibverbs and provider files remain intact; fail setup immediately if EFA installation fails." - "在 DSXE 上跳过 EFA 安装器对 rdma-core 的升级,保留 Pyxis 提供的 libibverbs 和 provider 文件;若 EFA 安装失败则立即终止初始化。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 + +- config-keys: + - dsv4-fp4-b300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Use only the Mooncake EFA CUDA 13 wheel in B300 disaggregated container setup. The EFA 1.50.0 installer fails in the DSXE Pyxis environment: libfabric1-aws requires rdma-core/ibverbs 64 while the mounted provider ABI is 50, and one launch reports a non-root container user. Rely on the cluster-provided verbs stack instead of modifying it in the container." + - "B300 分离式容器初始化仅安装 Mooncake EFA CUDA 13 wheel。EFA 1.50.0 安装器在 DSXE Pyxis 环境中失败:libfabric1-aws 要求 rdma-core/ibverbs 64,而挂载的 provider ABI 为 50,另有一次启动报告容器用户非 root。因此沿用集群提供的 verbs 软件栈,不在容器内改动。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 From 4366f34362ca0c7d0a7dcf3036bf568c1d26d37a Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 01:33:04 -0700 Subject: [PATCH 06/14] feat(dsxe): add B300 disaggregated c64 point MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace the aggregate TP4 c8 point with a 1P1D DEP8 c64 recipe and remove the explicit Prometheus multiprocess-directory workaround.\n\n以 1P1D DEP8 c64 配方替换聚合式 TP4 c8 测试点,并移除显式的 Prometheus 多进程目录 workaround。 --- .../configs/install-mooncake-efa-cu13.sh | 1 - .../b300-fp4/agentx/agg-tp4-c8-mtp.yaml | 128 ---------- ...agg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 2 - ...sagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml | 227 ++++++++++++++++++ ...agg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 2 - inferencex-e2e/configs/nvidia-master.yaml | 26 +- inferencex-e2e/perf-changelog.yaml | 13 +- 7 files changed, 245 insertions(+), 154 deletions(-) delete mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh index 09f5350fe2..252168b822 100755 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh @@ -3,4 +3,3 @@ set -eo pipefail python3 -m pip uninstall --break-system-packages -y mooncake-transfer-engine-cuda13 mooncake-transfer-engine-efa-cuda13 python3 -m pip install --break-system-packages --no-deps mooncake-transfer-engine-efa-cuda13==0.3.13.post1 -mkdir -p /tmp/sglang-prometheus-prefill /tmp/sglang-prometheus-decode diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml deleted file mode 100644 index 455ac340f1..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml +++ /dev/null @@ -1,128 +0,0 @@ -schema: 2 -name: "agg-b300-tp4-c8-mtp" - -# AgentX aggregate topology: one TP4 worker occupies half of one eight-GPU -# B300 node and serves both prefill and decode with DSpark K=6. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" - -dynamo: - install: false - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: "b300" - gpus_per_node: 8 -services: - - name: etcd - type: etcd - placement: - node: infra - - name: nats - type: nats - placement: - node: infra - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: false - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DEFAULT_THINKING: "1" - SGLANG_DSV4_REASONING_EFFORT: high - PIP_BREAK_SYSTEM_PACKAGES: "1" - SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" - SGLANG_OPT_USE_ONLINE_COMPRESS: "0" - SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" - SGLANG_OPT_USE_JIT_NORM: "1" - SGLANG_OPT_USE_TOPK_V2: "True" - - args: - served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" - enable-metrics: true - enable-cache-report: true - trust-remote-code: true - weight-loader-prefetch-checkpoints: true - stream-interval: 10 - watchdog-timeout: 1000000 - mem-fraction-static: 0.90 - page-size: 256 - chunked-prefill-size: 8192 - max-prefill-tokens: 8192 - moe-runner-backend: "flashinfer_mxfp4" - enable-deepseek-v4-fp4-indexer: true - disable-flashinfer-autotune: true - swa-full-tokens-ratio: 0.1 - max-running-requests: 16 - cuda-graph-max-bs-decode: 16 - scheduler-recv-interval: 30 - dp-size: 1 - tp-size: 4 - ep-size: 1 - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - -sbatch_directives: - mem: "0" - cpus-per-task: "192" - exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" - RESULT_DIR: "/logs/agentic" - PORT: "8000" - IS_MULTINODE: "false" - TP: "4" - PP_SIZE: "1" - PCP_SIZE: "1" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml index 0b5ccf6f42..dc11039597 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -68,7 +68,6 @@ roles: SGLANG_DSV4_MHC_PREWARM: '1' PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' - PROMETHEUS_MULTIPROC_DIR: /tmp/sglang-prometheus-prefill SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' @@ -144,7 +143,6 @@ roles: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' - PROMETHEUS_MULTIPROC_DIR: /tmp/sglang-prometheus-decode SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml new file mode 100644 index 0000000000..17d83b38ce --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml @@ -0,0 +1,227 @@ +schema: 2 +name: "disagg-b300-1p1d-dep8-dep8-c64-mtp-kvoffload" + +# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 on B300 +# (1P x DEP8 / 1D x DEP8, DSpark K=6 + hierarchical-cache KV offload), +# tuned for concurrency 64. Concurrency is exported by the master config. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +setup_script: install-mooncake-efa-cu13.sh + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: b300 + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_TCP_REQUEST_TIMEOUT: "60" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs-prefill: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 3 + hicache-io-backend: direct + + decode: + nodes: 1 + workers: 1 + gpus: 8 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 128 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "true" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml index 20ef349147..0c5f4a59bc 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -67,7 +67,6 @@ roles: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' - PROMETHEUS_MULTIPROC_DIR: /tmp/sglang-prometheus-prefill OMP_NUM_THREADS: '1' SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache @@ -144,7 +143,6 @@ roles: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' - PROMETHEUS_MULTIPROC_DIR: /tmp/sglang-prometheus-decode OMP_NUM_THREADS: '1' SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 5e970801a3..f9b9e315fd 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1194,16 +1194,6 @@ dsv4-fp4-b300-dynamo-sglang-agentic-agg: dp-attn: false additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml" - - spec-decoding: draft_model - conc-list: [8] - num-nodes: 1 - worker: - num-worker: 1 - tp: 4 - ep: 1 - dp-attn: false - additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp.yaml" dsv4-fp4-b300-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 @@ -1220,6 +1210,22 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: agentic-coding: - dram-utilization: 0.80 search-space: + - spec-decoding: draft_model + conc-list: [64] + kv-offloading: dram + kv-offload-backend: { name: hicache } + prefill: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml" + decode: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true - spec-decoding: draft_model conc-list: [240] kv-offloading: dram diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 23d7d5891b..ec13c5d880 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9155,10 +9155,10 @@ scenario-type: - agentic-coding description: - - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4/c8 recipes, plus disaggregated 1P1D DEP8 c240 and 2P1D DEP8 c480 recipes, under the refactored inferencex-e2e project layout." + - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4 recipes, plus disaggregated 1P1D DEP8 c64/c240 and 2P1D DEP8 c480 recipes, under the refactored inferencex-e2e project layout." - "Use EFA 1.50.0 with mooncake-transfer-engine-efa-cuda13 0.3.13.post1 on cluster:b300-dsxe; rely on the matching Pyxis-provided libfabric ABI without FI_* or LD_LIBRARY_PATH overrides." - "Use 192 logical CPUs per task, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE nodes." - - "在重构后的 inferencex-e2e 项目布局中新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4/c8,以及分离式 1P1D DEP8 c240 和 2P1D DEP8 c480。" + - "在重构后的 inferencex-e2e 项目布局中新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4,以及分离式 1P1D DEP8 c64/c240 和 2P1D DEP8 c480。" - "在 cluster:b300-dsxe 上使用 EFA 1.50.0 与 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 Pyxis 提供且 ABI 匹配的 libfabric,不设置 FI_* 或 LD_LIBRARY_PATH 覆盖。" - "每任务使用 192 个逻辑 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 @@ -9172,15 +9172,6 @@ - "从两个 B300 分离式配方中移除 LD_PRELOAD,避免 EFA 1.50.0 安装时向 dpkg 等初始化子进程注入 libfabric/libefa;EFA 安装命令和 Mooncake 包版本保持不变。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 -- config-keys: - - dsv4-fp4-b300-dynamo-sglang-agentic-disagg - scenario-type: - - agentic-coding - description: - - "Set stable, separate Prometheus multiprocess directories for the B300 prefill and decode roles, creating them in container setup before SGLang starts. This avoids the pinned SGLang build's temporary-directory lifetime race during scheduler metrics initialization." - - "为 B300 prefill 和 decode 角色设置稳定且分离的 Prometheus 多进程目录,并在容器初始化时预先创建,避免所用 SGLang 版本在调度器指标初始化期间出现临时目录生命周期竞争。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 - - config-keys: - dsv4-fp4-b300-dynamo-sglang-agentic-agg - dsv4-fp4-b300-dynamo-sglang-agentic-disagg From 12abcc398014103fd617f510e131e97ed4d99ff0 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 06:50:01 -0700 Subject: [PATCH 07/14] feat(dsv4): use NIXL for B300 disaggregation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将三个 B300 Dynamo+SGLang 解耦配置从 Mooncake 切换为基于 UCX 的 NIXL,并移除 Mooncake EFA 安装脚本及相关环境变量。 --- .../configs/install-mooncake-efa-cu13.sh | 5 ----- .../disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 14 ++++---------- .../disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml | 14 ++++---------- .../disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 14 ++++---------- inferencex-e2e/configs/nvidia-master.yaml | 2 +- 5 files changed, 13 insertions(+), 36 deletions(-) delete mode 100755 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh deleted file mode 100755 index 252168b822..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh +++ /dev/null @@ -1,5 +0,0 @@ -#!/usr/bin/env bash -set -eo pipefail - -python3 -m pip uninstall --break-system-packages -y mooncake-transfer-engine-cuda13 mooncake-transfer-engine-efa-cuda13 -python3 -m pip install --break-system-packages --no-deps mooncake-transfer-engine-efa-cuda13==0.3.13.post1 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml index dc11039597..7323365a12 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -19,8 +19,6 @@ identity: dynamo: install: false -setup_script: install-mooncake-efa-cu13.sh - health_check: max_attempts: 1440 interval_seconds: 10 @@ -65,6 +63,7 @@ roles: SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX SGLANG_DSV4_MHC_PREWARM: '1' PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' @@ -81,9 +80,6 @@ roles: SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' SGLANG_OPT_USE_ONLINE_COMPRESS: '0' NCCL_CUMEM_ENABLE: '1' - MOONCAKE_PROTOCOL: efa - MC_MS_FILTERS: rdmap191s0 - WITH_NVIDIA_PEERMEM: '0' NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' DYN_SKIP_SGLANG_LOG_FORMATTING: '1' @@ -111,7 +107,7 @@ roles: moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake + disaggregation-transfer-backend: nixl disaggregation-mode: prefill load-balance-method: total_tokens mem-fraction-static: 0.85 @@ -141,6 +137,7 @@ roles: env: SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' @@ -155,9 +152,6 @@ roles: SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' SGLANG_OPT_USE_ONLINE_COMPRESS: '0' NCCL_CUMEM_ENABLE: '1' - MOONCAKE_PROTOCOL: efa - MC_MS_FILTERS: rdmap191s0 - WITH_NVIDIA_PEERMEM: '0' NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' @@ -185,7 +179,7 @@ roles: moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake + disaggregation-transfer-backend: nixl disaggregation-mode: decode load-balance-method: total_tokens mem-fraction-static: 0.9 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml index 17d83b38ce..731d2a952b 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml @@ -19,8 +19,6 @@ identity: dynamo: install: false -setup_script: install-mooncake-efa-cu13.sh - health_check: max_attempts: 1440 interval_seconds: 10 @@ -65,6 +63,7 @@ roles: SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX SGLANG_DSV4_MHC_PREWARM: '1' PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' @@ -81,9 +80,6 @@ roles: SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' SGLANG_OPT_USE_ONLINE_COMPRESS: '0' NCCL_CUMEM_ENABLE: '1' - MOONCAKE_PROTOCOL: efa - MC_MS_FILTERS: rdmap191s0 - WITH_NVIDIA_PEERMEM: '0' NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' DYN_SKIP_SGLANG_LOG_FORMATTING: '1' @@ -111,7 +107,7 @@ roles: moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake + disaggregation-transfer-backend: nixl disaggregation-mode: prefill load-balance-method: total_tokens mem-fraction-static: 0.85 @@ -141,6 +137,7 @@ roles: env: SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' @@ -155,9 +152,6 @@ roles: SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' SGLANG_OPT_USE_ONLINE_COMPRESS: '0' NCCL_CUMEM_ENABLE: '1' - MOONCAKE_PROTOCOL: efa - MC_MS_FILTERS: rdmap191s0 - WITH_NVIDIA_PEERMEM: '0' NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' @@ -185,7 +179,7 @@ roles: moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake + disaggregation-transfer-backend: nixl disaggregation-mode: decode load-balance-method: total_tokens mem-fraction-static: 0.9 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml index 0c5f4a59bc..e39b0b8e94 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -19,8 +19,6 @@ identity: dynamo: install: false -setup_script: install-mooncake-efa-cu13.sh - health_check: max_attempts: 1440 interval_seconds: 10 @@ -65,6 +63,7 @@ roles: SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' OMP_NUM_THREADS: '1' @@ -81,9 +80,6 @@ roles: SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' SGLANG_OPT_USE_ONLINE_COMPRESS: '0' NCCL_CUMEM_ENABLE: '1' - MOONCAKE_PROTOCOL: efa - MC_MS_FILTERS: rdmap191s0 - WITH_NVIDIA_PEERMEM: '0' NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' DYN_SKIP_SGLANG_LOG_FORMATTING: '1' @@ -112,7 +108,7 @@ roles: moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake + disaggregation-transfer-backend: nixl disaggregation-mode: prefill load-balance-method: total_tokens mem-fraction-static: 0.85 @@ -141,6 +137,7 @@ roles: env: SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' OMP_NUM_THREADS: '1' @@ -156,9 +153,6 @@ roles: SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' SGLANG_OPT_USE_ONLINE_COMPRESS: '0' NCCL_CUMEM_ENABLE: '1' - MOONCAKE_PROTOCOL: efa - MC_MS_FILTERS: rdmap191s0 - WITH_NVIDIA_PEERMEM: '0' NCCL_TIMEOUT: '100000' SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' @@ -187,7 +181,7 @@ roles: moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake + disaggregation-transfer-backend: nixl disaggregation-mode: decode load-balance-method: total_tokens mem-fraction-static: 0.9 diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index f9b9e315fd..a70384d0e6 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1203,7 +1203,7 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: precision: fp4 framework: dynamo-sglang router: { name: dynamo-router, version: "1.5.0.dev20260914" } - kv-p2p-transfer: mooncake + kv-p2p-transfer: nixl multinode: true disagg: true scenarios: From 58453ff03a754beca0321e78c665e251cdfc5b82 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 13:40:13 -0700 Subject: [PATCH 08/14] fix(dsv4): use libfabric for B300 NIXL --- .../agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 4 ++-- .../agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml | 4 ++-- .../agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 4 ++-- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml index 7323365a12..e5034d7d3b 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -63,7 +63,7 @@ roles: SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX + SGLANG_DISAGGREGATION_NIXL_BACKEND: LIBFABRIC SGLANG_DSV4_MHC_PREWARM: '1' PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' @@ -137,7 +137,7 @@ roles: env: SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX + SGLANG_DISAGGREGATION_NIXL_BACKEND: LIBFABRIC PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml index 731d2a952b..0dbefbbe1f 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml @@ -63,7 +63,7 @@ roles: SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX + SGLANG_DISAGGREGATION_NIXL_BACKEND: LIBFABRIC SGLANG_DSV4_MHC_PREWARM: '1' PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' @@ -137,7 +137,7 @@ roles: env: SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX + SGLANG_DISAGGREGATION_NIXL_BACKEND: LIBFABRIC PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml index e39b0b8e94..c4d36771cd 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -63,7 +63,7 @@ roles: SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX + SGLANG_DISAGGREGATION_NIXL_BACKEND: LIBFABRIC PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' OMP_NUM_THREADS: '1' @@ -137,7 +137,7 @@ roles: env: SGLANG_RAGGED_VERIFY_MODE: "static" SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - SGLANG_DISAGGREGATION_NIXL_BACKEND: UCX + SGLANG_DISAGGREGATION_NIXL_BACKEND: LIBFABRIC PIP_BREAK_SYSTEM_PACKAGES: '1' PYTHONUNBUFFERED: '1' OMP_NUM_THREADS: '1' From 8dfc5d85d2ed1f7d604b9fd1141b0977b34e975e Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 13:45:50 -0700 Subject: [PATCH 09/14] docs(perf): link B300 NIXL submission --- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index ec13c5d880..52ba87e29d 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9199,3 +9199,12 @@ - "Use only the Mooncake EFA CUDA 13 wheel in B300 disaggregated container setup. The EFA 1.50.0 installer fails in the DSXE Pyxis environment: libfabric1-aws requires rdma-core/ibverbs 64 while the mounted provider ABI is 50, and one launch reports a non-root container user. Rely on the cluster-provided verbs stack instead of modifying it in the container." - "B300 分离式容器初始化仅安装 Mooncake EFA CUDA 13 wheel。EFA 1.50.0 安装器在 DSXE Pyxis 环境中失败:libfabric1-aws 要求 rdma-core/ibverbs 64,而挂载的 provider ABI 为 50,另有一次启动报告容器用户非 root。因此沿用集群提供的 verbs 软件栈,不在容器内改动。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 + +- config-keys: + - dsv4-fp4-b300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Switch the B300 DeepSeek-V4-Pro Dynamo+SGLang disaggregated c64, c240, and c480 recipes from Mooncake to NIXL with the LIBFABRIC backend for Amazon EFA, and remove the runtime Mooncake-EFA installer and setup hooks." + - "将 B300 DeepSeek-V4-Pro Dynamo+SGLang 分离式 c64、c240 和 c480 配方从 Mooncake 切换到使用 LIBFABRIC 后端的 NIXL(适配 Amazon EFA),并移除运行时 Mooncake-EFA 安装脚本和 setup hook。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3597 From 5b923d1f995fbfe263a904b5b5947128a97fa26a Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 18:33:43 -0700 Subject: [PATCH 10/14] fix(dsxe): migrate B300 cache mounts to Python launcher --- inferencex-e2e/configs/runners.yaml | 1 + inferencex-e2e/infx/launch/drivers/srt/lanes.py | 5 ++++- .../infx/tests/launch/test_srt_policy.py | 17 +++++++++++++++++ 3 files changed, 22 insertions(+), 1 deletion(-) diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index cf67501a63..7f25a1221b 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -541,6 +541,7 @@ clusters: cpus-per-gpu: 24 salloc-args: [--mem=0] volumes: + aiperf-cache: {path: /data/home/sa-gha-runner/aiperf-cache} hf-home: {path: ~/.cache/huggingface} hf-hub-cache: {path: ~/.cache/huggingface/hub} scratch: {path: /scratch/models, visibility: node-local} diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index e8217eb607..6ec7f1d0e1 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -64,7 +64,10 @@ class SrtLane: ), mounts=_AGENTIC_CACHES, ), - ("b300-dsxe", LaunchPath.SRT_MULTI): SrtLane(frameworks=_DYNAMO), + ("b300-dsxe", LaunchPath.SRT_MULTI): SrtLane( + frameworks=_DYNAMO, + mounts=_AGENTIC_CACHES, + ), ("gb200-nv", LaunchPath.SRT_MULTI): SrtLane( frameworks=_DYNAMO, setup_scripts={"dynamo-sglang": "install-torchao.sh"}, diff --git a/inferencex-e2e/infx/tests/launch/test_srt_policy.py b/inferencex-e2e/infx/tests/launch/test_srt_policy.py index ce4b3d00d5..2404c4423f 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_policy.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_policy.py @@ -73,6 +73,23 @@ def test_a_path_without_a_lane_on_the_cluster_is_refused(): srt_lane("b200-cw", LaunchPath.SRT_MULTI) +def test_b300_multinode_agentic_lane_mounts_standard_caches(): + lane = srt_lane("b300-dsxe", LaunchPath.SRT_MULTI) + agentic = request(IS_AGENTIC="1") + regular = request(IS_AGENTIC="0") + + matched = [ + (mount.volume, mount.target, mount.world_writable) + for mount in lane.mounts + if mount.when(agentic) + ] + assert matched == [ + ("aiperf-cache", "/aiperf_mmap_cache", True), + ("hf-hub-cache", "/hf_hub_cache", True), + ] + assert not any(mount.when(regular) for mount in lane.mounts) + + OVERRIDES = ( Override(Match(frameworks=any_of("sglang")), entry="M"), Override(Match(frameworks=any_of("trt")), served_name="served-m"), From 148567fec3ae9944e00865f9cafe669b14dd52be Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 19:19:33 -0700 Subject: [PATCH 11/14] docs(perf): describe final B300 NIXL configuration --- inferencex-e2e/perf-changelog.yaml | 50 ++---------------------------- 1 file changed, 2 insertions(+), 48 deletions(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 52ba87e29d..91c0917378 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9156,55 +9156,9 @@ - agentic-coding description: - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4 recipes, plus disaggregated 1P1D DEP8 c64/c240 and 2P1D DEP8 c480 recipes, under the refactored inferencex-e2e project layout." - - "Use EFA 1.50.0 with mooncake-transfer-engine-efa-cuda13 0.3.13.post1 on cluster:b300-dsxe; rely on the matching Pyxis-provided libfabric ABI without FI_* or LD_LIBRARY_PATH overrides." + - "Use NIXL with the LIBFABRIC backend for disaggregated KV transfer, relying on the networking stack provided by the pinned container and DSXE runtime without explicit host EFA/OFI mounts or runtime transport setup." - "Use 192 logical CPUs per task, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE nodes." - "在重构后的 inferencex-e2e 项目布局中新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4,以及分离式 1P1D DEP8 c64/c240 和 2P1D DEP8 c480。" - - "在 cluster:b300-dsxe 上使用 EFA 1.50.0 与 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 Pyxis 提供且 ABI 匹配的 libfabric,不设置 FI_* 或 LD_LIBRARY_PATH 覆盖。" + - "分离式 KV 传输使用 NIXL 的 LIBFABRIC 后端,依赖固定容器与 DSXE 运行时提供的网络栈,不显式挂载主机 EFA/OFI 目录,也不在运行时安装传输组件。" - "每任务使用 192 个逻辑 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 - -- config-keys: - - dsv4-fp4-b300-dynamo-sglang-agentic-disagg - scenario-type: - - agentic-coding - description: - - "Remove LD_PRELOAD from both B300 disaggregated recipes so the EFA 1.50.0 installer does not inject libfabric/libefa into dpkg and other setup subprocesses. Keep the EFA installer and Mooncake package pins unchanged." - - "从两个 B300 分离式配方中移除 LD_PRELOAD,避免 EFA 1.50.0 安装时向 dpkg 等初始化子进程注入 libfabric/libefa;EFA 安装命令和 Mooncake 包版本保持不变。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 - -- config-keys: - - dsv4-fp4-b300-dynamo-sglang-agentic-agg - - dsv4-fp4-b300-dynamo-sglang-agentic-disagg - scenario-type: - - agentic-coding - description: - - "Exclude DSXE gpu-11 from all B300 Dynamo+SGLang AgentX recipes after an otherwise identical TP8 c1 canary measured 175.69 P90 tokens/s/user there versus 291.12 on gpu-17." - - "在其他条件相同的 TP8 c1 预检中,DSXE gpu-11 的 P90 速度为每用户 175.69 token/s,而 gpu-17 为 291.12;因此在所有 B300 Dynamo+SGLang AgentX 配方中排除 gpu-11。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 - -- config-keys: - - dsv4-fp4-b300-dynamo-sglang-agentic-disagg - scenario-type: - - agentic-coding - description: - - "Skip the EFA installer's rdma-core upgrade on DSXE so the Pyxis-provided libibverbs and provider files remain intact; fail setup immediately if EFA installation fails." - - "在 DSXE 上跳过 EFA 安装器对 rdma-core 的升级,保留 Pyxis 提供的 libibverbs 和 provider 文件;若 EFA 安装失败则立即终止初始化。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 - -- config-keys: - - dsv4-fp4-b300-dynamo-sglang-agentic-disagg - scenario-type: - - agentic-coding - description: - - "Use only the Mooncake EFA CUDA 13 wheel in B300 disaggregated container setup. The EFA 1.50.0 installer fails in the DSXE Pyxis environment: libfabric1-aws requires rdma-core/ibverbs 64 while the mounted provider ABI is 50, and one launch reports a non-root container user. Rely on the cluster-provided verbs stack instead of modifying it in the container." - - "B300 分离式容器初始化仅安装 Mooncake EFA CUDA 13 wheel。EFA 1.50.0 安装器在 DSXE Pyxis 环境中失败:libfabric1-aws 要求 rdma-core/ibverbs 64,而挂载的 provider ABI 为 50,另有一次启动报告容器用户非 root。因此沿用集群提供的 verbs 软件栈,不在容器内改动。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 - -- config-keys: - - dsv4-fp4-b300-dynamo-sglang-agentic-disagg - scenario-type: - - agentic-coding - description: - - "Switch the B300 DeepSeek-V4-Pro Dynamo+SGLang disaggregated c64, c240, and c480 recipes from Mooncake to NIXL with the LIBFABRIC backend for Amazon EFA, and remove the runtime Mooncake-EFA installer and setup hooks." - - "将 B300 DeepSeek-V4-Pro Dynamo+SGLang 分离式 c64、c240 和 c480 配方从 Mooncake 切换到使用 LIBFABRIC 后端的 NIXL(适配 Amazon EFA),并移除运行时 Mooncake-EFA 安装脚本和 setup hook。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3597 From 3763b4a9e50c3f7cf6edd76a34eb3ffdf5cb5927 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 21:00:40 -0700 Subject: [PATCH 12/14] fix(dsxe): defer B300 CPU allocation to runner --- .../dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml | 1 - .../dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml | 1 - .../agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 1 - .../agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml | 1 - .../agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 1 - inferencex-e2e/perf-changelog.yaml | 4 ++-- 6 files changed, 2 insertions(+), 7 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml index 4a0de7e8b3..a33048ebc5 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml @@ -103,7 +103,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml index a7383fcac7..335d1082a4 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml @@ -103,7 +103,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml index e5034d7d3b..6975e0bb48 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -199,7 +199,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml index 0dbefbbe1f..6dc77ff5e6 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml @@ -199,7 +199,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml index c4d36771cd..5ef595d31b 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -201,7 +201,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 91c0917378..f572c7f5f9 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9157,8 +9157,8 @@ description: - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4 recipes, plus disaggregated 1P1D DEP8 c64/c240 and 2P1D DEP8 c480 recipes, under the refactored inferencex-e2e project layout." - "Use NIXL with the LIBFABRIC backend for disaggregated KV transfer, relying on the networking stack provided by the pinned container and DSXE runtime without explicit host EFA/OFI mounts or runtime transport setup." - - "Use 192 logical CPUs per task, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE nodes." + - "Use the B300 DSXE runner's 24-CPUs-per-GPU policy, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE nodes." - "在重构后的 inferencex-e2e 项目布局中新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4,以及分离式 1P1D DEP8 c64/c240 和 2P1D DEP8 c480。" - "分离式 KV 传输使用 NIXL 的 LIBFABRIC 后端,依赖固定容器与 DSXE 运行时提供的网络栈,不显式挂载主机 EFA/OFI 目录,也不在运行时安装传输组件。" - - "每任务使用 192 个逻辑 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" + - "使用 B300 DSXE 运行器的每 GPU 24 个 CPU 策略、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3597 From a27ae88e8ce5249450a5dffd58afad603ec38b83 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Wed, 30 Sep 2026 08:34:27 -0700 Subject: [PATCH 13/14] Raise B300 DSV4 agg AgentX health-check budget to 8h The non-EP TP MoE aggregate recipes need a long weight-load and post-load phase before the server is healthy: the TP8 c1 canary took ~3h20m to become healthy on an otherwise idle node, and the TP4 c4 worker was still loading when the 4h (1440 x 10s) health budget expired. Raise max_attempts to 2880 (8h) for both agg recipes; this stays within the 780-minute agentic job timeout and does not change serving or benchmark settings. Co-Authored-By: Claude Opus 5.5 --- .../dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml | 2 +- .../dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml index a33048ebc5..1a8e0defe0 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml @@ -19,7 +19,7 @@ dynamo: install: false health_check: - max_attempts: 1440 + max_attempts: 2880 # 8h: non-EP TP MoE startup took ~3h20m alone (TP8) interval_seconds: 10 resources: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml index 335d1082a4..0e27b14a78 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml @@ -19,7 +19,7 @@ dynamo: install: false health_check: - max_attempts: 1440 + max_attempts: 2880 # 8h: non-EP TP MoE startup took ~3h20m alone (TP8) interval_seconds: 10 resources: From 8ddfe842cf8dbf5c7aa9a4beed916382fb243827 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Wed, 30 Sep 2026 09:06:24 -0700 Subject: [PATCH 14/14] style(perf): restore changelog entry separator Co-Authored-By: Claude Opus 5.5 --- inferencex-e2e/perf-changelog.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index f572c7f5f9..7e8503c591 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9149,6 +9149,7 @@ - "Use AITER attention and allreduce fusion, FP8 KV (fp8_e4m3), and EAGLE MTP (3 steps, topk 1, 4 draft tokens); do not enable ROCm INT4 quick all-reduce. HiCache uses ratio 1.5, write_through_selective, kernel I/O and page_first layout. The matching cookbook recipe is sgl-project/sglang#41849." - "The embedded MTP head runs at its stored precision: block-FP8 expert weights and checkpoint dtype for mtp.fc and gates. No separate draft, draft dtype override or submission-side quantization is used." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3602 + - config-keys: - dsv4-fp4-b300-dynamo-sglang-agentic-agg - dsv4-fp4-b300-dynamo-sglang-agentic-disagg