Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
#!/usr/bin/env bash
set -eo pipefail

python3 -m pip uninstall --break-system-packages -y mooncake-transfer-engine-cuda13 mooncake-transfer-engine-efa-cuda13
python3 -m pip install --break-system-packages --no-deps mooncake-transfer-engine-efa-cuda13==0.3.13.post1
Comment on lines +4 to +5

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

can we get this in upstream?

@Ankur-singh Ankur-singh Oct 2, 2026 •

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@functionstackx This is already covered by SGLang’s official AWS EFA instructions. They explicitly prescribe replacing the standard CUDA 13 Mooncake package with mooncake-transfer-engine-efa-cuda13==0.3.13.post1 using --no-deps.

This PR doesn't make any other changes to the container. I can submit a waiver explaining/documenting it, if it helps.

Original file line number Diff line number Diff line change
@@ -0,0 +1,163 @@
schema: 2
name: "agg-b300-dep8-c384-mtp-kvoffload"

# AgentX aggregate topology: one DEP8 worker (attention DP8 + EP8 Mega-MoE +
# FP4 indexer) occupies one eight-GPU B300 node and serves both prefill and
# decode with DSpark K=6 and a HiCache DRAM tier, tuned for concurrency 384.
# Engine settings follow the single-node SGLang DEP8 c384 point
# (benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-mtp/agentic.yaml,
# override_dep8_c384). The Dynamo KV router replaces the SGLang router: each
# DP rank publishes KV events, and sessions stay sticky via X-Dynamo-Session-ID.
# Concurrency is exported by the master config.

model:
path: "deepseek-v4-pro-0813"
container: "dynamo-sglang"
precision: "fp4"

identity:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro-0813"
container:
image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9"

dynamo:
install: false

health_check:
max_attempts: 1440
interval_seconds: 10

resources:
gpu_type: "b300"
gpus_per_node: 8
services:
- name: etcd
type: etcd
placement:
node: infra
- name: nats
type: nats
placement:
node: infra
options:
max_payload_mb: 32
frontend:
type: dynamo
nginx_session_affinity: true
nginx_session_affinity_header: X-Dynamo-Session-ID
enable_multiple_frontends: false
env:
PIP_BREAK_SYSTEM_PACKAGES: "1"
DYN_NATS_REQUEST_TIMEOUT_SECS: "1800"
# The 5 s default request-plane ack timeout expires while the worker
# ingests the c384 warmup burst; match the disaggregated recipes.
DYN_TCP_REQUEST_TIMEOUT: "60"
args:
router-mode: "kv"
router-session-affinity-ttl-secs: "3600"
active-decode-blocks-threshold: "None"
active-prefill-tokens-threshold: "None"
active-prefill-tokens-threshold-frac: "None"

engine: sglang
roles:
agg:
nodes: 1
workers: 1
gpus: 8

env:
SGLANG_RAGGED_VERIFY_MODE: "static"
SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1"
SGLANG_DEFAULT_THINKING: "1"
SGLANG_DSV4_REASONING_EFFORT: high
PIP_BREAK_SYSTEM_PACKAGES: "1"
PYTHONNOUSERSITE: "1"
TORCH_CUDA_ARCH_LIST: "10.0"
# Triton compiles with the image's CUDA ptxas.
TRITON_PTXAS_PATH: /usr/local/cuda/bin/ptxas
SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1"
SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1"
SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1"
SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1"
SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1"
SGLANG_OPT_USE_ONLINE_COMPRESS: "0"
SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1"
SGLANG_OPT_USE_JIT_NORM: "1"
SGLANG_OPT_USE_TOPK_V2: "True"
# Covers the 8192-token per-rank prefill budget (65536 / 8 DP ranks).
SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: "8320"

args:
served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813"
enable-metrics: true
enable-cache-report: true
trust-remote-code: true
weight-loader-prefetch-checkpoints: true
watchdog-timeout: 1000000
allow-auto-truncate: true
attention-backend: dsv4
page-size: 256
disable-shared-experts-fusion: true
disable-flashinfer-autotune: true
tp-size: 8
dp-size: 8
ep-size: 8
enable-dp-attention: true
enable-dp-lm-head: true
enable-dp-attention-local-control-broadcast: true
moe-a2a-backend: megamoe
enable-deepseek-v4-fp4-indexer: true
enable-prefill-delayer: true
prefill-decode-interval: 20
incremental-streaming-output: true
stream-interval: 20
# Mega-MoE's transient workspace sits outside the static pool.
mem-fraction-static: 0.88
swa-full-tokens-ratio: 0.075
chunked-prefill-size: 65536
max-running-requests: 768
# Per-rank graph limit: max-running-requests / dp-size = 768 / 8.
cuda-graph-max-bs-decode: 96
kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}'
speculative-algorithm: DSPARK
speculative-dspark-block-size: 6
speculative-num-steps: 1
speculative-eagle-topk: 1
speculative-num-draft-tokens: 7
# HiCache capacity is a host/device ratio; 3 keeps the tier near 2 TB.
enable-hierarchical-cache: true
hicache-ratio: 3
hicache-write-policy: write_back
hicache-io-backend: direct
hicache-mem-layout: page_first_direct

sbatch_directives:
mem: "0"
exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16"

srun_options:
mem: "0"
container-remap-root: ""

benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace"
RESULT_DIR: "/logs/agentic"
PORT: "8000"
IS_MULTINODE: "false"
TP: "8"
EP_SIZE: "8"
DP_ATTENTION: "true"
PP_SIZE: "1"
PCP_SIZE: "1"
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:"
# Warmup can leave bytes unacknowledged past AIPerf's 30 s default.
AIPERF_HTTP_TCP_USER_TIMEOUT: "900000"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
Original file line number Diff line number Diff line change
@@ -0,0 +1,128 @@
schema: 2
name: "agg-b300-tp4-c4-mtp"

# AgentX aggregate topology: one TP4 worker occupies half of one eight-GPU
# B300 node and serves both prefill and decode with DSpark K=6.

model:
path: "deepseek-v4-pro-0813"
container: "dynamo-sglang"
precision: "fp4"

identity:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro-0813"
container:
image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9"

dynamo:
install: false

health_check:
max_attempts: 1440
interval_seconds: 10

resources:
gpu_type: "b300"
gpus_per_node: 8
services:
- name: etcd
type: etcd
placement:
node: infra
- name: nats
type: nats
placement:
node: infra
options:
max_payload_mb: 32
frontend:
type: dynamo
nginx_session_affinity: true
nginx_session_affinity_header: X-Dynamo-Session-ID
enable_multiple_frontends: false
env:
PIP_BREAK_SYSTEM_PACKAGES: "1"
DYN_NATS_REQUEST_TIMEOUT_SECS: "1800"
args:
router-mode: "kv"
router-session-affinity-ttl-secs: "3600"
active-decode-blocks-threshold: "None"
active-prefill-tokens-threshold: "None"
active-prefill-tokens-threshold-frac: "None"

engine: sglang
roles:
agg:
nodes: 1
workers: 1
gpus: 4

env:
SGLANG_RAGGED_VERIFY_MODE: "static"
SGLANG_DEFAULT_THINKING: "1"
SGLANG_DSV4_REASONING_EFFORT: high
PIP_BREAK_SYSTEM_PACKAGES: "1"
SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1"
SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1"
SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1"
SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1"
SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1"
SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1"
SGLANG_OPT_USE_ONLINE_COMPRESS: "0"
SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1"
SGLANG_OPT_USE_JIT_NORM: "1"
SGLANG_OPT_USE_TOPK_V2: "True"

args:
served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813"
enable-metrics: true
enable-cache-report: true
trust-remote-code: true
weight-loader-prefetch-checkpoints: true
stream-interval: 10
watchdog-timeout: 1000000
mem-fraction-static: 0.90
page-size: 256
chunked-prefill-size: 8192
max-prefill-tokens: 8192
moe-runner-backend: "flashinfer_mxfp4"
enable-deepseek-v4-fp4-indexer: true
disable-flashinfer-autotune: true
swa-full-tokens-ratio: 0.1
max-running-requests: 8
cuda-graph-max-bs-decode: 8
scheduler-recv-interval: 30
dp-size: 1
tp-size: 4
ep-size: 1
speculative-algorithm: DSPARK
speculative-dspark-block-size: 6
speculative-num-steps: 1
speculative-eagle-topk: 1
speculative-num-draft-tokens: 7

sbatch_directives:
mem: "0"
exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16"

srun_options:
mem: "0"
container-remap-root: ""

benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace"
RESULT_DIR: "/logs/agentic"
PORT: "8000"
IS_MULTINODE: "false"
TP: "4"
PP_SIZE: "1"
PCP_SIZE: "1"
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
Loading
Loading