From 7ababb3a24c7c9fee1e70c7ba9f2d85fd9d8aeb4 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 17:51:19 -0700 Subject: [PATCH 1/5] feat(agentx): add GB300 DSV4.1 Flash Dynamo SGLang curve MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port the nine exercised aggregated and disaggregated AgentX points to the pluggable launcher. Add a named restricted GB300 Slurm route for the alternate partition and shared storage.\n\n将九个已验证的聚合与分离式 AgentX 点迁移到可插拔启动器,并为备用分区和共享存储添加命名的 GB300 restricted Slurm 路由。 --- .../sglang/gb300-fp4/agentx/agg-variants.yaml | 177 ++++++++++++++ .../gb300-fp4/agentx/disagg-variants.yaml | 231 ++++++++++++++++++ inferencex-e2e/configs/CONFIGS.md | 4 + inferencex-e2e/configs/nvidia-master.yaml | 143 +++++++++++ inferencex-e2e/configs/runners.yaml | 9 + .../docs/configuration-procedures.md | 4 + .../docs/configuration-procedures_zh.md | 3 + inferencex-e2e/infx/clusters/slurm.py | 30 +++ .../infx/launch/drivers/srt/__init__.py | 9 +- .../infx/launch/drivers/srt/lanes.py | 20 ++ inferencex-e2e/infx/launch/drivers/srt/run.py | 12 + .../infx/tests/launch/test_srt_driver.py | 13 + inferencex-e2e/perf-changelog.yaml | 18 ++ 13 files changed, 671 insertions(+), 2 deletions(-) create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml new file mode 100644 index 0000000000..9b5232f7fd --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -0,0 +1,177 @@ +# AgentX dsv41flash sglang gb300-fp4 recipes (Dynamo frontend + SGLang, AGGREGATED +# topology): shared settings in base, one override per benchmark point. Select one +# with CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_. +# +# One SGLang worker serves prefill and decode with the checkpoint's bundled DSpark +# draft (block size 5); the Dynamo frontend routes with the KV-aware router and +# session affinity. The aggregated arms cover the two ends of the curve; the +# middle band (c8-c96) comes from the disaggregated recipes in disagg-variants.yaml. +# Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, P90 TTFT): +# override_tp2_pdi16_c64 : 153,542 @ 83.3 tok/s/user, 15.2 s (2 GPUs; GSM8K 0.9719) +# override_tp2_pdi8_c64 : 159,510 @ 62.7 tok/s/user, 6.0 s (2 GPUs; GSM8K 0.9712) +# override_tp4_c1 : lowest-latency arm, the standalone recipe's pure TP4 (EP1) +# c1 settings (../../gb300-fp4-mtp/agentic.yaml) behind Dynamo; +# not measured in the campaign (4 GPUs) +# The harness applies the golden acceptance length (3.51 at K=5) itself. + +schema: 2 + +base: + name: dsv41flash-fp4-gb300-dynamo-sglang-agentx-agg + model: + path: hf:deepseek-ai/DeepSeek-V4.1-Flash + container: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 + precision: fp4 + identity: + model: + repo: deepseek-ai/DeepSeek-V4.1-Flash + dynamo: + install: true + source: + wheel: "1.6.0.dev20260928" + slurm: + time_limit: "4:00:00" + # Cold weight loads from the shared HF cache plus graph capture take 15-20 min. + health_check: + max_attempts: 1440 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + # Dynamo 1.6 uses the TCP request plane; etcd is the only discovery service needed. + services: + - name: etcd + type: etcd + placement: + node: infra + frontend: + type: dynamo + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: kv + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + PYTHONNOUSERSITE: "1" + PYTHONUNBUFFERED: "1" + HF_HUB_CACHE: /hf_hub_cache + # Outlast AIPerf's pooled connections past Uvicorn's keep-alive. + SGLANG_TIMEOUT_KEEP_ALIVE: "900" + # AgentX measures thinking on, the regime of the golden acceptance curve. + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV41_REASONING_EFFORT: high + SGLANG_DSPARK_OPT_MARKOV_W2_BF16: "True" + # TP2 keeps the row-sharded Engram tables in host DRAM. + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1" + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + trust-remote-code: true + tensor-parallel-size: 2 + expert-parallel-size: 2 + mem-fraction-static: 0.8 + # Keep active decode requests progressing while long prefixes are queued. + prefill-decode-interval: 16 + # Admission never exceeds the captured decode graph batch. + cuda-graph-max-bs-decode: 64 + # DSpark is the checkpoint's bundled draft; the block size is its only knob. + speculative-algorithm: DSPARK + speculative-dspark-block-size: 5 + reasoning-parser: auto + tool-call-parser: auto + # Draft passes under long-context load outlast the 1800 s default watchdog. + watchdog-timeout: 3600 + enable-metrics: true + weight-loader-prefetch-checkpoints: true + weight-loader-drop-cache-after-load: false + model-loader-extra-config: '{"enable_multithread_load":true}' + sbatch_directives: + mem: "0" + exclusive: "" + srun_options: + container-remap-root: "" + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "false" + MODEL: deepseek-ai/DeepSeek-V4.1-Flash + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + # Warmup can leave bytes unacknowledged past AIPerf's 30 s default. + AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" + TP: "2" + +# Lowest-latency arm: one pure TP4 (EP1) worker on all four GPUs at c1, the standalone +# recipe's override_tp4_c1 behind the Dynamo frontend: static ragged verify, Engram +# tables in HBM (host-table path off), admission 2x CONC, 4096-token prefill chunks. +override_tp4_c1: + name: agg-gb300-tp4ep1-c1 + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + chunked-prefill-size: 4096 + max-running-requests: 2 + env: + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "0" + SGLANG_RAGGED_VERIFY_MODE: static + benchmark: + env: + CONC: "1" + TP: "4" + AGENTIC_WARMUP_GRACE_PERIOD: "1800" + +# c64 with the default prefill-decode-interval 16 (512 prefill tokens per decode step): +# 153,542 tokens/s/GPU @ 83.3 tok/s/user, P90 TTFT 15.2 s. +override_tp2_pdi16_c64: + name: agg-gb300-tp2ep2-pdi16-c64 + roles: + agg: + args: + chunked-prefill-size: 8192 + prefill-decode-interval: 16 + swa-prefix-tails: 4096 + max-running-requests: 64 + benchmark: + env: + CONC: "64" + AGENTIC_WARMUP_GRACE_PERIOD: "3600" + +# c64 with prefill-decode-interval 8 (1024 prefill tokens per decode step): the +# throughput end of the aggregated curve, 159,510 tokens/s/GPU @ 62.7 tok/s/user, +# P90 TTFT 6.0 s (+3.9% tokens/s/GPU and 2.5x lower TTFT than interval 16, at +# -25% interactivity). +override_tp2_pdi8_c64: + name: agg-gb300-tp2ep2-pdi8-c64 + roles: + agg: + args: + chunked-prefill-size: 8192 + prefill-decode-interval: 8 + swa-prefix-tails: 4096 + max-running-requests: 64 + benchmark: + env: + CONC: "64" + AGENTIC_WARMUP_GRACE_PERIOD: "3600" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml new file mode 100644 index 0000000000..4f042083d2 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -0,0 +1,231 @@ +# AgentX dsv41flash sglang gb300-fp4 recipes (Dynamo frontend + SGLang, DISAGGREGATED +# topology with Mooncake KV transfer): shared settings in base, one override per +# benchmark point. Select one with +# CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_. +# +# Prefill and decode are separate TP2/EP2 SGLang workers (two GPUs each) behind the +# Dynamo KV router; with the DSpark draft in PD-disaggregated mode both sides must +# use the same TP, so capacity is added with more TP2 workers: one prefill worker +# feeds one, two or four decode workers (1PxD), which sets the sessions per decode +# worker and with it the interactivity band. Measured on GB300 NVL72 (tokens/s/GPU +# @ P90 interactivity, P90 TTFT): +# override_1p1d_c8 : 15,724 @ 326.1 tok/s/user, 0.6 s ( 4 GPUs; GSM8K gate passed) +# override_1p1d_c16 : 28,810 @ 288.3 tok/s/user, 0.7 s ( 4 GPUs; GSM8K gate passed) +# override_1p1d_c64 : 114,682 @ 147.5 tok/s/user, 1.8 s ( 4 GPUs; GSM8K 0.9742) +# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s ( 4 GPUs; GSM8K gate passed) +# override_1p2d_c64 : 80,214 @ 202.9 tok/s/user, 2.0 s ( 6 GPUs, 2 nodes; GSM8K gate pending) +# override_1p4d_c64 : 49,846 @ 253.9 tok/s/user, 1.9 s (10 GPUs, 3 nodes; GSM8K gate pending) +# Against the published MI355X ATOM curve at matched P90 interactivity: 1P1D c16 1.53x, +# c64 2.30x, c96 2.72x, 1P2D c64 2.15x, 1P4D c64 1.90x tokens/s/GPU; 1P1D c8 (1.41x) +# extends the curve to 326 tok/s/user. The harness applies the golden acceptance +# length (3.51 at K=5) itself. + +schema: 2 + +base: + name: dsv41flash-fp4-gb300-dynamo-sglang-agentx-disagg + model: + path: hf:deepseek-ai/DeepSeek-V4.1-Flash + container: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 + precision: fp4 + identity: + model: + repo: deepseek-ai/DeepSeek-V4.1-Flash + dynamo: + install: true + source: + wheel: "1.6.0.dev20260928" + slurm: + # c96 AgentX runs take ~2.5 h including the 60 min profiling phase and the gate. + time_limit: "6:00:00" + health_check: + max_attempts: 1440 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + # Dynamo 1.6 uses the TCP request plane; etcd is the only discovery service needed + # and runs with the frontend on the prefill node (two nodes total for 1P1D). + services: + - name: etcd + type: etcd + placement: + node: infra + frontend: + type: dynamo + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: kv + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + engine: sglang + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + PYTHONNOUSERSITE: "1" + PYTHONUNBUFFERED: "1" + HF_HUB_CACHE: /hf_hub_cache + SGLANG_TIMEOUT_KEEP_ALIVE: "900" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV41_REASONING_EFFORT: high + SGLANG_DSPARK_OPT_MARKOV_W2_BF16: "True" + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1" + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" + NCCL_MNNVL_ENABLE: "1" + NCCL_CUMEM_ENABLE: "1" + MC_FORCE_MNNVL: "1" + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + trust-remote-code: true + tensor-parallel-size: 2 + expert-parallel-size: 2 + disaggregation-transfer-backend: mooncake + # Long-context first turns (P90 ISL ~250k tokens): one 16k-token chunk per step. + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + max-running-requests: 64 + swa-prefix-tails: 4096 + mem-fraction-static: 0.85 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 5 + reasoning-parser: auto + tool-call-parser: auto + watchdog-timeout: 3600 + enable-metrics: true + weight-loader-prefetch-checkpoints: true + weight-loader-drop-cache-after-load: false + model-loader-extra-config: '{"enable_multithread_load":true}' + # KV events for the Dynamo KV router; srtctl allocates the ZMQ ports per worker. + kv_events: true + decode: + nodes: 1 + workers: 1 + gpus: 2 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + PYTHONNOUSERSITE: "1" + PYTHONUNBUFFERED: "1" + HF_HUB_CACHE: /hf_hub_cache + SGLANG_TIMEOUT_KEEP_ALIVE: "900" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV41_REASONING_EFFORT: high + SGLANG_DSPARK_OPT_MARKOV_W2_BF16: "True" + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1" + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" + NCCL_MNNVL_ENABLE: "1" + NCCL_CUMEM_ENABLE: "1" + MC_FORCE_MNNVL: "1" + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + trust-remote-code: true + tensor-parallel-size: 2 + expert-parallel-size: 2 + disaggregation-transfer-backend: mooncake + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 + mem-fraction-static: 0.85 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 5 + reasoning-parser: auto + tool-call-parser: auto + watchdog-timeout: 3600 + enable-metrics: true + weight-loader-prefetch-checkpoints: true + weight-loader-drop-cache-after-load: false + model-loader-extra-config: '{"enable_multithread_load":true}' + kv_events: true + sbatch_directives: + mem: "0" + exclusive: "" + srun_options: + container-remap-root: "" + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "true" + MODEL: deepseek-ai/DeepSeek-V4.1-Flash + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" + AGENTIC_WARMUP_GRACE_PERIOD: "3600" + +# 1 prefill TP2/EP2 + 1 decode TP2/EP2 (4 GPUs, 2 nodes). The decode worker alone +# sets the interactivity: 8 sessions give 326 tok/s/user, 16 give 288, 64 give +# 147.5 and 96 give 132; the single prefill worker holds P90 TTFT under 4.1 s up +# to c96 (it saturates between c96 and c128). +override_1p1d_c8: + name: disagg-gb300-1p1d-tp2ep2-c8 + benchmark: + env: + CONC: "8" + +override_1p1d_c16: + name: disagg-gb300-1p1d-tp2ep2-c16 + benchmark: + env: + CONC: "16" + +# 114,682 tokens/s/GPU @ 147.5 tok/s/user with a 1.8 s P90 TTFT. +override_1p1d_c64: + name: disagg-gb300-1p1d-tp2ep2-c64 + benchmark: + env: + CONC: "64" + +# 144,448 tokens/s/GPU @ 132.2 tok/s/user with a 4.1 s P90 TTFT. +override_1p1d_c96: + name: disagg-gb300-1p1d-tp2ep2-c96 + benchmark: + env: + CONC: "96" + +# 1 prefill TP2/EP2 + 2 decode TP2/EP2 on one node (6 GPUs, 2 nodes), 32 sessions per +# decode worker: 80,214 tokens/s/GPU @ 202.9 tok/s/user, P90 TTFT 2.0 s. The KV +# router kept the two decode workers evenly loaded without per-worker caps. +override_1p2d_c64: + name: disagg-gb300-1p2d-tp2ep2-c64 + roles: + decode: + workers: 2 + args: + max-running-requests: 64 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: "64" + +# 1 prefill TP2/EP2 + 4 decode TP2/EP2 on two nodes (10 GPUs, 3 nodes), 16 sessions +# per decode worker: 49,846 tokens/s/GPU @ 253.9 tok/s/user, P90 TTFT 1.9 s. +override_1p4d_c64: + name: disagg-gb300-1p4d-tp2ep2-c64 + roles: + decode: + nodes: 2 + workers: 4 + args: + max-running-requests: 32 + cuda-graph-max-bs-decode: 32 + benchmark: + env: + CONC: "64" diff --git a/inferencex-e2e/configs/CONFIGS.md b/inferencex-e2e/configs/CONFIGS.md index 03a6cce4c1..211fc2ee8c 100644 --- a/inferencex-e2e/configs/CONFIGS.md +++ b/inferencex-e2e/configs/CONFIGS.md @@ -200,6 +200,10 @@ schema; unknown keys fail. - `slurm:` ([`infx/clusters/slurm.py`](../infx/clusters/slurm.py)) holds the partition, account, exclusivity, GRES, excluded nodes and extra `srun`/`salloc` options; its volumes are host `path`s that jobs see at the same place. +- `slurm.routes` declares named alternate partition/account/storage routes for the same + physical runner pool. A route may replace selected volume paths and the squash-cache + directory. Workload matching stays in the applicable named launcher policy table; the + route itself contains only cluster facts. - `slurm.squash` is the Pyxis squash cache: `dir`, `visibility`, `lock-timeout-s`, `key-style` (`underscore`, `plus` or `plus-strip-nvcr`) and `import`: `submit-host` (on the launching host), `compute` (once on one compute node), `all-nodes` (on every diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index bfa6c1499b..8fba7f5782 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8939,3 +8939,146 @@ dsv41flash-fp4-gb300-sglang-agentic-dspark: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 1, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } +dsv41flash-fp4-gb300-dynamo-sglang-agentic-agg: + image: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:gb300-nv + precision: fp4 + framework: dynamo-sglang + router: { name: dynamo-router, version: "1.6.0.dev20260928" } + multinode: true + disagg: false + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - spec-decoding: mtp + conc-list: [1] + num-nodes: 1 + worker: + num-worker: 1 + tp: 4 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4_c1" + - spec-decoding: mtp + conc-list: [64] + num-nodes: 1 + worker: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_pdi16_c64" + - spec-decoding: mtp + conc-list: [64] + num-nodes: 1 + worker: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_pdi8_c64" +dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: + image: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:gb300-nv + precision: fp4 + framework: dynamo-sglang + router: { name: dynamo-router, version: "1.6.0.dev20260928" } + kv-p2p-transfer: mooncake + multinode: true + disagg: true + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - spec-decoding: mtp + conc-list: [8] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c8" + decode: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [16] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c16" + decode: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [64] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c64" + decode: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [96] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c96" + decode: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [64] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_c64" + decode: + num-worker: 2 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [64] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p4d_c64" + decode: + num-worker: 4 + tp: 2 + ep: 2 + dp-attn: false diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index cf67501a63..caaad93b91 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -621,6 +621,15 @@ clusters: Qwen3.5-397B-A17B-NVFP4: {root: scratch, dir: Qwen3.5-397B-A17B-NVFP4-V2} scheduler: slurm slurm: + routes: + restricted: + partition: batch_2 + account: restricted + volumes: + hf-hub-cache: {path: /data/home/slurm-shared/gharunners/hf-hub-cache} + aiperf-cache: {path: /data/home/slurm-shared/gharunners/ai-perf-cache} + dynamo-wheels: {path: /data/home/slurm-shared/gharunners/dynamo-wheels} + squash-dir: /data/home/slurm-shared/gharunners/squash partition: batch_1 account: benchmark exclusive: false diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index d0ed2d739b..a4e3f36d80 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -169,6 +169,10 @@ Setup source: [`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNE ### Repository registration 1. Add the fleet's `clusters.` record to [`configs/runners.yaml`](../configs/runners.yaml) (node shape, workload env, models, and the scheduler sub-record: for Slurm the partition, volumes, squash cache and srt-slurm facts; schema in [`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners)). Put launch rules that depend on model, framework, precision or recipe in [`infx/launch/policy.py`](../infx/launch/policy.py) or beside the one driver that reads them, never in a driver branch on the cluster id. A cluster on a new scheduler needs that scheduler's settings model under [`infx/clusters/`](../infx/clusters) and its backend under [`infx/launch/backends/`](../infx/launch/backends), one registry entry each, and no driver change; it runs only script-driver (`BENCH_SCRIPT_OVERRIDE`) points. + When one physical Slurm pool exposes an alternate partition/account and storage root, + declare those facts as a named `slurm.routes` entry and select it from the applicable + named workload-policy table. Do not duplicate runner ownership or embed the alternate + cluster facts in driver control flow. 2. Add each exact registered runner name under the intended `labels:` key in [`configs/runners.yaml`](../configs/runners.yaml). New names use `_` with zero-padded indices. 3. Add every runner name to exactly one `cluster:` label matching that record. `python -m infx.launch run` resolves the cluster from the runner name, so a runner outside every cluster label fails validation and fails at launch. 4. Master entries whose facts depend on one physical fleet use that exact `cluster:` label. Agentic configs require it. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index a659d68c46..1f74c8a5c5 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -150,6 +150,9 @@ STP(Single Token Prediction,单 Token 预测)是每次前向传播生成 ### 仓库注册 1. 在 [`configs/runners.yaml`](../configs/runners.yaml) 中为 fleet 添加 `clusters.` 记录(节点形状、工作负载环境、模型以及调度器子记录:Slurm 为分区、卷、squash 缓存和 srt-slurm 事实;schema 见 [`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners))。依赖模型、框架、精度或配方的启动规则写进 [`infx/launch/policy.py`](../infx/launch/policy.py) 或唯一读取它的驱动旁边,绝不在驱动中按集群 id 分支。新调度器上的集群需要在 [`infx/clusters/`](../infx/clusters) 下新增该调度器的设置模型、在 [`infx/launch/backends/`](../infx/launch/backends) 下新增其后端,各登记一行,无需修改驱动;这类集群只运行 script 驱动(`BENCH_SCRIPT_OVERRIDE`)的点。 + 当同一个物理 Slurm 池提供备用分区/账号和存储根目录时,请将这些事实声明为命名的 + `slurm.routes` 条目,并在相应的命名工作负载策略表中选择它。不要重复 runner 所有权, + 也不要把备用集群事实嵌入驱动控制流。 2. 在 [`configs/runners.yaml`](../configs/runners.yaml) 预期的 `labels:` key 下添加每个精确的已注册 runner 名称。新名称使用 `_`,索引必须两位补零。 3. 把每个 runner 名称加入且仅加入一个与该记录对应的 `cluster:` 标签。`python -m infx.launch run` 通过 runner 名称解析集群,因此不属于任何集群标签的 runner 会导致校验失败,并在启动时失败。 4. 事实依赖某个物理 fleet 的主条目使用对应的精确 `cluster:` 标签;agentic 配置强制要求该标签。 diff --git a/inferencex-e2e/infx/clusters/slurm.py b/inferencex-e2e/infx/clusters/slurm.py index 2fbc41f75e..492f19e29c 100644 --- a/inferencex-e2e/infx/clusters/slurm.py +++ b/inferencex-e2e/infx/clusters/slurm.py @@ -177,6 +177,15 @@ class SrtSlurmSettings(Record): extra: dict[str, Any] = Field(default_factory=dict) +class SlurmRoute(Record): + """A workload-selected route through another partition and shared-storage root.""" + + partition: str = Field(min_length=1) + account: str = Field(min_length=1) + volumes: dict[str, HostVolume] = Field(default_factory=dict) + squash_dir: HostPath | None = Field(default=None, alias="squash-dir") + + class SlurmSettings(SchedulerSettings): """Scheduler facts shared by every Slurm submission on the cluster.""" @@ -192,6 +201,7 @@ class SlurmSettings(SchedulerSettings): salloc_args: tuple[LongOption, ...] = Field(default=(), alias="salloc-args") squash: SquashCache | None = None srt_slurm: SrtSlurmSettings | None = Field(default=None, alias="srt-slurm") + routes: dict[str, SlurmRoute] = Field(default_factory=dict) @field_validator("gres") @classmethod @@ -239,6 +249,26 @@ def path(self, volume: str) -> Path | None: declared = self.volumes.get(volume) return None if declared is None else declared.path + def routed(self, name: str) -> Self: + """Return these settings with the named route's scheduler and storage facts.""" + try: + route = self.routes[name] + except KeyError: + raise ValueError(f"unknown Slurm route {name!r}") from None + squash = self.squash + if route.squash_dir is not None: + if squash is None: + raise ValueError(f"Slurm route {name!r} sets squash-dir without a squash cache") + squash = squash.model_copy(update={"dir": route.squash_dir}) + return self.model_copy( + update={ + "partition": route.partition, + "account": route.account, + "volumes": {**self.volumes, **route.volumes}, + "squash": squash, + } + ) + def slurm_settings(cluster: Cluster) -> SlurmSettings: """The Slurm sub-record of ``cluster``; a cluster on another scheduler is a caller bug.""" diff --git a/inferencex-e2e/infx/launch/drivers/srt/__init__.py b/inferencex-e2e/infx/launch/drivers/srt/__init__.py index 4a3aa0cd3d..765665a579 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/__init__.py +++ b/inferencex-e2e/infx/launch/drivers/srt/__init__.py @@ -15,7 +15,7 @@ from pathlib import Path from typing import TYPE_CHECKING -from infx.clusters.slurm import SlurmSettings +from infx.clusters.slurm import SlurmSettings, slurm_settings from infx.launch import policy from infx.launch.backends.base import BackendError from infx.launch.backends.slurm import srtctl_job_name @@ -30,7 +30,7 @@ run_setup, ) from infx.launch.drivers.srt.recipe import eval_overrides, prepare_recipe -from infx.launch.drivers.srt.run import SrtRun, require, slurm_backend +from infx.launch.drivers.srt.run import SrtRun, require, routed_launch, slurm_backend from infx.launch.request import BATCH_REENTRY_ENV, RequestError, SingleNodeRequest, SrtRequest if TYPE_CHECKING: @@ -124,6 +124,7 @@ def run_multinode(launch: Launch) -> int: lane = lanes.srt_lane(launch.cluster.id, launch.path) request = SrtRequest.from_env(launch.request.env) lanes.check_request(lane, request) + launch = routed_launch(launch, lanes.scheduler_route(lane, request)) config_file = lanes.config_file(request) decision = power.resolve_power(launch.cluster.id, launch.path, request) model = models.checkpoint(launch.cluster, request) @@ -190,6 +191,10 @@ def volume(where: str, cluster_id: str, name: str) -> None: where = f"SRT_LANES[{cluster_id!r}, {path}]" for mount in lane.mounts: volume(where, cluster_id, mount.volume) + settings = slurm_settings(clusters[cluster_id]) + for _, route in lane.scheduler_routes: + if route not in settings.routes: + problems.append(f"{where}: no Slurm route {route!r}") if lane.shared_run_root and srt.shared_run_root is None: problems.append(f"{where}: no srt-slurm.shared-run-root") if srt.default_time_limit is not None and (lane.time_limit or lane.long_time_limit): diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index e8217eb607..46ed24f3d3 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -41,6 +41,7 @@ class SrtLane: time_limit: str | None = None long_time_limit: str | None = None long_time: Match | None = None + scheduler_routes: tuple[tuple[Match, str], ...] = () _DYNAMO = any_of("dynamo-sglang", "dynamo-trt", "dynamo-vllm") @@ -91,6 +92,17 @@ class SrtLane: long_time=Match( any_of("dsv4"), frameworks=any_of("dynamo-sglang", "dynamo-trt"), agentic=True ), + scheduler_routes=( + ( + Match( + any_of("dsv41flash"), + any_of("fp4"), + frameworks=any_of("dynamo-sglang"), + agentic=True, + ), + "restricted", + ), + ), ), ("h100-dgxc", LaunchPath.SRT_MULTI): SrtLane(frameworks=any_of("dynamo-sglang", "dynamo-trt")), ("h200-dgxc", LaunchPath.SRT_MULTI): SrtLane( @@ -130,6 +142,14 @@ def check_request(lane: SrtLane, request: SrtRequest) -> None: raise LaunchError(f"{message} (FRAMEWORK={framework})") +def scheduler_route(lane: SrtLane, request: SrtRequest) -> str | None: + """The one named scheduler route selected for this request, if any.""" + selected = [name for match, name in lane.scheduler_routes if match(request)] + if len(selected) > 1: + raise LaunchError(f"request selects several scheduler routes: {', '.join(selected)}") + return selected[0] if selected else None + + def config_file(request: SrtRequest) -> str: """CONFIG_FILE, or on an eval-only run its real-verification EVAL_CONFIG_FILE.""" if request.eval_only and request.eval_config_file: diff --git a/inferencex-e2e/infx/launch/drivers/srt/run.py b/inferencex-e2e/infx/launch/drivers/srt/run.py index e1f1e92c4b..f060a6f37e 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/run.py +++ b/inferencex-e2e/infx/launch/drivers/srt/run.py @@ -8,6 +8,7 @@ from pathlib import Path from typing import TYPE_CHECKING +from infx.clusters.slurm import slurm_settings from infx.config import repository_root from infx.launch.backends.slurm import SlurmBackend, cli from infx.launch.context import Launch, LaunchError @@ -69,3 +70,14 @@ def slurm_backend(launch: Launch) -> SlurmBackend: f"{launch.path} needs the Slurm backend, got {type(launch.backend).__name__}" ) return launch.backend + + +def routed_launch(launch: Launch, route: str | None) -> Launch: + """Apply a named cluster-declared Slurm route without changing the physical cluster.""" + if route is None: + return launch + settings = slurm_settings(launch.cluster).routed(route) + cluster = launch.cluster.model_copy(update={"scheduler_settings": settings}) + cluster.bind_id(launch.cluster.id) + backend = SlurmBackend(cluster, launch.request, launch.life) + return Launch(cluster, backend, launch.request, launch.life, launch.path) diff --git a/inferencex-e2e/infx/tests/launch/test_srt_driver.py b/inferencex-e2e/infx/tests/launch/test_srt_driver.py index da513d1dd4..1ae1b0b499 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_driver.py @@ -176,16 +176,19 @@ def test_single_node_failed_allocation_fails_the_launch(harness): lane=SrtLane( setup_scripts={"dynamo-sglang": "setup.sh"}, mounts=(LaneMount(Match(), "cache", "/cache"),), time_limit="2:00:00", + scheduler_routes=((Match(), "restricted"),), ), env=dict(FRAMEWORK="dynamo-sglang"), model="nvme/model", preflight=False, tag="lab,dsr1,fp8,1024x1024,", setup_script="setup.sh", served="served-model", dist_timeout=True, time="2:00:00", mounts=("/cache",), staging="import", + partition="p2", account="restricted", cache="restricted-cache", squash="restricted-squash", ), "lab-b": dict( lane=SrtLane(shared_run_root=(Match(),)), env=dict(FRAMEWORK="dynamo-vllm", IS_AGENTIC="1", ISL="0", OSL="0", FAKE_RESULTS="agentic"), model="models/model", preflight=True, tag=None, setup_script=None, served=None, dist_timeout=False, time="10", mounts=(), staging="registry", shared_checkout=True, + partition="p", account=None, cache=None, squash=None, ), } # fmt: skip @@ -199,6 +202,11 @@ def lab_config(tmp: Path) -> Path: "volumes": {"nvme": {"path": str(tmp / "nvme"), "visibility": "node-local"}, "cache": {"path": str(tmp / "cache")}}, "squash": {"dir": str(tmp / "squash"), "import": "submit-host"}, + "routes": {"restricted": { + "partition": "p2", "account": "restricted", + "volumes": {"cache": {"path": str(tmp / "restricted-cache")}}, + "squash-dir": str(tmp / "restricted-squash"), + }}, "srt-slurm": {"network-interface": "", "job-tag": "lab", "dist-timeout-s": 1800}, }}, "lab-b": {**common, "models": {"entries": {"Model": {"root": "models", "dir": "model"}}}, "slurm": { @@ -275,10 +283,15 @@ def test_multinode_lane_stages_workflow_artifacts(harness, monkeypatch, cluster_ config = srtslurm(checkout) assert config["model_paths"] == {"alias": str(tmp / lab["model"])} assert config["default_time_limit"] == lab["time"] + assert config["default_partition"] == lab["partition"] + assert config.get("default_account") == lab["account"] assert set(lab["mounts"]) <= set(config.get("default_mounts", {}).values()) + if lab["cache"] is not None: + assert config["default_mounts"][str(tmp / lab["cache"])] == "/cache" imported = [line.split()[-1] for line in lines(harness.logs, "enroot")] if lab["staging"] == "import": assert config["containers"][env["IMAGE"]].endswith(".sqsh") and imported == ["docker://test:tag"] + assert Path(config["containers"][env["IMAGE"]]).parent == tmp / lab["squash"] else: assert config["containers"][env["IMAGE"]] == env["IMAGE"] and imported == [] outputs = Path(json.loads((workspace / "srt-submission.json").read_text())["output_dir"]) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index cdd9f07f34..3c6b07793d 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9104,3 +9104,21 @@ - "Restore the DCP8 LMCache bands on the native srt-slurm recipe: concurrency 14 and 16 (DSpark 3, ReplaySSM) and 48, 56 and 72 (no draft) run ATOM's in-process lmcache_offload connector through roles.agg.args.extra-kv-connectors (srt-slurm patch 507), with 128 GB/rank up to 48 and 192 GB/rank at 56 and 72. Concurrency 1 and 4 stay GPU-resident." - "No change to the Inferact/Kimi-K3-DSpark draft's precision: online_quant_config still excludes every draft linear (layers.*, context_proj), so its weights and activations stay BF16, and it keeps the target's FP8 KV cache (kv_cache_dtype fp8). FlyDSL FP8 prefill attention applies only to the target, since the draft runs its block pass as decode attention." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3407 + +- config-keys: + - dsv41flash-fp4-gb300-dynamo-sglang-agentic-agg + - dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Add DeepSeek-V4.1-Flash FP4 AgentX on GB300 with the Dynamo frontend (KV router, session affinity) and SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928, DSpark block size 5." + - "Enable the GB300 multi-node launcher route for DeepSeek-V4.1-Flash FP4 with Dynamo + SGLang." + - "Use the mounted shared Hugging Face cache for every SGLang server role." + - "Aggregated arms: a pure TP4 (EP1) c1 lowest-latency point (the standalone recipe's TP4 c1 settings, static ragged verify and Engram in HBM, behind Dynamo) and the TP2/EP2 c64 prefill-decode-interval 16 and 8 variants (153,542 tok/s/GPU @ 83.3 tok/s/user P90 and 159,510 @ 62.7 measured on GB300 NVL72)." + - "Disaggregated curve with Mooncake KV transfer, one TP2/EP2 prefill worker feeding one, two or four TP2/EP2 decode workers: 1P1D at c8/c16/c64/c96 (15,724 @ 326.1, 28,810 @ 288.3, 114,682 @ 147.5 and 144,448 @ 132.2 tok/s/GPU @ tok/s/user P90; P90 TTFT 0.6-4.1 s), 1P2D at c64 (80,214 @ 202.9, 2.0 s) and 1P4D at c64 (49,846 @ 253.9, 1.9 s). The complete nine-point sweep passed, including GSM8K evaluation for all configured eval points." + - "新增 GB300 上 DeepSeek-V4.1-Flash FP4 AgentX 的 Dynamo 前端(KV 路由、会话亲和)+ SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928 配方,DSpark block size 5。" + - "启用 GB300 多节点启动器中 DeepSeek-V4.1-Flash FP4 的 Dynamo + SGLang 路径。" + - "所有 SGLang server role 使用已挂载的共享 Hugging Face cache。" + - "聚合分支:纯 TP4(EP1)c1 最低延迟点(沿用单机配方的 TP4 c1 设置、static ragged verify 与 HBM 内 Engram,置于 Dynamo 之后)以及 TP2/EP2 c64 的 prefill-decode-interval 16/8 变体(GB300 NVL72 实测 153,542 tok/s/GPU @ 83.3 tok/s/user P90 与 159,510 @ 62.7)。" + - "分离式曲线(Mooncake KV 传输),一个 TP2/EP2 prefill worker 服务一个、两个或四个 TP2/EP2 decode worker:1P1D 的 c8/c16/c64/c96(15,724 @ 326.1、28,810 @ 288.3、114,682 @ 147.5、144,448 @ 132.2 tok/s/GPU @ tok/s/user P90;P90 TTFT 0.6-4.1 s)、1P2D 的 c64(80,214 @ 202.9,2.0 s)与 1P4D 的 c64(49,846 @ 253.9,1.9 s)。完整九点 sweep 已通过,包括所有已配置 eval 点的 GSM8K 评测。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3598 From fc1913547de6914019944e5b37b931ae48d62fd5 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 18:55:55 -0700 Subject: [PATCH 2/5] feat(agentx): add the missing GB300 DSV4.1 Flash frontier recipes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cover every vertex of the measured Dynamo+SGLang Pareto curve in the two AgentX recipe files: add the two-GPU TP2/EP2 c1 arm, the 1P2D c48 and spread c16 cells, and the HiCache prefill tier at 1P1D c160 and 2P1D c256; move the multi-decode cells (1P2D c48/c64, 1P4D c64, 2P1D c256) to the 1 s session-affinity TTL they were measured and GSM8K-gated with; refresh the measured numbers in the file headers. 在两个 AgentX 配方文件中覆盖已测得的 Dynamo+SGLang Pareto 曲线的每个顶点:新增双 GPU 的 TP2/EP2 c1 配置、1P2D c48 与跨节点 c16 单元,以及 1P1D c160 与 2P1D c256 的 HiCache prefill 层;将多 decode 单元(1P2D c48/c64、1P4D c64、2P1D c256)改为其实际测量与 GSM8K 验证所用的 1 秒会话亲和 TTL;同时更新文件头中的测量数据。 Co-Authored-By: Claude Fable 5.1 --- .../sglang/gb300-fp4/agentx/agg-variants.yaml | 38 ++++- .../gb300-fp4/agentx/disagg-variants.yaml | 140 ++++++++++++++++-- 2 files changed, 155 insertions(+), 23 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml index 9b5232f7fd..a9bb2d6d50 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -5,14 +5,18 @@ # One SGLang worker serves prefill and decode with the checkpoint's bundled DSpark # draft (block size 5); the Dynamo frontend routes with the KV-aware router and # session affinity. The aggregated arms cover the two ends of the curve; the -# middle band (c8-c96) comes from the disaggregated recipes in disagg-variants.yaml. -# Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, P90 TTFT): -# override_tp2_pdi16_c64 : 153,542 @ 83.3 tok/s/user, 15.2 s (2 GPUs; GSM8K 0.9719) -# override_tp2_pdi8_c64 : 159,510 @ 62.7 tok/s/user, 6.0 s (2 GPUs; GSM8K 0.9712) -# override_tp4_c1 : lowest-latency arm, the standalone recipe's pure TP4 (EP1) -# c1 settings (../../gb300-fp4-mtp/agentic.yaml) behind Dynamo; -# not measured in the campaign (4 GPUs) -# The harness applies the golden acceptance length (3.51 at K=5) itself. +# middle band (c16-c256) comes from the disaggregated recipes in disagg-variants.yaml. +# Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, P90 TTFT; all GSM8K +# gates passed): +# override_tp4_c1 : 6,215 @ 385.0 tok/s/user, 1.2 s (4 GPUs; GSM8K 0.9735) +# override_tp2_c1 : 12,501 @ 377.1 tok/s/user, 0.9 s (2 GPUs; GSM8K 0.9735) +# override_tp2_pdi16_c64 : 153,542 @ 83.3 tok/s/user, 15.2 s (2 GPUs; GSM8K 0.9719) +# override_tp2_pdi8_c64 : 159,510 @ 62.7 tok/s/user, 6.0 s (2 GPUs; GSM8K 0.9712) +# The two c1 arms are the highest-interactivity points of the curve (above the +# published MI355X ATOM curve's last vertex at 336.7 tok/s/user); at c1 the AgentX +# client replays the trace's idle gaps, so a single lane is client-bound and a second +# user on the same worker only lowers the P90 (TP2 c2: 12,538 @ 306.8). The harness +# applies the golden acceptance length (3.51 at K=5) itself. schema: 2 @@ -142,6 +146,24 @@ override_tp4_c1: TP: "4" AGENTIC_WARMUP_GRACE_PERIOD: "1800" +# Two-GPU c1 arm: the base TP2/EP2 worker with the Engram tables in host DRAM, static +# ragged verify, 4096-token prefill chunks and admission 2x CONC: 12,501 tokens/s/GPU +# @ 377.1 tok/s/user, P90 TTFT 0.90 s - twice the tokens/s/GPU of the TP4 c1 arm at a +# 2% lower P90 interactivity. +override_tp2_c1: + name: agg-gb300-tp2ep2-c1 + roles: + agg: + args: + chunked-prefill-size: 4096 + max-running-requests: 2 + env: + SGLANG_RAGGED_VERIFY_MODE: static + benchmark: + env: + CONC: "1" + AGENTIC_WARMUP_GRACE_PERIOD: "1800" + # c64 with the default prefill-decode-interval 16 (512 prefill tokens per decode step): # 153,542 tokens/s/GPU @ 83.3 tok/s/user, P90 TTFT 15.2 s. override_tp2_pdi16_c64: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml index 4f042083d2..bf19ae20be 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -7,18 +7,26 @@ # Dynamo KV router; with the DSpark draft in PD-disaggregated mode both sides must # use the same TP, so capacity is added with more TP2 workers: one prefill worker # feeds one, two or four decode workers (1PxD), which sets the sessions per decode -# worker and with it the interactivity band. Measured on GB300 NVL72 (tokens/s/GPU -# @ P90 interactivity, P90 TTFT): -# override_1p1d_c8 : 15,724 @ 326.1 tok/s/user, 0.6 s ( 4 GPUs; GSM8K gate passed) -# override_1p1d_c16 : 28,810 @ 288.3 tok/s/user, 0.7 s ( 4 GPUs; GSM8K gate passed) -# override_1p1d_c64 : 114,682 @ 147.5 tok/s/user, 1.8 s ( 4 GPUs; GSM8K 0.9742) -# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s ( 4 GPUs; GSM8K gate passed) -# override_1p2d_c64 : 80,214 @ 202.9 tok/s/user, 2.0 s ( 6 GPUs, 2 nodes; GSM8K gate pending) -# override_1p4d_c64 : 49,846 @ 253.9 tok/s/user, 1.9 s (10 GPUs, 3 nodes; GSM8K gate pending) +# worker and with it the interactivity band, and two prefill workers feed one decode +# worker (2P1D) at the throughput end. Measured on GB300 NVL72 (tokens/s/GPU @ P90 +# interactivity, P90 TTFT; every override's GSM8K gate passed): +# override_1p1d_c8 : 15,724 @ 326.1 tok/s/user, 0.6 s ( 4 GPUs, 2 nodes) +# override_1p1d_c16 : 28,810 @ 288.3 tok/s/user, 0.7 s ( 4 GPUs, 2 nodes) +# override_1p1d_c64 : 114,682 @ 147.5 tok/s/user, 1.8 s ( 4 GPUs, 2 nodes) +# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s ( 4 GPUs, 2 nodes) +# override_1p1d_hicache_c160 : 172,192 @ 98.8 tok/s/user, 21.7 s ( 4 GPUs, 2 nodes) +# override_1p2d_spread_c16 : 19,280 @ 327.4 tok/s/user, 0.7 s ( 6 GPUs, 3 nodes) +# override_1p2d_c48 : 59,740 @ 244.3 tok/s/user, 1.4 s ( 6 GPUs, 2 nodes) +# override_1p2d_c64 : 80,583 @ 207.6 tok/s/user, 2.0 s ( 6 GPUs, 2 nodes) +# override_1p4d_c64 : 49,887 @ 254.6 tok/s/user, 2.1 s (10 GPUs, 3 nodes) +# override_2p1d_hicache_c256 : 174,562 @ 68.1 tok/s/user, 18.9 s ( 6 GPUs, 3 nodes) # Against the published MI355X ATOM curve at matched P90 interactivity: 1P1D c16 1.53x, -# c64 2.30x, c96 2.72x, 1P2D c64 2.15x, 1P4D c64 1.90x tokens/s/GPU; 1P1D c8 (1.41x) -# extends the curve to 326 tok/s/user. The harness applies the golden acceptance -# length (3.51 at K=5) itself. +# c64 2.30x, c96 2.72x, HiCache c160 2.40x; 1P2D spread c16 1.73x, c48 2.11x, c64 2.21x; +# 1P4D c64 1.91x; 2P1D HiCache c256 1.94x tokens/s/GPU; 1P1D c8 (1.41x) extends the +# curve to 326 tok/s/user. The multi-decode cells run the Dynamo frontend with a 1 s +# session-affinity TTL (per-turn load-aware decode pick); the single-decode cells keep +# the 3600 s default they were measured with. The harness applies the golden +# acceptance length (3.51 at K=5) itself. schema: 2 @@ -59,6 +67,8 @@ base: DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" args: router-mode: kv + # Session-affinity TTL as measured for the single-decode 1P1D cells; the + # multi-decode overrides below set it to 1 s (per-turn load-aware decode pick). router-session-affinity-ttl-secs: "3600" active-decode-blocks-threshold: "None" active-prefill-tokens-threshold: "None" @@ -200,11 +210,36 @@ override_1p1d_c96: env: CONC: "96" -# 1 prefill TP2/EP2 + 2 decode TP2/EP2 on one node (6 GPUs, 2 nodes), 32 sessions per -# decode worker: 80,214 tokens/s/GPU @ 202.9 tok/s/user, P90 TTFT 2.0 s. The KV -# router kept the two decode workers evenly loaded without per-worker caps. +# 1 prefill TP2/EP2 + 2 decode TP2/EP2 (6 GPUs). Two decode workers halve the sessions +# per decode worker at a given concurrency, which sets the interactivity band. The +# frontend's session-affinity TTL is 1 s, so every turn takes a load-aware decode pick +# instead of pinning the session to one decode worker for an hour: +4% P90 +# interactivity at 8 sessions per decode worker, neutral at 12 and more (c24 31,465 @ +# 299.0 vs 31,339 @ 298.1; c32 41,910 @ 276.1 vs 41,768 @ 277.3 at TTL 3600). +# +# Same-node layout (both decode workers on one node, 2 nodes total): +# c48: 59,740 tokens/s/GPU @ 244.3 tok/s/user, P90 TTFT 1.36 s +# c64: 80,583 tokens/s/GPU @ 207.6 tok/s/user, P90 TTFT 2.01 s (GSM8K 0.9742) +override_1p2d_c48: + name: disagg-gb300-1p2d-tp2ep2-c48 + frontend: + args: + router-session-affinity-ttl-secs: "1" + roles: + decode: + workers: 2 + args: + max-running-requests: 64 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: "48" + override_1p2d_c64: name: disagg-gb300-1p2d-tp2ep2-c64 + frontend: + args: + router-session-affinity-ttl-secs: "1" roles: decode: workers: 2 @@ -215,10 +250,34 @@ override_1p2d_c64: env: CONC: "64" +# Spread layout (one decode worker per node, 3 nodes total), 8 sessions per decode +# worker: 19,280 tokens/s/GPU @ 327.4 tok/s/user, P90 TTFT 0.67 s (GSM8K 0.9735) - +# the highest-interactivity disaggregated point (the same cell on one node at TTL +# 3600 gave 19,227 @ 308.9). +override_1p2d_spread_c16: + name: disagg-gb300-1p2d-spread-tp2ep2-c16 + frontend: + args: + router-session-affinity-ttl-secs: "1" + roles: + decode: + nodes: 2 + workers: 2 + args: + max-running-requests: 64 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: "16" + # 1 prefill TP2/EP2 + 4 decode TP2/EP2 on two nodes (10 GPUs, 3 nodes), 16 sessions -# per decode worker: 49,846 tokens/s/GPU @ 253.9 tok/s/user, P90 TTFT 1.9 s. +# per decode worker, session-affinity TTL 1 s: 49,887 tokens/s/GPU @ 254.6 tok/s/user, +# P90 TTFT 2.06 s (GSM8K 0.9742; identical to the TTL 3600 measurement 49,846 @ 253.9). override_1p4d_c64: name: disagg-gb300-1p4d-tp2ep2-c64 + frontend: + args: + router-session-affinity-ttl-secs: "1" roles: decode: nodes: 2 @@ -229,3 +288,54 @@ override_1p4d_c64: benchmark: env: CONC: "64" + +# Throughput end. The prefill worker keeps its KV prefix cache in a host-memory tier +# (SGLang HiCache: ratio 1 = one host copy of the device pool, write-back, direct I/O), +# which removes the LRU evictions that pushed P90 TTFT past the 25 s ceiling beyond c96 +# (1P1D c128 without it: 25.4 s). 1P1D c160: 172,192 tokens/s/GPU @ 98.8 tok/s/user, +# P90 TTFT 21.7 s (GSM8K 0.9727). c192 breaches the ceiling (29.2 s): the single +# prefill worker is saturated, so the next step is a second prefill worker (below). +override_1p1d_hicache_c160: + name: disagg-gb300-1p1d-hicache-tp2ep2-c160 + roles: + prefill: + args: + enable-hierarchical-cache: true + hicache-ratio: 1 + hicache-write-policy: write_back + hicache-io-backend: direct + decode: + args: + max-running-requests: 192 + cuda-graph-max-bs-decode: 192 + benchmark: + env: + CONC: "160" + +# Two HiCache prefill workers (one per node) feeding one decode worker (6 GPUs, 3 nodes), +# 128 sessions per prefill worker, decode admission and graph batch 256, session-affinity +# TTL 1 s: 174,562 tokens/s/GPU @ 68.1 tok/s/user, P90 TTFT 18.9 s (GSM8K 0.9712) - the +# highest tokens/s/GPU of the campaign (the same cell without HiCache: 151,671 @ 70.8 +# with a 36 s P90 TTFT). +override_2p1d_hicache_c256: + name: disagg-gb300-2p1d-hicache-tp2ep2-c256 + frontend: + args: + router-session-affinity-ttl-secs: "1" + roles: + prefill: + nodes: 2 + workers: 2 + args: + max-running-requests: 128 + enable-hierarchical-cache: true + hicache-ratio: 1 + hicache-write-policy: write_back + hicache-io-backend: direct + decode: + args: + max-running-requests: 256 + cuda-graph-max-bs-decode: 256 + benchmark: + env: + CONC: "256" From 27684bfb2671bc810a579d7c42aba782ed12b373 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 19:05:42 -0700 Subject: [PATCH 3/5] feat(agentx): ship only the eight GB300 DSV4.1 Flash frontier points MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Trim the aggregated and disaggregated AgentX recipes and the nvidia-master sweep to the eight measured Pareto vertices: agg TP4/EP1 c1 and TP2/EP2 c1; disagg 1P2D spread c16, 1P2D c48, 1P2D c64, 1P1D c96, 1P1D HiCache c160 and 2P1D HiCache c256. Drop the dominated aggregated c64 (prefill-decode-interval 8/16), 1P1D c8/c16/c64 and 1P4D c64 arms, add master entries for the new overrides, and update the perf-changelog description to the shipped set. 将聚合与分离式 AgentX 配方以及 nvidia-master sweep 精简为实测 Pareto 曲线的八个顶点:聚合 TP4/EP1 c1 与 TP2/EP2 c1;分离式 1P2D 跨节点 c16、1P2D c48、1P2D c64、1P1D c96、1P1D HiCache c160 与 2P1D HiCache c256。移除被支配的聚合 c64(prefill-decode-interval 8/16)、1P1D c8/c16/c64 与 1P4D c64 配置,为新增 override 添加 master 条目,并将 perf-changelog 描述更新为最终交付集合。 Co-Authored-By: Claude Fable 5.1 --- .../sglang/gb300-fp4/agentx/agg-variants.yaml | 53 +++---------- .../gb300-fp4/agentx/disagg-variants.yaml | 78 +++++-------------- inferencex-e2e/configs/nvidia-master.yaml | 44 ++++------- inferencex-e2e/perf-changelog.yaml | 8 +- 4 files changed, 49 insertions(+), 134 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml index a9bb2d6d50..910d9d1617 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -4,15 +4,16 @@ # # One SGLang worker serves prefill and decode with the checkpoint's bundled DSpark # draft (block size 5); the Dynamo frontend routes with the KV-aware router and -# session affinity. The aggregated arms cover the two ends of the curve; the -# middle band (c16-c256) comes from the disaggregated recipes in disagg-variants.yaml. -# Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, P90 TTFT; all GSM8K -# gates passed): -# override_tp4_c1 : 6,215 @ 385.0 tok/s/user, 1.2 s (4 GPUs; GSM8K 0.9735) -# override_tp2_c1 : 12,501 @ 377.1 tok/s/user, 0.9 s (2 GPUs; GSM8K 0.9735) -# override_tp2_pdi16_c64 : 153,542 @ 83.3 tok/s/user, 15.2 s (2 GPUs; GSM8K 0.9719) -# override_tp2_pdi8_c64 : 159,510 @ 62.7 tok/s/user, 6.0 s (2 GPUs; GSM8K 0.9712) -# The two c1 arms are the highest-interactivity points of the curve (above the +# session affinity. The aggregated arms are the two lowest-latency points of the +# curve; everything from c16 to c256 comes from the disaggregated recipes in +# disagg-variants.yaml. Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, +# P90 TTFT; both GSM8K gates passed): +# override_tp4_c1 : 6,215 @ 385.0 tok/s/user, 1.2 s (4 GPUs; GSM8K 0.9735) +# override_tp2_c1 : 12,501 @ 377.1 tok/s/user, 0.9 s (2 GPUs; GSM8K 0.9735) +# The aggregated c64 cells (prefill-decode-interval 8: 159,510 @ 62.7; interval 16: +# 153,542 @ 83.3) were measured as well but are dominated by the disaggregated +# HiCache cells and are not shipped. The two c1 arms are the highest-interactivity +# points of the curve (above the # published MI355X ATOM curve's last vertex at 336.7 tok/s/user); at c1 the AgentX # client replays the trace's idle gaps, so a single lane is client-bound and a second # user on the same worker only lowers the P90 (TP2 c2: 12,538 @ 306.8). The harness @@ -163,37 +164,3 @@ override_tp2_c1: env: CONC: "1" AGENTIC_WARMUP_GRACE_PERIOD: "1800" - -# c64 with the default prefill-decode-interval 16 (512 prefill tokens per decode step): -# 153,542 tokens/s/GPU @ 83.3 tok/s/user, P90 TTFT 15.2 s. -override_tp2_pdi16_c64: - name: agg-gb300-tp2ep2-pdi16-c64 - roles: - agg: - args: - chunked-prefill-size: 8192 - prefill-decode-interval: 16 - swa-prefix-tails: 4096 - max-running-requests: 64 - benchmark: - env: - CONC: "64" - AGENTIC_WARMUP_GRACE_PERIOD: "3600" - -# c64 with prefill-decode-interval 8 (1024 prefill tokens per decode step): the -# throughput end of the aggregated curve, 159,510 tokens/s/GPU @ 62.7 tok/s/user, -# P90 TTFT 6.0 s (+3.9% tokens/s/GPU and 2.5x lower TTFT than interval 16, at -# -25% interactivity). -override_tp2_pdi8_c64: - name: agg-gb300-tp2ep2-pdi8-c64 - roles: - agg: - args: - chunked-prefill-size: 8192 - prefill-decode-interval: 8 - swa-prefix-tails: 4096 - max-running-requests: 64 - benchmark: - env: - CONC: "64" - AGENTIC_WARMUP_GRACE_PERIOD: "3600" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml index bf19ae20be..dbd3e5d12a 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -6,27 +6,23 @@ # Prefill and decode are separate TP2/EP2 SGLang workers (two GPUs each) behind the # Dynamo KV router; with the DSpark draft in PD-disaggregated mode both sides must # use the same TP, so capacity is added with more TP2 workers: one prefill worker -# feeds one, two or four decode workers (1PxD), which sets the sessions per decode +# feeds one or two decode workers (1P1D, 1P2D), which sets the sessions per decode # worker and with it the interactivity band, and two prefill workers feed one decode # worker (2P1D) at the throughput end. Measured on GB300 NVL72 (tokens/s/GPU @ P90 # interactivity, P90 TTFT; every override's GSM8K gate passed): -# override_1p1d_c8 : 15,724 @ 326.1 tok/s/user, 0.6 s ( 4 GPUs, 2 nodes) -# override_1p1d_c16 : 28,810 @ 288.3 tok/s/user, 0.7 s ( 4 GPUs, 2 nodes) -# override_1p1d_c64 : 114,682 @ 147.5 tok/s/user, 1.8 s ( 4 GPUs, 2 nodes) -# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s ( 4 GPUs, 2 nodes) -# override_1p1d_hicache_c160 : 172,192 @ 98.8 tok/s/user, 21.7 s ( 4 GPUs, 2 nodes) -# override_1p2d_spread_c16 : 19,280 @ 327.4 tok/s/user, 0.7 s ( 6 GPUs, 3 nodes) -# override_1p2d_c48 : 59,740 @ 244.3 tok/s/user, 1.4 s ( 6 GPUs, 2 nodes) -# override_1p2d_c64 : 80,583 @ 207.6 tok/s/user, 2.0 s ( 6 GPUs, 2 nodes) -# override_1p4d_c64 : 49,887 @ 254.6 tok/s/user, 2.1 s (10 GPUs, 3 nodes) -# override_2p1d_hicache_c256 : 174,562 @ 68.1 tok/s/user, 18.9 s ( 6 GPUs, 3 nodes) -# Against the published MI355X ATOM curve at matched P90 interactivity: 1P1D c16 1.53x, -# c64 2.30x, c96 2.72x, HiCache c160 2.40x; 1P2D spread c16 1.73x, c48 2.11x, c64 2.21x; -# 1P4D c64 1.91x; 2P1D HiCache c256 1.94x tokens/s/GPU; 1P1D c8 (1.41x) extends the -# curve to 326 tok/s/user. The multi-decode cells run the Dynamo frontend with a 1 s -# session-affinity TTL (per-turn load-aware decode pick); the single-decode cells keep -# the 3600 s default they were measured with. The harness applies the golden -# acceptance length (3.51 at K=5) itself. +# override_1p2d_spread_c16 : 19,280 @ 327.4 tok/s/user, 0.7 s (6 GPUs, 3 nodes) +# override_1p2d_c48 : 59,740 @ 244.3 tok/s/user, 1.4 s (6 GPUs, 2 nodes) +# override_1p2d_c64 : 80,583 @ 207.6 tok/s/user, 2.0 s (6 GPUs, 2 nodes) +# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s (4 GPUs, 2 nodes) +# override_1p1d_hicache_c160 : 172,192 @ 98.8 tok/s/user, 21.7 s (4 GPUs, 2 nodes) +# override_2p1d_hicache_c256 : 174,562 @ 68.1 tok/s/user, 18.9 s (6 GPUs, 3 nodes) +# Against the published MI355X ATOM curve at matched P90 interactivity: 1P2D spread +# c16 1.73x, c48 2.11x, c64 2.21x; 1P1D c96 2.72x, HiCache c160 2.40x; 2P1D HiCache +# c256 1.94x tokens/s/GPU. Other measured cells (1P1D c8/c16/c64, 1P2D c24/c32, 1P4D +# c64) lie on or under the line through these vertices and are not shipped. The +# multi-decode cells run the Dynamo frontend with a 1 s session-affinity TTL (per-turn +# load-aware decode pick); the single-decode cells keep the 3600 s default they were +# measured with. The harness applies the golden acceptance length (3.51 at K=5) itself. schema: 2 @@ -181,29 +177,10 @@ base: AGENTIC_WARMUP_GRACE_PERIOD: "3600" # 1 prefill TP2/EP2 + 1 decode TP2/EP2 (4 GPUs, 2 nodes). The decode worker alone -# sets the interactivity: 8 sessions give 326 tok/s/user, 16 give 288, 64 give -# 147.5 and 96 give 132; the single prefill worker holds P90 TTFT under 4.1 s up -# to c96 (it saturates between c96 and c128). -override_1p1d_c8: - name: disagg-gb300-1p1d-tp2ep2-c8 - benchmark: - env: - CONC: "8" - -override_1p1d_c16: - name: disagg-gb300-1p1d-tp2ep2-c16 - benchmark: - env: - CONC: "16" - -# 114,682 tokens/s/GPU @ 147.5 tok/s/user with a 1.8 s P90 TTFT. -override_1p1d_c64: - name: disagg-gb300-1p1d-tp2ep2-c64 - benchmark: - env: - CONC: "64" - -# 144,448 tokens/s/GPU @ 132.2 tok/s/user with a 4.1 s P90 TTFT. +# sets the interactivity (96 sessions give 132 tok/s/user); the single prefill worker +# holds P90 TTFT under 4.1 s up to c96 and saturates between c96 and c128, where the +# HiCache override below takes over. 144,448 tokens/s/GPU @ 132.2 tok/s/user with a +# 4.1 s P90 TTFT. override_1p1d_c96: name: disagg-gb300-1p1d-tp2ep2-c96 benchmark: @@ -270,25 +247,6 @@ override_1p2d_spread_c16: env: CONC: "16" -# 1 prefill TP2/EP2 + 4 decode TP2/EP2 on two nodes (10 GPUs, 3 nodes), 16 sessions -# per decode worker, session-affinity TTL 1 s: 49,887 tokens/s/GPU @ 254.6 tok/s/user, -# P90 TTFT 2.06 s (GSM8K 0.9742; identical to the TTL 3600 measurement 49,846 @ 253.9). -override_1p4d_c64: - name: disagg-gb300-1p4d-tp2ep2-c64 - frontend: - args: - router-session-affinity-ttl-secs: "1" - roles: - decode: - nodes: 2 - workers: 4 - args: - max-running-requests: 32 - cuda-graph-max-bs-decode: 32 - benchmark: - env: - CONC: "64" - # Throughput end. The prefill worker keeps its KV prefix cache in a host-memory tier # (SGLang HiCache: ratio 1 = one host copy of the device pool, write-back, direct I/O), # which removes the LRU evictions that pushed P90 TTFT past the 25 s ceiling beyond c96 diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 8fba7f5782..4a35d9349d 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8964,17 +8964,7 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-agg: additional-settings: - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4_c1" - spec-decoding: mtp - conc-list: [64] - num-nodes: 1 - worker: - num-worker: 1 - tp: 2 - ep: 2 - dp-attn: false - additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_pdi16_c64" - - spec-decoding: mtp - conc-list: [64] + conc-list: [1] num-nodes: 1 worker: num-worker: 1 @@ -8982,7 +8972,7 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-agg: ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_pdi8_c64" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_c1" dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 model: deepseek-ai/DeepSeek-V4.1-Flash @@ -8999,30 +8989,30 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: - dram-utilization: 0.80 search-space: - spec-decoding: mtp - conc-list: [8] + conc-list: [16] prefill: num-worker: 1 tp: 2 ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c8" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_spread_c16" decode: - num-worker: 1 + num-worker: 2 tp: 2 ep: 2 dp-attn: false - spec-decoding: mtp - conc-list: [16] + conc-list: [48] prefill: num-worker: 1 tp: 2 ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c16" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_c48" decode: - num-worker: 1 + num-worker: 2 tp: 2 ep: 2 dp-attn: false @@ -9034,9 +9024,9 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c64" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_c64" decode: - num-worker: 1 + num-worker: 2 tp: 2 ep: 2 dp-attn: false @@ -9055,30 +9045,30 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 2 dp-attn: false - spec-decoding: mtp - conc-list: [64] + conc-list: [160] prefill: num-worker: 1 tp: 2 ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_c64" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_hicache_c160" decode: - num-worker: 2 + num-worker: 1 tp: 2 ep: 2 dp-attn: false - spec-decoding: mtp - conc-list: [64] + conc-list: [256] prefill: - num-worker: 1 + num-worker: 2 tp: 2 ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p4d_c64" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_2p1d_hicache_c256" decode: - num-worker: 4 + num-worker: 1 tp: 2 ep: 2 dp-attn: false diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 3c6b07793d..2132df7a88 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9114,11 +9114,11 @@ - "Add DeepSeek-V4.1-Flash FP4 AgentX on GB300 with the Dynamo frontend (KV router, session affinity) and SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928, DSpark block size 5." - "Enable the GB300 multi-node launcher route for DeepSeek-V4.1-Flash FP4 with Dynamo + SGLang." - "Use the mounted shared Hugging Face cache for every SGLang server role." - - "Aggregated arms: a pure TP4 (EP1) c1 lowest-latency point (the standalone recipe's TP4 c1 settings, static ragged verify and Engram in HBM, behind Dynamo) and the TP2/EP2 c64 prefill-decode-interval 16 and 8 variants (153,542 tok/s/GPU @ 83.3 tok/s/user P90 and 159,510 @ 62.7 measured on GB300 NVL72)." - - "Disaggregated curve with Mooncake KV transfer, one TP2/EP2 prefill worker feeding one, two or four TP2/EP2 decode workers: 1P1D at c8/c16/c64/c96 (15,724 @ 326.1, 28,810 @ 288.3, 114,682 @ 147.5 and 144,448 @ 132.2 tok/s/GPU @ tok/s/user P90; P90 TTFT 0.6-4.1 s), 1P2D at c64 (80,214 @ 202.9, 2.0 s) and 1P4D at c64 (49,846 @ 253.9, 1.9 s). The complete nine-point sweep passed, including GSM8K evaluation for all configured eval points." + - "Aggregated arms: the two lowest-latency points of the curve, a pure TP4 (EP1) c1 point (the standalone recipe's TP4 c1 settings, static ragged verify and Engram in HBM, behind Dynamo; 6,215 tok/s/GPU @ 385.0 tok/s/user P90) and a TP2/EP2 c1 point with static ragged verify and Engram in host DRAM (12,501 @ 377.1, P90 TTFT 0.9 s); both passed their GSM8K gates." + - "Disaggregated curve with Mooncake KV transfer and TP2/EP2 workers: 1P2D with a 1 s Dynamo session-affinity TTL at c16 (one decode worker per node; 19,280 tok/s/GPU @ 327.4 tok/s/user P90, P90 TTFT 0.7 s), c48 (59,740 @ 244.3, 1.4 s) and c64 (80,583 @ 207.6, 2.0 s); 1P1D at c96 (144,448 @ 132.2, 4.1 s) and, with the SGLang HiCache host-memory prefix-cache tier on the prefill worker, at c160 (172,192 @ 98.8, 21.7 s); 2P1D HiCache at c256 (174,562 @ 68.1, 18.9 s). Every point passed its GSM8K gate; at matched P90 interactivity the curve is 1.7-2.7x the published MI355X ATOM tokens/s/GPU." - "新增 GB300 上 DeepSeek-V4.1-Flash FP4 AgentX 的 Dynamo 前端(KV 路由、会话亲和)+ SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928 配方,DSpark block size 5。" - "启用 GB300 多节点启动器中 DeepSeek-V4.1-Flash FP4 的 Dynamo + SGLang 路径。" - "所有 SGLang server role 使用已挂载的共享 Hugging Face cache。" - - "聚合分支:纯 TP4(EP1)c1 最低延迟点(沿用单机配方的 TP4 c1 设置、static ragged verify 与 HBM 内 Engram,置于 Dynamo 之后)以及 TP2/EP2 c64 的 prefill-decode-interval 16/8 变体(GB300 NVL72 实测 153,542 tok/s/GPU @ 83.3 tok/s/user P90 与 159,510 @ 62.7)。" - - "分离式曲线(Mooncake KV 传输),一个 TP2/EP2 prefill worker 服务一个、两个或四个 TP2/EP2 decode worker:1P1D 的 c8/c16/c64/c96(15,724 @ 326.1、28,810 @ 288.3、114,682 @ 147.5、144,448 @ 132.2 tok/s/GPU @ tok/s/user P90;P90 TTFT 0.6-4.1 s)、1P2D 的 c64(80,214 @ 202.9,2.0 s)与 1P4D 的 c64(49,846 @ 253.9,1.9 s)。完整九点 sweep 已通过,包括所有已配置 eval 点的 GSM8K 评测。" + - "聚合分支:曲线上延迟最低的两个点——纯 TP4(EP1)c1(沿用单机配方的 TP4 c1 设置、static ragged verify 与 HBM 内 Engram,置于 Dynamo 之后;6,215 tok/s/GPU @ 385.0 tok/s/user P90)与采用 static ragged verify、Engram 置于主机内存的 TP2/EP2 c1(12,501 @ 377.1,P90 TTFT 0.9 s);两者均通过 GSM8K 验证。" + - "分离式曲线(Mooncake KV 传输,TP2/EP2 worker):1P2D 配合 1 秒 Dynamo 会话亲和 TTL 的 c16(每节点一个 decode worker;19,280 tok/s/GPU @ 327.4 tok/s/user P90,P90 TTFT 0.7 s)、c48(59,740 @ 244.3,1.4 s)与 c64(80,583 @ 207.6,2.0 s);1P1D 的 c96(144,448 @ 132.2,4.1 s)以及在 prefill worker 上启用 SGLang HiCache 主机内存前缀缓存层后的 c160(172,192 @ 98.8,21.7 s);2P1D HiCache 的 c256(174,562 @ 68.1,18.9 s)。所有点均通过 GSM8K 验证;在相同 P90 交互性下,该曲线为已发布 MI355X ATOM tokens/s/GPU 的 1.7-2.7 倍。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3598 From fde7b9da34d7859e26ad5c6f8402393109194fea Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Thu, 1 Oct 2026 02:27:56 -0700 Subject: [PATCH 4/5] fix(agentx): run GB300 DSV4.1 Flash Dynamo SGLang on the default Slurm route Drop the named `restricted` GB300 Slurm route (batch_2 partition, restricted account, alternate shared caches). The CI runners are not associated with that account, so the canary failed at image import with "Invalid account or account/partition combination specified". The recipes now use the default gb300-nv partition, account and shared caches, like the other GB300 DeepSeek-V4.1-Flash configs. Revert the route plumbing in the Slurm settings, SRT launcher, tests and docs, and drop the matching changelog bullet. Co-Authored-By: Claude Opus 5.5 --- inferencex-e2e/configs/CONFIGS.md | 4 --- inferencex-e2e/configs/runners.yaml | 9 ------ .../docs/configuration-procedures.md | 4 --- .../docs/configuration-procedures_zh.md | 3 -- inferencex-e2e/infx/clusters/slurm.py | 30 ------------------- .../infx/launch/drivers/srt/__init__.py | 9 ++---- .../infx/launch/drivers/srt/lanes.py | 20 ------------- inferencex-e2e/infx/launch/drivers/srt/run.py | 12 -------- .../infx/tests/launch/test_srt_driver.py | 13 -------- inferencex-e2e/perf-changelog.yaml | 2 -- 10 files changed, 2 insertions(+), 104 deletions(-) diff --git a/inferencex-e2e/configs/CONFIGS.md b/inferencex-e2e/configs/CONFIGS.md index 211fc2ee8c..03a6cce4c1 100644 --- a/inferencex-e2e/configs/CONFIGS.md +++ b/inferencex-e2e/configs/CONFIGS.md @@ -200,10 +200,6 @@ schema; unknown keys fail. - `slurm:` ([`infx/clusters/slurm.py`](../infx/clusters/slurm.py)) holds the partition, account, exclusivity, GRES, excluded nodes and extra `srun`/`salloc` options; its volumes are host `path`s that jobs see at the same place. -- `slurm.routes` declares named alternate partition/account/storage routes for the same - physical runner pool. A route may replace selected volume paths and the squash-cache - directory. Workload matching stays in the applicable named launcher policy table; the - route itself contains only cluster facts. - `slurm.squash` is the Pyxis squash cache: `dir`, `visibility`, `lock-timeout-s`, `key-style` (`underscore`, `plus` or `plus-strip-nvcr`) and `import`: `submit-host` (on the launching host), `compute` (once on one compute node), `all-nodes` (on every diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index caaad93b91..cf67501a63 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -621,15 +621,6 @@ clusters: Qwen3.5-397B-A17B-NVFP4: {root: scratch, dir: Qwen3.5-397B-A17B-NVFP4-V2} scheduler: slurm slurm: - routes: - restricted: - partition: batch_2 - account: restricted - volumes: - hf-hub-cache: {path: /data/home/slurm-shared/gharunners/hf-hub-cache} - aiperf-cache: {path: /data/home/slurm-shared/gharunners/ai-perf-cache} - dynamo-wheels: {path: /data/home/slurm-shared/gharunners/dynamo-wheels} - squash-dir: /data/home/slurm-shared/gharunners/squash partition: batch_1 account: benchmark exclusive: false diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index a4e3f36d80..d0ed2d739b 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -169,10 +169,6 @@ Setup source: [`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNE ### Repository registration 1. Add the fleet's `clusters.` record to [`configs/runners.yaml`](../configs/runners.yaml) (node shape, workload env, models, and the scheduler sub-record: for Slurm the partition, volumes, squash cache and srt-slurm facts; schema in [`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners)). Put launch rules that depend on model, framework, precision or recipe in [`infx/launch/policy.py`](../infx/launch/policy.py) or beside the one driver that reads them, never in a driver branch on the cluster id. A cluster on a new scheduler needs that scheduler's settings model under [`infx/clusters/`](../infx/clusters) and its backend under [`infx/launch/backends/`](../infx/launch/backends), one registry entry each, and no driver change; it runs only script-driver (`BENCH_SCRIPT_OVERRIDE`) points. - When one physical Slurm pool exposes an alternate partition/account and storage root, - declare those facts as a named `slurm.routes` entry and select it from the applicable - named workload-policy table. Do not duplicate runner ownership or embed the alternate - cluster facts in driver control flow. 2. Add each exact registered runner name under the intended `labels:` key in [`configs/runners.yaml`](../configs/runners.yaml). New names use `_` with zero-padded indices. 3. Add every runner name to exactly one `cluster:` label matching that record. `python -m infx.launch run` resolves the cluster from the runner name, so a runner outside every cluster label fails validation and fails at launch. 4. Master entries whose facts depend on one physical fleet use that exact `cluster:` label. Agentic configs require it. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 1f74c8a5c5..a659d68c46 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -150,9 +150,6 @@ STP(Single Token Prediction,单 Token 预测)是每次前向传播生成 ### 仓库注册 1. 在 [`configs/runners.yaml`](../configs/runners.yaml) 中为 fleet 添加 `clusters.` 记录(节点形状、工作负载环境、模型以及调度器子记录:Slurm 为分区、卷、squash 缓存和 srt-slurm 事实;schema 见 [`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners))。依赖模型、框架、精度或配方的启动规则写进 [`infx/launch/policy.py`](../infx/launch/policy.py) 或唯一读取它的驱动旁边,绝不在驱动中按集群 id 分支。新调度器上的集群需要在 [`infx/clusters/`](../infx/clusters) 下新增该调度器的设置模型、在 [`infx/launch/backends/`](../infx/launch/backends) 下新增其后端,各登记一行,无需修改驱动;这类集群只运行 script 驱动(`BENCH_SCRIPT_OVERRIDE`)的点。 - 当同一个物理 Slurm 池提供备用分区/账号和存储根目录时,请将这些事实声明为命名的 - `slurm.routes` 条目,并在相应的命名工作负载策略表中选择它。不要重复 runner 所有权, - 也不要把备用集群事实嵌入驱动控制流。 2. 在 [`configs/runners.yaml`](../configs/runners.yaml) 预期的 `labels:` key 下添加每个精确的已注册 runner 名称。新名称使用 `_`,索引必须两位补零。 3. 把每个 runner 名称加入且仅加入一个与该记录对应的 `cluster:` 标签。`python -m infx.launch run` 通过 runner 名称解析集群,因此不属于任何集群标签的 runner 会导致校验失败,并在启动时失败。 4. 事实依赖某个物理 fleet 的主条目使用对应的精确 `cluster:` 标签;agentic 配置强制要求该标签。 diff --git a/inferencex-e2e/infx/clusters/slurm.py b/inferencex-e2e/infx/clusters/slurm.py index 492f19e29c..2fbc41f75e 100644 --- a/inferencex-e2e/infx/clusters/slurm.py +++ b/inferencex-e2e/infx/clusters/slurm.py @@ -177,15 +177,6 @@ class SrtSlurmSettings(Record): extra: dict[str, Any] = Field(default_factory=dict) -class SlurmRoute(Record): - """A workload-selected route through another partition and shared-storage root.""" - - partition: str = Field(min_length=1) - account: str = Field(min_length=1) - volumes: dict[str, HostVolume] = Field(default_factory=dict) - squash_dir: HostPath | None = Field(default=None, alias="squash-dir") - - class SlurmSettings(SchedulerSettings): """Scheduler facts shared by every Slurm submission on the cluster.""" @@ -201,7 +192,6 @@ class SlurmSettings(SchedulerSettings): salloc_args: tuple[LongOption, ...] = Field(default=(), alias="salloc-args") squash: SquashCache | None = None srt_slurm: SrtSlurmSettings | None = Field(default=None, alias="srt-slurm") - routes: dict[str, SlurmRoute] = Field(default_factory=dict) @field_validator("gres") @classmethod @@ -249,26 +239,6 @@ def path(self, volume: str) -> Path | None: declared = self.volumes.get(volume) return None if declared is None else declared.path - def routed(self, name: str) -> Self: - """Return these settings with the named route's scheduler and storage facts.""" - try: - route = self.routes[name] - except KeyError: - raise ValueError(f"unknown Slurm route {name!r}") from None - squash = self.squash - if route.squash_dir is not None: - if squash is None: - raise ValueError(f"Slurm route {name!r} sets squash-dir without a squash cache") - squash = squash.model_copy(update={"dir": route.squash_dir}) - return self.model_copy( - update={ - "partition": route.partition, - "account": route.account, - "volumes": {**self.volumes, **route.volumes}, - "squash": squash, - } - ) - def slurm_settings(cluster: Cluster) -> SlurmSettings: """The Slurm sub-record of ``cluster``; a cluster on another scheduler is a caller bug.""" diff --git a/inferencex-e2e/infx/launch/drivers/srt/__init__.py b/inferencex-e2e/infx/launch/drivers/srt/__init__.py index 765665a579..4a3aa0cd3d 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/__init__.py +++ b/inferencex-e2e/infx/launch/drivers/srt/__init__.py @@ -15,7 +15,7 @@ from pathlib import Path from typing import TYPE_CHECKING -from infx.clusters.slurm import SlurmSettings, slurm_settings +from infx.clusters.slurm import SlurmSettings from infx.launch import policy from infx.launch.backends.base import BackendError from infx.launch.backends.slurm import srtctl_job_name @@ -30,7 +30,7 @@ run_setup, ) from infx.launch.drivers.srt.recipe import eval_overrides, prepare_recipe -from infx.launch.drivers.srt.run import SrtRun, require, routed_launch, slurm_backend +from infx.launch.drivers.srt.run import SrtRun, require, slurm_backend from infx.launch.request import BATCH_REENTRY_ENV, RequestError, SingleNodeRequest, SrtRequest if TYPE_CHECKING: @@ -124,7 +124,6 @@ def run_multinode(launch: Launch) -> int: lane = lanes.srt_lane(launch.cluster.id, launch.path) request = SrtRequest.from_env(launch.request.env) lanes.check_request(lane, request) - launch = routed_launch(launch, lanes.scheduler_route(lane, request)) config_file = lanes.config_file(request) decision = power.resolve_power(launch.cluster.id, launch.path, request) model = models.checkpoint(launch.cluster, request) @@ -191,10 +190,6 @@ def volume(where: str, cluster_id: str, name: str) -> None: where = f"SRT_LANES[{cluster_id!r}, {path}]" for mount in lane.mounts: volume(where, cluster_id, mount.volume) - settings = slurm_settings(clusters[cluster_id]) - for _, route in lane.scheduler_routes: - if route not in settings.routes: - problems.append(f"{where}: no Slurm route {route!r}") if lane.shared_run_root and srt.shared_run_root is None: problems.append(f"{where}: no srt-slurm.shared-run-root") if srt.default_time_limit is not None and (lane.time_limit or lane.long_time_limit): diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index 46ed24f3d3..e8217eb607 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -41,7 +41,6 @@ class SrtLane: time_limit: str | None = None long_time_limit: str | None = None long_time: Match | None = None - scheduler_routes: tuple[tuple[Match, str], ...] = () _DYNAMO = any_of("dynamo-sglang", "dynamo-trt", "dynamo-vllm") @@ -92,17 +91,6 @@ class SrtLane: long_time=Match( any_of("dsv4"), frameworks=any_of("dynamo-sglang", "dynamo-trt"), agentic=True ), - scheduler_routes=( - ( - Match( - any_of("dsv41flash"), - any_of("fp4"), - frameworks=any_of("dynamo-sglang"), - agentic=True, - ), - "restricted", - ), - ), ), ("h100-dgxc", LaunchPath.SRT_MULTI): SrtLane(frameworks=any_of("dynamo-sglang", "dynamo-trt")), ("h200-dgxc", LaunchPath.SRT_MULTI): SrtLane( @@ -142,14 +130,6 @@ def check_request(lane: SrtLane, request: SrtRequest) -> None: raise LaunchError(f"{message} (FRAMEWORK={framework})") -def scheduler_route(lane: SrtLane, request: SrtRequest) -> str | None: - """The one named scheduler route selected for this request, if any.""" - selected = [name for match, name in lane.scheduler_routes if match(request)] - if len(selected) > 1: - raise LaunchError(f"request selects several scheduler routes: {', '.join(selected)}") - return selected[0] if selected else None - - def config_file(request: SrtRequest) -> str: """CONFIG_FILE, or on an eval-only run its real-verification EVAL_CONFIG_FILE.""" if request.eval_only and request.eval_config_file: diff --git a/inferencex-e2e/infx/launch/drivers/srt/run.py b/inferencex-e2e/infx/launch/drivers/srt/run.py index f060a6f37e..e1f1e92c4b 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/run.py +++ b/inferencex-e2e/infx/launch/drivers/srt/run.py @@ -8,7 +8,6 @@ from pathlib import Path from typing import TYPE_CHECKING -from infx.clusters.slurm import slurm_settings from infx.config import repository_root from infx.launch.backends.slurm import SlurmBackend, cli from infx.launch.context import Launch, LaunchError @@ -70,14 +69,3 @@ def slurm_backend(launch: Launch) -> SlurmBackend: f"{launch.path} needs the Slurm backend, got {type(launch.backend).__name__}" ) return launch.backend - - -def routed_launch(launch: Launch, route: str | None) -> Launch: - """Apply a named cluster-declared Slurm route without changing the physical cluster.""" - if route is None: - return launch - settings = slurm_settings(launch.cluster).routed(route) - cluster = launch.cluster.model_copy(update={"scheduler_settings": settings}) - cluster.bind_id(launch.cluster.id) - backend = SlurmBackend(cluster, launch.request, launch.life) - return Launch(cluster, backend, launch.request, launch.life, launch.path) diff --git a/inferencex-e2e/infx/tests/launch/test_srt_driver.py b/inferencex-e2e/infx/tests/launch/test_srt_driver.py index 1ae1b0b499..da513d1dd4 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_driver.py @@ -176,19 +176,16 @@ def test_single_node_failed_allocation_fails_the_launch(harness): lane=SrtLane( setup_scripts={"dynamo-sglang": "setup.sh"}, mounts=(LaneMount(Match(), "cache", "/cache"),), time_limit="2:00:00", - scheduler_routes=((Match(), "restricted"),), ), env=dict(FRAMEWORK="dynamo-sglang"), model="nvme/model", preflight=False, tag="lab,dsr1,fp8,1024x1024,", setup_script="setup.sh", served="served-model", dist_timeout=True, time="2:00:00", mounts=("/cache",), staging="import", - partition="p2", account="restricted", cache="restricted-cache", squash="restricted-squash", ), "lab-b": dict( lane=SrtLane(shared_run_root=(Match(),)), env=dict(FRAMEWORK="dynamo-vllm", IS_AGENTIC="1", ISL="0", OSL="0", FAKE_RESULTS="agentic"), model="models/model", preflight=True, tag=None, setup_script=None, served=None, dist_timeout=False, time="10", mounts=(), staging="registry", shared_checkout=True, - partition="p", account=None, cache=None, squash=None, ), } # fmt: skip @@ -202,11 +199,6 @@ def lab_config(tmp: Path) -> Path: "volumes": {"nvme": {"path": str(tmp / "nvme"), "visibility": "node-local"}, "cache": {"path": str(tmp / "cache")}}, "squash": {"dir": str(tmp / "squash"), "import": "submit-host"}, - "routes": {"restricted": { - "partition": "p2", "account": "restricted", - "volumes": {"cache": {"path": str(tmp / "restricted-cache")}}, - "squash-dir": str(tmp / "restricted-squash"), - }}, "srt-slurm": {"network-interface": "", "job-tag": "lab", "dist-timeout-s": 1800}, }}, "lab-b": {**common, "models": {"entries": {"Model": {"root": "models", "dir": "model"}}}, "slurm": { @@ -283,15 +275,10 @@ def test_multinode_lane_stages_workflow_artifacts(harness, monkeypatch, cluster_ config = srtslurm(checkout) assert config["model_paths"] == {"alias": str(tmp / lab["model"])} assert config["default_time_limit"] == lab["time"] - assert config["default_partition"] == lab["partition"] - assert config.get("default_account") == lab["account"] assert set(lab["mounts"]) <= set(config.get("default_mounts", {}).values()) - if lab["cache"] is not None: - assert config["default_mounts"][str(tmp / lab["cache"])] == "/cache" imported = [line.split()[-1] for line in lines(harness.logs, "enroot")] if lab["staging"] == "import": assert config["containers"][env["IMAGE"]].endswith(".sqsh") and imported == ["docker://test:tag"] - assert Path(config["containers"][env["IMAGE"]]).parent == tmp / lab["squash"] else: assert config["containers"][env["IMAGE"]] == env["IMAGE"] and imported == [] outputs = Path(json.loads((workspace / "srt-submission.json").read_text())["output_dir"]) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 1f56e93a3e..0e38a32609 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9128,12 +9128,10 @@ - agentic-coding description: - "Add DeepSeek-V4.1-Flash FP4 AgentX on GB300 with the Dynamo frontend (KV router, session affinity) and SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928, DSpark block size 5." - - "Enable the GB300 multi-node launcher route for DeepSeek-V4.1-Flash FP4 with Dynamo + SGLang." - "Use the mounted shared Hugging Face cache for every SGLang server role." - "Aggregated arms: the two lowest-latency points of the curve, a pure TP4 (EP1) c1 point (the standalone recipe's TP4 c1 settings, static ragged verify and Engram in HBM, behind Dynamo; 6,215 tok/s/GPU @ 385.0 tok/s/user P90) and a TP2/EP2 c1 point with static ragged verify and Engram in host DRAM (12,501 @ 377.1, P90 TTFT 0.9 s); both passed their GSM8K gates." - "Disaggregated curve with Mooncake KV transfer and TP2/EP2 workers: 1P2D with a 1 s Dynamo session-affinity TTL at c16 (one decode worker per node; 19,280 tok/s/GPU @ 327.4 tok/s/user P90, P90 TTFT 0.7 s), c48 (59,740 @ 244.3, 1.4 s) and c64 (80,583 @ 207.6, 2.0 s); 1P1D at c96 (144,448 @ 132.2, 4.1 s) and, with the SGLang HiCache host-memory prefix-cache tier on the prefill worker, at c160 (172,192 @ 98.8, 21.7 s); 2P1D HiCache at c256 (174,562 @ 68.1, 18.9 s). Every point passed its GSM8K gate; at matched P90 interactivity the curve is 1.7-2.7x the published MI355X ATOM tokens/s/GPU." - "新增 GB300 上 DeepSeek-V4.1-Flash FP4 AgentX 的 Dynamo 前端(KV 路由、会话亲和)+ SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928 配方,DSpark block size 5。" - - "启用 GB300 多节点启动器中 DeepSeek-V4.1-Flash FP4 的 Dynamo + SGLang 路径。" - "所有 SGLang server role 使用已挂载的共享 Hugging Face cache。" - "聚合分支:曲线上延迟最低的两个点——纯 TP4(EP1)c1(沿用单机配方的 TP4 c1 设置、static ragged verify 与 HBM 内 Engram,置于 Dynamo 之后;6,215 tok/s/GPU @ 385.0 tok/s/user P90)与采用 static ragged verify、Engram 置于主机内存的 TP2/EP2 c1(12,501 @ 377.1,P90 TTFT 0.9 s);两者均通过 GSM8K 验证。" - "分离式曲线(Mooncake KV 传输,TP2/EP2 worker):1P2D 配合 1 秒 Dynamo 会话亲和 TTL 的 c16(每节点一个 decode worker;19,280 tok/s/GPU @ 327.4 tok/s/user P90,P90 TTFT 0.7 s)、c48(59,740 @ 244.3,1.4 s)与 c64(80,583 @ 207.6,2.0 s);1P1D 的 c96(144,448 @ 132.2,4.1 s)以及在 prefill worker 上启用 SGLang HiCache 主机内存前缀缓存层后的 c160(172,192 @ 98.8,21.7 s);2P1D HiCache 的 c256(174,562 @ 68.1,18.9 s)。所有点均通过 GSM8K 验证;在相同 P90 交互性下,该曲线为已发布 MI355X ATOM tokens/s/GPU 的 1.7-2.7 倍。" From 59530ec7a89080029a315c8d1c1b5269c8360caf Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Thu, 1 Oct 2026 03:14:51 -0700 Subject: [PATCH 5/5] fix(agentx): set PP_SIZE and PCP_SIZE for the GB300 DSV4.1 Flash aggregated recipe The aggregated AgentX power path now requires TP, PP_SIZE and PCP_SIZE before replay. Set PP_SIZE=1 and PCP_SIZE=1 in the base benchmark env, as the Qwen3.5 GB300 aggregated recipe does, so the replay no longer exits on missing inputs. Co-Authored-By: Claude Opus 5.5 --- .../dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml index 910d9d1617..e855c7db4c 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -124,6 +124,8 @@ base: # Warmup can leave bytes unacknowledged past AIPerf's 30 s default. AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" TP: "2" + PP_SIZE: "1" + PCP_SIZE: "1" # Lowest-latency arm: one pure TP4 (EP1) worker on all four GPUs at c1, the standalone # recipe's override_tp4_c1 behind the Dynamo frontend: static ragged verify, Engram