diff --git a/inferencex-e2e/benchmarks/llm-d/envoy.yaml b/inferencex-e2e/benchmarks/llm-d/envoy.yaml index ccba51ba0b..8bc10fe595 100644 --- a/inferencex-e2e/benchmarks/llm-d/envoy.yaml +++ b/inferencex-e2e/benchmarks/llm-d/envoy.yaml @@ -44,6 +44,13 @@ static_resources: - name: vh domains: ["*"] routes: + # Scrape metrics per node, never through load balancing. + - match: { path: "/metrics" } + direct_response: { status: 404 } + typed_per_filter_config: + envoy.filters.http.ext_proc: + "@type": type.googleapis.com/envoy.extensions.filters.http.ext_proc.v3.ExtProcPerRoute + disabled: true - match: { prefix: "/" } route: cluster: original_dst diff --git a/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-agg.sh b/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-agg.sh new file mode 100644 index 0000000000..772ce13802 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-agg.sh @@ -0,0 +1,9 @@ +#!/usr/bin/env bash +set -eo pipefail + +export GPUS_PER_NODE=4 TIME_LIMIT=08:00:00 CONTAINER_IMAGE="$IMAGE" +export PREFILL_WORKERS=1 DECODE_WORKERS=1 + +cd "$GITHUB_WORKSPACE/benchmarks/multi_node/llm-d" +exec bash ./submit.sh "$PREFILL_NODES" "$DECODE_NODES" \ + "$ISL" "$OSL" "${CONC_LIST// /x}" inf "$RANDOM_RANGE_RATIO" diff --git a/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh b/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh new file mode 100755 index 0000000000..58ce0262a5 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh @@ -0,0 +1,9 @@ +#!/usr/bin/env bash +set -eo pipefail + +export GPUS_PER_NODE=4 TIME_LIMIT=08:00:00 CONTAINER_IMAGE="$IMAGE" +export PREFILL_WORKERS="$PREFILL_NUM_WORKERS" DECODE_WORKERS="$DECODE_NUM_WORKERS" + +cd "$GITHUB_WORKSPACE/benchmarks/multi_node/llm-d" +exec bash ./submit.sh "$PREFILL_NODES" "$DECODE_NODES" \ + "$ISL" "$OSL" "${CONC_LIST// /x}" inf "$RANDOM_RANGE_RATIO" diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/agg-gb200-dep8-dspark-agentic.yaml b/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/agg-gb200-dep8-dspark-agentic.yaml new file mode 100644 index 0000000000..b22a33791a --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/agg-gb200-dep8-dspark-agentic.yaml @@ -0,0 +1,101 @@ +# DeepSeek-V4-Pro-0813 (DSpark) FP4 GB200, aggregated DEP8, 2 nodes. +# DSpark speculative decoding: 3 tokens (adaptive verification). No Mooncake. +apiVersion: llm-d.ai/v1alpha1 +kind: EndpointPickerConfig + +plugins: + - name: file-disc + type: file-discovery + parameters: + path: /tmp/endpoints.yaml + watchFile: false + + - type: inflight-load-producer + - type: approx-prefix-cache-producer + parameters: + autoTune: false + blockSizeTokens: 256 + maxPrefixTokensToMatch: 1048576 + maxPrefixBlocksToMatch: 4096 + lruCapacityPerServer: 5858 + - type: prefix-cache-affinity-filter + parameters: + peakPrefillThroughput: 20000 + maxTTFTPenaltyMs: 30000 + - type: prefix-cache-scorer + - type: token-load-scorer + parameters: + queueThresholdTokens: 1499703 + - type: active-request-scorer + - type: queue-scorer + - type: max-score-picker + +schedulingProfiles: + - name: default + plugins: + - pluginRef: prefix-cache-affinity-filter + - pluginRef: prefix-cache-scorer + weight: 6 + - pluginRef: token-load-scorer + weight: 3 + - pluginRef: active-request-scorer + weight: 2 + - pluginRef: queue-scorer + weight: 3 + - pluginRef: max-score-picker + +dataLayer: + discovery: + pluginRef: file-disc + +prefill: + tp: 1 + enable-expert-parallel: true + extra-args: >- + --kv-cache-dtype fp8 + --gpu-memory-utilization 0.88 + --max-num-batched-tokens 8192 + --block-size 256 + --tokenizer-mode deepseek_v4 + --tool-call-parser deepseek_v4 + --enable-auto-tool-choice + --reasoning-parser deepseek_v4 + --moe-backend deep_gemm_mega_moe + --enable-ep-weight-filter + --enable-cumem-allocator + --no-disable-hybrid-kv-cache-manager + --no-enable-flashinfer-autotune + --numa-bind + --compilation-config {"cudagraph_mode":"FULL_DECODE_ONLY","mode":0} + --speculative-config {"method":"dspark","num_speculative_tokens":3,"enable_adaptive_verification":true,"draft_sample_method":"probabilistic","attention_backend":"FLASHINFER_MLA_SPARSE_DSV4"} + --attention-config {"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true} + env: + PYTHONHASHSEED: "0" + VLLM_USE_RUST_FRONTEND: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_SERVER_DEV_MODE: "1" + VLLM_USE_NCCL_SYMM_MEM: "1" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_RPC_TIMEOUT: "600000" + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" + VLLM_DSV4_MEGA_FP8_COMBINE: "1" + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" + VLLM_USE_DEEP_GEMM: "1" + TILELANG_CLEANUP_TEMP_FILES: "1" + TORCH_SYMMMEM: NVSHMEM + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-agentx + NVSHMEM_REMOTE_TRANSPORT: "none" + NVSHMEM_IB_ENABLE_IBGDA: "false" + NVSHMEM_CUMEM_HANDLE_TYPE: FABRIC + NVSHMEM_DISABLE_CUDA_VMM: "0" + UCX_TLS: cuda_copy,cuda_ipc,rc,tcp + UCX_CUDA_IPC_ENABLE_MNNVL: "y" + UCX_MEMTYPE_CACHE: "n" + NCCL_CUMEM_ENABLE: "1" + NCCL_MNNVL_ENABLE: "1" + NCCL_NVLS_ENABLE: "1" + NCCL_P2P_LEVEL: NVL + +slurm: + time_limit: "08:00:00" diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/agg-gb200-dep8-dspark-mooncake-agentic.yaml b/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/agg-gb200-dep8-dspark-mooncake-agentic.yaml new file mode 100644 index 0000000000..87f4fca833 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/agg-gb200-dep8-dspark-mooncake-agentic.yaml @@ -0,0 +1,120 @@ +# DeepSeek-V4-Pro-0813 (DSpark) FP4 GB200, aggregated DEP8 with Mooncake prefix-cache. +# Sibling of agg-gb200-dep8-dspark-agentic.yaml; adds Mooncake (P2PHANDSHAKE embedded RDMA) +# so server.sh wires MultiConnector (NixlConnector + SimpleCPUOffloadConnector + MooncakeStoreConnector, kv_both). +# DSpark 3 speculative tokens (adaptive verification). conc-list [52,72]. +apiVersion: llm-d.ai/v1alpha1 +kind: EndpointPickerConfig + +plugins: + - name: file-disc + type: file-discovery + parameters: + path: /tmp/endpoints.yaml + watchFile: false + + - type: inflight-load-producer + - type: approx-prefix-cache-producer + parameters: + autoTune: false + blockSizeTokens: 256 + maxPrefixTokensToMatch: 1048576 + maxPrefixBlocksToMatch: 4096 + lruCapacityPerServer: 40810 + - type: prefix-cache-affinity-filter + parameters: + peakPrefillThroughput: 20000 + maxTTFTPenaltyMs: 30000 + - type: prefix-cache-scorer + - type: token-load-scorer + parameters: + queueThresholdTokens: 1499703 + - type: active-request-scorer + - type: queue-scorer + - type: max-score-picker + +schedulingProfiles: + - name: default + plugins: + - pluginRef: prefix-cache-affinity-filter + - pluginRef: prefix-cache-scorer + weight: 6 + - pluginRef: token-load-scorer + weight: 3 + - pluginRef: active-request-scorer + weight: 2 + - pluginRef: queue-scorer + weight: 3 + - pluginRef: max-score-picker + +dataLayer: + discovery: + pluginRef: file-disc + +# ---- Per-role vLLM flags ---- +prefill: + tp: 1 + enable-expert-parallel: true + extra-args: >- + --kv-cache-dtype fp8 + --gpu-memory-utilization 0.87 + --max-num-batched-tokens 8192 + --block-size 256 + --tokenizer-mode deepseek_v4 + --tool-call-parser deepseek_v4 + --enable-auto-tool-choice + --reasoning-parser deepseek_v4 + --moe-backend deep_gemm_mega_moe + --enable-ep-weight-filter + --enable-cumem-allocator + --no-disable-hybrid-kv-cache-manager + --no-enable-flashinfer-autotune + --numa-bind + --compilation-config {"cudagraph_mode":"FULL_DECODE_ONLY","mode":0} + --speculative-config {"method":"dspark","num_speculative_tokens":3,"enable_adaptive_verification":true,"draft_sample_method":"probabilistic","attention_backend":"FLASHINFER_MLA_SPARSE_DSV4"} + --attention-config {"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true} + env: + PYTHONHASHSEED: "0" + VLLM_USE_RUST_FRONTEND: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_SERVER_DEV_MODE: "1" + VLLM_USE_NCCL_SYMM_MEM: "1" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_RPC_TIMEOUT: "600000" + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" + VLLM_DSV4_MEGA_FP8_COMBINE: "1" + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" + VLLM_USE_DEEP_GEMM: "1" + TILELANG_CLEANUP_TEMP_FILES: "1" + TORCH_SYMMMEM: NVSHMEM + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-agentx + NVSHMEM_REMOTE_TRANSPORT: "none" + NVSHMEM_IB_ENABLE_IBGDA: "false" + NVSHMEM_CUMEM_HANDLE_TYPE: FABRIC + NVSHMEM_DISABLE_CUDA_VMM: "0" + UCX_TLS: cuda_copy,cuda_ipc,rc,tcp + UCX_CUDA_IPC_ENABLE_MNNVL: "y" + UCX_MEMTYPE_CACHE: "n" + NCCL_CUMEM_ENABLE: "1" + NCCL_MNNVL_ENABLE: "1" + NCCL_NVLS_ENABLE: "1" + NCCL_P2P_LEVEL: NVL + VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" + MC_ENABLE_DEST_DEVICE_AFFINITY: "1" + MC_STORE_CLIENT_METRIC: "1" + MC_STORE_CLIENT_METRIC_INTERVAL: "5" + MC_TE_METRIC: "0" + +# ---- Mooncake KV store config ---- +mooncake: + store_config: + metadata_server: "P2PHANDSHAKE" + local_buffer_size: "4GB" + protocol: "rdma" + device_name: "mlx5_0,mlx5_1,mlx5_3,mlx5_4" + mode: "embedded" + enable_offload: false # SSD only; the embedded DRAM pool is still enabled. + +# ---- SLURM resource directives ---- +slurm: + time_limit: "08:00:00" diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/agg-gb200-tp8-dspark-agentic.yaml b/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/agg-gb200-tp8-dspark-agentic.yaml new file mode 100644 index 0000000000..b62994470a --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/agg-gb200-tp8-dspark-agentic.yaml @@ -0,0 +1,77 @@ +# DeepSeek-V4-Pro-0813 (DSpark) FP4 GB200, aggregated TP8, 2 nodes. +# DSpark speculative decoding: 5 tokens. No Mooncake (two-node TP8). +apiVersion: llm-d.ai/v1alpha1 +kind: EndpointPickerConfig + +plugins: + - name: file-disc + type: file-discovery + parameters: + path: /tmp/endpoints.yaml + watchFile: false + + - type: active-request-scorer + - type: queue-scorer + - type: weighted-random-picker + +schedulingProfiles: + - name: default + plugins: + - pluginRef: active-request-scorer + weight: 2 + - pluginRef: queue-scorer + weight: 2 + - pluginRef: weighted-random-picker + +dataLayer: + discovery: + pluginRef: file-disc + +prefill: + tp: 8 + # Must be explicit false: server.sh defaults enable-expert-parallel to true. + enable-expert-parallel: false + extra-args: >- + --kv-cache-dtype fp8 + --gpu-memory-utilization 0.85 + --max-num-batched-tokens 8192 + --block-size 256 + --tokenizer-mode deepseek_v4 + --tool-call-parser deepseek_v4 + --enable-auto-tool-choice + --reasoning-parser deepseek_v4 + --disable-custom-all-reduce + --enable-cumem-allocator + --no-disable-hybrid-kv-cache-manager + --no-enable-flashinfer-autotune + --compilation-config {"cudagraph_mode":"FULL_DECODE_ONLY","mode":0} + --speculative-config {"method":"dspark","num_speculative_tokens":5,"enable_adaptive_verification":false,"draft_sample_method":"probabilistic","attention_backend":"FLASHINFER_MLA_SPARSE_DSV4"} + --attention-config {"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true} + env: + VLLM_USE_RUST_FRONTEND: "1" + VLLM_SERVER_DEV_MODE: "1" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_RPC_TIMEOUT: "600000" + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" + VLLM_DSV4_MEGA_FP8_COMBINE: "1" + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" + VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: "1024" + TILELANG_CLEANUP_TEMP_FILES: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_USE_NCCL_SYMM_MEM: "0" + VLLM_ALLREDUCE_USE_SYMM_MEM: "0" + VLLM_ALLREDUCE_USE_FLASHINFER: "1" + VLLM_FLASHINFER_ALLREDUCE_BACKEND: "auto" + NCCL_P2P_LEVEL: "NVL" + TORCH_SYMMMEM: NVSHMEM + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-agentx + UCX_TLS: cuda_copy,cuda_ipc,tcp + UCX_MEMTYPE_CACHE: "n" + UCX_MEMTYPE_REG_WHOLE: "n" + UCX_CUDA_IPC_ENABLE_MNNVL: "y" + NCCL_CUMEM_ENABLE: "1" + NCCL_MNNVL_ENABLE: "1" + NCCL_NVLS_ENABLE: "1" + +slurm: + time_limit: "08:00:00" diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/disagg-gb200-1p1d-dep8-dep8-dspark-agentic.yaml b/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/disagg-gb200-1p1d-dep8-dep8-dspark-agentic.yaml new file mode 100644 index 0000000000..7dcd10af7d --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/llm-d-recipes/agentic/disagg-gb200-1p1d-dep8-dep8-dspark-agentic.yaml @@ -0,0 +1,208 @@ +# DeepSeek-V4-Pro-0813 (DSpark) FP4 GB200, P/D disagg 1P-DEP8/1D-DEP8 (4 nodes). +# Always uses Mooncake (P2PHANDSHAKE embedded RDMA prefix-cache). +# DSpark speculative decoding: prefill 1 token, decode 3 tokens (adaptive verification). +# Requires EPP/pd-sidecar v0.10.0 (disagg-profile-handler deciders shape). +apiVersion: llm-d.ai/v1alpha1 +kind: EndpointPickerConfig + +plugins: + - name: file-disc + type: file-discovery + parameters: + path: /tmp/endpoints.yaml + watchFile: false + + - type: prefill-filter + - type: decode-filter + - type: inflight-load-producer + - type: approx-prefix-cache-producer + parameters: + autoTune: false + blockSizeTokens: 256 + maxPrefixTokensToMatch: 1048576 + maxPrefixBlocksToMatch: 4096 + lruCapacityPerServer: 43949 + - type: prefix-cache-affinity-filter + parameters: + peakPrefillThroughput: 4783 + maxTTFTPenaltyMs: 30000 + - type: prefix-cache-scorer + - type: token-load-scorer + parameters: + queueThresholdTokens: 3000000 + - type: active-request-scorer + - type: queue-scorer + - type: always-disagg-pd-decider + - type: disagg-profile-handler + parameters: + deciders: + prefill: always-disagg-pd-decider + - type: max-score-picker + name: prefill-picker + - type: max-score-picker + name: decode-picker + +schedulingProfiles: + - name: prefill + plugins: + - pluginRef: prefill-filter + - pluginRef: prefix-cache-affinity-filter + - pluginRef: prefix-cache-scorer + weight: 6 + - pluginRef: token-load-scorer + weight: 3 + - pluginRef: queue-scorer + weight: 3 + - pluginRef: prefill-picker + - name: decode + plugins: + - pluginRef: decode-filter + - pluginRef: active-request-scorer + - pluginRef: decode-picker + +dataLayer: + discovery: + pluginRef: file-disc + +prefill: + tp: 1 + enable-expert-parallel: true + extra-args: >- + --kv-cache-dtype fp8 + --gpu-memory-utilization 0.95 + --max-num-batched-tokens 8192 + --long-prefill-token-threshold 1024 + --max-num-seqs 16 + --max-model-len 1048576 + --speculative-config {"method":"dspark","num_speculative_tokens":1,"draft_sample_method":"probabilistic","attention_backend":"FLASHINFER_MLA_SPARSE_DSV4","enable_adaptive_verification":true} + --enable-cumem-allocator + --block-size 256 + --tokenizer-mode deepseek_v4 + --tool-call-parser deepseek_v4 + --enable-auto-tool-choice + --reasoning-parser deepseek_v4 + --moe-backend deep_gemm_mega_moe + --enable-ep-weight-filter + --no-disable-hybrid-kv-cache-manager + --no-enable-flashinfer-autotune + --numa-bind + --attention-config {"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true} + env: + PYTHONHASHSEED: "0" + VLLM_USE_RUST_FRONTEND: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_SERVER_DEV_MODE: "1" + VLLM_USE_NCCL_SYMM_MEM: "1" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" + VLLM_RPC_TIMEOUT: "600000" + VLLM_LOG_STATS_INTERVAL: "1" + VLLM_DSV4_MEGA_FP8_COMBINE: "1" + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" + VLLM_HTTP_TIMEOUT_KEEP_ALIVE: "120" + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" + VLLM_USE_DEEP_GEMM: "1" + VLLM_CONNECTOR_PREFETCH_DEPTH: "8" + VLLM_CONNECTOR_PREFETCH_KV_CAP: "0.65" + VLLM_USE_BREAKABLE_CUDAGRAPH: "0" + FLASH_ATTENTION_CUTE_DSL_CACHE_ENABLED: "1" + VLLM_NO_USAGE_STATS: "1" + VLLM_LOGGING_LEVEL: INFO + TQDM_DISABLE: "1" + TILELANG_CLEANUP_TEMP_FILES: "1" + TORCH_SYMMMEM: NVSHMEM + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-agentx + NVSHMEM_REMOTE_TRANSPORT: "none" + NVSHMEM_IB_ENABLE_IBGDA: "false" + NVSHMEM_CUMEM_HANDLE_TYPE: FABRIC + NVSHMEM_DISABLE_CUDA_VMM: "0" + UCX_TLS: cuda_copy,cuda_ipc,rc,tcp + UCX_CUDA_IPC_ENABLE_MNNVL: "y" + UCX_MEMTYPE_CACHE: "n" + NCCL_CUMEM_ENABLE: "1" + NCCL_MNNVL_ENABLE: "1" + NCCL_NVLS_ENABLE: "1" + NCCL_P2P_LEVEL: NVL + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" + VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" + MC_ENABLE_DEST_DEVICE_AFFINITY: "1" + MC_STORE_RPC_CLIENT_IO_THREADS: "32" + MC_TE_RPC_CLIENT_IO_THREADS: "32" + MC_STORE_CLIENT_METRIC: "1" + MC_STORE_CLIENT_METRIC_INTERVAL: "5" + MC_TE_METRIC: "0" + +decode: + tp: 1 + enable-expert-parallel: true + extra-args: >- + --kv-cache-dtype fp8 + --gpu-memory-utilization 0.95 + --max-num-batched-tokens 256 + --max-num-seqs 32 + --max-cudagraph-capture-size 256 + --max-model-len 1048576 + --speculative-config {"method":"dspark","num_speculative_tokens":3,"draft_sample_method":"probabilistic","attention_backend":"FLASHINFER_MLA_SPARSE_DSV4","enable_adaptive_verification":true} + --enable-cumem-allocator + --block-size 256 + --compilation-config {"cudagraph_mode":"FULL_DECODE_ONLY","mode":0} + --tokenizer-mode deepseek_v4 + --tool-call-parser deepseek_v4 + --enable-auto-tool-choice + --reasoning-parser deepseek_v4 + --moe-backend deep_gemm_mega_moe + --enable-ep-weight-filter + --no-disable-hybrid-kv-cache-manager + --no-enable-flashinfer-autotune + --numa-bind + --attention-config {"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true} + env: + PYTHONHASHSEED: "0" + VLLM_USE_RUST_FRONTEND: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_SERVER_DEV_MODE: "1" + VLLM_USE_NCCL_SYMM_MEM: "1" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" + VLLM_RPC_TIMEOUT: "600000" + VLLM_LOG_STATS_INTERVAL: "1" + VLLM_DSV4_MEGA_FP8_COMBINE: "1" + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768" + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" + VLLM_USE_DEEP_GEMM: "1" + VLLM_CONNECTOR_PREFETCH_DEPTH: "8" + VLLM_NO_USAGE_STATS: "1" + VLLM_LOGGING_LEVEL: INFO + TQDM_DISABLE: "1" + TILELANG_CLEANUP_TEMP_FILES: "1" + TORCH_SYMMMEM: NVSHMEM + DG_JIT_CACHE_DIR: /tmp/dg-cache-dsv4-agentx + NVSHMEM_REMOTE_TRANSPORT: "none" + NVSHMEM_IB_ENABLE_IBGDA: "false" + NVSHMEM_CUMEM_HANDLE_TYPE: FABRIC + NVSHMEM_DISABLE_CUDA_VMM: "0" + UCX_TLS: cuda_copy,cuda_ipc,rc,tcp + UCX_CUDA_IPC_ENABLE_MNNVL: "y" + UCX_MEMTYPE_CACHE: "n" + NCCL_CUMEM_ENABLE: "1" + NCCL_MNNVL_ENABLE: "1" + NCCL_NVLS_ENABLE: "1" + NCCL_P2P_LEVEL: NVL + TORCH_DISTRIBUTED_DEFAULT_TIMEOUT: "1800" + VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" + MC_ENABLE_DEST_DEVICE_AFFINITY: "1" + MC_STORE_CLIENT_METRIC: "1" + MC_STORE_CLIENT_METRIC_INTERVAL: "5" + MC_TE_METRIC: "0" + +mooncake: + store_config: + metadata_server: "P2PHANDSHAKE" + local_buffer_size: "4GB" + protocol: "rdma" + device_name: "mlx5_0,mlx5_1,mlx5_3,mlx5_4" + mode: "embedded" + enable_offload: false # SSD only; the embedded DRAM pool is still enabled. + +slurm: + time_limit: "08:00:00" diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/README.md b/inferencex-e2e/benchmarks/multi_node/llm-d/README.md index 81dbd51995..37180d4bf6 100644 --- a/inferencex-e2e/benchmarks/multi_node/llm-d/README.md +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/README.md @@ -25,6 +25,17 @@ the coordinator (EPP + Envoy + bench), exactly like the AMD path's | `xP` | decode leader + pd-sidecar + EPP + Envoy + benchmark client | | `xP+1 .. xP+yD-1` | decode workers | +### Aggregated mode (`yD = 0`) + +`DECODE_NODES=0` runs one engine for prefill and decode. Rank 0 runs EPP, +Envoy, and the client; no P/D sidecar is started. Only the Mooncake variant +uses a KV connector. The master config uses `disagg: false`, `worker`, and +`num-nodes`; discovery labels its serving endpoints `prefill`. + +The GB200 AgentX recipes live under `llm-d-recipes/agentic/`. TP8/DEP8 uses +2 nodes (8 GPUs); 1P-DEP8/1D-DEP8 uses 4 nodes (16 GPUs). Pure TP publishes +only its leader's API endpoint; DEP publishes an endpoint on every node. + Each instance (prefill or decode) is one vLLM engine spanning multiple nodes via `--data-parallel-hybrid-lb`. With `xP=2, yD=2, GPUS_PER_NODE=8` you get DP=16 prefill + DP=16 decode (the wide-EP diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/agentic.sh b/inferencex-e2e/benchmarks/multi_node/llm-d/agentic.sh new file mode 100644 index 0000000000..1dfa04d91e --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/agentic.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +# Client-only AgentX adapter for an already-ready llm-d Envoy frontend. +set -eo pipefail + +: "${INFMAX_CONTAINER_WORKSPACE:?Set the repository mount path}" +export MODEL="$MODEL_NAME" +export SERVED_MODEL_NAME="$MODEL_NAME" +export PORT="$VLLM_PORT" +export AIPERF_SERVER_URL="http://localhost:$ENVOY_PORT" +export RESULT_DIR="$BENCHMARK_LOGS_DIR/agentic" +export AGENTIC_OUTPUT_DIR="$BENCHMARK_LOGS_DIR" +export CONC_LIST="${BENCH_MAX_CONCURRENCY//x/ }" +export CONC="${CONC_LIST%% *}" + +# Use discovery's serving nodes, but scrape vLLM rather than the decode sidecar. +mkdir -p "$RESULT_DIR" +AIPERF_METRIC_URLS=$(python3 - "$LLMD_ENDPOINTS_FILE" "$VLLM_PORT" \ + "$RESULT_DIR/llmd_metrics_endpoints.json" "$DECODE_NODES" "${SIDECAR_PORT:-8000}" <<'PY' +import json +import sys +import yaml + +with open(sys.argv[1]) as source: + endpoints = yaml.safe_load(source)["endpoints"] +vllm_base = int(sys.argv[2]) +sidecar_base = int(sys.argv[5]) + +def vllm_metrics_port(endpoint): + role = endpoint["labels"]["llm-d.ai/role"] + endpoint_port = int(endpoint["port"]) + if role == "decode": + return vllm_base + (endpoint_port - sidecar_base) + return endpoint_port + +metrics_endpoints = { + f"http://{endpoint['address']}:{vllm_metrics_port(endpoint)}/metrics": { + "name": endpoint["name"], + "role": endpoint["labels"]["llm-d.ai/role"] if int(sys.argv[4]) else "combined", + } + for endpoint in endpoints +} +if not metrics_endpoints: + raise SystemExit("No llm-d serving endpoints available for metrics") +with open(sys.argv[3], "w") as output: + json.dump(metrics_endpoints, output, indent=2) + output.write("\n") +print(",".join(metrics_endpoints)) +PY +) +export AIPERF_METRIC_URLS +# benchmark_lib.sh forwards this name to AIPerf's --server-metrics argument. +export AIPERF_SERVER_METRICS_URLS="$AIPERF_METRIC_URLS" +export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="vllm:" + +# AIPerf also probes the inference URL; it must not return load-balanced counters. +frontend_metrics_status=$(curl --silent --show-error --connect-timeout 5 --max-time 10 \ + --output /dev/null --write-out '%{http_code}' "$AIPERF_SERVER_URL/metrics") +if [[ "$frontend_metrics_status" != "404" ]]; then + echo "ERROR: llm-d frontend /metrics must return 404, got $frontend_metrics_status" >&2 + exit 1 +fi + +IFS=',' read -r -a metrics_urls <<< "$AIPERF_METRIC_URLS" +metrics_probe=$(mktemp /tmp/llmd-metrics.XXXXXX) +trap 'rm -f "$metrics_probe"' EXIT +for metrics_url in "${metrics_urls[@]}"; do + curl --fail --silent --show-error --connect-timeout 5 --max-time 10 \ + --retry 6 --retry-connrefused --retry-delay 2 \ + "$metrics_url" --output "$metrics_probe" + if ! grep -q '^vllm:' "$metrics_probe"; then + echo "ERROR: no vLLM metrics exposed at $metrics_url" >&2 + exit 1 + fi + echo "vLLM metrics ready: $metrics_url" +done +rm -f "$metrics_probe" +trap - EXIT + +exec bash "$INFMAX_CONTAINER_WORKSPACE/benchmarks/srt_agentic.sh" diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm b/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm index fe36301b6a..3f1e3d0f28 100644 --- a/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm @@ -48,8 +48,13 @@ fi # prefill workers = ranks 1 .. PREFILL_NODES-1 # decode leader = rank PREFILL_NODES (also coordinator: EPP + Envoy + bench) # decode workers = ranks PREFILL_NODES+1 .. NUM_NODES-1 +# Aggregated mode has no decode address. PREFILL_LEADER_IP="${IPS[0]}" -DECODE_LEADER_IP="${IPS[$PREFILL_NODES]}" +if [[ "$DECODE_NODES" -gt 0 ]]; then + DECODE_LEADER_IP="${IPS[$PREFILL_NODES]}" +else + DECODE_LEADER_IP="" +fi # DP leader addresses for vLLM --data-parallel-address (rank 0 of each instance). PREFILL_DP_ADDR="$PREFILL_LEADER_IP" @@ -62,10 +67,11 @@ DOCKER_CONT_NAME="llmd_bench_${SANITIZED_USER}_${SLURM_JOB_ID}" export DOCKER_CONT_NAME : "${BENCHMARK_LOGS_DIR:?BENCHMARK_LOGS_DIR not set}" DOCKER_MOUNT_PATH="/workspace" +export INFMAX_CONTAINER_WORKSPACE="$DOCKER_MOUNT_PATH" cleanup() { echo "[${SLURM_JOB_ID}] cleanup on $(hostname)" - [[ -n "${WATCHER_PID:-}" ]] && kill "$WATCHER_PID" 2>/dev/null || true + [[ -n "${WATCHER_PID}" ]] && kill "$WATCHER_PID" 2>/dev/null || true } trap cleanup INT TERM HUP EXIT @@ -100,6 +106,27 @@ for setting in $INFERENCEX_RUNTIME_ENV_VARS; do RUNTIME_DOCKER_ENV+=(-e "$setting") done +# Preserve the workflow's AgentX protocol, provenance, offload and physical GPU +# metadata on both engines. Forward names, not interpolated values: JSON and +# HF_TOKEN must never be embedded in the nested shell command or printed. +AGENTIC_ENV_NAMES=( + INFMAX_CONTAINER_WORKSPACE + IS_AGENTIC SCENARIO_TYPE CONC CONC_LIST DURATION AIPERF_EXPERIMENTAL_FAST + IMAGE RECIPE_FINGERPRINT DISAGG HF_TOKEN + KV_OFFLOADING KV_OFFLOAD_BACKEND KV_OFFLOAD_BACKEND_METADATA + ROUTER_METADATA KV_P2P_TRANSFER TOTAL_CPU_DRAM_GB + PREFILL_NUM_WORKERS PREFILL_TP PREFILL_PP_SIZE PREFILL_PCP_SIZE + PREFILL_DCP_SIZE PREFILL_EP PREFILL_DP_ATTN + DECODE_NUM_WORKERS DECODE_TP DECODE_PP_SIZE DECODE_PCP_SIZE + DECODE_DCP_SIZE DECODE_EP DECODE_DP_ATTN +) +AGENTIC_DOCKER_ENV="" +for env_name in "${AGENTIC_ENV_NAMES[@]}"; do + export "$env_name" + AGENTIC_DOCKER_ENV+=" -e $env_name" +done +AGENTIC_PYXIS_ENV=$(IFS=,; echo "${AGENTIC_ENV_NAMES[*]}") + if [[ "$LLMD_CONTAINER_ENGINE" == "docker" ]]; then # One docker run per node, one task per node. server.sh dispatches by NODE_RANK. srun \ @@ -178,12 +205,14 @@ exec docker run --rm \ -e CONFIG_FILE=$CONFIG_FILE \ ${RUNTIME_DOCKER_ENV[*]} \ -e VLLM_RANDOMIZE_DP_DUMMY_INPUTS=$VLLM_RANDOMIZE_DP_DUMMY_INPUTS \ + -e VLLM_ENGINE_READY_TIMEOUT_S=$VLLM_ENGINE_READY_TIMEOUT_S \ -e VLLM_LOGGING_LEVEL=$VLLM_LOGGING_LEVEL \ -e UCX_TLS=$UCX_TLS \ -e NVSHMEM_REMOTE_TRANSPORT=$NVSHMEM_REMOTE_TRANSPORT \ -e NVSHMEM_IB_ENABLE_IBGDA=$NVSHMEM_IB_ENABLE_IBGDA \ -e NVSHMEM_SYMMETRIC_SIZE=$NVSHMEM_SYMMETRIC_SIZE \ -e LLMD_API_SERVER_COUNT=$LLMD_API_SERVER_COUNT \ + $AGENTIC_DOCKER_ENV \ --name \"${DOCKER_CONT_NAME}_\$SLURM_PROCID\" \ \"\$DOCKER_IMAGE_NAME\" -lc ' set -o pipefail @@ -232,6 +261,7 @@ elif [[ "$LLMD_CONTAINER_ENGINE" == "pyxis" ]]; then PYXIS_ENV_LIST+=",VLLM_RANDOMIZE_DP_DUMMY_INPUTS,VLLM_ENGINE_READY_TIMEOUT_S,VLLM_LOGGING_LEVEL,UCX_TLS,NVSHMEM_REMOTE_TRANSPORT,NVSHMEM_IB_ENABLE_IBGDA,NVSHMEM_SYMMETRIC_SIZE,LLMD_API_SERVER_COUNT" PYXIS_ENV_LIST+=",${INFERENCEX_RUNTIME_ENV_VARS// /,}" + PYXIS_ENV_LIST+=",$AGENTIC_PYXIS_ENV" PYXIS_MOUNTS="${MODEL_DIR}:/models:ro" PYXIS_MOUNTS+=",${BENCHMARK_LOGS_DIR}:/benchmark_logs" @@ -240,23 +270,15 @@ elif [[ "$LLMD_CONTAINER_ENGINE" == "pyxis" ]]; then PYXIS_MOUNTS+=",${DI_REPO_DIR}/benchmarks/llm-d/epp-config.yaml:/etc/epp/config.yaml:ro" PYXIS_MOUNTS+=",${DI_REPO_DIR}/benchmarks/llm-d/envoy.yaml:/etc/envoy/envoy.yaml:ro" - # Optional: mount the epp / pd-sidecar / envoy binaries from a shared - # filesystem instead of relying on them being baked into the image. - # This lets a STOCK vllm/vllm-openai image be used directly (no - # combined-image rebuild per vLLM version bump) - see - # benchmarks/llm-d/binaries.env + extract-binaries.sh. Each mount is - # gated on the file existing, so this is a no-op when the binaries - # have not been extracted (the baked-image path keeps working), and - # harmless when they have (mounting a binary over the identical one). - # shellcheck source=/dev/null - [[ -f "${DI_REPO_DIR}/benchmarks/llm-d/binaries.env" ]] && \ + # AgentX uses the image-bundled router; legacy runs can mount extracted binaries. + if [[ "$IS_AGENTIC" != "1" ]]; then source "${DI_REPO_DIR}/benchmarks/llm-d/binaries.env" - for _bin in epp pd-sidecar envoy; do - if [[ -n "${LLMD_BIN_DIR:-}" && -x "${LLMD_BIN_DIR}/${_bin}" ]]; then - PYXIS_MOUNTS+=",${LLMD_BIN_DIR}/${_bin}:/usr/local/bin/${_bin}:ro" - echo "Mounting ${LLMD_BIN_DIR}/${_bin} -> /usr/local/bin/${_bin}" - fi - done + for _bin in epp pd-sidecar envoy; do + if [[ -x "${LLMD_BIN_DIR}/${_bin}" ]]; then + PYXIS_MOUNTS+=",${LLMD_BIN_DIR}/${_bin}:/usr/local/bin/${_bin}:ro" + fi + done + fi # MODEL_DIR / BENCHMARK_LOGS_DIR / NODE_RANK are translated to their # in-container values inside bash -lc (host MODEL_DIR is the source diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/recipe.py b/inferencex-e2e/benchmarks/multi_node/llm-d/recipe.py new file mode 100644 index 0000000000..3f2a4fab67 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/recipe.py @@ -0,0 +1,115 @@ +"""Render llm-d role arguments and enforce AgentX benchmark metadata.""" + +import argparse +import json +import os +from pathlib import Path +import re +import shlex +import sys + +import yaml + +from infx.golden_al_distribution import golden_length + + +def validate_agentic_offload(recipe: dict, env: dict) -> None: + """An embedded Mooncake store is DRAM offload even with SSD disabled.""" + if env.get("IS_AGENTIC") != "1": + return + store = recipe.get("mooncake", {}).get("store_config") + expected = "dram" if store else "none" + if env.get("KV_OFFLOADING") != expected: + raise ValueError(f"Recipe requires KV_OFFLOADING={expected}; fix the master YAML") + if store and env.get("KV_OFFLOAD_BACKEND") != "mooncake": + raise ValueError("Mooncake recipe requires KV_OFFLOAD_BACKEND=mooncake") + + +def role_assignments(recipe: dict, role: str, env: dict) -> str: + validate_agentic_offload(recipe, env) + section = recipe.get(role) or {} + extra = (section.get("extra-args") or "").strip() + if env.get("IS_AGENTIC") == "1": + match = re.search(r"--speculative-config\s+", extra) + if match: + config, length = json.JSONDecoder().raw_decode(extra[match.end():]) + if config.get("method") == "dspark": + if env.get("SPEC_DECODING") != "mtp": + raise ValueError("DSpark requires SPEC_DECODING=mtp in the master YAML") + if env.get("EVAL_ONLY") == "true": + config.pop("synthetic_acceptance_length", None) + config.pop("rejection_sample_method", None) + elif config.get("enable_adaptive_verification"): + config.pop("synthetic_acceptance_length", None) + config.pop("rejection_sample_method", None) + print( + f"DSpark {role}: K={config['num_speculative_tokens']}, adaptive verification", + file=sys.stderr, + ) + elif config.get("rejection_sample_method") == "block": + config.pop("synthetic_acceptance_length", None) + print( + f"DSpark {role}: K={config['num_speculative_tokens']}, real verification (block)", + file=sys.stderr, + ) + else: + if env.get("RUN_EVAL") == "true": + raise ValueError("Run accuracy evals separately with EVAL_ONLY=true, not synthetic AL") + model_prefix = env.get("MODEL_PREFIX") + thinking = env.get("THINKING_MODE", "thinking_on") + if not model_prefix: + raise ValueError("Missing MODEL_PREFIX for DSpark golden AL lookup") + k = config["num_speculative_tokens"] + al = golden_length(model_prefix, config, thinking) + config.update( + rejection_sample_method="synthetic", + synthetic_acceptance_length=al, + ) + print( + f"DSpark {role}: K={k}, golden AL={al} (thinking={thinking})", + file=sys.stderr, + ) + extra = extra[:match.end()] + json.dumps(config, separators=(",", ":")) + extra[match.end() + length:] + assignments = [f"ROLE_EXTRA_ARGS={shlex.quote(extra)}", + f"PREFILL_ENABLE_EP={str(recipe.get('prefill', {}).get('enable-expert-parallel', True)).lower()}"] + if section.get("tp") is not None: + assignments.append(f"TP_SIZE={int(section['tp'])}") + if section.get("enable-expert-parallel") is not None: + assignments.append(f"ROLE_ENABLE_EP={str(section['enable-expert-parallel']).lower()}") + for key, value in (section.get("env") or {}).items(): + if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", key): + raise ValueError(f"Invalid recipe environment variable: {key}") + assignments.append(f"export {key}={shlex.quote(str(value))}") + return "\n".join(assignments) + + +def mooncake_config(recipe: dict, env: dict) -> str: + validate_agentic_offload(recipe, env) + config = dict(recipe.get("mooncake", {}).get("store_config") or {}) + if not config: + return "" + config["master_server_address"] = f"{env['ALL_IPS'].split(',')[0]}:50051" + if env.get("IS_AGENTIC") == "1": + budget_gb = int(env["TOTAL_CPU_DRAM_GB"]) + gpus_per_node = int(env["GPUS_PER_NODE"]) + if budget_gb <= 0 or gpus_per_node <= 0: + raise ValueError("Mooncake requires a positive per-node DRAM budget and GPU count") + # The master budget is per node. Each embedded per-GPU store owns a + # share; transfer buffers are separate from the reusable KV pool. + config["global_segment_size"] = budget_gb * 10**9 // gpus_per_node + return json.dumps(config) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("recipe", type=Path) + output = parser.add_mutually_exclusive_group(required=True) + output.add_argument("--role", choices=("prefill", "decode")) + output.add_argument("--mooncake", action="store_true") + args = parser.parse_args() + recipe = yaml.safe_load(args.recipe.read_text()) + print(mooncake_config(recipe, os.environ) if args.mooncake else role_assignments(recipe, args.role, os.environ)) + + +if __name__ == "__main__": + main() diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh b/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh index a4e8a94047..1be77d6b9c 100755 --- a/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh @@ -20,7 +20,8 @@ source /workspace/benchmarks/benchmark_lib.sh check_env_vars \ NODE_RANK PREFILL_NODES DECODE_NODES GPUS_PER_NODE PREFILL_WORKERS \ - DECODE_WORKERS EVAL_ONLY RUN_EVAL + DECODE_WORKERS EVAL_ONLY RUN_EVAL ALL_IPS +IS_AGGREGATED=$(( DECODE_NODES == 0 )) VLLM_PORT=8200 SIDECAR_PORT=8000 ENVOY_PORT=8080 @@ -32,6 +33,9 @@ EPP_METRICS_PORT=9090 # the served-model-name, not a filesystem path. MODEL="${MODEL_DIR}" +# ---------------------------------------------------------------- +# Host IP + default interface +# ---------------------------------------------------------------- # Resolved without iproute2 (`ip` is absent on the arm64 vLLM base); python3's # socket layer exposes the kernel's source-IP / iface choice. _HOST_INFO=$(python3 -c ' @@ -65,6 +69,9 @@ ENVOY_LOG="/benchmark_logs/envoy.log" echo "=== rank=$NODE_RANK host=$HOST_IP model=$MODEL ===" +# ---------------------------------------------------------------- +# Role + topology (Option B engine grouping) +# ---------------------------------------------------------------- # A role's nodes split into PREFILL_WORKERS / DECODE_WORKERS independent DP/EP # engines, each spanning (role_nodes / role_workers) nodes with its own DP # coordinator (leader IP) and rank range. workers=1 => one engine over all role @@ -92,8 +99,7 @@ else exit 1 fi -# Each engine's DP coordinator = its leader node's IP (ALL_IPS[leader rank]); -# fall back to the role leader env when ALL_IPS is unset. +# Each engine's DP coordinator = its leader node's IP; fall back to role leaders. if [[ -n "${_ALL_IPS[${_group_leader_rank}]:-}" ]]; then DP_ADDR="${_ALL_IPS[${_group_leader_rank}]}" elif [[ "$ROLE" == "prefill" ]]; then @@ -108,45 +114,15 @@ START_RANK=$((LWS_WORKER_INDEX * DP_SIZE_LOCAL)) # Defaults: TP=1, DP=role_total, EP on (the H200 1P+1D shape). Recipe overrides below. TP_SIZE=1 ROLE_ENABLE_EP=true +PREFILL_ENABLE_EP=true echo "ROLE=$ROLE DP_SIZE=$DP_SIZE DP_ADDR=$DP_ADDR LWS_WORKER_INDEX=$LWS_WORKER_INDEX START_RANK=$START_RANK" -# Per-role keys: tp (int -> --tensor-parallel-size), enable-expert-parallel -# (bool -> --enable-expert-parallel + DP/wide-EP knobs), extra-args (appended -# verbatim), env (map, exported before vllm serve). Absent keys keep the -# defaults above, so a recipe with neither tp nor EP is a plain TP=1 DP+EP run. -ROLE_EXTRA_ARGS="" -if [[ -n "${CONFIG_FILE:-}" ]]; then - RECIPE_PATH="/etc/llmd-recipes/${CONFIG_FILE}" - if [[ -f "$RECIPE_PATH" ]]; then - echo "Loading $ROLE recipe from $RECIPE_PATH" - eval "$(python3 - <&2 - fi -fi -echo "Resolved $ROLE TP_SIZE=$TP_SIZE ROLE_ENABLE_EP=$ROLE_ENABLE_EP" - export GLOO_SOCKET_IFNAME=${GLOO_SOCKET_IFNAME:-$DEFAULT_IFACE} export NCCL_SOCKET_IFNAME=${NCCL_SOCKET_IFNAME:-$DEFAULT_IFACE} check_env_vars \ - VLLM_RANDOMIZE_DP_DUMMY_INPUTS VLLM_ENGINE_READY_TIMEOUT_S VLLM_LOGGING_LEVEL UCX_TLS NVSHMEM_REMOTE_TRANSPORT \ - NVSHMEM_IB_ENABLE_IBGDA NVSHMEM_SYMMETRIC_SIZE LLMD_API_SERVER_COUNT + VLLM_RANDOMIZE_DP_DUMMY_INPUTS VLLM_ENGINE_READY_TIMEOUT_S VLLM_LOGGING_LEVEL UCX_TLS \ + NVSHMEM_REMOTE_TRANSPORT NVSHMEM_IB_ENABLE_IBGDA NVSHMEM_SYMMETRIC_SIZE export VLLM_SKIP_P2P_CHECK=1 # Randomized DP dummy inputs make idle DP ranks fan their lockstep dummy passes # across all experts (full MoE all-to-all), wasting prefill bandwidth; a recipe @@ -172,8 +148,7 @@ export VLLM_LOGGING_LEVEL # exposes /dev/infiniband + IPC_LOCK); cuda_copy/cuda_ipc cover intra-node. export UCX_TLS -# Single-node-per-role recipes avoid DeepEP / NVSHMEM ibgda, so leave these off -# there to avoid triggering ibgda code paths that are not needed. + if [[ "$LWS_GROUP_SIZE" -gt 1 ]]; then export NVIDIA_GDRCOPY=enabled # ibgda default kept for future DeepEP/wide-EP recipes; a recipe may override @@ -189,34 +164,110 @@ if [[ "$LWS_GROUP_SIZE" -gt 1 ]]; then fi fi -if [[ -n "${KV_ROLE_OVERRIDE:-}" ]]; then - KV_ROLE="$KV_ROLE_OVERRIDE" -elif [[ "$ROLE" == "prefill" ]]; then - KV_ROLE="kv_producer" -else - KV_ROLE="kv_consumer" +# ---------------------------------------------------------------- +# Recipe: per-role serve args + env (/etc/llmd-recipes/$CONFIG_FILE) +# ---------------------------------------------------------------- +# Per-role keys: tp (int -> --tensor-parallel-size), enable-expert-parallel +# (bool -> --enable-expert-parallel + DP/wide-EP knobs), extra-args (appended +# verbatim), env (map, exported before vllm serve). Absent keys keep the +# defaults above, so a recipe with neither tp nor EP is a plain TP=1 DP+EP run. +ROLE_EXTRA_ARGS="" +if [[ -n "${CONFIG_FILE}" ]]; then + RECIPE_PATH="/etc/llmd-recipes/${CONFIG_FILE}" + if [[ -f "$RECIPE_PATH" ]]; then + echo "Loading $ROLE recipe from $RECIPE_PATH" + # Keep command substitution separate from eval so renderer failures + # (missing golden AL or incorrect offload metadata) stop server startup. + ROLE_ASSIGNMENTS=$(PYTHONPATH="$INFERENCEX_REPO_ROOT${PYTHONPATH:+:$PYTHONPATH}" \ + python3 /workspace/benchmarks/multi_node/llm-d/recipe.py \ + "$RECIPE_PATH" --role "$ROLE") + eval "$ROLE_ASSIGNMENTS" + else + if [[ "${IS_AGENTIC}" == "1" ]]; then + echo "ERROR: AgentX recipe not found: $RECIPE_PATH" >&2 + exit 1 + fi + echo "WARNING: CONFIG_FILE=$CONFIG_FILE but $RECIPE_PATH not found; using defaults" >&2 + fi fi -KV_TRANSFER_CONFIG="{\"kv_connector\":\"NixlConnector\",\"kv_role\":\"$KV_ROLE\",\"kv_load_failure_policy\":\"fail\"}" +echo "Resolved $ROLE TP_SIZE=$TP_SIZE ROLE_ENABLE_EP=$ROLE_ENABLE_EP" +# ---------------------------------------------------------------- +# Mooncake KV store (optional, from recipe top-level `mooncake:` key) +# ---------------------------------------------------------------- +# Embedded stores contribute per-rank DRAM to one job-local Mooncake master. +MOONCAKE_CONFIG_PATH="" +if [[ -n "${CONFIG_FILE}" && -f "/etc/llmd-recipes/${CONFIG_FILE}" ]]; then + _MC_JSON=$(PYTHONPATH="$INFERENCEX_REPO_ROOT${PYTHONPATH:+:$PYTHONPATH}" \ + python3 /workspace/benchmarks/multi_node/llm-d/recipe.py \ + "/etc/llmd-recipes/${CONFIG_FILE}" --mooncake) + if [[ -n "$_MC_JSON" ]]; then + echo "$_MC_JSON" > /tmp/mooncake_config.json + MOONCAKE_CONFIG_PATH=/tmp/mooncake_config.json + export MOONCAKE_CONFIG_PATH + echo "Mooncake enabled: config at $MOONCAKE_CONFIG_PATH" + if [[ "$NODE_RANK" -eq 0 ]]; then + mooncake_master --rpc_port=50051 --metrics_port=50052 \ + > "$BENCHMARK_LOGS_DIR/mooncake_master.log" 2>&1 & + fi + curl --fail --silent --show-error --connect-timeout 5 --max-time 10 \ + --retry 30 --retry-connrefused --retry-delay 1 \ + "http://${_ALL_IPS[0]}:50052/metrics" > /dev/null + fi +fi + +# ---------------------------------------------------------------- +# Bring up vLLM engine (every node) +# ---------------------------------------------------------------- COMMON_ARGS=( + --host 0.0.0.0 --port "$VLLM_PORT" --served-model-name "$MODEL_NAME" --trust-remote-code --disable-access-log-for-endpoints=/health,/metrics --tensor-parallel-size "$TP_SIZE" - --kv_transfer_config "$KV_TRANSFER_CONFIG" ) -# One frontend (HTTP + tokenize + DP load-balance) is CPU-bound and caps throughput, -# so run several (LLMD_API_SERVER_COUNT). Incompatible with --headless, so the -# headless-worker branch below drops it. With --data-parallel-hybrid-lb each node's -# api-server balances its local DP ranks, so VLLM_PORT is also the health port. +# Aggregated engines need KV transfer only when Mooncake is enabled. +if [[ "$IS_AGGREGATED" -eq 0 ]]; then + if [[ -n "${KV_ROLE_OVERRIDE}" ]]; then + KV_ROLE="$KV_ROLE_OVERRIDE" + elif [[ "$ROLE" == "prefill" ]]; then + KV_ROLE="kv_producer" + else + KV_ROLE="kv_consumer" + fi + if [[ -n "${MOONCAKE_CONFIG_PATH}" ]]; then + # MultiConnector on prefill: NixlConnector handles direct P/D KV transfer; + # MooncakeStoreConnector enables cross-node prefix-cache lookup via RDMA. + # Decode uses NixlConnector only (matches agentX v13): Mooncake on decode + # would pollute the prefix cache with non-reusable decode blocks. kv_both + # on decode so it can serve speculative-decode prefills in DSpark. + _MC_EXTRA='"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false' + _NIXL_EXTRA='"enforce_handshake_compat":false,"enable_cross_layers_blocks":false,"kv_lease_duration":1800' + if [[ "$ROLE" == "prefill" ]]; then + KV_TRANSFER_CONFIG="{\"kv_connector\":\"MultiConnector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{\"connectors\":[{\"kv_connector\":\"NixlConnector\",\"kv_role\":\"kv_both\",\"kv_load_failure_policy\":\"fail\",\"kv_buffer_device\":\"cuda\",\"kv_connector_extra_config\":{${_NIXL_EXTRA}}},{\"kv_connector\":\"MooncakeStoreConnector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{${_MC_EXTRA}}}]}}" + else + KV_TRANSFER_CONFIG="{\"kv_connector\":\"NixlConnector\",\"kv_role\":\"kv_both\",\"kv_load_failure_policy\":\"fail\",\"kv_buffer_device\":\"cuda\",\"kv_connector_extra_config\":{${_NIXL_EXTRA}}}" + fi + else + KV_TRANSFER_CONFIG="{\"kv_connector\":\"NixlConnector\",\"kv_role\":\"$KV_ROLE\",\"kv_load_failure_policy\":\"fail\"}" + fi + COMMON_ARGS+=(--kv_transfer_config "$KV_TRANSFER_CONFIG") +elif [[ -n "${MOONCAKE_CONFIG_PATH}" ]]; then + # Aggregated + Mooncake: single role acts as kv_both (stores new KV and + # loads cache hits from the Mooncake RDMA store for prefix-cache sharing). + # SimpleCPUOffloadConnector stages freshly computed KV blocks into CPU DRAM + # (~40 GB) before writing to MooncakeStore, matching the local DEP8 v1 setup. + KV_TRANSFER_CONFIG="{\"kv_connector\":\"MultiConnector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{\"connectors\":[{\"kv_connector\":\"NixlConnector\",\"kv_role\":\"kv_both\",\"kv_load_failure_policy\":\"fail\",\"kv_buffer_device\":\"cuda\",\"kv_connector_extra_config\":{\"enforce_handshake_compat\":false,\"enable_cross_layers_blocks\":false}},{\"kv_connector\":\"SimpleCPUOffloadConnector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{\"cpu_bytes_to_use\":42949672960}},{\"kv_connector\":\"MooncakeStoreConnector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{\"load_async\":true,\"lookup_async\":true,\"enable_cross_layers_blocks\":false,\"enable_offload\":false}}]}}" + COMMON_ARGS+=(--kv_transfer_config "$KV_TRANSFER_CONFIG") +fi +# EP roles use multi-port-external-lb: each local DP rank gets its own serving +# port starting at VLLM_PORT (8200, 8201, ...). The vLLM supervisor binds 8100 +# and serves /health once all engines are ready -> health check uses port 8100. +# Pure-TP roles serve on a single VLLM_PORT with the standard health check. HEALTH_PORT="$VLLM_PORT" -API_SERVER_COUNT="${LLMD_API_SERVER_COUNT}" -# Multiple frontends only help the DP (wide-EP) path. A pure-TP engine has a single -# core with one frontend, so it keeps the default (and avoids --api-server-count -# interacting with the --headless multi-node TP launch below). if [[ "$ROLE_ENABLE_EP" == "true" ]]; then - COMMON_ARGS+=(--api-server-count "$API_SERVER_COUNT") + HEALTH_PORT="8100" fi # Set to 1 by the pure-TP multi-node branch below on --headless followers, which # run no local api-server; gates the post-launch health wait. @@ -230,10 +281,11 @@ if [[ "$ROLE_ENABLE_EP" == "true" ]]; then COMMON_ARGS+=( --enable-expert-parallel --data-parallel-size "$DP_SIZE" + --data-parallel-multi-port-external-lb + --data-parallel-supervisor-port 8100 ) if [[ "$LWS_GROUP_SIZE" -gt 1 ]]; then COMMON_ARGS+=( - --data-parallel-hybrid-lb --data-parallel-size-local "$DP_SIZE_LOCAL" --data-parallel-address "$DP_ADDR" --data-parallel-rpc-port 5555 @@ -273,6 +325,9 @@ else echo "vLLM ready on rank $NODE_RANK ($ROLE worker_index=$LWS_WORKER_INDEX, health port $HEALTH_PORT)" fi +# ---------------------------------------------------------------- +# Bring up pd-sidecar (every decode node) +# ---------------------------------------------------------------- # The sidecar forwards a prefill request, reads kv_transfer_params from vLLM's # response, then hits its local decode vLLM, whose NIXLv2 connector pulls KV # directly from prefill vLLM. @@ -282,11 +337,13 @@ fi # endpoint per node. Pure-TP: only the TP-group leader has an api-server # (followers are --headless), so only the leader runs a sidecar and only leaders # are listed as endpoints. +# if [[ "$ROLE" == "decode" && ( "$ROLE_ENABLE_EP" == "true" || "$LWS_WORKER_INDEX" -eq 0 ) ]]; then SIDECAR_CONNECTOR="nixlv2" SIDECAR_FLAGS=(--port="$SIDECAR_PORT" --vllm-port="$VLLM_PORT" --kv-connector="$SIDECAR_CONNECTOR" --secure-proxy=false - --enable-prefiller-sampling) + --enable-prefiller-sampling + --data-parallel-size="$DP_SIZE_LOCAL") SIDECAR_HEALTH_PORT="$SIDECAR_PORT" echo "Starting pd-sidecar (decode node_rank=$NODE_RANK worker_index=$LWS_WORKER_INDEX): ${SIDECAR_FLAGS[*]}" pd-sidecar "${SIDECAR_FLAGS[@]}" > "$SIDECAR_LOG" 2>&1 & @@ -295,55 +352,51 @@ if [[ "$ROLE" == "decode" && ( "$ROLE_ENABLE_EP" == "true" || "$LWS_WORKER_INDEX echo "pd-sidecar ready on $HOST_IP:$SIDECAR_HEALTH_PORT" fi -# Coordinator (decode leader): endpoints, EPP, Envoy, bench, eval -if [[ "$ROLE" == "decode" && "$LWS_WORKER_INDEX" -eq 0 ]]; then +# ================================================================ +# Coordinator: endpoints, EPP, Envoy, bench, eval +# ================================================================ +# Rank 0 coordinates aggregated runs; the decode leader coordinates P/D runs. +if [[ ( "$ROLE" == "decode" && "$LWS_WORKER_INDEX" -eq 0 ) || \ + ( "$IS_AGGREGATED" -eq 1 && "$ROLE" == "prefill" && "$NODE_RANK" -eq 0 ) ]]; then # Release the allocation whenever the coordinator exits. BENCH_DONE_MARKER="$BENCHMARK_LOGS_DIR/.bench_done.$SLURM_JOB_ID" trap 'touch "$BENCH_DONE_MARKER" 2>/dev/null || true' EXIT - # namespace must match EPP's --pool-namespace (file-discovery filters by it; - # the schema default 'default' would drop every entry). See README.md. + # DEP registers every node; pure TP registers only each engine's API leader. + export LLMD_ENDPOINTS_FILE=/tmp/endpoints.yaml python3 - < DEP8 hybrid-LB (an api-server per node); -# EP off => pure-TP (only each TP-group leader has an api-server). -decode_ep = ('$ROLE_ENABLE_EP' == 'true') -VLLM_PORT = int('$VLLM_PORT') -SIDECAR_PORT = int('$SIDECAR_PORT') -# ALL_IPS is rank-ordered: ranks [0:pn] are prefill nodes, [pn:pn+dn] decode. -prefill_ips = all_ips[:pn] or [os.environ['PREFILL_LEADER_IP']] -decode_ips = all_ips[pn:pn + dn] or [os.environ['DECODE_LEADER_IP']] +ips = os.environ['ALL_IPS'].split(',') +pn = int(os.environ['PREFILL_NODES']) +dn = int(os.environ['DECODE_NODES']) +gpus_per_node = int('$GPUS_PER_NODE') endpoints = [] -def add_role(role, ips, base_port, group_size=1): - # group_size == 1: one endpoint per node (DEP8 hybrid-LB: each node's - # api-server / sidecar load-balances its local DP ranks). - # group_size > 1: one endpoint per TP-group leader (pure-TP: followers are - # --headless with no api-server), i.e. every group_size-th node IP. - serving_ips = ips[::group_size] if group_size > 1 else ips - for i, ip in enumerate(serving_ips): - endpoints.append({'name': f'{role}-{i}', 'namespace': NS, 'address': ip, - 'port': str(base_port), 'labels': {'llm-d.ai/role': role}}) - -# Prefill (DEP8 in every current recipe): one endpoint per node, EPP hits vLLM -# directly (VLLM_PORT). Decode: EPP hits the pd-sidecar (SIDECAR_PORT); one -# endpoint per node for DEP8, or one per TP-group leader for pure-TP. -add_role('prefill', prefill_ips, VLLM_PORT) -decode_group = 1 if decode_ep else max(1, dn // decode_workers) -add_role('decode', decode_ips, SIDECAR_PORT, group_size=decode_group) -yaml.safe_dump({'endpoints': endpoints}, open('/tmp/endpoints.yaml', 'w')) -print(f'endpoints.yaml ({len(endpoints)} endpoints):') -print(open('/tmp/endpoints.yaml').read()) +def add_role(role, addresses, port, group_size, dp_local=1): + idx = 0 + for address in addresses[::group_size]: + for rank in range(dp_local): + endpoints.append({'name': f'{role}-{idx}', 'namespace': 'inferencex', + 'address': address, 'port': str(port + rank), + 'labels': {'llm-d.ai/role': role}}) + idx += 1 + +prefill_ep = '$PREFILL_ENABLE_EP' == 'true' +prefill_group = 1 if prefill_ep else pn // int('$PREFILL_WORKERS') +prefill_dp_local = gpus_per_node if prefill_ep else 1 +add_role('prefill', ips[:pn], int('$VLLM_PORT'), prefill_group, prefill_dp_local) +if dn: + decode_ep = '$ROLE_ENABLE_EP' == 'true' + decode_group = 1 if decode_ep else dn // int('$DECODE_WORKERS') + decode_dp_local = gpus_per_node if decode_ep else 1 + add_role('decode', ips[pn:pn + dn], int('$SIDECAR_PORT'), decode_group, decode_dp_local) +with open(os.environ['LLMD_ENDPOINTS_FILE'], 'w') as output: + yaml.safe_dump({'endpoints': endpoints}, output) +print(yaml.safe_dump({'endpoints': endpoints})) PY - # EPP + # ---- Bring up EPP ---- # Config: when a recipe is set, project it down to the keys EPP's strict # decoder accepts (it rejects the per-role vLLM / slurm keys); else use the # default mounted at /etc/epp/config.yaml. @@ -391,7 +444,7 @@ PY done echo "EPP listening on $EPP_GRPC_PORT" - # Envoy + # ---- Bring up Envoy ---- envoy -c /etc/envoy/envoy.yaml > "$ENVOY_LOG" 2>&1 & ENVOY_PID=$! @@ -415,13 +468,31 @@ PY done echo "Envoy admin ready; listener should be on $ENVOY_PORT" - # Gate on ALL prefill vLLM /health endpoints. Prefill ranks only wait on their own - # local /health, and with PREFILL_WORKERS>1 every prefill node must be probed, not - # just IPS[0]. curl gets explicit connect/max timeouts so a blackholed endpoint - # trips the deadline instead of hanging the run (a timeout-less curl once wedged - # a 2P run for 7h). - _prefill_ips=( "${_ALL_IPS[@]:0:${PREFILL_NODES}}" ) - [[ ${#_prefill_ips[@]} -gt 0 ]] || _prefill_ips=( "$PREFILL_LEADER_IP" ) + # ---- Gate on ALL prefill vLLM /health endpoints (cross-node) ---- + # Prefill ranks wait on their own local /health; wait_for_server_ready only + # probes localhost, so the coordinator polls every prefill node here. + # External LB registers one EPP endpoint per DP rank, so dedupe by node IP. + # EP roles expose /health on the DP supervisor (8100), not the serving port. + # curl gets an explicit connect/max timeout so a blackholed endpoint trips the + # deadline instead of hanging the whole run (a single timeout-less curl once + # wedged a 2P run for 7h before it was cancelled). + mapfile -t _prefill_ips < <(python3 - "$LLMD_ENDPOINTS_FILE" <<'PY' +import sys, yaml +seen = set() +for endpoint in yaml.safe_load(open(sys.argv[1]))['endpoints']: + if endpoint['labels']['llm-d.ai/role'] != 'prefill': + continue + address = endpoint['address'] + if address in seen: + continue + seen.add(address) + print(address) +PY + ) + _PREFILL_HEALTH_PORT="$VLLM_PORT" + if [[ "$PREFILL_ENABLE_EP" == "true" ]]; then + _PREFILL_HEALTH_PORT="8100" + fi # On failure, dump enough to tell a server-not-ready problem (TCP connects but # /health is slow) apart from a network/subnet problem (TCP connect refused or @@ -432,7 +503,7 @@ PY { echo "=== NET DIAG: decode -> prefill ${ip}:${port} ===" echo "[diag] decode node: $(hostname -f 2>/dev/null || hostname) local-ips: $(hostname -I 2>/dev/null)" - echo "[diag] ifaces: DEFAULT_IFACE=${DEFAULT_IFACE:-} NCCL_SOCKET_IFNAME=${NCCL_SOCKET_IFNAME:-} GLOO_SOCKET_IFNAME=${GLOO_SOCKET_IFNAME:-}" + echo "[diag] ifaces: DEFAULT_IFACE=${DEFAULT_IFACE} NCCL_SOCKET_IFNAME=${NCCL_SOCKET_IFNAME} GLOO_SOCKET_IFNAME=${GLOO_SOCKET_IFNAME}" # Local source address the kernel would pick to reach ip: reveals which # subnet/interface the route uses, without needing iproute2. python3 - "$ip" <<'PY' 2>&1 || true @@ -483,77 +554,99 @@ PY else echo "[diag] TCP connect ${ip}:${port} FAILED/timed out -> closed, filtered, or unreachable (LIKELY network/subnet/firewall issue)" fi + # L3: ICMP reachability, if ping is present. if command -v ping >/dev/null 2>&1; then ping -c 2 -W 2 "$ip" 2>&1 || echo "[diag] ping ${ip} failed (ICMP blocked or host down)" fi + # Verbose HTTP connect detail (DNS/connect/TLS timing, HTTP status). curl -v --connect-timeout 5 --max-time 8 "http://${ip}:${port}/health" 2>&1 || true echo "=== END NET DIAG ${ip}:${port} ===" } >&2 } - # Log the decode->prefill target layout up front so a subnet/interface - # mismatch is visible even on a run that eventually succeeds. Every prefill - # node serves on VLLM_PORT (hybrid LB). - echo "[diag] decode-leader $(hostname 2>/dev/null) local-ips: $(hostname -I 2>/dev/null); prefill targets: ${_prefill_ips[*]}" - echo "Waiting for prefill vLLM /health on ${#_prefill_ips[@]} node(s): ${_prefill_ips[*]}" - PREFILL_WAIT_DEADLINE=$(( $(date +%s) + 300 )) + # Log the coordinator->prefill target layout up front so a subnet/interface + # mismatch is visible even on a run that eventually succeeds. + echo "[diag] coordinator $(hostname 2>/dev/null) local-ips: $(hostname -I 2>/dev/null); prefill targets: ${_prefill_ips[*]}:${_PREFILL_HEALTH_PORT}" + echo "Waiting for prefill vLLM /health on ${#_prefill_ips[@]} node(s) (port ${_PREFILL_HEALTH_PORT}): ${_prefill_ips[*]}" + PREFILL_WAIT_DEADLINE=$(( $(date +%s) + 600 )) for _pidx in "${!_prefill_ips[@]}"; do _pip="${_prefill_ips[$_pidx]}" - _pport="$VLLM_PORT" until curl --output /dev/null --silent --fail \ --connect-timeout 5 --max-time 10 \ - "http://$_pip:$_pport/health"; do + "http://$_pip:${_PREFILL_HEALTH_PORT}/health"; do if [[ "$(date +%s)" -ge "$PREFILL_WAIT_DEADLINE" ]]; then - echo "ERROR: prefill vLLM at $_pip:$_pport not ready within 5 min" >&2 - _diag_prefill_endpoint "$_pip" "$_pport" + echo "ERROR: prefill vLLM at $_pip:${_PREFILL_HEALTH_PORT} not ready within 10 min" >&2 + _diag_prefill_endpoint "$_pip" "$_PREFILL_HEALTH_PORT" exit 1 fi sleep 5 done - echo "Prefill vLLM at $_pip:$_pport is ready" - done - echo "All ${#_prefill_ips[@]} prefill vLLM endpoint(s) ready" - - # Benchmark sweep. BENCH_MAX_CONCURRENCY is 'x'-delimited from submit.sh (e.g. "1024x512"). - IFS='x' read -r -a CONCURRENCIES <<< "$BENCH_MAX_CONCURRENCY" - # GPU counts are embedded in the result filename as _gpus_/_ctx_/_gen_ so the CI - # "Process result" step can parse them. - # ctx = prefill GPUs, gen = decode GPUs. - _bench_prefill_gpus=$(( PREFILL_NODES * GPUS_PER_NODE )) - _bench_decode_gpus=$(( DECODE_NODES * GPUS_PER_NODE )) - _bench_total_gpus=$(( _bench_prefill_gpus + _bench_decode_gpus )) - if [[ "${EVAL_ONLY}" != "true" ]]; then - for max_concurrency in "${CONCURRENCIES[@]}"; do - num_prompts=$(( max_concurrency * BENCH_NUM_PROMPTS_MULTIPLIER )) - [[ "$num_prompts" -lt 16 ]] && num_prompts=16 - # Bench against Envoy (EPP routes to decode; the sidecar pulls from - # prefill via NIXL). --bench-serving-dir = the /workspace repo bind-mount; - # --tokenizer = /models (served-model-name is not a valid HF repo id). - # Non-fatal: a failed or timed-out conc point must not abort the sweep or (under - # set -e) skip the allocation release below. - run_benchmark_serving \ - --bench-serving-dir /workspace \ - --tokenizer /models \ - --model "$MODEL_NAME" \ - --port "$ENVOY_PORT" \ - --backend openai \ - --input-len "$BENCH_INPUT_LEN" \ - --output-len "$BENCH_OUTPUT_LEN" \ - --random-range-ratio "$BENCH_RANDOM_RANGE_RATIO" \ - --num-prompts "$num_prompts" \ - --max-concurrency "$max_concurrency" \ - --result-filename "${RESULT_FILENAME}_c${max_concurrency}_gpus_${_bench_total_gpus}_ctx_${_bench_prefill_gpus}_gen_${_bench_decode_gpus}" \ - --result-dir "$BENCHMARK_LOGS_DIR/" \ - || echo "WARNING: benchmark conc=$max_concurrency failed/timed out (rc=$?)" + echo "Prefill vLLM at $_pip:${_PREFILL_HEALTH_PORT} is ready" done + echo "All ${#_prefill_ips[@]} prefill vLLM node(s) ready" + + if [[ "${IS_AGENTIC}" == "1" && "${EVAL_ONLY}" != "true" ]]; then + export ENVOY_PORT VLLM_PORT INFMAX_CONTAINER_WORKSPACE=/workspace + bash /workspace/benchmarks/multi_node/llm-d/agentic.sh + elif [[ "${EVAL_ONLY}" != "true" ]]; then + # Benchmark sweep. BENCH_MAX_CONCURRENCY is 'x'-delimited from submit.sh (e.g. "1024x512"). + IFS='x' read -r -a CONCURRENCIES <<< "$BENCH_MAX_CONCURRENCY" + # GPU counts are embedded in the result filename as _gpus_/_ctx_/_gen_ so the CI + # "Process result" step can parse them (same convention as amd_utils/bench.sh). + # ctx = prefill GPUs, gen = decode GPUs. + _bench_prefill_gpus=$(( PREFILL_NODES * GPUS_PER_NODE )) + _bench_decode_gpus=$(( DECODE_NODES * GPUS_PER_NODE )) + _bench_total_gpus=$(( _bench_prefill_gpus + _bench_decode_gpus )) + for max_concurrency in "${CONCURRENCIES[@]}"; do + num_prompts=$(( max_concurrency * BENCH_NUM_PROMPTS_MULTIPLIER )) + [[ "$num_prompts" -lt 16 ]] && num_prompts=16 + # Bench against Envoy (EPP routes to decode; the sidecar pulls from + # prefill via NIXL). --bench-serving-dir = the /workspace repo bind-mount; + # --tokenizer = /models (served-model-name is not a valid HF repo id). + # DSV4-Pro needs trust-remote-code + tokenizer-mode deepseek_v4 (the older + # transformers wheel does not register it) + chat template / --dsv4 to + # match the dynamo-vllm bench prompt formatting. + bench_extra_args=() + if [[ "${MODEL_NAME,,}" == *"deepseek-v4"* ]]; then + bench_extra_args+=( + --trust-remote-code + --tokenizer-mode deepseek_v4 + --use-chat-template + --dsv4 + ) + fi + # Non-fatal: a failed or timed-out conc point must not abort the sweep or (under + # set -e) skip the allocation release below. + run_benchmark_serving \ + --bench-serving-dir /workspace \ + --tokenizer /models \ + --model "$MODEL_NAME" \ + --port "$ENVOY_PORT" \ + --backend openai \ + --input-len "$BENCH_INPUT_LEN" \ + --output-len "$BENCH_OUTPUT_LEN" \ + --random-range-ratio "$BENCH_RANDOM_RANGE_RATIO" \ + --num-prompts "$num_prompts" \ + --max-concurrency "$max_concurrency" \ + --result-filename "${RESULT_FILENAME}_c${max_concurrency}_gpus_${_bench_total_gpus}_ctx_${_bench_prefill_gpus}_gen_${_bench_decode_gpus}" \ + --result-dir "$BENCHMARK_LOGS_DIR/" \ + "${bench_extra_args[@]}" \ + || echo "WARNING: benchmark conc=$max_concurrency failed/timed out (rc=$?)" + done fi - # Eval (optional) + # ---- Eval (optional) ---- if [[ "${RUN_EVAL}" == "true" ]]; then - # run_eval/append_lm_eval_summary read EVAL_CONCURRENT_REQUESTS and CONC (not - # EVAL_CONC). Exporting CONC makes meta_env.json's "conc" match what - # infx.evals.validate_scores --expected-concs verifies; without it the - # metadata records conc=1 and score verification fails even when accuracy passes. + # Concurrency for the eval and, crucially, for the concurrency stamped + # into meta_env.json. run_eval/append_lm_eval_summary read + # EVAL_CONCURRENT_REQUESTS and CONC (not EVAL_CONC), so mirror the AMD + # multi-node servers: use the workflow-provided EVAL_CONC when set, else + # fall back to the max of the (x-delimited) BENCH_MAX_CONCURRENCY list. + # Exporting CONC makes meta_env.json's "conc" match what + # infx.evals.validate_scores --expected-concs verifies; without it + # CONC is empty, the metadata records conc=1, and score verification + # fails ("eval metadata concurrency does not match workflow request") + # even when accuracy passes. if [[ -n "${EVAL_CONC:-}" ]]; then export EVAL_CONCURRENT_REQUESTS="${EVAL_CONC}" else diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/test_agentic_adapter.py b/inferencex-e2e/benchmarks/multi_node/llm-d/test_agentic_adapter.py new file mode 100644 index 0000000000..28331dc5fc --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/test_agentic_adapter.py @@ -0,0 +1,148 @@ +import json +import os +import subprocess +from pathlib import Path + +import pytest +import yaml + +REPO_ROOT = Path(__file__).resolve().parents[3] + + +@pytest.mark.parametrize("metrics_body", ["vllm:num_requests_running 0\n", "envoy_http_requests_total 0\n"]) +def test_llmd_agentic_adapter_uses_discovered_worker_metrics(tmp_path: Path, metrics_body: str) -> None: + """Check endpoint selection, preflight failure, and the real AIPerf CLI builder.""" + client = tmp_path / "benchmarks/srt_agentic.sh" + client.parent.mkdir(parents=True) + client.write_text('''source "$REAL_BENCHMARK_LIB" +build_replay_cmd "$RESULT_DIR" +export REPLAY_CMD +python3 - <<'PY' +import json, os +keys = ["AIPERF_METRIC_URLS", "AIPERF_SERVER_METRICS_URLS", "REPLAY_CMD"] +print(json.dumps({key: os.environ[key] for key in keys})) +PY +''') + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + curl = bin_dir / "curl" + curl.write_text( + '#!/usr/bin/env python3\n' + 'import os, sys\nfrom pathlib import Path\n' + 'args = sys.argv[1:]\n' + 'url = next((a for a in args if a.startswith("http://")), "")\n' + 'if "--write-out" in args:\n' + ' print("404", end="")\n' + 'else:\n' + ' out_path = args[args.index("--output") + 1]\n' + ' if out_path != "/dev/null":\n' + ' Path(out_path).write_text(os.environ["METRICS_BODY"])\n' + ' with open(os.environ["METRICS_REQUESTS"], "a") as f:\n' + ' f.write(url + "\\n")\n' + ) + curl.chmod(0o755) + endpoints = tmp_path / "endpoints.yaml" + endpoints.write_text(yaml.safe_dump({"endpoints": [ + {"address": "10.0.0.1", "port": "8200", "name": "vllm-node-0", + "labels": {"llm-d.ai/role": "combined"}}, + {"address": "10.0.0.2", "port": "8201", "name": "vllm-node-1", + "labels": {"llm-d.ai/role": "combined"}}, + {"address": "10.0.0.3", "port": "8202", "name": "vllm-node-2", + "labels": {"llm-d.ai/role": "combined"}}, + ]})) + requests = tmp_path / "metrics-requests.txt" + env = dict(os.environ, INFMAX_CONTAINER_WORKSPACE=str(tmp_path), + REAL_BENCHMARK_LIB=str(REPO_ROOT / "benchmarks/benchmark_lib.sh"), + PATH=str(bin_dir) + os.pathsep + os.environ["PATH"], + METRICS_BODY=metrics_body, METRICS_REQUESTS=str(requests), + LLMD_ENDPOINTS_FILE=str(endpoints), MODEL_NAME="test-model", MODEL_PREFIX="dsv4", + FRAMEWORK="llmd-vllm", DURATION="3600", IS_AGENTIC="1", KV_OFFLOADING="none", + ENVOY_PORT="8080", VLLM_PORT="8200", SIDECAR_PORT="8000", + BENCHMARK_LOGS_DIR=str(tmp_path / "logs"), + BENCH_MAX_CONCURRENCY="64", DECODE_NODES="0") + result = subprocess.run(["bash", str(REPO_ROOT / "benchmarks/multi_node/llm-d/agentic.sh")], + env=env, text=True, capture_output=True) + if metrics_body.startswith("envoy_"): + assert result.returncode != 0 + assert "no vLLM metrics exposed" in result.stderr + return + assert result.returncode == 0, result.stderr + recorded = json.loads(result.stdout.splitlines()[-1]) + expected_urls = [ + "http://10.0.0.1:8200/metrics", + "http://10.0.0.2:8201/metrics", + "http://10.0.0.3:8202/metrics", + ] + assert requests.read_text().splitlines() == expected_urls + assert recorded["AIPERF_METRIC_URLS"].split(",") == expected_urls + assert recorded["AIPERF_SERVER_METRICS_URLS"].split(",") == expected_urls + assert "--url http://localhost:8080 " in recorded["REPLAY_CMD"] + assert "--server-metrics " + " ".join(expected_urls) + " " in recorded["REPLAY_CMD"] + + +def test_llmd_agentic_adapter_maps_decode_sidecar_ports_to_vllm_metrics( + tmp_path: Path, +) -> None: + """Disagg decode endpoints list sidecar ports; metrics scrape vLLM DP ranks.""" + client = tmp_path / "benchmarks/srt_agentic.sh" + client.parent.mkdir(parents=True) + client.write_text('''source "$REAL_BENCHMARK_LIB" +build_replay_cmd "$RESULT_DIR" +export REPLAY_CMD +python3 - <<'PY' +import json, os +keys = ["AIPERF_METRIC_URLS", "AIPERF_SERVER_METRICS_URLS", "REPLAY_CMD"] +print(json.dumps({key: os.environ[key] for key in keys})) +PY +''') + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + curl = bin_dir / "curl" + curl.write_text( + '#!/usr/bin/env python3\n' + 'import os, sys\nfrom pathlib import Path\n' + 'args = sys.argv[1:]\n' + 'url = next((a for a in args if a.startswith("http://")), "")\n' + 'if "--write-out" in args:\n' + ' print("404", end="")\n' + 'else:\n' + ' out_path = args[args.index("--output") + 1]\n' + ' if out_path != "/dev/null":\n' + ' Path(out_path).write_text(os.environ["METRICS_BODY"])\n' + ' with open(os.environ["METRICS_REQUESTS"], "a") as f:\n' + ' f.write(url + "\\n")\n' + ) + curl.chmod(0o755) + endpoints = tmp_path / "endpoints.yaml" + endpoints.write_text(yaml.safe_dump({"endpoints": [ + {"address": "10.0.0.10", "port": "8200", "name": "prefill-0", + "labels": {"llm-d.ai/role": "prefill"}}, + {"address": "10.0.0.10", "port": "8201", "name": "prefill-1", + "labels": {"llm-d.ai/role": "prefill"}}, + {"address": "10.0.0.20", "port": "8000", "name": "decode-0", + "labels": {"llm-d.ai/role": "decode"}}, + {"address": "10.0.0.20", "port": "8001", "name": "decode-1", + "labels": {"llm-d.ai/role": "decode"}}, + ]})) + requests = tmp_path / "metrics-requests.txt" + env = dict(os.environ, INFMAX_CONTAINER_WORKSPACE=str(tmp_path), + REAL_BENCHMARK_LIB=str(REPO_ROOT / "benchmarks/benchmark_lib.sh"), + PATH=str(bin_dir) + os.pathsep + os.environ["PATH"], + METRICS_BODY="vllm:num_requests_running 0\n", METRICS_REQUESTS=str(requests), + LLMD_ENDPOINTS_FILE=str(endpoints), MODEL_NAME="test-model", MODEL_PREFIX="dsv4", + FRAMEWORK="llmd-vllm", DURATION="3600", IS_AGENTIC="1", KV_OFFLOADING="none", + ENVOY_PORT="8080", VLLM_PORT="8200", SIDECAR_PORT="8000", + BENCHMARK_LOGS_DIR=str(tmp_path / "logs"), + BENCH_MAX_CONCURRENCY="64", DECODE_NODES="2") + result = subprocess.run(["bash", str(REPO_ROOT / "benchmarks/multi_node/llm-d/agentic.sh")], + env=env, text=True, capture_output=True) + assert result.returncode == 0, result.stderr + recorded = json.loads(result.stdout.splitlines()[-1]) + expected_urls = [ + "http://10.0.0.10:8200/metrics", + "http://10.0.0.10:8201/metrics", + "http://10.0.0.20:8200/metrics", + "http://10.0.0.20:8201/metrics", + ] + assert requests.read_text().splitlines() == expected_urls + assert recorded["AIPERF_METRIC_URLS"].split(",") == expected_urls diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 07c47e6e40..0f28652615 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -4076,6 +4076,134 @@ dsr1-fp4-b200-dynamo-sglang-mtp: ep: 8 dp-attn: true +# DeepSeek-V4-Pro-0813 (DSpark) FP4 GB200, P/D disagg via llmd-vllm. +# Long-context workspace headroom: GPU-memory budgets are 0.88 for DEP8, 0.85 for TP8. +# Always uses Mooncake; DSpark with adaptive verification (no golden AL) on disagg. +# Aggregated TP8 uses committed golden AL; aggregated DEP8 uses adaptive verification. +# The DSpark image bundles EPP/pd-sidecar v0.10.0. +dsv4-fp4-gb200-llmd-vllm-agentx: + image: quay.io/rh-ee-imarkov/llm-d-nokube-vllm:dspark-0814-nightly@sha256:00742fdd10e572172d49559e44c357a68c388933c74db2c4e819bf61377a6a95 + model: deepseek-ai/DeepSeek-V4-Pro-0813 + model-prefix: dsv4 + runner: cluster:gb200-nv + precision: fp4 + framework: llmd-vllm + router: { name: llm-d-router, version: "0.10.0" } + kv-p2p-transfer: nixl + multinode: true + disagg: true + scenarios: + agentic-coding: + - dram-utilization: 0.666667 + search-space: + # 1P DEP8 + 1D DEP8 (4 nodes / 16 GPUs). Always Mooncake. + - spec-decoding: mtp + kv-offloading: dram + kv-offload-backend: { name: mooncake } + conc-list: [64, 128, 160, 256] + prefill: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "PREFILL_NODES=2" + - "GPUS_PER_NODE=4" + - "CONFIG_FILE=agentic/disagg-gb200-1p1d-dep8-dep8-dspark-agentic.yaml" + decode: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "DECODE_NODES=2" + - "GPUS_PER_NODE=4" + + # 2P DEP8 + 1D DEP16 (8 nodes / 32 GPUs). agentX long EPP routing. + - spec-decoding: mtp + kv-offloading: dram + kv-offload-backend: { name: mooncake } + conc-list: [256, 384, 576] + prefill: + num-worker: 2 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "PREFILL_NODES=4" + - "GPUS_PER_NODE=4" + - "CONFIG_FILE=agentic/disagg-gb200-1p1d-dep8-dep8-dspark-agentic.yaml" + decode: + num-worker: 1 + tp: 8 + ep: 16 + dp-attn: true + additional-settings: + - "DECODE_NODES=4" + - "GPUS_PER_NODE=4" + +# Aggregated TP8/DEP8; only the Mooncake arm enables DRAM offloading. +dsv4-fp4-gb200-llmd-vllm-agentx-agg: + image: quay.io/rh-ee-imarkov/llm-d-nokube-vllm:dspark-0814-nightly@sha256:00742fdd10e572172d49559e44c357a68c388933c74db2c4e819bf61377a6a95 + model: deepseek-ai/DeepSeek-V4-Pro-0813 + model-prefix: dsv4 + runner: cluster:gb200-nv + precision: fp4 + framework: llmd-vllm + router: { name: llm-d-router, version: "0.10.0" } + multinode: true + disagg: false + scenarios: + agentic-coding: + - dram-utilization: 0.45 + search-space: + # Aggregated TP8 (2 nodes / 8 GPUs; pure tensor-parallel, no EP). + - spec-decoding: mtp + kv-offloading: none + conc-list: [1, 2, 4, 8, 12] + num-nodes: 2 + worker: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "PREFILL_NODES=2" + - "DECODE_NODES=0" + - "GPUS_PER_NODE=4" + - "CONFIG_FILE=agentic/agg-gb200-tp8-dspark-agentic.yaml" + # Aggregated DEP8 (2 nodes / 8 GPUs; DP=8 + EP), no Mooncake. + - spec-decoding: mtp + kv-offloading: none + conc-list: [12, 16, 32] + num-nodes: 2 + worker: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "PREFILL_NODES=2" + - "DECODE_NODES=0" + - "GPUS_PER_NODE=4" + - "CONFIG_FILE=agentic/agg-gb200-dep8-dspark-agentic.yaml" + # Aggregated DEP8 with Mooncake prefix-cache KV store. + - spec-decoding: mtp + kv-offloading: dram + kv-offload-backend: { name: mooncake } + conc-list: [52, 72] + num-nodes: 2 + worker: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "PREFILL_NODES=2" + - "DECODE_NODES=0" + - "GPUS_PER_NODE=4" + - "CONFIG_FILE=agentic/agg-gb200-dep8-dspark-mooncake-agentic.yaml" + qwen3.5-fp8-gb200-dynamo-sglang: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 model: Qwen/Qwen3.5-397B-A17B-FP8 diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index cf67501a63..f17fdd1cb0 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -566,6 +566,7 @@ clusters: deepseek-r1-0528: {root: lustre-models, dir: deepseek-r1-0528} deepseek-r1-0528-fp4-v2: {root: lustre-models, dir: deepseek-r1-0528-fp4-v2} DeepSeek-V4-Pro: {root: lustre-models, dir: DeepSeek-V4-Pro} + DeepSeek-V4-Pro-0813: {root: sa-shared-models, dir: DeepSeek-V4-Pro-0813} MiniMax-M3-MXFP8: {root: lustre-models, dir: MiniMax-M3-MXFP8} MiniMax-M3-NVFP4: {root: lustre-models, dir: MiniMax-M3-NVFP4} Qwen3.5-397B-A17B-FP8: {root: lustre-models, dir: Qwen3.5-397B-A17B-FP8} diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index d0ed2d739b..42101d006c 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -228,6 +228,12 @@ llm-d is not the srt-slurm path: InferenceX owns the Slurm allocation and starts A missing/unset `CONFIG_FILE` silently selects the image's `/etc/epp/config.yaml` fallback and removes recipe-specific vLLM flags. Treat that as a validation failure unless fallback is explicitly intended. +The GB200 DSpark AgentX keys use `cluster:gb200-nv`: aggregated TP8/DEP8 spans two nodes (8 GPUs); disagg 1P-DEP8/1D-DEP8 spans four nodes (16 GPUs); disagg 2P-DEP8/1D-DEP16 spans eight nodes (32 GPUs). The ARM64 image digest includes router v0.10.0; AgentX does not mount the legacy router binaries. Recipes using `token-load-scorer` must explicitly include `inflight-load-producer` to supply its uncached-token dependency. The launcher uses the staged `DeepSeek-V4-Pro-0813` checkpoint, not the older V4-Pro weights. Use `gpu-memory-utilization=0.88` for DEP8 and `0.85` for TP8. Long-context replay exhausted sparse-attention indexer memory at 0.92 and 0.90 respectively; retain headroom without shortening the model context or filtering traces. NIXL+Mooncake P/D uses a 1,800-second KV lease, matching the model-execution timeout; the default 30-second lease expired during long-context decode stalls at c192. + +For DSpark AgentX, `recipe.py` selects the acceptance-length mode from each role's `--speculative-config`. The disagg recipe (`agentic/disagg-gb200-1p1d-dep8-dep8-dspark-agentic.yaml`) and aggregated DEP8 recipes (with and without Mooncake) set `enable_adaptive_verification: true` (disagg: prefill K=1, decode K=3; agg DEP8: K=5). Throughput and `EVAL_ONLY` on those arms use real target verification; `recipe.py` does not inject golden AL. The aggregated TP8 recipe keeps `enable_adaptive_verification: false`; throughput injects the committed golden AL via `infx.golden_al_distribution.golden_length` from [`infx/golden_al_distribution/dsv4-pro-0813-dspark.yaml`](../infx/golden_al_distribution/dsv4-pro-0813-dspark.yaml), while `EVAL_ONLY=true` strips synthetic acceptance and keeps real verification. Do not mix modes on one recipe: adaptive verification trims the draft budget at runtime and is incompatible with a fixed-K golden AL target. Mooncake's embedded store is DRAM offloading even with `enable_offload: false` (the flag controls SSD). Declare `kv-offloading: dram`, `kv-offload-backend: { name: mooncake }`, and `dram-utilization`; the runtime divides each node's budget among its four GPU ranks. Rank 0 starts a job-local Mooncake master on port 50051 (metrics on 50052); every rank waits for it before starting vLLM. `P2PHANDSHAKE` does not replace the store master. Mooncake uses InfiniBand HCAs `mlx5_0,mlx5_1,mlx5_3,mlx5_4`; `mlx5_2` and `mlx5_5` are Ethernet. Plain TP8/DEP8 declares `none`. + +Discovery excludes headless TP followers. `agentic.sh` checks every serving node's vLLM `/metrics`, exports `AIPERF_METRIC_URLS` and `AIPERF_SERVER_METRICS_URLS`, and forwards them through AIPerf's `--server-metrics`. Envoy rejects `/metrics` before EPP routing so AIPerf's automatic frontend scrape cannot duplicate worker counters. The adapter verifies this 404 before replay and requires exported `vllm:` metrics. Raw AgentX artifacts include `llmd_metrics_endpoints.json`, mapping each scrape URL to its discovery name and `prefill`, `decode`, or `combined` role; engine IDs alone are not unique across P/D groups. This manifest does not add Prometheus labels or change app ingestion. Envoy and P/D-sidecar metrics are not substitutes for vLLM metrics. + ## Update an image Sources: [`AGENTS.md#non-negotiable-benchmark-invariants`](../../AGENTS.md#non-negotiable-benchmark-invariants), the matching master configs, runtime scripts, and checked-in recipes. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index a659d68c46..b4d9e44273 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -209,6 +209,12 @@ llm-d 不是 srt-slurm 路径:InferenceX 自己持有 Slurm allocation,并 `CONFIG_FILE` 未设置或文件缺失时,会静默选择镜像内 `/etc/epp/config.yaml` fallback,并移除配方特定 vLLM 参数。除非明确打算使用 fallback,否则应将其视为验证失败。 +GB200 DSpark AgentX 配置使用 `cluster:gb200-nv`:聚合 TP8/DEP8 跨两个节点(8 GPU);解耦 1P-DEP8/1D-DEP8 跨四个节点(16 GPU);解耦 2P-DEP8/1D-DEP16 跨八个节点(32 GPU)。ARM64 镜像摘要包含 router v0.10.0;AgentX 不挂载旧版路由器二进制文件。使用 `token-load-scorer` 的配置必须显式声明 `inflight-load-producer`,以提供其依赖的未缓存 token 数据。launcher 使用预先存储的 `DeepSeek-V4-Pro-0813` checkpoint,而非旧版 V4-Pro 权重。DEP8 使用 `gpu-memory-utilization=0.88`,TP8 使用 `0.85`。长上下文回放分别在 0.92 和 0.90 时耗尽稀疏注意力索引器内存;应保留工作区余量,而不是缩短模型上下文或过滤轨迹。 NIXL+Mooncake P/D 使用 1,800 秒 KV 租约,与模型执行超时一致;默认的 30 秒租约在 c192 长上下文解码停顿期间过期。 + +DSpark AgentX 的接受长度模式由 `recipe.py` 根据各角色的 `--speculative-config` 选择。解耦配方(`agentic/disagg-gb200-1p1d-dep8-dep8-dspark-agentic.yaml`)与聚合 DEP8 配方(含 Mooncake 与无 Mooncake)均设置 `enable_adaptive_verification: true`(解耦:prefill K=1、decode K=3;聚合 DEP8:K=5)。这些臂的吞吐与 `EVAL_ONLY` 均使用真实目标验证;`recipe.py` 不注入 golden AL。聚合 TP8 配方保持 `enable_adaptive_verification: false`;吞吐通过 `infx.golden_al_distribution.golden_length` 从 [`infx/golden_al_distribution/dsv4-pro-0813-dspark.yaml`](../infx/golden_al_distribution/dsv4-pro-0813-dspark.yaml) 注入已提交的 golden AL,`EVAL_ONLY=true` 则移除 synthetic 接受长度并保留真实验证。同一配方不要混用两种模式:自适应验证会在运行时裁剪草稿预算,与固定 K 的 golden AL 目标不兼容。Mooncake 嵌入式存储即使设置 `enable_offload: false` 也属于 DRAM 卸载(该开关控制 SSD)。需声明 `kv-offloading: dram`、`kv-offload-backend: { name: mooncake }` 和 `dram-utilization`;运行时将每节点预算均分给四个 GPU rank。Rank 0 在端口 50051 启动本次作业专用的 Mooncake master(指标端口为 50052);所有 rank 等待其就绪后再启动 vLLM。`P2PHANDSHAKE` 不能替代存储 master。Mooncake 使用 InfiniBand HCA `mlx5_0,mlx5_1,mlx5_3,mlx5_4`;`mlx5_2` 和 `mlx5_5` 为以太网设备。普通 TP8/DEP8 声明 `none`。 + +服务发现排除无 API 的 TP follower。`agentic.sh` 检查各服务节点的 vLLM `/metrics`,导出 `AIPERF_METRIC_URLS` 和 `AIPERF_SERVER_METRICS_URLS`,再通过 AIPerf 的 `--server-metrics` 传递。Envoy 在 EPP 路由前拒绝 `/metrics`,避免 AIPerf 自动抓取前端时重复统计 worker 计数器。适配器在回放前验证此端点返回 404,并要求导出结果包含 `vllm:` 指标。原始 AgentX 工件包含 `llmd_metrics_endpoints.json`,将各抓取 URL 映射到服务发现名称及 `prefill`、`decode` 或 `combined` 角色;仅凭 engine ID 无法区分 P/D 组。此清单不会添加 Prometheus 标签或改变应用的摄取逻辑。Envoy 和 P/D sidecar 的指标不能替代 vLLM 指标。 + ## 更新镜像 来源:[`AGENTS.md#non-negotiable-benchmark-invariants`](../../AGENTS.md#non-negotiable-benchmark-invariants)、对应主配置、运行时脚本与检入的 Recipe。 diff --git a/inferencex-e2e/infx/launch/drivers/__init__.py b/inferencex-e2e/infx/launch/drivers/__init__.py index bab33b1cb6..996bb5738d 100644 --- a/inferencex-e2e/infx/launch/drivers/__init__.py +++ b/inferencex-e2e/infx/launch/drivers/__init__.py @@ -13,7 +13,7 @@ from infx.launch import policy from infx.launch.backends import backend_class from infx.launch.context import Launch, LaunchError -from infx.launch.drivers import legacy, script, srt +from infx.launch.drivers import legacy, llmd, script, srt from infx.launch.policy import LaunchPath, launch_path if TYPE_CHECKING: @@ -38,6 +38,7 @@ class Route: LaunchPath.SCRIPT: Route(None, script.run), LaunchPath.LEGACY_TILERT: Route("slurm", legacy.run_tilert), LaunchPath.LEGACY_AMD_UTILS: Route("slurm", legacy.run_amd_utils), + LaunchPath.LLMD: Route("slurm", llmd.run), } diff --git a/inferencex-e2e/infx/launch/drivers/llmd.py b/inferencex-e2e/infx/launch/drivers/llmd.py new file mode 100644 index 0000000000..809104458a --- /dev/null +++ b/inferencex-e2e/infx/launch/drivers/llmd.py @@ -0,0 +1,152 @@ +"""GB200 llm-d vLLM multinode jobs submitted through benchmarks/multi_node/llm-d/submit.sh.""" + +from __future__ import annotations + +import os +import shutil +import subprocess +import sys +from pathlib import Path + +from infx.launch import artifacts, policy, proc +from infx.launch.backends.base import BackendError +from infx.launch.backends.slurm import cli +from infx.launch.context import Launch, LaunchError +from infx.launch.drivers.srt import models +from infx.launch.drivers.srt.run import slurm_backend +from infx.launch.request import LlmdRequest, RequestError + +CANCEL_TIMEOUT_S = 600.0 + + +def _bench_script(request: LlmdRequest) -> Path: + model_tag = request.exp_name.split("_", 1)[0] + kind = "disagg" if request.disagg else "agg" + script = ( + request.workspace + / f"benchmarks/multi_node/{model_tag}_{request.precision}_gb200_llmd-vllm-{kind}.sh" + ) + if not script.is_file(): + raise LaunchError(f"llm-d wrapper not found: {script}") + return script + + +def _find_eval_dir(logs_dir: Path) -> Path | None: + for root, dirs, _files in os.walk(logs_dir): + if "eval_results" in dirs: + return Path(root) / "eval_results" + return None + + +def _stage_agentic(logs_dir: Path, workspace: Path) -> None: + agentic = logs_dir / "agentic" + if not agentic.is_dir(): + return + staged = workspace / "LOGS" / "agentic" + staged.mkdir(parents=True, exist_ok=True) + for entry in agentic.iterdir(): + destination = staged / entry.name + if entry.is_dir(): + shutil.copytree(entry, destination, dirs_exist_ok=True) + elif entry.is_file(): + shutil.copy2(entry, destination) + + +def run(launch: Launch) -> int: + """Submit the llm-d Slurm job, follow its log, and stage benchmark artifacts.""" + backend = slurm_backend(launch) + request = LlmdRequest.from_env(launch.request.env) + if launch.cluster.id not in policy.LLMD_CLUSTERS: + raise LaunchError(f"llmd-vllm is not configured for cluster {launch.cluster.id!r}") + + checkpoint = models.checkpoint(launch.cluster, request) + if checkpoint is None: + raise LaunchError( + f"cluster {launch.cluster.id!r} stages no checkpoint for MODEL={request.model}" + ) + model_path = models.host_path(launch.cluster, checkpoint) + if not (model_path / "config.json").is_file(): + raise LaunchError(f"model checkpoint is unavailable: {model_path / 'config.json'}") + + squash = backend.prepare_image(request.image) + logs_dir = request.workspace / "benchmark_logs" + logs_dir.mkdir(parents=True, exist_ok=True) + + account = backend.settings.account or cli.default_account() + if not account: + raise RequestError.missing("SLURM_ACCOUNT") + + env = policy.runtime_env( + launch.cluster, + request, + models.job_env(launch.cluster, request, str(model_path)), + { + "SLURM_PARTITION": backend.settings.partition, + "SLURM_ACCOUNT": account, + "MODEL_PATH": str(model_path), + "MODEL_NAME": request.model, + "LLMD_CONTAINER_ENGINE": "pyxis", + "LLMD_SQUASH_FILE": squash.reference, + "BENCHMARK_LOGS_DIR": str(logs_dir), + }, + ) + + script = _bench_script(request) + argv = ["bash", str(script)] + proc.echo(argv, env) + submitted = subprocess.run( + argv, + stdout=subprocess.PIPE, + stderr=sys.stderr, + text=True, + env=env, + cwd=request.workspace, + check=False, + ) + job_id = submitted.stdout.strip() + if submitted.returncode != 0 or not job_id: + print("ERROR: llm-d submit wrapper failed before returning a Slurm job id", file=sys.stderr) + return 1 + if not (job_id.isascii() and job_id.isdigit()): + print( + f"ERROR: llm-d submit wrapper printed {job_id!r} instead of a Slurm job id", + file=sys.stderr, + ) + return 1 + + log_file = logs_dir / f"slurm_job-{job_id}.out" + job = backend.attach(job_id, log=log_file, outputs=logs_dir) + print(f"Submitted llm-d job: {job_id}", flush=True) + + launch.life.callback( + artifacts.bundle_server_logs, logs_dir, request.workspace / "multinode_server_logs.tar.gz" + ) + launch.life.callback(backend.cancel, job, wait_s=CANCEL_TIMEOUT_S) + + try: + backend.stream_logs(job) + except BackendError: + return 1 + + status = backend.state(job) + rc = 0 if status.succeeded else 1 + + for result_file in sorted(logs_dir.glob(f"{request.result_filename}*.json")): + try: + artifacts.copy_to_workspace(result_file, request.workspace / result_file.name) + except artifacts.ArtifactError as error: + print(f"ERROR: {error}", file=sys.stderr) + rc = 1 + + if request.is_agentic and not request.eval_only: + _stage_agentic(logs_dir, request.workspace) + + if request.run_eval: + eval_dir = _find_eval_dir(logs_dir) or logs_dir / "eval_results" + try: + artifacts.copy_eval_artifacts(eval_dir, request.workspace) + except artifacts.ArtifactError as error: + print(f"ERROR: {error}", file=sys.stderr) + rc = 1 + + return rc diff --git a/inferencex-e2e/infx/launch/drivers/srt/models.py b/inferencex-e2e/infx/launch/drivers/srt/models.py index a8f1dad3fc..1f66e4d3b1 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/models.py +++ b/inferencex-e2e/infx/launch/drivers/srt/models.py @@ -46,6 +46,15 @@ class Override: Override(Match(model_glob="*/DeepSeek-V4-Pro-0813"), entry="DeepSeek-V4-Pro-0813"), ), "gb200-nv": ( + Override( + Match(any_of("dsv4"), any_of("fp4"), any_of("llmd-vllm"), model_glob="*0813*"), + entry="DeepSeek-V4-Pro-0813", + ), + Override( + Match(any_of("dsv4"), any_of("fp4"), any_of("llmd-vllm")), + entry="DeepSeek-V4-Pro@numa1", + served_name="deepseek-ai/DeepSeek-V4-Pro", + ), Override( Match(any_of("dsr1"), any_of("fp4"), any_of("dynamo-sglang")), entry="deepseek-r1-0528-fp4-v2", diff --git a/inferencex-e2e/infx/launch/policy.py b/inferencex-e2e/infx/launch/policy.py index f9d26ab88c..76c135500e 100644 --- a/inferencex-e2e/infx/launch/policy.py +++ b/inferencex-e2e/infx/launch/policy.py @@ -15,6 +15,7 @@ from typing import TYPE_CHECKING from infx.clusters.slurm import SlurmSettings +from infx.launch.context import LaunchError if TYPE_CHECKING: from infx.clusters import Cluster @@ -61,6 +62,7 @@ class LaunchPath(StrEnum): SCRIPT = "script" LEGACY_TILERT = "legacy-tilert" LEGACY_AMD_UTILS = "legacy-amd-utils" + LLMD = "llmd" NATIVE_SRT_LANES: dict[str, tuple[Match, ...]] = { @@ -79,8 +81,18 @@ class LaunchPath(StrEnum): } +LLMD_CLUSTERS: frozenset[str] = frozenset({"gb200-nv"}) + + def launch_path(cluster_id: str, request: LaunchRequest) -> LaunchPath: if request.is_multinode: + if request.framework == "llmd-vllm": + if cluster_id not in LLMD_CLUSTERS: + raise LaunchError( + f"llmd-vllm is not configured for cluster {cluster_id!r}; " + f"supported clusters: {', '.join(sorted(LLMD_CLUSTERS))}" + ) + return LaunchPath.LLMD if any(lane(request) for lane in NATIVE_SRT_LANES.get(cluster_id, ())): return LaunchPath.SRT_NATIVE if cluster_id in LEGACY_TILERT and request.framework == "tilert": @@ -228,6 +240,11 @@ def keys(table: Mapping[str, object]) -> list[str]: for key in keys(table) if key not in clusters ] + problems += [ + f"LLMD_CLUSTERS[{cluster_id!r}]: no such cluster" + for cluster_id in LLMD_CLUSTERS + if only in (None, cluster_id) and cluster_id not in clusters + ] for cluster_id in keys(LEGACY_TILERT): cluster = clusters.get(cluster_id) settings = cluster.scheduler_settings if cluster is not None else None @@ -248,4 +265,13 @@ def keys(table: Mapping[str, object]) -> list[str]: for name in lane.host_setup_env if name not in host_env ] + for cluster_id in LLMD_CLUSTERS: + if only not in (None, cluster_id): + continue + cluster = clusters.get(cluster_id) + settings = cluster.scheduler_settings if cluster is not None else None + if isinstance(settings, SlurmSettings) and settings.squash is None: + problems.append( + f"LLMD_CLUSTERS[{cluster_id!r}]: no slurm.squash for Pyxis image import" + ) return problems diff --git a/inferencex-e2e/infx/launch/request.py b/inferencex-e2e/infx/launch/request.py index fe8863b2ec..db7f35ff8c 100644 --- a/inferencex-e2e/infx/launch/request.py +++ b/inferencex-e2e/infx/launch/request.py @@ -176,3 +176,11 @@ class AmdUtilsRequest(LegacyRequest): model: str = Field(alias="MODEL") user: str | None = Field(None, alias="USER") keep_logs: OneFlag = Field(False, alias="KEEP_LOGS") + + +class LlmdRequest(SrtRequest): + """A GB200 llm-d vLLM multinode job submitted through benchmarks/multi_node/llm-d.""" + + model: str = Field(alias="MODEL") + exp_name: str = Field(alias="EXP_NAME") + disagg: TrueFlag = Field(alias="DISAGG") diff --git a/inferencex-e2e/infx/tests/launch/test_llmd_driver.py b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py new file mode 100644 index 0000000000..473e0dc583 --- /dev/null +++ b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py @@ -0,0 +1,115 @@ +"""GB200 llm-d vLLM launches through infx.launch instead of runners/launch_gb200-nv.sh.""" + +import json +import subprocess +import tarfile +from pathlib import Path + +import pytest + +from infx.tests.launch.fake_slurm import ( + base_env, + install_fakes, + launch, + make_workspace, + runner_for, + sandbox_runner_config, +) + +LLMD_SUBMIT = """#!/usr/bin/env bash +set -e +env > "$GITHUB_WORKSPACE/submitted.env" +logs="$BENCHMARK_LOGS_DIR" +job="$logs/slurm_job-4299" +mkdir -p "$logs/agentic/conc_128" "$job/eval_results" +echo '{"conc": 128}' > "$logs/point-identity_conc128.json" +echo trace > "$logs/agentic/conc_128/profile.json" +echo '{"score": 1}' > "$job/eval_results/results_gsm8k.json" +echo 'server log' > "$logs/server.log" +echo 'benchmark done' > "$logs/slurm_job-4299.out" +echo 'worker warning' > "$logs/slurm_job-4299.err" +echo 'submitting' >&2 +[[ "${NO_JOB_ID:-}" == 1 ]] && exit 1 +echo 4299 +""" + + +@pytest.fixture +def harness(tmp_path): + """Sandboxed gb200-nv cluster, fake Slurm binaries, and a workspace with the llm-d wrapper.""" + sandbox = tmp_path / "sandbox" + sandbox.mkdir() + config = sandbox_runner_config(sandbox) + workspace = make_workspace(tmp_path / "workspace") + wrapper = workspace / "benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh" + wrapper.parent.mkdir(parents=True, exist_ok=True) + wrapper.write_text(LLMD_SUBMIT) + wrapper.chmod(0o755) + + model_root = sandbox / "mnt/lustre01/users-public/sa-shared/models/DeepSeek-V4-Pro-0813" + model_root.mkdir(parents=True) + (model_root / "config.json").write_text("{}\n") + + logs = tmp_path / "logs" + env = base_env( + fakes=install_fakes(tmp_path / "bin"), logs=logs, workspace=workspace, sandbox=sandbox + ) + env.update( + RUNNER_NAME=runner_for("gb200-nv"), + IS_MULTINODE="true", + IS_AGENTIC="1", + RUN_EVAL="true", + EVAL_ONLY="false", + FRAMEWORK="llmd-vllm", + MODEL="deepseek-ai/DeepSeek-V4-Pro-0813", + MODEL_PREFIX="dsv4", + PRECISION="fp4", + SPEC_DECODING="mtp", + THINKING_MODE="thinking_on", + EXP_NAME="dsv4_agentic_p1x8", + DISAGG="true", + IMAGE="vllm/vllm-openai:v0.21.0", + RESULT_FILENAME="point-identity", + ENROOT_IMPORT_TIME_LIMIT="10", + ) + return config, workspace, env + + +def run_launch(harness) -> subprocess.CompletedProcess[str]: + config, workspace, env = harness + return launch(env, config, workspace) + + +def test_llmd_driver_submits_the_wrapper_and_stages_artifacts(harness): + result = run_launch(harness) + config, workspace, env = harness + + assert result.returncode == 0, result.stdout + result.stderr + submitted = dict( + line.split("=", 1) for line in (workspace / "submitted.env").read_text().splitlines() if "=" in line + ) + model_path = f"{config.parent}/mnt/lustre01/users-public/sa-shared/models/DeepSeek-V4-Pro-0813" + assert submitted["MODEL_PATH"] == model_path + assert submitted["MODEL_NAME"] == env["MODEL"] + assert submitted["LLMD_CONTAINER_ENGINE"] == "pyxis" + assert submitted["LLMD_SQUASH_FILE"] + assert submitted["BENCHMARK_LOGS_DIR"] == f"{workspace}/benchmark_logs" + assert submitted["SLURM_PARTITION"] == "batch" + assert submitted["SLURM_ACCOUNT"] == "benchmark" + + assert json.loads((workspace / "point-identity_conc128.json").read_text()) == {"conc": 128} + assert (workspace / "LOGS/agentic/conc_128/profile.json").read_text() == "trace\n" + assert json.loads((workspace / "results_gsm8k.json").read_text()) == {"score": 1} + with tarfile.open(workspace / "multinode_server_logs.tar.gz") as bundle: + assert "./server.log" in bundle.getnames() + assert "submitting" in result.stderr + + +def test_llmd_driver_fails_when_the_wrapper_prints_no_job_id(harness): + _, workspace, env = harness + env["NO_JOB_ID"] = "1" + + result = launch(env, harness[0], workspace) + + assert result.returncode == 1 + assert "failed before returning a Slurm job id" in result.stderr diff --git a/inferencex-e2e/infx/tests/launch/test_srt_policy.py b/inferencex-e2e/infx/tests/launch/test_srt_policy.py index ce4b3d00d5..d8d8200a6c 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_policy.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_policy.py @@ -56,6 +56,7 @@ def cluster(tmp_path, single_node_models: str = "staged") -> Cluster: ("b200-nscale", dict(MULTI, MODEL_PREFIX="glm5.1", PRECISION="fp8", FRAMEWORK="tilert", SPEC_DECODING="mtp", IS_AGENTIC="0"), LaunchPath.LEGACY_TILERT), ("mi355x-amds", dict(IS_MULTINODE="true", FRAMEWORK="atom-disagg"), LaunchPath.LEGACY_AMD_UTILS), ("mi355x-amds", dict(MULTI, FRAMEWORK="sglang-disagg"), LaunchPath.SRT_MULTI), + ("gb200-nv", dict(MULTI, FRAMEWORK="llmd-vllm", MODEL_PREFIX="dsv4", PRECISION="fp4", SPEC_DECODING="mtp"), LaunchPath.LLMD), ("gb200-nv", dict(MULTI, FRAMEWORK="tilert"), LaunchPath.SRT_MULTI), ("b300-dsxe", dict(SINGLE, MODEL_PREFIX="dsv41flash", FRAMEWORK="sglang", IS_AGENTIC="1"), LaunchPath.SRT_BATCH), ("b300-dsxe", dict(SINGLE, MODEL_PREFIX="dsv41flash", FRAMEWORK="sglang", IS_AGENTIC="1", INFX_BATCH_REENTRY="1"), LaunchPath.SRT_SINGLE), diff --git a/inferencex-e2e/infx/tests/llm_d/test_recipe.py b/inferencex-e2e/infx/tests/llm_d/test_recipe.py new file mode 100644 index 0000000000..b6ca2af3b5 --- /dev/null +++ b/inferencex-e2e/infx/tests/llm_d/test_recipe.py @@ -0,0 +1,69 @@ +"""Behavioral checks for llm-d recipe acceptance-length selection.""" + +import importlib.util +import json +import re +import sys +from pathlib import Path + +import pytest +import yaml + +E2E_ROOT = Path(__file__).resolve().parents[3] +RECIPE_MODULE = E2E_ROOT / "benchmarks/multi_node/llm-d/recipe.py" +RECIPE_DIR = E2E_ROOT / "benchmarks/multi_node/llm-d-recipes/agentic" + + +def _load_recipe_module(): + spec = importlib.util.spec_from_file_location("llm_d_recipe", RECIPE_MODULE) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + sys.path.insert(0, str(E2E_ROOT)) + spec.loader.exec_module(module) + return module + + +def _spec_config(output: str) -> dict: + match = re.search(r"--speculative-config (\{.*?\})(?:\s|$)", output) + assert match is not None, output + return json.loads(match.group(1)) + + +@pytest.mark.parametrize( + ("recipe_name", "role", "expect_golden", "kv_offloading"), + [ + ("agg-gb200-tp8-dspark-agentic.yaml", "prefill", True, "none"), + ("agg-gb200-dep8-dspark-agentic.yaml", "prefill", False, "none"), + ("agg-gb200-dep8-dspark-mooncake-agentic.yaml", "prefill", False, "dram"), + ("disagg-gb200-1p1d-dep8-dep8-dspark-agentic.yaml", "decode", False, "dram"), + ], +) +def test_role_assignments_use_golden_al_only_for_tp8( + recipe_name: str, role: str, expect_golden: bool, kv_offloading: str, +) -> None: + recipe = _load_recipe_module() + env = { + "IS_AGENTIC": "1", + "SPEC_DECODING": "mtp", + "EVAL_ONLY": "false", + "RUN_EVAL": "false", + "MODEL_PREFIX": "dsv4", + "THINKING_MODE": "thinking_on", + "KV_OFFLOADING": kv_offloading, + } + if kv_offloading == "dram": + env["KV_OFFLOAD_BACKEND"] = "mooncake" + output = recipe.role_assignments( + yaml.safe_load((RECIPE_DIR / recipe_name).read_text()), + role, + env, + ) + config = _spec_config(output) + if expect_golden: + assert config["rejection_sample_method"] == "synthetic" + assert config["synthetic_acceptance_length"] == 3.61 + assert config["enable_adaptive_verification"] is False + else: + assert config["enable_adaptive_verification"] is True + assert "synthetic_acceptance_length" not in config + assert "rejection_sample_method" not in config diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 2b184929a0..0b95b3e5bd 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9127,3 +9127,19 @@ - "Add Kimi-K3 MXFP4 vLLM agentic-coding on MI355X (TP8, DSpark): dcp1 c1/c4 GPU-resident and c8-c14 with SimpleCPUOffload DRAM offload, plus a new dcp8 c44/c48/c70 throughput band (mtp synthetic acceptance, no draft); vLLM ROCm image vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89" - "The Inferact/Kimi-K3-DSpark draft (K3DSparkModel, model_type k3_dspark, 5 layers, hidden 7168, torch_dtype bfloat16, no quantization_config) keeps every layer at its pristine dtype. vLLM builds all its modules with quant_config from get_draft_quant_config() (vllm/models/kimi_k3/nvidia/dspark_mla.py), which returns None because K3DSparkModel is explicitly excluded from the DeepSeek-V4 branch that would set draft.quantization = target.quantization (vllm/config/speculative.py:1431); so context_proj (ReplicatedLinear), context_kv_proj (MergedColumnParallelLinear) and each decoder layer's MLA q/kv projections and dense KimiMLP (gate/up/down) load unquantized in BF16 -- no ptpc_fp8, no mxfp4, no INT4 weights. The draft has no FusedMoE/block_sparse_moe, so VLLM_ROCM_USE_AITER_MOE_SITUV2 (A8W4) never touches it, and its KV cache stays fp8 (kv_cache_dtype). This PR only changes draft depth (num_speculative_tokens 4->7) and enables INT4 custom quick all-reduce (VLLM_ROCM_QUICK_REDUCE_QUANTIZATION=INT4, vllm/distributed/device_communicators/quick_all_reduce.py); because the draft shares the target's TP8 group (vllm/v1/spec_decode/draft_model.py), that INT4 applies to its tensor-parallel reductions too -- a collective-reduction transport precision, not any draft weight, activation, or KV dtype." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3561 + +- config-keys: + - dsv4-fp4-gb200-llmd-vllm-agentx + - dsv4-fp4-gb200-llmd-vllm-agentx-agg + scenario-type: + - agentic-coding + description: + - "Add GB200 llm-d vLLM AgentX benchmarks for DeepSeek-V4-Pro-0813 DSpark FP4/FP8-KV: disagg Mooncake 1P1D DEP8/DEP8 (conc 64/128/160/256) and 2P1D DEP8/DEP16 (conc 256/384/576); agg TP8 (conc 1-12), DEP8 (conc 12-32), and DEP8 Mooncake (conc 52/72). DSpark uses adaptive verification with prefill K=1 and decode/agg K=3 (no golden AL)." + - "Disagg: NIXL P/D KV (1800s leases), multi-port external LB for wide-EP DEP8, llm-d 0.10.0 prefix/token-load EPP routing, Mooncake RDMA prefix cache at 66.7% node DRAM (~140 GiB/GPU, matching Dynamo); prefill Nixl+MooncakeStore, decode Nixl-only. Multi-prefill coordinator waits up to 600s for prefill /health after decode readiness." + - "Agg DEP8 Mooncake: gpu-memory-utilization 0.87, Mooncake DRAM segment at 45% utilization, MultiConnector (Nixl + SimpleCPUOffload + Mooncake). Route DeepGEMM MegaMoE symmetric memory over GB200 NVLink (NVSHMEM remote/IBGDA off, CUDA fabric handles, NCCL P2P NVL, UCX CUDA IPC over MNNVL). Scrape per-rank vLLM metrics without duplicate frontend counters." + - "Route llm-d launches through infx.launch (gb200-nv LLMD driver) instead of runners/launch_gb200-nv.sh; re-remove archived dsv4-fp4-gb200-llmd-vllm 8k1k master key reintroduced by upstream merge." + - "新增 GB200 llm-d vLLM AgentX 基准:DeepSeek-V4-Pro-0813 DSpark FP4/FP8-KV;分离式 Mooncake 1P1D DEP8/DEP8(并发 64/128/160/256)与 2P1D DEP8/DEP16(并发 256/384/576);聚合 TP8(并发 1-12)、DEP8(并发 12-32)及 DEP8 Mooncake(并发 52/72)。DSpark 启用自适应验证,prefill K=1、decode/聚合 K=3(无 golden AL)。" + - "分离式:NIXL P/D KV(租约 1800s)、DEP8 宽 EP 多端口外部 LB、llm-d 0.10.0 前缀/令牌负载 EPP 路由、66.7% 节点 DRAM 的 Mooncake RDMA 前缀缓存(约 140 GiB/GPU,与 Dynamo 对齐);prefill 为 Nixl+MooncakeStore,decode 仅 Nixl。多预填充协调器在 decode 就绪后最多等待 600s 以轮询 prefill /health。" + - "聚合 DEP8 Mooncake:gpu-memory-utilization 0.87、Mooncake DRAM 段 45% 利用率、MultiConnector(Nixl + SimpleCPUOffload + Mooncake)。DeepGEMM MegaMoE 对称内存经 GB200 NVLink 路由(关闭 NVSHMEM 远程/IBGDA、启用 CUDA fabric 句柄、NCCL P2P NVL、MNNVL 上 UCX CUDA IPC)。按 rank 抓取 vLLM 指标,避免重复的前端计数器。" + - "通过 infx.launch(gb200-nv LLMD 驱动)路由 llm-d 启动,替代 runners/launch_gb200-nv.sh;重新移除 upstream 合并误带回的已归档 dsv4-fp4-gb200-llmd-vllm 8k1k 主配置项。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2719