From 9b49adfebdee5b37ea72d2b4f57f1b061c858902 Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:00:36 +0800 Subject: [PATCH 1/4] feat(bench): add explicit fixed-sequence DSpark comparisons MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add eight-GPU 8k256 comparison points with seven draft tokens and caller-supplied mean acceptance. Reject comparison overrides for AgentX and evals, preserve golden defaults, and forward exact-length controls through CI. 中文:新增八 GPU、8k256、七个草稿 token 的对比配置,由调用方指定平均接受长度;拒绝覆盖 AgentX 黄金曲线及评测,并在 CI 中传递精确长度控制。 --- .../workflows/benchmark-multinode-tmpl.yml | 13 +++- .github/workflows/benchmark-tmpl.yml | 8 ++- .github/workflows/e2e-tests.yml | 36 ++++++++++ .../sglang/b200-fp4-mtp/matched-8k256.yaml | 67 +++++++++++++++++ .../sglang/b300-fp4-mtp/matched-8k256.yaml | 67 +++++++++++++++++ .../sglang/h200-fp4-mtp/matched-8k256.yaml | 68 ++++++++++++++++++ .../single_node/srt_fixed_sequence.sh | 2 +- inferencex-e2e/configs/nvidia-master.yaml | 62 ++++++++++++++++ inferencex-e2e/docs/ci-procedures.md | 6 ++ inferencex-e2e/docs/ci-procedures_zh.md | 6 ++ .../infx/srt_slurm/synthetic_acceptance.py | 38 +++++++++- .../srt_slurm/test_synthetic_acceptance.py | 71 +++++++++++++++++++ inferencex-e2e/perf-changelog.yaml | 11 +++ 13 files changed, 449 insertions(+), 6 deletions(-) create mode 100644 inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml create mode 100644 inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml create mode 100644 inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml diff --git a/.github/workflows/benchmark-multinode-tmpl.yml b/.github/workflows/benchmark-multinode-tmpl.yml index 6c996e3a9b..6745fb4bf6 100644 --- a/.github/workflows/benchmark-multinode-tmpl.yml +++ b/.github/workflows/benchmark-multinode-tmpl.yml @@ -77,6 +77,16 @@ on: required: false type: string default: "" + random-range-ratio: + description: "Fixed-sequence length variation; 1.0 uses exact lengths" + required: false + type: string + default: "0.8" + fixed-sequence-acceptance-length: + description: "Explicit research-only fixed-sequence acceptance target; never used for AgentX or evals" + required: false + type: string + default: "" require-power: description: "Fail fixed-sequence result processing when GPU power is invalid" type: boolean @@ -119,6 +129,7 @@ on: type: string env: + FIXED_SEQUENCE_ACCEPTANCE_LENGTH: ${{ inputs.fixed-sequence-acceptance-length }} PYTHONPATH: ${{ github.workspace }}/inferencex-e2e INFERENCEX_E2E_ROOT: ${{ github.workspace }} PORT: '8888' @@ -131,7 +142,7 @@ env: GPU_METRICS_CSV: 'gpu_metrics.csv' KEEP_LOGS: '0' IS_MULTINODE: 'true' - RANDOM_RANGE_RATIO: 0.8 + RANDOM_RANGE_RATIO: ${{ inputs.random-range-ratio }} # Day-zero models resolved via hf: ids download from the Hub inside the # slurm job (srtctl pre-download + dynamo hub fetch). Anonymous requests # get 429-rate-limited when several workers pull a 444 GB snapshot at diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index 35548863fe..7b2fdad718 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -74,6 +74,11 @@ on: required: false type: string default: '3600' + fixed-sequence-acceptance-length: + description: "Explicit research-only fixed-sequence acceptance target; never used for AgentX or evals" + required: false + type: string + default: "" require-power: description: "Fail result processing when GPU power is invalid" required: false @@ -95,6 +100,7 @@ on: type: string default: "" env: + FIXED_SEQUENCE_ACCEPTANCE_LENGTH: ${{ inputs.fixed-sequence-acceptance-length }} PYTHONPATH: ${{ github.workspace }}/inferencex-e2e INFERENCEX_E2E_ROOT: ${{ github.workspace }} PORT: '8888' @@ -107,7 +113,7 @@ env: GPU_METRICS_CSV: 'gpu_metrics.csv' KEEP_LOGS: '0' IS_MULTINODE: 'false' - RANDOM_RANGE_RATIO: 0.8 + RANDOM_RANGE_RATIO: ${{ inputs.random-range-ratio }} HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} HF_HUB_CACHE: '/mnt/hf_hub_cache/' EXP_NAME: ${{ fromJSON(inputs.config).exp-name }} diff --git a/.github/workflows/e2e-tests.yml b/.github/workflows/e2e-tests.yml index 3310fee119..94a203ad30 100644 --- a/.github/workflows/e2e-tests.yml +++ b/.github/workflows/e2e-tests.yml @@ -31,6 +31,16 @@ on: # zizmor: ignore[concurrency-limits] required: false type: string default: "" + random-range-ratio: + description: "Fixed-sequence length variation (1.0 means exact configured lengths)" + required: false + type: string + default: "0.8" + fixed-sequence-acceptance-length: + description: "Research-only mean committed tokens per verification step; fixed-sequence throughput only" + required: false + type: string + default: "" require-power: description: "Fail benchmark jobs when GPU power is invalid" required: false @@ -133,6 +143,16 @@ on: # zizmor: ignore[concurrency-limits] required: false type: string default: "" + random-range-ratio: + description: "Fixed-sequence length variation (1.0 means exact configured lengths)" + required: false + type: string + default: "0.8" + fixed-sequence-acceptance-length: + description: "Research-only mean committed tokens per verification step; fixed-sequence throughput only" + required: false + type: string + default: "" require-power: description: "Fail benchmark jobs when GPU power is invalid" required: false @@ -373,6 +393,8 @@ jobs: MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} + fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }} + random-range-ratio: ${{ inputs.random-range-ratio }} klaud-run: ${{ inputs.klaud-run }} runner: ${{ matrix.config.runner }} node-count: ${{ matrix.config.node-count }} @@ -399,6 +421,8 @@ jobs: MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} + fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }} + random-range-ratio: ${{ inputs.random-range-ratio }} klaud-run: ${{ inputs.klaud-run }} runner: ${{ matrix.config.runner }} node-count: ${{ matrix.config.node-count }} @@ -429,6 +453,8 @@ jobs: MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} + fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }} + random-range-ratio: ${{ inputs.random-range-ratio }} klaud-run: ${{ inputs.klaud-run }} runner: ${{ matrix.config.runner }} priority: ${{ matrix.config.priority }} @@ -456,6 +482,8 @@ jobs: MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} + fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }} + random-range-ratio: ${{ inputs.random-range-ratio }} klaud-run: ${{ inputs.klaud-run }} runner: ${{ matrix.config.runner }} priority: ${{ matrix.config.priority }} @@ -487,6 +515,8 @@ jobs: MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} + fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }} + random-range-ratio: ${{ inputs.random-range-ratio }} klaud-run: ${{ inputs.klaud-run }} runner: ${{ matrix.config.runner }} node-count: ${{ matrix.config.node-count }} @@ -517,6 +547,8 @@ jobs: MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} + fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }} + random-range-ratio: ${{ inputs.random-range-ratio }} klaud-run: ${{ inputs.klaud-run }} runner: ${{ matrix.config.runner }} node-count: ${{ matrix.config.node-count }} @@ -550,6 +582,8 @@ jobs: MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} + fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }} + random-range-ratio: ${{ inputs.random-range-ratio }} klaud-run: ${{ inputs.klaud-run }} runner: ${{ matrix.config.runner }} priority: ${{ matrix.config.priority }} @@ -574,6 +608,8 @@ jobs: MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} + fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }} + random-range-ratio: ${{ inputs.random-range-ratio }} klaud-run: ${{ inputs.klaud-run }} runner: ${{ matrix.config.runner }} priority: ${{ matrix.config.priority }} diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml new file mode 100644 index 0000000000..cd1fa69a94 --- /dev/null +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml @@ -0,0 +1,67 @@ +# Fixed-sequence DSpark comparison; acceptance is supplied explicitly at dispatch. +base: + schema: 2 + name: dsv41flash-fp4-b200-matched-8k256 + model: + path: hf:deepseek-ai/DeepSeek-V4.1-Flash + container: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:0e1b14e302619a42ef5581b87db6804651d4f946301cf543d74a3a1eb1c33b40 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + trust-remote-code: true + prefill-decode-interval: 16 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 7 + cuda-graph-max-bs-decode: 1 + reasoning-parser: auto + tool-call-parser: auto + watchdog-timeout: 3600 + enable-metrics: true + tensor-parallel-size: 8 + expert-parallel-size: 8 + data-parallel-size: 1 + moe-runner-backend: deep_gemm + disable-shared-experts-fusion: true + speculative-num-draft-tokens: 8 + chunked-prefill-size: 8192 + context-length: 16384 + max-running-requests: 1 + mem-fraction-static: 0.8 + swa-prefix-tails: 128 + stream-interval: 1 + env: + PYTHONNOUSERSITE: '1' + PYTHONUNBUFFERED: '1' + SGLANG_TIMEOUT_KEEP_ALIVE: '900' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV41_REASONING_EFFORT: high + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1' + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' + gpus: 8 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code --dsv4 + env: + MODEL: deepseek-ai/DeepSeek-V4.1-Flash + ISL: '8192' + OSL: '256' + RANDOM_RANGE_RATIO: '1.0' + USE_CHAT_TEMPLATE: 'true' + CONC: '1' +override_tp8_c1: {} diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml new file mode 100644 index 0000000000..f1cb1c4d04 --- /dev/null +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml @@ -0,0 +1,67 @@ +# Fixed-sequence DSpark comparison; acceptance is supplied explicitly at dispatch. +base: + schema: 2 + name: dsv41flash-fp4-b300-matched-8k256 + model: + path: hf:deepseek-ai/DeepSeek-V4.1-Flash + container: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:35ea4d321b0735051dcce2599fc495853ad1ef30f6a1908f0a75e74204362d14 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + trust-remote-code: true + chunked-prefill-size: 8192 + prefill-decode-interval: 16 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 7 + cuda-graph-max-bs-decode: 1 + reasoning-parser: auto + tool-call-parser: auto + watchdog-timeout: 3600 + enable-metrics: true + tensor-parallel-size: 8 + expert-parallel-size: 8 + data-parallel-size: 1 + moe-runner-backend: deep_gemm + disable-shared-experts-fusion: true + speculative-num-draft-tokens: 8 + context-length: 16384 + max-running-requests: 1 + mem-fraction-static: 0.8 + swa-prefix-tails: 128 + stream-interval: 1 + env: + PYTHONNOUSERSITE: '1' + PYTHONUNBUFFERED: '1' + SGLANG_TIMEOUT_KEEP_ALIVE: '900' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV41_REASONING_EFFORT: high + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1' + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' + gpus: 8 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code --dsv4 + env: + MODEL: deepseek-ai/DeepSeek-V4.1-Flash + ISL: '8192' + OSL: '256' + RANDOM_RANGE_RATIO: '1.0' + USE_CHAT_TEMPLATE: 'true' + CONC: '1' +override_tp8_c1: {} diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml new file mode 100644 index 0000000000..775451483a --- /dev/null +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml @@ -0,0 +1,68 @@ +# Fixed-sequence DSpark comparison; acceptance is supplied explicitly at dispatch. +base: + schema: 2 + name: dsv41flash-fp4-h200-matched-8k256 + model: + path: hf:deepseek-ai/DeepSeek-V4.1-Flash + container: lmsysorg/sglang:nightly-dev-cu13-20260923-06008c17 + precision: fp4 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + trust-remote-code: true + data-parallel-size: 1 + expert-parallel-size: 8 + attention-backend: dsv4 + moe-runner-backend: deep_gemm + mem-fraction-static: 0.8 + chunked-prefill-size: 8192 + prefill-decode-interval: 16 + swa-prefix-tails: 128 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 7 + cuda-graph-max-bs-decode: 1 + reasoning-parser: auto + tool-call-parser: auto + watchdog-timeout: 3600 + enable-metrics: true + tensor-parallel-size: 8 + disable-shared-experts-fusion: true + speculative-num-draft-tokens: 8 + context-length: 16384 + max-running-requests: 1 + stream-interval: 1 + env: + PYTHONNOUSERSITE: '1' + PYTHONUNBUFFERED: '1' + SGLANG_TIMEOUT_KEEP_ALIVE: '900' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV41_REASONING_EFFORT: high + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1' + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' + gpus: 8 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code --dsv4 + env: + MODEL: deepseek-ai/DeepSeek-V4.1-Flash + ISL: '8192' + OSL: '256' + RANDOM_RANGE_RATIO: '1.0' + USE_CHAT_TEMPLATE: 'true' + CONC: '1' +override_tp8_c1: {} diff --git a/inferencex-e2e/benchmarks/single_node/srt_fixed_sequence.sh b/inferencex-e2e/benchmarks/single_node/srt_fixed_sequence.sh index 0bc79b9ab4..0a4c511444 100644 --- a/inferencex-e2e/benchmarks/single_node/srt_fixed_sequence.sh +++ b/inferencex-e2e/benchmarks/single_node/srt_fixed_sequence.sh @@ -20,7 +20,7 @@ SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" CLIENT_ARGS=() for argument in "$@"; do case "$argument" in - --trust-remote-code) CLIENT_ARGS+=("$argument") ;; + --trust-remote-code|--dsv4) CLIENT_ARGS+=("$argument") ;; *) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;; esac done diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 07c47e6e40..345df96c5a 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8939,3 +8939,65 @@ dsv41flash-fp4-gb300-sglang-agentic-dspark: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 1, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } + +# Explicit fixed-sequence comparison points; not AgentX workloads. +dsv41flash-fp4-h200-sglang-matched-8k256: + image: lmsysorg/sglang:nightly-dev-cu13-20260923-06008c17 + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:h200-dgxc + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 256 + search-space: + - tp: 8 + ep: 8 + dp-attn: false + spec-decoding: mtp + conc-list: + - 1 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml +dsv41flash-fp4-b200-sglang-matched-8k256: + image: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:0e1b14e302619a42ef5581b87db6804651d4f946301cf543d74a3a1eb1c33b40 + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:b200-nscale + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 256 + search-space: + - tp: 8 + ep: 8 + dp-attn: false + spec-decoding: mtp + conc-list: + - 1 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml +dsv41flash-fp4-b300-sglang-matched-8k256: + image: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:35ea4d321b0735051dcce2599fc495853ad1ef30f6a1908f0a75e74204362d14 + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:b300-dsxe + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 256 + search-space: + - tp: 8 + ep: 8 + dp-attn: false + spec-decoding: mtp + conc-list: + - 1 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml diff --git a/inferencex-e2e/docs/ci-procedures.md b/inferencex-e2e/docs/ci-procedures.md index 94a205dace..6c5d7503ea 100644 --- a/inferencex-e2e/docs/ci-procedures.md +++ b/inferencex-e2e/docs/ci-procedures.md @@ -580,3 +580,9 @@ See [OperatorX GitHub Actions](../../operatorx/CI.md) for dispatch, coverage, artifacts, cancellation, and validation. For H200 DeepSeek-V4.1 Flash SGLang AgentX performance at concurrency 64 or above, the launch policy (`SALLOC_TIME_BUMPS` in `infx/launch/policy.py`) allows a 1440-minute Slurm allocation and the reusable workflow allows 1470 minutes. This accommodates normal warmup and the unchanged 3600-second profile; lower concurrencies and eval-only jobs retain the standard deadlines. Run `35775895782` exhausted the previous eight-hour allocation during progressing, error-free warmup. A failed-only retry retains the original workflow deadline, so deadline changes require a new workflow run. + +### Explicit fixed-sequence DSpark comparisons + +For requested research comparisons of `dsv41flash`, `e2e-tests.yml` accepts `fixed-sequence-acceptance-length` (mean committed tokens per verification step, including the verification token) and `random-range-ratio` (`1.0` uses exact configured lengths). Dispatch the workflow from the branch that defines these inputs, and use `--no-evals`. This optional mode requires fixed-sequence throughput, an explicit positive DSpark draft count, and a finite AL in `[1, draft count + 1]`; AgentX and eval requests reject it. The SRT connector injects native server settings and records the target and draft count in the benchmark environment. It does not modify or add measured golden curves. Without the input, existing golden-AgentX and real-verification behavior is unchanged. + +These are research artifacts, not evidence that acceptance was measured on the synthetic prompt dataset. Record checkpoint, effective target/draft precision, cache formats, GPU count, exact prompt/output lengths, concurrency, configured and realized AL, and measured TPOT separately from throughput derived as concurrency divided by TPOT. diff --git a/inferencex-e2e/docs/ci-procedures_zh.md b/inferencex-e2e/docs/ci-procedures_zh.md index 26fd59d79f..8ed682ff68 100644 --- a/inferencex-e2e/docs/ci-procedures_zh.md +++ b/inferencex-e2e/docs/ci-procedures_zh.md @@ -561,3 +561,9 @@ attention 支持 torch 和 AITER。 [OperatorX GitHub Actions](../../operatorx/CI_zh.md)。 H200 DeepSeek-V4.1 Flash SGLang AgentX 在并发 64 及以上的性能任务由启动策略(`infx/launch/policy.py` 中的 `SALLOC_TIME_BUMPS`)允许 1440 分钟 Slurm 分配,并允许 1470 分钟 GitHub 任务,以容纳正常预热及保持不变的 3600 秒正式测试;更低并发和 eval-only 任务仍使用标准期限。运行 `35775895782` 在持续推进、请求无错误的预热期间耗尽了原有八小时分配。仅重试失败任务会保留原工作流期限,因此修改期限后必须启动新运行。 + +### 显式固定序列 DSpark 对比 + +针对用户要求的 `dsv41flash` 研究对比,`e2e-tests.yml` 接受 `fixed-sequence-acceptance-length`(每次验证步骤平均提交的 token 数,包含验证 token)及 `random-range-ratio`(`1.0` 使用配置中的精确长度)。请从定义这些输入的分支调度工作流,并使用 `--no-evals`。此可选模式只允许固定序列吞吐测试,要求显式的正整数 DSpark 草稿数量,以及 `[1, 草稿数量 + 1]` 范围内的有限 AL;AgentX 和评测请求会被拒绝。SRT 连接器注入原生服务端设置,并在 benchmark 环境中记录目标 AL 和草稿数量,不修改或新增实测黄金曲线。未提供该输入时,现有 AgentX 黄金曲线和真实验证行为保持不变。 + +这些是研究产物,不代表在合成提示词数据集上实测得到的接受长度。需记录 checkpoint、目标和草稿的实际精度、缓存格式、GPU 数量、精确输入和输出长度、并发数、配置与实测 AL,并区分实测 TPOT 和由并发数除以 TPOT 推算的吞吐量。 diff --git a/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py b/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py index 45a9f2e153..4db73c864a 100644 --- a/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py +++ b/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py @@ -1,4 +1,4 @@ -"""Select AgentX golden acceptance automatically and pass native srtctl overrides.""" +"""Plan native acceptance for AgentX and explicit fixed-sequence comparisons.""" from __future__ import annotations @@ -6,6 +6,7 @@ import copy import fnmatch import json +import math import os import re import subprocess @@ -37,6 +38,28 @@ TRT_VARIABLE = "TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS" +def fixed_sequence_acceptance( + spec: Mapping[str, Any], environment: Mapping[str, str] +) -> float | None: + """Validate an explicit research comparison, separate from AgentX golden curves.""" + value = environment.get("FIXED_SEQUENCE_ACCEPTANCE_LENGTH", "") + if not value: + return None + if environment.get("IS_AGENTIC", "").lower() not in {"0", "false"}: + raise ValueError("Fixed-sequence acceptance cannot override AgentX golden curves") + if any(environment.get(key) != "false" for key in ("RUN_EVAL", "EVAL_ONLY")): + raise ValueError("Fixed-sequence acceptance requires throughput-only execution") + if environment.get("MODEL_PREFIX") != "dsv41flash" or spec.get("method") != "dspark": + raise ValueError("Fixed-sequence acceptance supports dsv41flash DSpark comparisons only") + drafts = spec.get("num_speculative_tokens") + if type(drafts) is not int or drafts < 1: + raise ValueError("Fixed-sequence acceptance requires an explicit positive draft count") + acceptance = float(value) + if not math.isfinite(acceptance) or not 1 <= acceptance <= drafts + 1: + raise ValueError("Fixed-sequence acceptance must be finite and within [1, drafts + 1]") + return acceptance + + def spec_parameters(role: Mapping[str, Any], engine: str) -> dict[str, Any]: args = role.get("args", {}) if engine == "atom": @@ -97,9 +120,11 @@ def build_overrides( *, golden_dir: Path = GOLDEN_DIR, ) -> list[str]: - """Infer acceptance from generation parameters; never accept a caller AL.""" + """Use golden AgentX acceptance or an explicit, isolated fixed-sequence comparison.""" engine = ENGINES.get(framework) if engine is None: + if environment.get("FIXED_SEQUENCE_ACCEPTANCE_LENGTH"): + raise ValueError("Framework does not support fixed-sequence acceptance") return [] roles = recipe.get("roles", {}) synthetic = ( @@ -110,12 +135,19 @@ def build_overrides( # Prefill may have a different MTP depth; generation defines the AL target. generation = roles.get("decode", roles.get("agg", {})) spec = spec_parameters(generation, engine) - al = None + al = fixed_sequence_acceptance(spec, environment) if synthetic and spec: al = golden_length( environment["MODEL_PREFIX"], spec, environment["THINKING_MODE"], golden_dir ) overrides = [] + if environment.get("FIXED_SEQUENCE_ACCEPTANCE_LENGTH"): + overrides += [ + "--set", + f"benchmark.env.FIXED_SEQUENCE_ACCEPTANCE_LENGTH={json.dumps(str(al))}", + "--set", + f"benchmark.env.FIXED_SEQUENCE_DRAFT_TOKENS={json.dumps(str(spec['num_speculative_tokens']))}", + ] variables = {"sglang": SGLANG_VARIABLES, "trtllm": (TRT_VARIABLE,)}.get(engine, ()) # SRT applies recipe-wide environment after role environment. Keep simulation # role-local so global values cannot override the golden AL or leak into evals. diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py b/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py index aa6cbf5be4..a90212e53d 100644 --- a/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py +++ b/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py @@ -86,6 +86,77 @@ def vllm_recipe(method: str = "mtp", **extra: Any) -> dict[str, Any]: } +@pytest.mark.parametrize("framework", ["vllm", "sglang"]) +def test_fixed_sequence_comparison_preserves_draft_work(framework: str, tmp_path: Path) -> None: + recipe = vllm_recipe("dspark", num_speculative_tokens=7, draft_sample_method="greedy") + if framework == "sglang": + recipe = { + "roles": { + "agg": { + "args": { + "speculative-algorithm": "DSPARK", + "speculative-dspark-block-size": 7, + "speculative-num-draft-tokens": 8, + } + } + } + } + env = { + **ENV, + "MODEL_PREFIX": "dsv41flash", + "IS_AGENTIC": "0", + "RUN_EVAL": "false", + "FIXED_SEQUENCE_ACCEPTANCE_LENGTH": "5.7", + } + result = apply_native(recipe, build_overrides(recipe, framework, env, golden_dir=tmp_path)) + assert result["benchmark"]["env"] == { + "FIXED_SEQUENCE_ACCEPTANCE_LENGTH": "5.7", + "FIXED_SEQUENCE_DRAFT_TOKENS": "7", + } + if framework == "vllm": + assert json.loads(result["roles"]["agg"]["args"]["speculative-config"]) == { + "method": "dspark", + "num_speculative_tokens": 7, + "draft_sample_method": "greedy", + "rejection_sample_method": "synthetic", + "synthetic_acceptance_length": 5.7, + } + else: + assert result["roles"]["agg"]["args"]["speculative-num-draft-tokens"] == 8 + assert result["roles"]["agg"]["env"] == { + "SGLANG_SIMULATE_ACC_LEN": "5.7", + "SGLANG_SIMULATE_ACC_METHOD": "match-expected", + "SGLANG_SIMULATE_ACC_TOKEN_MODE": "real-draft-token", + } + + +@pytest.mark.parametrize( + ("changed", "message"), + [ + ({"IS_AGENTIC": "1"}, "AgentX golden"), + ({"RUN_EVAL": "true"}, "throughput-only"), + ({"EVAL_ONLY": "true"}, "throughput-only"), + ({"MODEL_PREFIX": "other"}, "dsv41flash DSpark"), + ({"FIXED_SEQUENCE_ACCEPTANCE_LENGTH": "nan"}, "finite"), + ({"FIXED_SEQUENCE_ACCEPTANCE_LENGTH": "0.9"}, "within"), + ({"FIXED_SEQUENCE_ACCEPTANCE_LENGTH": "4.1"}, "within"), + ], +) +def test_fixed_sequence_comparison_rejects_incompatible_work( + changed: dict[str, str], message: str, tmp_path: Path +) -> None: + env = { + **ENV, + "MODEL_PREFIX": "dsv41flash", + "IS_AGENTIC": "0", + "RUN_EVAL": "false", + "FIXED_SEQUENCE_ACCEPTANCE_LENGTH": "3.0", + **changed, + } + with pytest.raises(ValueError, match=message): + build_overrides(vllm_recipe("dspark"), "vllm", env, golden_dir=tmp_path) + + @pytest.mark.parametrize( ("prefix", "method", "extra", "expected"), [ diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 2b184929a0..17278d8fe1 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9127,3 +9127,14 @@ - "Add Kimi-K3 MXFP4 vLLM agentic-coding on MI355X (TP8, DSpark): dcp1 c1/c4 GPU-resident and c8-c14 with SimpleCPUOffload DRAM offload, plus a new dcp8 c44/c48/c70 throughput band (mtp synthetic acceptance, no draft); vLLM ROCm image vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89" - "The Inferact/Kimi-K3-DSpark draft (K3DSparkModel, model_type k3_dspark, 5 layers, hidden 7168, torch_dtype bfloat16, no quantization_config) keeps every layer at its pristine dtype. vLLM builds all its modules with quant_config from get_draft_quant_config() (vllm/models/kimi_k3/nvidia/dspark_mla.py), which returns None because K3DSparkModel is explicitly excluded from the DeepSeek-V4 branch that would set draft.quantization = target.quantization (vllm/config/speculative.py:1431); so context_proj (ReplicatedLinear), context_kv_proj (MergedColumnParallelLinear) and each decoder layer's MLA q/kv projections and dense KimiMLP (gate/up/down) load unquantized in BF16 -- no ptpc_fp8, no mxfp4, no INT4 weights. The draft has no FusedMoE/block_sparse_moe, so VLLM_ROCM_USE_AITER_MOE_SITUV2 (A8W4) never touches it, and its KV cache stays fp8 (kv_cache_dtype). This PR only changes draft depth (num_speculative_tokens 4->7) and enables INT4 custom quick all-reduce (VLLM_ROCM_QUICK_REDUCE_QUANTIZATION=INT4, vllm/distributed/device_communicators/quick_all_reduce.py); because the draft shares the target's TP8 group (vllm/v1/spec_decode/draft_model.py), that INT4 applies to its tensor-parallel reductions too -- a collective-reduction transport precision, not any draft weight, activation, or KV dtype." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3561 + +- config-keys: + - dsv41flash-fp4-h200-sglang-matched-8k256 + - dsv41flash-fp4-b200-sglang-matched-8k256 + - dsv41flash-fp4-b300-sglang-matched-8k256 + scenario-type: + - fixed-seq-len + description: + - "Add fixed 8192/256, eight-GPU, seven-draft DSpark comparison points and an explicit throughput-only acceptance input, isolated from AgentX golden curves and evals." + - "新增固定 8192/256、八 GPU、七个草稿 token 的 DSpark 对比配置,以及仅用于吞吐测试的显式接受长度输入;与 AgentX 黄金曲线和评测隔离。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX From 1669d9761f310e90050d37d3c7a037c4f6ed75e2 Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:07:38 +0800 Subject: [PATCH 2/4] fix(bench): match the V4.1 single-user chat prefix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Use the released numeric high-effort header and account for its tokens in fixed-sequence requests. Link the research changelog entries to PR 3607. 中文:使用 V4.1 发布版本的数字化 high-effort 前缀,并在固定序列请求中计入其 token;将研究日志关联至 PR 3607。 --- inferencex-e2e/docs/ci-procedures.md | 2 ++ inferencex-e2e/docs/ci-procedures_zh.md | 2 ++ .../infx/bench_serving/benchmark_serving.py | 16 ++++++++++++++- .../bench_serving/test_benchmark_serving.py | 20 +++++++++++++++++++ inferencex-e2e/perf-changelog.yaml | 13 +++++++++++- 5 files changed, 51 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/docs/ci-procedures.md b/inferencex-e2e/docs/ci-procedures.md index 6c5d7503ea..15f61871ee 100644 --- a/inferencex-e2e/docs/ci-procedures.md +++ b/inferencex-e2e/docs/ci-procedures.md @@ -586,3 +586,5 @@ For H200 DeepSeek-V4.1 Flash SGLang AgentX performance at concurrency 64 or abov For requested research comparisons of `dsv41flash`, `e2e-tests.yml` accepts `fixed-sequence-acceptance-length` (mean committed tokens per verification step, including the verification token) and `random-range-ratio` (`1.0` uses exact configured lengths). Dispatch the workflow from the branch that defines these inputs, and use `--no-evals`. This optional mode requires fixed-sequence throughput, an explicit positive DSpark draft count, and a finite AL in `[1, draft count + 1]`; AgentX and eval requests reject it. The SRT connector injects native server settings and records the target and draft count in the benchmark environment. It does not modify or add measured golden curves. Without the input, existing golden-AgentX and real-verification behavior is unchanged. These are research artifacts, not evidence that acceptance was measured on the synthetic prompt dataset. Record checkpoint, effective target/draft precision, cache formats, GPU count, exact prompt/output lengths, concurrency, configured and realized AL, and measured TPOT separately from throughput derived as concurrency divided by TPOT. + +The fixed-sequence `--dsv4` client path recognizes the V4.1 tokenizer name and includes its released numeric high-effort header (75), including that header in prompt-length accounting. diff --git a/inferencex-e2e/docs/ci-procedures_zh.md b/inferencex-e2e/docs/ci-procedures_zh.md index 8ed682ff68..8a486fb6e1 100644 --- a/inferencex-e2e/docs/ci-procedures_zh.md +++ b/inferencex-e2e/docs/ci-procedures_zh.md @@ -567,3 +567,5 @@ H200 DeepSeek-V4.1 Flash SGLang AgentX 在并发 64 及以上的性能任务由 针对用户要求的 `dsv41flash` 研究对比,`e2e-tests.yml` 接受 `fixed-sequence-acceptance-length`(每次验证步骤平均提交的 token 数,包含验证 token)及 `random-range-ratio`(`1.0` 使用配置中的精确长度)。请从定义这些输入的分支调度工作流,并使用 `--no-evals`。此可选模式只允许固定序列吞吐测试,要求显式的正整数 DSpark 草稿数量,以及 `[1, 草稿数量 + 1]` 范围内的有限 AL;AgentX 和评测请求会被拒绝。SRT 连接器注入原生服务端设置,并在 benchmark 环境中记录目标 AL 和草稿数量,不修改或新增实测黄金曲线。未提供该输入时,现有 AgentX 黄金曲线和真实验证行为保持不变。 这些是研究产物,不代表在合成提示词数据集上实测得到的接受长度。需记录 checkpoint、目标和草稿的实际精度、缓存格式、GPU 数量、精确输入和输出长度、并发数、配置与实测 AL,并区分实测 TPOT 和由并发数除以 TPOT 推算的吞吐量。 + +固定序列客户端的 `--dsv4` 路径会识别 V4.1 tokenizer 名称,加入发布版本的数字化 high-effort 前缀(75),并将该前缀计入提示词长度。 diff --git a/inferencex-e2e/infx/bench_serving/benchmark_serving.py b/inferencex-e2e/infx/bench_serving/benchmark_serving.py index 6c8a185d05..e8fca37728 100644 --- a/inferencex-e2e/infx/bench_serving/benchmark_serving.py +++ b/inferencex-e2e/infx/bench_serving/benchmark_serving.py @@ -173,8 +173,22 @@ def _apply_chat_template(prompt: str, tokenizer: PreTrainedTokenizerBase, dsv4: fall back to the tokenizer's built-in jinja chat template. """ if dsv4: + messages = [{"role": "user", "content": prompt}] + if "deepseek-v4.1" in getattr(tokenizer, "name_or_path", "").lower(): + # V4.1's released text encoder adds its numeric high-effort header. + # This client renders one user message; no multimodal/tool encoding is involved. + messages.insert( + 0, + { + "role": "system", + "content": ( + "<|System|>Reasoning Effort: 75 " # noqa: RUF001 + "(range 1-100, the higher the value, the more thorough the reasoning)\n\n" + ), + }, + ) return dsv4_encode_messages( - [{"role": "user", "content": prompt}], + messages, thinking_mode="thinking", ) return tokenizer.apply_chat_template( diff --git a/inferencex-e2e/infx/tests/bench_serving/test_benchmark_serving.py b/inferencex-e2e/infx/tests/bench_serving/test_benchmark_serving.py index d7a506c10d..8fd481d05b 100644 --- a/inferencex-e2e/infx/tests/bench_serving/test_benchmark_serving.py +++ b/inferencex-e2e/infx/tests/bench_serving/test_benchmark_serving.py @@ -69,3 +69,23 @@ def test_client_preserves_outcome_before_failure( assert 'error' not in outcome assert outcome['failed'] == {100: 0, 95: 5, 94: 6, 0: 100}[completed] assert outcome['max_failure_rate'] == 0.05 + + +@pytest.mark.parametrize( + ('model', 'expected'), + [ + ( + 'deepseek-ai/DeepSeek-V4-Pro', + '<|begin▁of▁sentence|><|User|>hello\nworld<|Assistant|>', + ), + ( + 'deepseek-ai/DeepSeek-V4.1-Flash', + '<|begin▁of▁sentence|><|System|>Reasoning Effort: 75 ' + '(range 1-100, the higher the value, the more thorough the reasoning)\n\n' + '<|User|>hello\nworld<|Assistant|>', + ), + ], +) +def test_fixed_sequence_chat_uses_checkpoint_family_header(model: str, expected: str) -> None: + tokenizer = Namespace(name_or_path=model) + assert client._apply_chat_template('hello\nworld', tokenizer, dsv4=True) == expected diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 17278d8fe1..6c6711c259 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9137,4 +9137,15 @@ description: - "Add fixed 8192/256, eight-GPU, seven-draft DSpark comparison points and an explicit throughput-only acceptance input, isolated from AgentX golden curves and evals." - "新增固定 8192/256、八 GPU、七个草稿 token 的 DSpark 对比配置,以及仅用于吞吐测试的显式接受长度输入;与 AgentX 黄金曲线和评测隔离。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3607 + +- config-keys: + - dsv41flash-fp4-h200-sglang-matched-8k256 + - dsv41flash-fp4-b200-sglang-matched-8k256 + - dsv41flash-fp4-b300-sglang-matched-8k256 + scenario-type: + - fixed-seq-len + description: + - "Render the released V4.1 numeric high-effort system header in the fixed-sequence client; include the header in exact prompt-length accounting." + - "在固定序列客户端中渲染 V4.1 发布版本的数字化 high-effort 系统前缀,并将其计入精确提示词长度。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3607 From c806a8518d393f41df2369cffe8f4e0c8c3780a0 Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:36:16 +0800 Subject: [PATCH 3/4] fix(bench): use native W4A8 MoE on Blackwell comparisons MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit B200/B300 对比配置改用原生 W4A8 MoE 后端,避开 DeepGEMM 缩放布局断言,并记录 H200 草稿精度限制。 --- .../dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml | 2 +- .../dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml | 2 +- inferencex-e2e/docs/ci-procedures.md | 2 ++ inferencex-e2e/docs/ci-procedures_zh.md | 2 ++ inferencex-e2e/perf-changelog.yaml | 10 ++++++++++ 5 files changed, 16 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml index cd1fa69a94..895691462c 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/matched-8k256.yaml @@ -35,7 +35,7 @@ base: tensor-parallel-size: 8 expert-parallel-size: 8 data-parallel-size: 1 - moe-runner-backend: deep_gemm + moe-runner-backend: flashinfer_mxfp4 disable-shared-experts-fusion: true speculative-num-draft-tokens: 8 chunked-prefill-size: 8192 diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml index f1cb1c4d04..8bcef86c71 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/matched-8k256.yaml @@ -36,7 +36,7 @@ base: tensor-parallel-size: 8 expert-parallel-size: 8 data-parallel-size: 1 - moe-runner-backend: deep_gemm + moe-runner-backend: flashinfer_mxfp4 disable-shared-experts-fusion: true speculative-num-draft-tokens: 8 context-length: 16384 diff --git a/inferencex-e2e/docs/ci-procedures.md b/inferencex-e2e/docs/ci-procedures.md index 15f61871ee..5510b5dbc5 100644 --- a/inferencex-e2e/docs/ci-procedures.md +++ b/inferencex-e2e/docs/ci-procedures.md @@ -588,3 +588,5 @@ For requested research comparisons of `dsv41flash`, `e2e-tests.yml` accepts `fix These are research artifacts, not evidence that acceptance was measured on the synthetic prompt dataset. Record checkpoint, effective target/draft precision, cache formats, GPU count, exact prompt/output lengths, concurrency, configured and realized AL, and measured TPOT separately from throughput derived as concurrency divided by TPOT. The fixed-sequence `--dsv4` client path recognizes the V4.1 tokenizer name and includes its released numeric high-effort header (75), including that header in prompt-length accounting. + +The B200/B300 comparison recipes use `flashinfer_mxfp4` with its default MXFP8 activations and checkpoint MXFP4 expert weights. The pinned DeepGEMM path failed its packed-scale shape assertion. On H200, this backend defaults to W4A16; selecting `flashinfer-mxfp4-moe-precision: fp8` also lowers DSpark activations because the pinned DSpark worker shares the target MoE settings. Do not label the default H200 path W4A8 or silently apply that flag under the draft-as-shipped policy. diff --git a/inferencex-e2e/docs/ci-procedures_zh.md b/inferencex-e2e/docs/ci-procedures_zh.md index 8a486fb6e1..2aa07bd118 100644 --- a/inferencex-e2e/docs/ci-procedures_zh.md +++ b/inferencex-e2e/docs/ci-procedures_zh.md @@ -569,3 +569,5 @@ H200 DeepSeek-V4.1 Flash SGLang AgentX 在并发 64 及以上的性能任务由 这些是研究产物,不代表在合成提示词数据集上实测得到的接受长度。需记录 checkpoint、目标和草稿的实际精度、缓存格式、GPU 数量、精确输入和输出长度、并发数、配置与实测 AL,并区分实测 TPOT 和由并发数除以 TPOT 推算的吞吐量。 固定序列客户端的 `--dsv4` 路径会识别 V4.1 tokenizer 名称,加入发布版本的数字化 high-effort 前缀(75),并将该前缀计入提示词长度。 + +B200/B300 对比配置使用 `flashinfer_mxfp4`,保留其默认 MXFP8 激活与检查点中的 MXFP4 专家权重。固定版本的 DeepGEMM 路径在打包缩放因子形状断言处失败。H200 上该后端默认为 W4A16;设置 `flashinfer-mxfp4-moe-precision: fp8` 也会降低 DSpark 激活精度,因为固定版本的 DSpark worker 与目标模型共享 MoE 设置。不得将 H200 默认路径标为 W4A8,也不得在草稿保持发布精度的规则下悄然启用该参数。 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 6c6711c259..c9b967ee15 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9149,3 +9149,13 @@ - "Render the released V4.1 numeric high-effort system header in the fixed-sequence client; include the header in exact prompt-length accounting." - "在固定序列客户端中渲染 V4.1 发布版本的数字化 high-effort 系统前缀,并将其计入精确提示词长度。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3607 + +- config-keys: + - dsv41flash-fp4-b200-sglang-matched-8k256 + - dsv41flash-fp4-b300-sglang-matched-8k256 + scenario-type: + - fixed-seq-len + description: + - "Use the native FlashInfer MXFP4 MoE runner on B200/B300 to avoid the DeepGEMM packed-scale layout assertion. Retain its default MXFP8 activations, checkpoint expert weights, seven draft tokens, and explicit dispatch AL." + - "B200/B300 改用原生 FlashInfer MXFP4 MoE 后端,避开 DeepGEMM 打包缩放因子的布局断言;保留默认 MXFP8 激活、检查点专家权重、七个草稿 token 及显式调度的 AL。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3607 From 032cc4f568f3f38ca881f28550d557340897ddc8 Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Wed, 30 Sep 2026 19:07:05 +0800 Subject: [PATCH 4/4] fix(bench): enable native H200 W4A8 research comparison MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 按用户明确授权,研究对比中的 H200 目标模型和 DSpark 使用原生 W4A8 路径;保留发布权重和 wo_a 默认转换,不作为正式基准贡献。 --- .../dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml | 3 ++- inferencex-e2e/docs/ci-procedures.md | 2 +- inferencex-e2e/docs/ci-procedures_zh.md | 2 +- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 13 insertions(+), 3 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml index 775451483a..a5c03e33bb 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h200-fp4-mtp/matched-8k256.yaml @@ -27,7 +27,8 @@ base: data-parallel-size: 1 expert-parallel-size: 8 attention-backend: dsv4 - moe-runner-backend: deep_gemm + moe-runner-backend: flashinfer_mxfp4 + flashinfer-mxfp4-moe-precision: fp8 mem-fraction-static: 0.8 chunked-prefill-size: 8192 prefill-decode-interval: 16 diff --git a/inferencex-e2e/docs/ci-procedures.md b/inferencex-e2e/docs/ci-procedures.md index 5510b5dbc5..6904d682a4 100644 --- a/inferencex-e2e/docs/ci-procedures.md +++ b/inferencex-e2e/docs/ci-procedures.md @@ -589,4 +589,4 @@ These are research artifacts, not evidence that acceptance was measured on the s The fixed-sequence `--dsv4` client path recognizes the V4.1 tokenizer name and includes its released numeric high-effort header (75), including that header in prompt-length accounting. -The B200/B300 comparison recipes use `flashinfer_mxfp4` with its default MXFP8 activations and checkpoint MXFP4 expert weights. The pinned DeepGEMM path failed its packed-scale shape assertion. On H200, this backend defaults to W4A16; selecting `flashinfer-mxfp4-moe-precision: fp8` also lowers DSpark activations because the pinned DSpark worker shares the target MoE settings. Do not label the default H200 path W4A8 or silently apply that flag under the draft-as-shipped policy. +The B200/B300 comparison recipes use `flashinfer_mxfp4` with its default MXFP8 activations and checkpoint MXFP4 expert weights. The pinned DeepGEMM path failed its packed-scale shape assertion. On H200, this backend defaults to W4A16; selecting `flashinfer-mxfp4-moe-precision: fp8` also lowers DSpark activations because the pinned DSpark worker shares the target MoE settings. The research-only H200 comparison explicitly selects this flag with user authorization for target and draft W4A8. It is excluded from contribution/official submission intent and does not satisfy the draft-as-shipped contribution rule. Do not use this recipe as precedent for production submissions. diff --git a/inferencex-e2e/docs/ci-procedures_zh.md b/inferencex-e2e/docs/ci-procedures_zh.md index 2aa07bd118..f5994f2485 100644 --- a/inferencex-e2e/docs/ci-procedures_zh.md +++ b/inferencex-e2e/docs/ci-procedures_zh.md @@ -570,4 +570,4 @@ H200 DeepSeek-V4.1 Flash SGLang AgentX 在并发 64 及以上的性能任务由 固定序列客户端的 `--dsv4` 路径会识别 V4.1 tokenizer 名称,加入发布版本的数字化 high-effort 前缀(75),并将该前缀计入提示词长度。 -B200/B300 对比配置使用 `flashinfer_mxfp4`,保留其默认 MXFP8 激活与检查点中的 MXFP4 专家权重。固定版本的 DeepGEMM 路径在打包缩放因子形状断言处失败。H200 上该后端默认为 W4A16;设置 `flashinfer-mxfp4-moe-precision: fp8` 也会降低 DSpark 激活精度,因为固定版本的 DSpark worker 与目标模型共享 MoE 设置。不得将 H200 默认路径标为 W4A8,也不得在草稿保持发布精度的规则下悄然启用该参数。 +B200/B300 对比配置使用 `flashinfer_mxfp4`,保留其默认 MXFP8 激活与检查点中的 MXFP4 专家权重。固定版本的 DeepGEMM 路径在打包缩放因子形状断言处失败。H200 上该后端默认为 W4A16;设置 `flashinfer-mxfp4-moe-precision: fp8` 也会降低 DSpark 激活精度,因为固定版本的 DSpark worker 与目标模型共享 MoE 设置。本研究专用 H200 对比经用户明确授权,启用该参数以使目标模型和草稿均采用 W4A8。本配置不用于贡献或正式提交,也不满足贡献流程中草稿保持发布精度的规则;不得将其作为生产提交的先例。 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index c9b967ee15..c90c6ed962 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9159,3 +9159,12 @@ - "Use the native FlashInfer MXFP4 MoE runner on B200/B300 to avoid the DeepGEMM packed-scale layout assertion. Retain its default MXFP8 activations, checkpoint expert weights, seven draft tokens, and explicit dispatch AL." - "B200/B300 改用原生 FlashInfer MXFP4 MoE 后端,避开 DeepGEMM 打包缩放因子的布局断言;保留默认 MXFP8 激活、检查点专家权重、七个草稿 token 及显式调度的 AL。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3607 + +- config-keys: + - dsv41flash-fp4-h200-sglang-matched-8k256 + scenario-type: + - fixed-seq-len + description: + - "Research-only H200 comparison: select native FlashInfer MXFP4 MoE with FP8 activations for target and DSpark. Explicitly authorized for this comparison; not an official benchmark submission. Checkpoint weights and default wo_a conversion are retained." + - "仅用于研究的 H200 对比:目标模型和 DSpark 均选择原生 FlashInfer MXFP4 MoE 与 FP8 激活。用户已明确授权此对比用途,不作为正式基准提交;保留检查点权重及默认 wo_a 转换。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3607