Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 12 additions & 1 deletion .github/workflows/benchmark-multinode-tmpl.yml
Original file line number Diff line number Diff line change
Expand Up @@ -77,6 +77,16 @@ on:
required: false
type: string
default: ""
random-range-ratio:
description: "Fixed-sequence length variation; 1.0 uses exact lengths"
required: false
type: string
default: "0.8"
fixed-sequence-acceptance-length:
description: "Explicit research-only fixed-sequence acceptance target; never used for AgentX or evals"
required: false
type: string
default: ""
require-power:
description: "Fail fixed-sequence result processing when GPU power is invalid"
type: boolean
Expand Down Expand Up @@ -119,6 +129,7 @@ on:
type: string

env:
FIXED_SEQUENCE_ACCEPTANCE_LENGTH: ${{ inputs.fixed-sequence-acceptance-length }}
PYTHONPATH: ${{ github.workspace }}/inferencex-e2e
INFERENCEX_E2E_ROOT: ${{ github.workspace }}
PORT: '8888'
Expand All @@ -131,7 +142,7 @@ env:
GPU_METRICS_CSV: 'gpu_metrics.csv'
KEEP_LOGS: '0'
IS_MULTINODE: 'true'
RANDOM_RANGE_RATIO: 0.8
RANDOM_RANGE_RATIO: ${{ inputs.random-range-ratio }}
# Day-zero models resolved via hf: ids download from the Hub inside the
# slurm job (srtctl pre-download + dynamo hub fetch). Anonymous requests
# get 429-rate-limited when several workers pull a 444 GB snapshot at
Expand Down
8 changes: 7 additions & 1 deletion .github/workflows/benchmark-tmpl.yml
Original file line number Diff line number Diff line change
Expand Up @@ -74,6 +74,11 @@ on:
required: false
type: string
default: '3600'
fixed-sequence-acceptance-length:
description: "Explicit research-only fixed-sequence acceptance target; never used for AgentX or evals"
required: false
type: string
default: ""
require-power:
description: "Fail result processing when GPU power is invalid"
required: false
Expand All @@ -95,6 +100,7 @@ on:
type: string
default: ""
env:
FIXED_SEQUENCE_ACCEPTANCE_LENGTH: ${{ inputs.fixed-sequence-acceptance-length }}
PYTHONPATH: ${{ github.workspace }}/inferencex-e2e
INFERENCEX_E2E_ROOT: ${{ github.workspace }}
PORT: '8888'
Expand All @@ -107,7 +113,7 @@ env:
GPU_METRICS_CSV: 'gpu_metrics.csv'
KEEP_LOGS: '0'
IS_MULTINODE: 'false'
RANDOM_RANGE_RATIO: 0.8
RANDOM_RANGE_RATIO: ${{ inputs.random-range-ratio }}
HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
HF_HUB_CACHE: '/mnt/hf_hub_cache/'
EXP_NAME: ${{ fromJSON(inputs.config).exp-name }}
Expand Down
36 changes: 36 additions & 0 deletions .github/workflows/e2e-tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,16 @@ on: # zizmor: ignore[concurrency-limits]
required: false
type: string
default: ""
random-range-ratio:
description: "Fixed-sequence length variation (1.0 means exact configured lengths)"
required: false
type: string
default: "0.8"
fixed-sequence-acceptance-length:
description: "Research-only mean committed tokens per verification step; fixed-sequence throughput only"
required: false
type: string
default: ""
require-power:
description: "Fail benchmark jobs when GPU power is invalid"
required: false
Expand Down Expand Up @@ -133,6 +143,16 @@ on: # zizmor: ignore[concurrency-limits]
required: false
type: string
default: ""
random-range-ratio:
description: "Fixed-sequence length variation (1.0 means exact configured lengths)"
required: false
type: string
default: "0.8"
fixed-sequence-acceptance-length:
description: "Research-only mean committed tokens per verification step; fixed-sequence throughput only"
required: false
type: string
default: ""
require-power:
description: "Fail benchmark jobs when GPU power is invalid"
required: false
Expand Down Expand Up @@ -373,6 +393,8 @@ jobs:
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }}
random-range-ratio: ${{ inputs.random-range-ratio }}
klaud-run: ${{ inputs.klaud-run }}
runner: ${{ matrix.config.runner }}
node-count: ${{ matrix.config.node-count }}
Expand All @@ -399,6 +421,8 @@ jobs:
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }}
random-range-ratio: ${{ inputs.random-range-ratio }}
klaud-run: ${{ inputs.klaud-run }}
runner: ${{ matrix.config.runner }}
node-count: ${{ matrix.config.node-count }}
Expand Down Expand Up @@ -429,6 +453,8 @@ jobs:
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }}
random-range-ratio: ${{ inputs.random-range-ratio }}
klaud-run: ${{ inputs.klaud-run }}
runner: ${{ matrix.config.runner }}
priority: ${{ matrix.config.priority }}
Expand Down Expand Up @@ -456,6 +482,8 @@ jobs:
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }}
random-range-ratio: ${{ inputs.random-range-ratio }}
klaud-run: ${{ inputs.klaud-run }}
runner: ${{ matrix.config.runner }}
priority: ${{ matrix.config.priority }}
Expand Down Expand Up @@ -487,6 +515,8 @@ jobs:
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }}
random-range-ratio: ${{ inputs.random-range-ratio }}
klaud-run: ${{ inputs.klaud-run }}
runner: ${{ matrix.config.runner }}
node-count: ${{ matrix.config.node-count }}
Expand Down Expand Up @@ -517,6 +547,8 @@ jobs:
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }}
random-range-ratio: ${{ inputs.random-range-ratio }}
klaud-run: ${{ inputs.klaud-run }}
runner: ${{ matrix.config.runner }}
node-count: ${{ matrix.config.node-count }}
Expand Down Expand Up @@ -550,6 +582,8 @@ jobs:
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }}
random-range-ratio: ${{ inputs.random-range-ratio }}
klaud-run: ${{ inputs.klaud-run }}
runner: ${{ matrix.config.runner }}
priority: ${{ matrix.config.priority }}
Expand All @@ -574,6 +608,8 @@ jobs:
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
fixed-sequence-acceptance-length: ${{ inputs.fixed-sequence-acceptance-length }}
random-range-ratio: ${{ inputs.random-range-ratio }}
klaud-run: ${{ inputs.klaud-run }}
runner: ${{ matrix.config.runner }}
priority: ${{ matrix.config.priority }}
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
# Fixed-sequence DSpark comparison; acceptance is supplied explicitly at dispatch.
base:
schema: 2
name: dsv41flash-fp4-b200-matched-8k256
model:
path: hf:deepseek-ai/DeepSeek-V4.1-Flash
container: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:0e1b14e302619a42ef5581b87db6804651d4f946301cf543d74a3a1eb1c33b40
precision: fp4
resources:
gpu_type: b200
gpus_per_node: 8
frontend:
type: sglang
enable_multiple_frontends: false
observability:
enabled: false
tachometer:
enabled: false
engine: sglang
roles:
agg:
nodes: 1
workers: 1
args:
served-model-name: deepseek-ai/DeepSeek-V4.1-Flash
trust-remote-code: true
prefill-decode-interval: 16
speculative-algorithm: DSPARK
speculative-dspark-block-size: 7
cuda-graph-max-bs-decode: 1
reasoning-parser: auto
tool-call-parser: auto
watchdog-timeout: 3600
enable-metrics: true
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 1
moe-runner-backend: flashinfer_mxfp4
disable-shared-experts-fusion: true
speculative-num-draft-tokens: 8
chunked-prefill-size: 8192
context-length: 16384
max-running-requests: 1
mem-fraction-static: 0.8
swa-prefix-tails: 128
stream-interval: 1
env:
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
SGLANG_TIMEOUT_KEEP_ALIVE: '900'
SGLANG_DEFAULT_THINKING: '1'
SGLANG_DSV41_REASONING_EFFORT: high
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1'
SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0'
gpus: 8
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code --dsv4
env:
MODEL: deepseek-ai/DeepSeek-V4.1-Flash
ISL: '8192'
OSL: '256'
RANDOM_RANGE_RATIO: '1.0'
USE_CHAT_TEMPLATE: 'true'
CONC: '1'
override_tp8_c1: {}
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
# Fixed-sequence DSpark comparison; acceptance is supplied explicitly at dispatch.
base:
schema: 2
name: dsv41flash-fp4-b300-matched-8k256
model:
path: hf:deepseek-ai/DeepSeek-V4.1-Flash
container: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:35ea4d321b0735051dcce2599fc495853ad1ef30f6a1908f0a75e74204362d14
precision: fp4
resources:
gpu_type: b300
gpus_per_node: 8
frontend:
type: sglang
enable_multiple_frontends: false
observability:
enabled: false
tachometer:
enabled: false
engine: sglang
roles:
agg:
nodes: 1
workers: 1
args:
served-model-name: deepseek-ai/DeepSeek-V4.1-Flash
trust-remote-code: true
chunked-prefill-size: 8192
prefill-decode-interval: 16
speculative-algorithm: DSPARK
speculative-dspark-block-size: 7
cuda-graph-max-bs-decode: 1
reasoning-parser: auto
tool-call-parser: auto
watchdog-timeout: 3600
enable-metrics: true
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 1
moe-runner-backend: flashinfer_mxfp4
disable-shared-experts-fusion: true
speculative-num-draft-tokens: 8
context-length: 16384
max-running-requests: 1
mem-fraction-static: 0.8
swa-prefix-tails: 128
stream-interval: 1
env:
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
SGLANG_TIMEOUT_KEEP_ALIVE: '900'
SGLANG_DEFAULT_THINKING: '1'
SGLANG_DSV41_REASONING_EFFORT: high
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1'
SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0'
gpus: 8
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code --dsv4
env:
MODEL: deepseek-ai/DeepSeek-V4.1-Flash
ISL: '8192'
OSL: '256'
RANDOM_RANGE_RATIO: '1.0'
USE_CHAT_TEMPLATE: 'true'
CONC: '1'
override_tp8_c1: {}
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
# Fixed-sequence DSpark comparison; acceptance is supplied explicitly at dispatch.
base:
schema: 2
name: dsv41flash-fp4-h200-matched-8k256
model:
path: hf:deepseek-ai/DeepSeek-V4.1-Flash
container: lmsysorg/sglang:nightly-dev-cu13-20260923-06008c17
precision: fp4
resources:
gpu_type: h200
gpus_per_node: 8
frontend:
type: sglang
enable_multiple_frontends: false
observability:
enabled: false
tachometer:
enabled: false
engine: sglang
roles:
agg:
nodes: 1
workers: 1
args:
served-model-name: deepseek-ai/DeepSeek-V4.1-Flash
trust-remote-code: true
data-parallel-size: 1
expert-parallel-size: 8
attention-backend: dsv4
moe-runner-backend: flashinfer_mxfp4
flashinfer-mxfp4-moe-precision: fp8
mem-fraction-static: 0.8
chunked-prefill-size: 8192
prefill-decode-interval: 16
swa-prefix-tails: 128
speculative-algorithm: DSPARK
speculative-dspark-block-size: 7
cuda-graph-max-bs-decode: 1
reasoning-parser: auto
tool-call-parser: auto
watchdog-timeout: 3600
enable-metrics: true
tensor-parallel-size: 8
disable-shared-experts-fusion: true
speculative-num-draft-tokens: 8
context-length: 16384
max-running-requests: 1
stream-interval: 1
env:
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
SGLANG_TIMEOUT_KEEP_ALIVE: '900'
SGLANG_DEFAULT_THINKING: '1'
SGLANG_DSV41_REASONING_EFFORT: high
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1'
SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0'
gpus: 8
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code --dsv4
env:
MODEL: deepseek-ai/DeepSeek-V4.1-Flash
ISL: '8192'
OSL: '256'
RANDOM_RANGE_RATIO: '1.0'
USE_CHAT_TEMPLATE: 'true'
CONC: '1'
override_tp8_c1: {}
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL"
CLIENT_ARGS=()
for argument in "$@"; do
case "$argument" in
--trust-remote-code) CLIENT_ARGS+=("$argument") ;;
--trust-remote-code|--dsv4) CLIENT_ARGS+=("$argument") ;;
*) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;;
esac
done
Expand Down
Loading
Loading