Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ base:
name: dsv41flash-fp4-gb300-vllm-agentic
model:
path: hf:deepseek-ai/DeepSeek-V4.1-Flash
container: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e
container: vllm/vllm-openai:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d
precision: fp4
resources:
gpu_type: gb300
Expand Down Expand Up @@ -37,12 +37,17 @@ base:
enable-auto-tool-choice: true
reasoning-parser: deepseek_v41
engram-config: '{"cpu_offload":true}'
# FlashInfer sparse attention with MXFP4 indexer KV and sparse logits.
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}'
kv-cache-dtype: fp8
# Five-token DSpark with probabilistic drafting. Throughput runs replace
# block rejection with the golden acceptance length.
speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}'
max-model-len: 1048576
max-num-seqs: 256
disable-uvicorn-access-log: true
env:
VLLM_DISABLED_KERNELS: FlashInferCutedslMxfp8LinearKernel
VLLM_USE_RUST_FRONTEND: '1'
PYTHONUNBUFFERED: '1'
VLLM_ENGINE_READY_TIMEOUT_S: '7200'
Expand All @@ -53,18 +58,20 @@ base:
MODEL: deepseek-ai/DeepSeek-V4.1-Flash
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:'

# One variant per point. Graph capture starts at 64 tokens and doubles until it
# covers CONC x (1 + 5 drafts), up to 2048. TP2 leaves ~175 GiB of weights on
# each 277 GiB GPU, so it caps batched tokens at 4096 (the indexer's 1M-wide
# logits buffer), bounds the scheduler at 2x CONC within [16, 256] (FlashInfer
# autotune fails below 16) and stops capturing above 512 tokens.
# One variant per point. Piecewise graph capture sizes are multiples of the
# six-token DSpark verification block, denser for small batches, and each
# batched-token limit matches the largest captured graph: CONC <= 4 and TP2
# CONC 128 capture up to 2046 tokens (the latter at 0.97 memory utilization),
# every other point up to 8190.
override_tp4_c1:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '1'
Expand All @@ -75,7 +82,9 @@ override_tp4_c2:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '2'
Expand All @@ -86,7 +95,9 @@ override_tp4_c4:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '4'
Expand All @@ -97,7 +108,9 @@ override_tp4_c8:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '8'
Expand All @@ -108,7 +121,9 @@ override_tp4_c16:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 128
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '16'
Expand All @@ -119,7 +134,9 @@ override_tp4_c32:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 256
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '32'
Expand All @@ -130,7 +147,9 @@ override_tp4_c64:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 512
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '64'
Expand All @@ -141,33 +160,22 @@ override_tp4_c128:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 1024
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '128'

override_tp2_c1:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '1'

override_tp2_c2:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '2'
Expand All @@ -178,9 +186,9 @@ override_tp2_c4:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
benchmark:
env:
CONC: '4'
Expand All @@ -191,9 +199,9 @@ override_tp2_c8:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '8'
Expand All @@ -204,9 +212,9 @@ override_tp2_c16:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 128
max-num-batched-tokens: 4096
max-num-seqs: 32
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '16'
Expand All @@ -217,9 +225,9 @@ override_tp2_c32:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 256
max-num-batched-tokens: 4096
max-num-seqs: 64
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '32'
Expand All @@ -230,9 +238,9 @@ override_tp2_c64:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 512
max-num-batched-tokens: 4096
max-num-seqs: 128
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
benchmark:
env:
CONC: '64'
Expand All @@ -243,9 +251,23 @@ override_tp2_c128:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 512
max-num-batched-tokens: 4096
max-num-seqs: 256
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: 2046
max-num-batched-tokens: 2048
gpu-memory-utilization: 0.97
benchmark:
env:
CONC: '128'

override_tp2_c1:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '1'
2 changes: 1 addition & 1 deletion inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8384,7 +8384,7 @@ dsv41flash-fp4-b200-sglang-agentic-dspark:
- { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml }

dsv41flash-fp4-gb300-vllm-agentic-dspark:
image: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e
image: vllm/vllm-openai:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d
model: deepseek-ai/DeepSeek-V4.1-Flash
model-prefix: dsv41flash
runner: cluster:gb300-nv
Expand Down
6 changes: 6 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9219,3 +9219,9 @@
description:
- "Update Kimi-K3 B200 AgentX to vllm/vllm-openai:nightly-ac9126e58aa7bbab1856ba6593ba4d5003fea516, set bfloat16 KDA state, and add concurrency 32/56/72/80 with TP8/PP2/DCP8 and DSpark K4."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3087

- config-keys:
- dsv41flash-fp4-gb300-vllm-agentic-dspark
description:
- "Pin GB300 DeepSeek-V4.1-Flash vLLM to nightly ac68c308 with FlashInfer sparse attention, MXFP4 indexer KV, sparse logits, fp8 KV cache, and B300's batching and CUDA graph tiers; use FlashInfer CUTLASS for MXFP8 GEMMs to avoid the CuTe-DSL split-K startup failure."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3396
Loading