Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ base:
name: minimaxm3-fp4-b300-vllm-agentic
model:
path: hf:nvidia/MiniMax-M3-NVFP4
container: vllm/vllm-openai:nightly-1dc464d42681d22f38caf1fdc1eb632dc4421c45
container: vllm/vllm-openai:nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b
precision: fp4
resources:
gpu_type: b300
Expand All @@ -30,24 +30,24 @@ base:
workers: 1
args:
served-model-name: nvidia/MiniMax-M3-NVFP4
gpu-memory-utilization: 0.9
gpu-memory-utilization: 0.95
block-size: 128
language-model-only: true
enable-prefix-caching: true
enable-chunked-prefill: true
no-enable-flashinfer-autotune: true
reasoning-parser: minimax_m3
tool-call-parser: minimax_m3
enable-auto-tool-choice: true
default-chat-template-kwargs: '{"thinking_mode":"enabled"}'
attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}'
kv-cache-dtype: fp8
max-cudagraph-capture-size: 512
max-num-batched-tokens: 16384
stream-interval: 20
trust-remote-code: true
# Three-token EAGLE3 with the GQA draft head. Throughput runs switch to
# synthetic rejection at the golden acceptance length.
speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}'
# Three-token EAGLE3 with the GQA draft head. Throughput uses
# golden-AL synthetic rejection.
speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER","use_local_argmax_reduction":true}'
env:
PYTHONNOUSERSITE: '1'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
Expand All @@ -61,13 +61,15 @@ base:
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:'

# One variant per point. DRAM points give SimpleCPUOffload the whole host budget
# (TOTAL_CPU_DRAM_GB GiB) with lazy offload.
# (TOTAL_CPU_DRAM_GB decimal GB) with lazy offload.
override_tp8_c1:
roles:
agg:
gpus: 8
args:
tensor-parallel-size: 8
max-num-seqs: 2
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,256,512,1024,2048]}'
benchmark:
env:
CONC: '1'
Expand All @@ -79,6 +81,8 @@ override_tp4_c1:
gpus: 4
args:
tensor-parallel-size: 4
max-num-seqs: 2
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,256,512,1024,2048]}'
benchmark:
env:
CONC: '1'
Expand All @@ -90,6 +94,8 @@ override_tp4_c5:
gpus: 4
args:
tensor-parallel-size: 4
max-num-seqs: 8
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,256,512,1024,2048]}'
benchmark:
env:
CONC: '5'
Expand All @@ -101,6 +107,8 @@ override_tp4_c10:
gpus: 4
args:
tensor-parallel-size: 4
max-num-seqs: 15
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,256,512,1024,2048]}'
benchmark:
env:
CONC: '10'
Expand All @@ -112,6 +120,8 @@ override_tp4_c15:
gpus: 4
args:
tensor-parallel-size: 4
max-num-seqs: 23
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,256,512,1024,2048]}'
benchmark:
env:
CONC: '15'
Expand All @@ -123,6 +133,8 @@ override_tp4_c20:
gpus: 4
args:
tensor-parallel-size: 4
max-num-seqs: 30
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,256,512,1024,2048]}'
benchmark:
env:
CONC: '20'
Expand All @@ -134,21 +146,26 @@ override_tp4_c30_dram:
gpus: 4
args:
tensor-parallel-size: 4
kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1099511627776,"lazy_offload":true}}'
max-num-seqs: 45
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,256,512,1024,2048]}'
kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}'
env:
VLLM_USE_SIMPLE_KV_OFFLOAD: '1'
benchmark:
env:
CONC: '30'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '1024'
TOTAL_CPU_DRAM_GB: '1499'

override_tp2_c24_dram:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
gpu-memory-utilization: 0.92
max-num-seqs: 36
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,256,512,1024,2048]}'
kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}'
env:
VLLM_USE_SIMPLE_KV_OFFLOAD: '1'
Expand All @@ -157,3 +174,89 @@ override_tp2_c24_dram:
CONC: '24'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '749'

override_tp4_c40_dram:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-num-seqs: 60
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,256,512,1024,2048]}'
kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}'
env:
VLLM_USE_SIMPLE_KV_OFFLOAD: '1'
benchmark:
env:
CONC: '40'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '1499'

override_tp4_c48_dram:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-num-seqs: 72
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,244,248,252,256,260,264,268,272,276,280,284,288,512,1024,2048]}'
kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}'
env:
VLLM_USE_SIMPLE_KV_OFFLOAD: '1'
benchmark:
env:
CONC: '48'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '1499'

override_tp4_c64_dram:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-num-seqs: 96
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,244,248,252,256,260,264,268,272,276,280,284,288,292,296,300,304,308,312,316,320,324,328,332,336,340,344,348,352,356,360,364,368,372,376,380,384,512,1024,2048]}'
kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}'
env:
VLLM_USE_SIMPLE_KV_OFFLOAD: '1'
benchmark:
env:
CONC: '64'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '1499'

override_tp2_c32_dram:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
gpu-memory-utilization: 0.92
max-num-seqs: 48
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,256,512,1024,2048]}'
kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}'
env:
VLLM_USE_SIMPLE_KV_OFFLOAD: '1'
benchmark:
env:
CONC: '32'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '749'
override_tp2_c40_dram:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
gpu-memory-utilization: 0.92
max-num-seqs: 60
compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,256,512,1024,2048]}'
kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}'
env:
VLLM_USE_SIMPLE_KV_OFFLOAD: '1'
benchmark:
env:
CONC: '40'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '749'
8 changes: 4 additions & 4 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5520,7 +5520,7 @@ qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp:
additional-settings:
- "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-variants.yaml:override_tp2ep2_hicache_cap48"
minimaxm3-fp4-b300-vllm-agentic-mtp:
image: vllm/vllm-openai:nightly-1dc464d42681d22f38caf1fdc1eb632dc4421c45
image: vllm/vllm-openai:nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b
model: nvidia/MiniMax-M3-NVFP4
model-prefix: minimaxm3
runner: cluster:b300-dsxe
Expand All @@ -5529,14 +5529,14 @@ minimaxm3-fp4-b300-vllm-agentic-mtp:
multinode: false
scenarios:
agentic-coding:
- dram-utilization: 0.683
- dram-utilization: 1.0
search-space:
- { tp: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [1], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml }
- { tp: 4, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 5, 10, 15, 20], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml }
- { tp: 4, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [30], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml }
- { tp: 4, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [30, 40, 48, 64], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml }
- dram-utilization: 1.0
search-space:
- { tp: 2, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [24], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml }
- { tp: 2, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [24, 32, 40], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml }

minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg:
image: vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7
Expand Down
8 changes: 8 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9062,6 +9062,14 @@
- "Update the GB300 DeepSeek-V4.1-Flash SGLang AgentX curve to lmsysorg/sglang:dev-cu13-nightly-0924 (digest pinned): pure TP4 (EP1) at C1/C2 with Engram in HBM, and TP4/EP4 at C4+ with per-rank host Engram, 16K prefill chunks, no prefill-decode interval, 4096 SWA prefix tails and a 128 decode graph batch; all TP4 points use static ragged verify; TP2 is unchanged."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3421

- config-keys:
- minimaxm3-fp4-b300-vllm-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Update B300 AgentX to vLLM nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b with local argmax reduction."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3547

- config-keys:
- minimaxm3-fp8-mi300x-vllm-agentic-mtp
scenario-type:
Expand Down
Loading