From 88a560341a3496cca21cb538e672ceff920a21c7 Mon Sep 17 00:00:00 2001 From: Xin Li Date: Mon, 28 Sep 2026 15:12:32 -0400 Subject: [PATCH 1/5] perf: tune MiniMax-M3 AgentX on B300 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 更新 B300 MiniMax-M3 AgentX 的 vLLM 镜像,并为 EAGLE3 草稿启用本地 argmax 归约。 --- .../minimaxm3/vllm/b300-fp4-mtp/agentic.yaml | 120 ++++++++++++++++-- inferencex-e2e/configs/nvidia-master.yaml | 8 +- inferencex-e2e/perf-changelog.yaml | 8 ++ 3 files changed, 124 insertions(+), 12 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml index d775524820..f31979b6e8 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml @@ -5,7 +5,7 @@ base: name: minimaxm3-fp4-b300-vllm-agentic model: path: hf:nvidia/MiniMax-M3-NVFP4 - container: vllm/vllm-openai:nightly-1dc464d42681d22f38caf1fdc1eb632dc4421c45 + container: vllm/vllm-openai:nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b precision: fp4 resources: gpu_type: b300 @@ -30,10 +30,11 @@ base: workers: 1 args: served-model-name: nvidia/MiniMax-M3-NVFP4 - gpu-memory-utilization: 0.9 + gpu-memory-utilization: 0.95 block-size: 128 language-model-only: true enable-prefix-caching: true + enable-chunked-prefill: true no-enable-flashinfer-autotune: true reasoning-parser: minimax_m3 tool-call-parser: minimax_m3 @@ -42,12 +43,12 @@ base: attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' kv-cache-dtype: fp8 max-cudagraph-capture-size: 512 - max-num-batched-tokens: 16384 + max-num-batched-tokens: 32768 stream-interval: 20 trust-remote-code: true # Three-token EAGLE3 with the GQA draft head. Throughput runs switch to # synthetic rejection at the golden acceptance length. - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER","use_local_argmax_reduction":true}' env: PYTHONNOUSERSITE: '1' VLLM_ENGINE_READY_TIMEOUT_S: '3600' @@ -61,13 +62,15 @@ base: AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' # One variant per point. DRAM points give SimpleCPUOffload the whole host budget -# (TOTAL_CPU_DRAM_GB GiB) with lazy offload. +# (TOTAL_CPU_DRAM_GB decimal GB) with lazy offload. override_tp8_c1: roles: agg: gpus: 8 args: tensor-parallel-size: 8 + max-num-seqs: 2 + max-cudagraph-capture-size: 8 benchmark: env: CONC: '1' @@ -79,6 +82,8 @@ override_tp4_c1: gpus: 4 args: tensor-parallel-size: 4 + max-num-seqs: 2 + max-cudagraph-capture-size: 8 benchmark: env: CONC: '1' @@ -90,6 +95,8 @@ override_tp4_c5: gpus: 4 args: tensor-parallel-size: 4 + max-num-seqs: 10 + max-cudagraph-capture-size: 40 benchmark: env: CONC: '5' @@ -101,6 +108,8 @@ override_tp4_c10: gpus: 4 args: tensor-parallel-size: 4 + max-num-seqs: 20 + max-cudagraph-capture-size: 80 benchmark: env: CONC: '10' @@ -112,6 +121,8 @@ override_tp4_c15: gpus: 4 args: tensor-parallel-size: 4 + max-num-seqs: 30 + max-cudagraph-capture-size: 120 benchmark: env: CONC: '15' @@ -123,6 +134,8 @@ override_tp4_c20: gpus: 4 args: tensor-parallel-size: 4 + max-num-seqs: 40 + max-cudagraph-capture-size: 160 benchmark: env: CONC: '20' @@ -134,14 +147,16 @@ override_tp4_c30_dram: gpus: 4 args: tensor-parallel-size: 4 - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1099511627776,"lazy_offload":true}}' + max-num-seqs: 60 + max-cudagraph-capture-size: 240 + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' benchmark: env: CONC: '30' KV_OFFLOADING: dram - TOTAL_CPU_DRAM_GB: '1024' + TOTAL_CPU_DRAM_GB: '1499' override_tp2_c24_dram: roles: @@ -149,7 +164,10 @@ override_tp2_c24_dram: gpus: 2 args: tensor-parallel-size: 2 - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}' + gpu-memory-utilization: 0.92 + max-num-seqs: 48 + max-cudagraph-capture-size: 192 + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":749000000000,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' benchmark: @@ -157,3 +175,89 @@ override_tp2_c24_dram: CONC: '24' KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '749' + +override_tp4_c40_dram: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-num-seqs: 80 + max-cudagraph-capture-size: 320 + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}' + env: + VLLM_USE_SIMPLE_KV_OFFLOAD: '1' + benchmark: + env: + CONC: '40' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1499' + +override_tp4_c48_dram: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-num-seqs: 96 + max-cudagraph-capture-size: 384 + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}' + env: + VLLM_USE_SIMPLE_KV_OFFLOAD: '1' + benchmark: + env: + CONC: '48' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1499' + +override_tp4_c64_dram: + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + max-num-seqs: 128 + max-cudagraph-capture-size: 512 + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}' + env: + VLLM_USE_SIMPLE_KV_OFFLOAD: '1' + benchmark: + env: + CONC: '64' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1499' + +override_tp2_c32_dram: + roles: + agg: + gpus: 2 + args: + tensor-parallel-size: 2 + gpu-memory-utilization: 0.92 + max-num-seqs: 64 + max-cudagraph-capture-size: 256 + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":749000000000,"lazy_offload":true}}' + env: + VLLM_USE_SIMPLE_KV_OFFLOAD: '1' + benchmark: + env: + CONC: '32' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '749' +override_tp2_c40_dram: + roles: + agg: + gpus: 2 + args: + tensor-parallel-size: 2 + gpu-memory-utilization: 0.92 + max-num-seqs: 80 + max-cudagraph-capture-size: 320 + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":749000000000,"lazy_offload":true}}' + env: + VLLM_USE_SIMPLE_KV_OFFLOAD: '1' + benchmark: + env: + CONC: '40' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '749' diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index cc8c56827f..935c60e0ea 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -5504,7 +5504,7 @@ qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp: additional-settings: - "CONFIG_FILE=recipes/qwen3.5/sglang/gb200-fp4/agentx/agg-variants.yaml:override_tp2ep2_hicache_cap48" minimaxm3-fp4-b300-vllm-agentic-mtp: - image: vllm/vllm-openai:nightly-1dc464d42681d22f38caf1fdc1eb632dc4421c45 + image: vllm/vllm-openai:nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b model: nvidia/MiniMax-M3-NVFP4 model-prefix: minimaxm3 runner: cluster:b300-dsxe @@ -5513,14 +5513,14 @@ minimaxm3-fp4-b300-vllm-agentic-mtp: multinode: false scenarios: agentic-coding: - - dram-utilization: 0.683 + - dram-utilization: 1.0 search-space: - { tp: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [1], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml } - { tp: 4, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 5, 10, 15, 20], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml } - - { tp: 4, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [30], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml } + - { tp: 4, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [30, 40, 48, 64], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml } - dram-utilization: 1.0 search-space: - - { tp: 2, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [24], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml } + - { tp: 2, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [24, 32, 40], srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml } minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg: image: vllm/vllm-openai:nightly-5e35a6f4f9bbc217c599692157ca985c894373f7 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index bf12737e67..48454d1dd4 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9035,3 +9035,11 @@ - "Enable breakable CUDA-graph prefill (BCG) at conc <= 4." - "Enable EP8 + MegaMoE for DP-attention arms at conc >= 128 (including conc 384)." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3430 + +- config-keys: + - minimaxm3-fp4-b300-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Update B300 AgentX to vLLM nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b and enable local argmax reduction for the FlashInfer EAGLE3 draft." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3547 From 167ccae12d02e0ae6650a24c6192c449e5a02112 Mon Sep 17 00:00:00 2001 From: Xin Li Date: Mon, 28 Sep 2026 15:39:04 -0400 Subject: [PATCH 2/5] perf: restore 16K B300 AgentX batch token cap MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 B300 AgentX 的每批 token 上限恢复为先前使用的 16384,以单独评估其他配置改动。 --- .../srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml index f31979b6e8..6b3d4e52b3 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml @@ -43,7 +43,7 @@ base: attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' kv-cache-dtype: fp8 max-cudagraph-capture-size: 512 - max-num-batched-tokens: 32768 + max-num-batched-tokens: 16384 stream-interval: 20 trust-remote-code: true # Three-token EAGLE3 with the GQA draft head. Throughput runs switch to From ab6e4edec60c7a891c2be1de5f85ac4971cfb57f Mon Sep 17 00:00:00 2001 From: Xin Li Date: Mon, 28 Sep 2026 20:22:01 -0400 Subject: [PATCH 3/5] perf: use FP8 Triton EAGLE3 draft on B300 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 B300 EAGLE3 草稿注意力切换到支持 FP8 KV 与融合多步解码的 Triton 后端,目标模型注意力和 KV 精度保持不变。 --- .../minimaxm3/vllm/b300-fp4-mtp/agentic.yaml | 6 +++--- inferencex-e2e/perf-changelog.yaml | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml index 6b3d4e52b3..f01ebcd415 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml @@ -46,9 +46,9 @@ base: max-num-batched-tokens: 16384 stream-interval: 20 trust-remote-code: true - # Three-token EAGLE3 with the GQA draft head. Throughput runs switch to - # synthetic rejection at the golden acceptance length. - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER","use_local_argmax_reduction":true}' + # Three-token EAGLE3 with the GQA draft head. Triton keeps FP8 draft KV + # and supports fused multi-step decode. Throughput uses golden-AL synthetic rejection. + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"TRITON_ATTN","use_local_argmax_reduction":true}' env: PYTHONNOUSERSITE: '1' VLLM_ENGINE_READY_TIMEOUT_S: '3600' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 48454d1dd4..a77eb0ce40 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9041,5 +9041,5 @@ scenario-type: - agentic-coding description: - - "Update B300 AgentX to vLLM nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b and enable local argmax reduction for the FlashInfer EAGLE3 draft." + - "Update B300 AgentX to vLLM nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b with local argmax reduction and an FP8 Triton EAGLE3 draft." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3547 From dbbc1cb68de0e460e83576d2be9122ef2b7d5ee9 Mon Sep 17 00:00:00 2001 From: Xin Li Date: Mon, 28 Sep 2026 21:28:50 -0400 Subject: [PATCH 4/5] perf: restore B300 TP2 host KV budget and FlashInfer draft MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 恢复 B300 TP2 主机 KV 缓存预算,并将 EAGLE3 草稿注意力切回 FlashInfer。 --- .../minimaxm3/vllm/b300-fp4-mtp/agentic.yaml | 12 ++++++------ inferencex-e2e/perf-changelog.yaml | 8 ++++++++ 2 files changed, 14 insertions(+), 6 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml index f01ebcd415..3aa71aeaea 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml @@ -46,9 +46,9 @@ base: max-num-batched-tokens: 16384 stream-interval: 20 trust-remote-code: true - # Three-token EAGLE3 with the GQA draft head. Triton keeps FP8 draft KV - # and supports fused multi-step decode. Throughput uses golden-AL synthetic rejection. - speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"TRITON_ATTN","use_local_argmax_reduction":true}' + # Three-token EAGLE3 with the GQA draft head. Throughput uses + # golden-AL synthetic rejection. + speculative-config: '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3-GQA","num_speculative_tokens":3,"attention_backend":"FLASHINFER","use_local_argmax_reduction":true}' env: PYTHONNOUSERSITE: '1' VLLM_ENGINE_READY_TIMEOUT_S: '3600' @@ -167,7 +167,7 @@ override_tp2_c24_dram: gpu-memory-utilization: 0.92 max-num-seqs: 48 max-cudagraph-capture-size: 192 - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":749000000000,"lazy_offload":true}}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' benchmark: @@ -236,7 +236,7 @@ override_tp2_c32_dram: gpu-memory-utilization: 0.92 max-num-seqs: 64 max-cudagraph-capture-size: 256 - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":749000000000,"lazy_offload":true}}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' benchmark: @@ -253,7 +253,7 @@ override_tp2_c40_dram: gpu-memory-utilization: 0.92 max-num-seqs: 80 max-cudagraph-capture-size: 320 - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":749000000000,"lazy_offload":true}}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' benchmark: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index a77eb0ce40..fd267f3a63 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9043,3 +9043,11 @@ description: - "Update B300 AgentX to vLLM nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b with local argmax reduction and an FP8 Triton EAGLE3 draft." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3547 + +- config-keys: + - minimaxm3-fp4-b300-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Restore the B300 TP2 host KV cache to 749 GiB and use FlashInfer for EAGLE3 draft attention." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3547 From ce431236be8be4f26f34aba5d490cf701933bc08 Mon Sep 17 00:00:00 2001 From: Xin Li Date: Tue, 29 Sep 2026 08:20:30 -0400 Subject: [PATCH 5/5] perf: tune B300 AgentX CUDA graphs and sequence caps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 B300 AgentX 的最大序列数调整为并发的 1.5 倍,并使用显式 CUDA 图捕获尺寸;精简性能变更日志。 --- .../minimaxm3/vllm/b300-fp4-mtp/agentic.yaml | 49 +++++++++---------- inferencex-e2e/perf-changelog.yaml | 10 +--- 2 files changed, 25 insertions(+), 34 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml index 3aa71aeaea..a1924efa51 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-mtp/agentic.yaml @@ -42,7 +42,6 @@ base: default-chat-template-kwargs: '{"thinking_mode":"enabled"}' attention-config: '{"backend":"FLASHINFER","use_trtllm_attention":true,"indexer_kv_dtype":"fp8"}' kv-cache-dtype: fp8 - max-cudagraph-capture-size: 512 max-num-batched-tokens: 16384 stream-interval: 20 trust-remote-code: true @@ -70,7 +69,7 @@ override_tp8_c1: args: tensor-parallel-size: 8 max-num-seqs: 2 - max-cudagraph-capture-size: 8 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,256,512,1024,2048]}' benchmark: env: CONC: '1' @@ -83,7 +82,7 @@ override_tp4_c1: args: tensor-parallel-size: 4 max-num-seqs: 2 - max-cudagraph-capture-size: 8 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,256,512,1024,2048]}' benchmark: env: CONC: '1' @@ -95,8 +94,8 @@ override_tp4_c5: gpus: 4 args: tensor-parallel-size: 4 - max-num-seqs: 10 - max-cudagraph-capture-size: 40 + max-num-seqs: 8 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,256,512,1024,2048]}' benchmark: env: CONC: '5' @@ -108,8 +107,8 @@ override_tp4_c10: gpus: 4 args: tensor-parallel-size: 4 - max-num-seqs: 20 - max-cudagraph-capture-size: 80 + max-num-seqs: 15 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,256,512,1024,2048]}' benchmark: env: CONC: '10' @@ -121,8 +120,8 @@ override_tp4_c15: gpus: 4 args: tensor-parallel-size: 4 - max-num-seqs: 30 - max-cudagraph-capture-size: 120 + max-num-seqs: 23 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,256,512,1024,2048]}' benchmark: env: CONC: '15' @@ -134,8 +133,8 @@ override_tp4_c20: gpus: 4 args: tensor-parallel-size: 4 - max-num-seqs: 40 - max-cudagraph-capture-size: 160 + max-num-seqs: 30 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,256,512,1024,2048]}' benchmark: env: CONC: '20' @@ -147,8 +146,8 @@ override_tp4_c30_dram: gpus: 4 args: tensor-parallel-size: 4 - max-num-seqs: 60 - max-cudagraph-capture-size: 240 + max-num-seqs: 45 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,256,512,1024,2048]}' kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' @@ -165,8 +164,8 @@ override_tp2_c24_dram: args: tensor-parallel-size: 2 gpu-memory-utilization: 0.92 - max-num-seqs: 48 - max-cudagraph-capture-size: 192 + max-num-seqs: 36 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,256,512,1024,2048]}' kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' @@ -182,8 +181,8 @@ override_tp4_c40_dram: gpus: 4 args: tensor-parallel-size: 4 - max-num-seqs: 80 - max-cudagraph-capture-size: 320 + max-num-seqs: 60 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,256,512,1024,2048]}' kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' @@ -199,8 +198,8 @@ override_tp4_c48_dram: gpus: 4 args: tensor-parallel-size: 4 - max-num-seqs: 96 - max-cudagraph-capture-size: 384 + max-num-seqs: 72 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,244,248,252,256,260,264,268,272,276,280,284,288,512,1024,2048]}' kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' @@ -216,8 +215,8 @@ override_tp4_c64_dram: gpus: 4 args: tensor-parallel-size: 4 - max-num-seqs: 128 - max-cudagraph-capture-size: 512 + max-num-seqs: 96 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,244,248,252,256,260,264,268,272,276,280,284,288,292,296,300,304,308,312,316,320,324,328,332,336,340,344,348,352,356,360,364,368,372,376,380,384,512,1024,2048]}' kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1499000000000,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' @@ -234,8 +233,8 @@ override_tp2_c32_dram: args: tensor-parallel-size: 2 gpu-memory-utilization: 0.92 - max-num-seqs: 64 - max-cudagraph-capture-size: 256 + max-num-seqs: 48 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,256,512,1024,2048]}' kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' @@ -251,8 +250,8 @@ override_tp2_c40_dram: args: tensor-parallel-size: 2 gpu-memory-utilization: 0.92 - max-num-seqs: 80 - max-cudagraph-capture-size: 320 + max-num-seqs: 60 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,256,512,1024,2048]}' kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":804232626176,"lazy_offload":true}}' env: VLLM_USE_SIMPLE_KV_OFFLOAD: '1' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index fd267f3a63..53a3a9aec2 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9041,13 +9041,5 @@ scenario-type: - agentic-coding description: - - "Update B300 AgentX to vLLM nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b with local argmax reduction and an FP8 Triton EAGLE3 draft." - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3547 - -- config-keys: - - minimaxm3-fp4-b300-vllm-agentic-mtp - scenario-type: - - agentic-coding - description: - - "Restore the B300 TP2 host KV cache to 749 GiB and use FlashInfer for EAGLE3 draft attention." + - "Update B300 AgentX to vLLM nightly-af7f9488c2210d67e1033ecdc845b087ee7fe92b with local argmax reduction." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3547