From 5f0511d7945bec5b24730d04915d10897d40f833 Mon Sep 17 00:00:00 2001 From: Xin Li Date: Sun, 27 Sep 2026 18:12:50 -0400 Subject: [PATCH] perf(dsv41): align GB200 with passing B200 vLLM config MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 GB200 的 vLLM 镜像和服务参数与已通过的 B200 配置对齐。 --- .../dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml | 7 +++++-- inferencex-e2e/configs/nvidia-master.yaml | 2 +- inferencex-e2e/perf-changelog.yaml | 6 ++++++ 3 files changed, 12 insertions(+), 3 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml index 64e0e7cca0..9b5ac6e0a2 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml @@ -6,7 +6,7 @@ base: name: dsv41flash-fp4-gb200-vllm-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - container: vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3 + container: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4 precision: fp4 resources: gpu_type: gb200 @@ -37,6 +37,9 @@ base: enable-auto-tool-choice: true reasoning-parser: deepseek_v41 engram-config: '{"cpu_offload":true}' + # FlashInfer sparse attention with MXFP4 indexer KV and sparse logits. + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + kv-cache-dtype: fp8 # Five-token DSpark with probabilistic drafting. Throughput runs replace # block rejection with the golden acceptance length. speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}' @@ -45,7 +48,7 @@ base: env: VLLM_USE_RUST_FRONTEND: '1' PYTHONUNBUFFERED: '1' - VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_ENGINE_READY_TIMEOUT_S: '7200' benchmark: type: custom command: bash /infmax-workspace/benchmarks/srt_agentic.sh diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index b89ea8e9f7..d4427c2ebe 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8429,7 +8429,7 @@ dsv41flash-fp4-h100-sglang-agentic-dspark-dpa: - { tp: 8, ep: 8, dp-attn: true, kv-offloading: none, spec-decoding: mtp, conc-list: [4, 8, 16, 20], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h100-fp4-mtp/agentic.yaml } dsv41flash-fp4-gb200-vllm-agentic-dspark: - image: vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3 + image: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4 model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:gb200-nv diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index db5662f5b8..4ba4c35e8e 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9292,3 +9292,9 @@ - "Update the B200 DeepSeek-V4.1-Flash vLLM AgentX image from nightly ddd6fbca to nightly-dev-x86_64-cu130-ac9126e58aa7 and enable FlashInfer autotuning." - "Run TP4 at concurrency 1-128 and replace TP2 with DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) at concurrency 8-32 and DEP4 (TP1 x DP4 + EP4, MegaMoE) at concurrency 64-128, both behind a consistent-hash vLLM Router. All points set gpu-memory-utilization 0.97." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3686 + +- config-keys: + - dsv41flash-fp4-gb200-vllm-agentic-dspark + description: + - "Align GB200 DeepSeek-V4.1-Flash with B200 using vLLM nightly ddd6fbca and the same serving flags." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3395