From 32ac79536b66cfa06663dc6b9ddc5308c37513b2 Mon Sep 17 00:00:00 2001 From: Juntian Liu Date: Fri, 2 Oct 2026 00:32:35 +0000 Subject: [PATCH] config: update GB300 DSV41flash vLLM image and retune TP4/DEP2 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Move the GB300 DeepSeek-V4.1-Flash vLLM AgentX recipe to nightly-dev-arm64-cu130-ac9126e58aa7 and enable FlashInfer autotuning. Run TP4 at CONC 1-16 and replace TP2 with DEP2 (TP1 x DP2 + EP2, MegaMoE) at CONC 8-192 behind a consistent-hash vLLM Router. Append the perf-changelog entry and document the GB300 settings. 将 GB300 DeepSeek-V4.1-Flash vLLM AgentX 配方更新到 nightly-dev-arm64-cu130-ac9126e58aa7,并开启 FlashInfer autotune。 TP4 覆盖 CONC 1-16;用 DEP2(TP1 x DP2 + EP2,MegaMoE)替代 TP2,覆盖 CONC 8-192,前置一致性哈希 vLLM Router。追加 perf-changelog 条目,并补充 GB300 配置文档。 Co-Authored-By: Claude Opus 5.5 --- .../vllm/gb300-fp4-mtp/agentic.yaml | 250 +++++------------- inferencex-e2e/configs/nvidia-master.yaml | 14 +- .../docs/configuration-procedures.md | 9 +- .../docs/configuration-procedures_zh.md | 9 +- inferencex-e2e/perf-changelog.yaml | 9 + 5 files changed, 92 insertions(+), 199 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml index 1b5e2787e2..1d3be2acdf 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml @@ -6,7 +6,7 @@ base: name: dsv41flash-fp4-gb300-vllm-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - container: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e + container: vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7 precision: fp4 resources: gpu_type: gb300 @@ -37,12 +37,21 @@ base: enable-auto-tool-choice: true reasoning-parser: deepseek_v41 engram-config: '{"cpu_offload":true}' + kernel-config: '{"enable_flashinfer_autotune":true}' + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + kv-cache-dtype: fp8 # Five-token DSpark with probabilistic drafting. Throughput runs replace # block rejection with the golden acceptance length. speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}' max-model-len: 1048576 + # Piecewise graph capture sizes are multiples of the six-token DSpark + # verification block, denser for small batches. + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 disable-uvicorn-access-log: true env: + VLLM_USE_V2_MODEL_RUNNER: '1' VLLM_USE_RUST_FRONTEND: '1' PYTHONUNBUFFERED: '1' VLLM_ENGINE_READY_TIMEOUT_S: '7200' @@ -53,199 +62,66 @@ base: MODEL: deepseek-ai/DeepSeek-V4.1-Flash AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' -# One variant per point. Graph capture starts at 64 tokens and doubles until it -# covers CONC x (1 + 5 drafts), up to 2048. TP2 leaves ~175 GiB of weights on -# each 277 GiB GPU, so it caps batched tokens at 4096 (the indexer's 1M-wide -# logits buffer), bounds the scheduler at 2x CONC within [16, 256] (FlashInfer -# autotune fails below 16) and stops capturing above 512 tokens. -override_tp4_c1: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-cudagraph-capture-size: 64 - benchmark: - env: - CONC: '1' - -override_tp4_c2: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-cudagraph-capture-size: 64 - benchmark: - env: - CONC: '2' - -override_tp4_c4: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-cudagraph-capture-size: 64 - benchmark: - env: - CONC: '4' - -override_tp4_c8: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-cudagraph-capture-size: 64 - benchmark: - env: - CONC: '8' - -override_tp4_c16: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-cudagraph-capture-size: 128 - benchmark: - env: - CONC: '16' - -override_tp4_c32: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-cudagraph-capture-size: 256 - benchmark: - env: - CONC: '32' - -override_tp4_c64: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - max-cudagraph-capture-size: 512 - benchmark: - env: - CONC: '64' +# Native zip expansion creates one variant per concurrency. -override_tp4_c128: +# TP4 covers low concurrency with THP Engram tables. +zip_override_tp4: roles: agg: gpus: 4 args: tensor-parallel-size: 4 - max-cudagraph-capture-size: 1024 - benchmark: - env: - CONC: '128' - -override_tp2_c1: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 64 - max-num-batched-tokens: 4096 - max-num-seqs: 16 - benchmark: - env: - CONC: '1' - -override_tp2_c2: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 64 - max-num-batched-tokens: 4096 - max-num-seqs: 16 - benchmark: - env: - CONC: '2' - -override_tp2_c4: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 64 - max-num-batched-tokens: 4096 - max-num-seqs: 16 - benchmark: - env: - CONC: '4' - -override_tp2_c8: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 64 - max-num-batched-tokens: 4096 - max-num-seqs: 16 - benchmark: - env: - CONC: '8' - -override_tp2_c16: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 128 - max-num-batched-tokens: 4096 - max-num-seqs: 32 - benchmark: - env: - CONC: '16' - -override_tp2_c32: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 256 - max-num-batched-tokens: 4096 - max-num-seqs: 64 - benchmark: - env: - CONC: '32' - -override_tp2_c64: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 4096 - max-num-seqs: 128 - benchmark: - env: - CONC: '64' - -override_tp2_c128: + engram-config: '{"cpu_offload":true,"use_thp":true}' + max-num-seqs: [8, 8, 16, 32] + compilation-config: + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}' + max-cudagraph-capture-size: [4092, 4092, 8190, 8190] + max-num-batched-tokens: [4096, 4096, 8192, 8192] + benchmark: + env: + CONC: ['1', '4', '8', '16'] + +# DEP2: TP1 x DP2 + EP2 with MegaMoE behind a consistent-hash router; +# MegaAttention at CONC 128 and above. +zip_override_dep2: + frontend: + type: vllm-router + args: + policy: consistent_hash + prometheus-host: 127.0.0.1 + prometheus-port: 18000 + request-timeout-secs: 14400 + disable-retries: true + setup_script: vllm-router-0.1.14.sh roles: agg: gpus: 2 args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 512 - max-num-batched-tokens: 4096 - max-num-seqs: 256 - benchmark: - env: - CONC: '128' + tensor-parallel-size: 1 + data-parallel-size: 2 + enable-expert-parallel: true + kernel-config: '{"enable_flashinfer_autotune":true,"moe_backend":"deep_gemm_mega_moe"}' + engram-config: '{"cpu_offload":true,"embedding_across_dp":true}' + max-num-seqs: [16, 16, 32, 64, 128, 192] + attention-config: + - '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHMLA_MEGA_ATTN_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHMLA_MEGA_ATTN_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + compilation-config: + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048]}' + max-cudagraph-capture-size: [8190, 8190, 8190, 8190, 8190, 4092] + max-num-batched-tokens: [8192, 8192, 8192, 8192, 16384, 16384] + benchmark: + env: + CONC: ['8', '16', '32', '64', '128', '192'] diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 052eb90726..a63136cfe1 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8332,7 +8332,7 @@ dsv41flash-fp4-b200-sglang-agentic-dspark: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml } dsv41flash-fp4-gb300-vllm-agentic-dspark: - image: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e + image: vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7 model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:gb300-nv @@ -8344,13 +8344,11 @@ dsv41flash-fp4-gb300-vllm-agentic-dspark: - dram-utilization: 0.80 search-space: # Engram weights use UVA DRAM; the KV cache stays GPU-resident. - - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml } - # TP2 halves the GPU count per replica. Weights rise to ~175 GiB on each - # 277 GiB GPU with the Engram tables in pinned host DRAM, so the arm - # takes the same caps as the B200 TP2 arm: batched tokens 4096 (the - # indexer's logits buffer is 32 GiB at the upstream 16384) and graph - # capture stopped at 512, leaving ~36 GiB of KV per GPU. - - { tp: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml } + # TP4 covers low concurrency with FlashInfer attention and THP Engram tables. + - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 4, 8, 16], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml } + # DEP2 is native TP1 x DP2 + EP2 with MegaMoE behind a consistent-hash vLLM + # Router; FlashInfer attention up to conc 64, MegaAttention above. + - { tp: 2, ep: 2, dp-attn: true, kv-offloading: none, spec-decoding: mtp, conc-list: [8, 16, 32, 64, 128, 192], router: { name: vllm-router, version: "0.1.14" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml } # H200 AgentX arm for DeepSeek-V4.1-Flash. Upstream marks h200 verified and says # the GB200 NVL4 TP4 layout becomes TP8 on 8-GPU nodes, so this is TP8. diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 4e8a7d0d23..98a908ae9c 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -328,7 +328,7 @@ EAGLE3 K3, golden AL 2.78 and indexer CP are unchanged. The GB200 DSpark recipe uses a minimum CUDA graph capture size of 64 tokens to cover concurrent AgentX subagents. This raises c1/c2/c4 from 8/16/32 to 64; c8 and above retain their existing sizes. The full trace, AL 3.51, and Engram UVA settings are preserved; low-concurrency tail latency improvements require CI confirmation. The B200 DSpark recipe uses the same minimum capture size and preserves the same workload settings. -The GB300 DSpark recipe uses the same minimum capture size and preserves the same workload settings. +The GB300 DSpark recipe sets explicit capture sizes per point, described below. The H200 DSpark recipe uses the same minimum capture size and preserves the same workload settings. B300 uses the same minimum capture size at c1/c2/c4. Its c1 CI comparison reduced request ITL P90/P99 from 38.74/41.42 ms to 2.62/3.45 ms; c2/c4 require CI confirmation. @@ -342,7 +342,7 @@ weights determine the recipe's `precision: fp4` label. The GPU-specific entry points share the text-only serving behavior, `deepseek_v41` tokenizer and parsers, 1M context, and the shared AgentX trace replay, power, metrics, and eval -helpers. The TP4 concurrency range is 1–128. The shared script sizes graph capture +helpers. Unless noted below, the TP4 concurrency range is 1–128. The shared script sizes graph capture for the six-token DSpark verification block. The srt-slurm single-node path mounts the checkout at `/infmax-workspace`, so AgentX runtime directories are not created under `/workspace`. Cluster model paths and persistent caches are reused. The recipe probes the serving port on the compute node and selects an available @@ -356,6 +356,11 @@ at 2046 or 8190 tokens. It sets `--max-num-batched-tokens` to 2048 for concurren TP2 concurrency-128 variant also sets `--gpu-memory-utilization 0.97`. Other SKUs continue to use the shared script. +The GB300 entry uses `vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7` with FlashInfer +autotuning. TP4 covers concurrency 1–16; DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) covers 8–192 +behind a consistent-hash vLLM Router, switching to MegaAttention at 128 and above. All points use +`FULL_AND_PIECEWISE` CUDA graphs sized in multiples of the six-token verification block. + The GB300 launcher allows 7200 seconds for engine readiness. In [run 34504969146](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34504969146), the Rust frontend exhausted its 3600-second deadline while the engine was still capturing graphs; model loading alone took 18–23 minutes. This extends startup time without changing the benchmark duration or decoding settings. GPU sweep and eval evidence is required before calling any recipe validated. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index efbf9347ba..a96367f2c8 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -302,7 +302,7 @@ TP4 约 1.23 TB)全部来自节点 0 的 1.5 TB 内存。`runners/srt-slurm/ho GB200 的 DSpark 配方将 CUDA graph 最小捕获范围设为 64 tokens,以覆盖 AgentX 子代理并发。这会将 c1/c2/c4 的上限从 8/16/32 提升至 64;c8 及以上保持原有大小。完整轨迹、AL 3.51 和 Engram UVA 配置保持不变;需通过 CI 验证低并发尾延迟改善。 B200 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工作负载配置。 -GB300 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工作负载配置。 +GB300 的 DSpark 配方按测试点显式设置捕获尺寸,详见下文。 H200 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工作负载配置。 B300 在 c1/c2/c4 使用相同的最小捕获范围。其 c1 CI 对比中,请求 ITL P90/P99 从 38.74/41.42 ms 降至 2.62/3.45 ms;c2/c4 仍需 CI 验证。 @@ -315,7 +315,7 @@ B300 在 c1/c2/c4 使用相同的最小捕获范围。其 c1 CI 对比中,请 专家权重为 MXFP4,因此配方标记为 `precision: fp4`。 各 GPU 入口共用纯文本服务行为,使用 `deepseek_v41` tokenizer 和解析器、1M 上下文, -以及共享的 AgentX 轨迹回放、功耗、指标和 eval helper。TP4 的并发范围为 1–128。 +以及共享的 AgentX 轨迹回放、功耗、指标和 eval helper。除下文另有说明外,TP4 的并发范围为 1–128。 共享脚本按六 token DSpark 验证块设置 CUDA graph capture。srt-slurm 单节点路径将检出挂载到 `/infmax-workspace`,避免在 `/workspace` 下创建 AgentX 运行目录。沿用集群的模型路径和持久化缓存。配方在计算节点探测服务端口,首选端口被占用时选择可用端口, 服务、回放、指标和 eval 共用同一端点。所有配方都必须获得 GPU sweep 和 eval @@ -327,6 +327,11 @@ TP2 并发 128 使用 `--max-num-batched-tokens 2048`,其余情况使用 8192 `--max-num-seqs` 固定为 256。TP2 并发 128 还设置 `--gpu-memory-utilization 0.97`。其他 SKU 继续使用共享脚本。 +GB300 条目使用 `vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7`,开启 FlashInfer autotune。 +TP4 覆盖并发 1–16;DEP2(TP1 x DP2 + EP2,DeepGEMM MegaMoE)覆盖 8–192,前置一致性哈希 vLLM +Router,并发 128 及以上改用 MegaAttention。所有测试点使用 `FULL_AND_PIECEWISE` CUDA graph,捕获尺寸为 +六 token 验证块的倍数。 + GB300 launcher 将引擎就绪等待时间设为 7200 秒。在[运行 34504969146](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34504969146) 中,仅模型加载就耗时 18–23 分钟;Rust frontend 达到 3600 秒期限时,引擎仍在捕获 CUDA graph。此次仅延长启动等待时间,基准测试时长和解码设置保持不变。 来源:[上游配方](https://recipes.vllm.ai/deepseek-ai/DeepSeek-V4.1-Flash)。 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index db5eaa6965..32b90474aa 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9211,3 +9211,12 @@ - "Add --attention-config '{\"indexer_kv_dtype\":\"mxfp4\",\"indexer_sparse_logits\":true}' and --block-size 128 to enable the vllm-project/vllm#58671 ROCm paged MXFP4 sparse-logits indexer, replacing the dense fp8 indexer path. A live A/B test (TP2 c16, matched 900s window, vllm-project/vllm#58208 reverted via vllm-project/vllm#59125 so the dense fallback doesn't crash) measured +14.7/+14.9% p50/p90 interactivity and -9.5/-10.8% p50/p90 e2e latency over the dense path, with throughput/GPU unchanged." - "Drop c128 from both TP2 and TP4. Neither c128 point was on the Pareto frontier in #3555's run 36528242520: TP2 c64 dominated both (P90 E2EL 58 s against 185 s and 85 s, at 111k against 85k and 79k total tok/s/GPU)." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3571 + +- config-keys: + - dsv41flash-fp4-gb300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Update the GB300 DeepSeek-V4.1-Flash vLLM AgentX image from nightly af1c0149 to nightly-dev-arm64-cu130-ac9126e58aa7 and enable FlashInfer autotuning." + - "Run TP4 at concurrency 1-16 and replace TP2 with DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) at concurrency 8-192 behind a consistent-hash vLLM Router." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3652