From 3e3ad544d7c799ded5305064e51260c3b4b67e10 Mon Sep 17 00:00:00 2001 From: Fangzhou Ai <31551580+Fangzhou-Ai@users.noreply.github.com> Date: Tue, 29 Sep 2026 04:01:14 +0000 Subject: [PATCH 1/5] Drop vllm-project/vllm#58671 from the DSv4.1-Flash MI355X recipe #58671 isn't merging in time. Keep #53492's VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA env var and the image unpin for #58208/#58655, which are unrelated. Drop the --attention-config flag and its --block-size 128 workaround, both of which existed only to exercise #58671's ROCm paged MXFP4 sparse-logits indexer. Signed-off-by: Fangzhou Ai <31551580+Fangzhou-Ai@users.noreply.github.com> Co-authored-by: Cursor --- .../dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml | 7 ++++++- inferencex-e2e/configs/amd-master.yaml | 5 +++-- inferencex-e2e/perf-changelog.yaml | 7 +++++++ 3 files changed, 16 insertions(+), 3 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml index 9ab84be6dd..2d09591d2b 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml @@ -6,7 +6,11 @@ base: name: dsv41flash-fp4-mi355x-vllm-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - container: vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 + # TODO: repin once vllm-project/vllm#58655 (mHC fused Triton seams), + # #53492 (sparse MLA Gluon kernel) and #58208 (DSA candidate + # block-selection topk) are merged upstream and released in a nightly + # image. + container: vllm/vllm-openai-rocm:nightly-rocm100-TBD precision: fp4 resources: gpu_type: mi355x @@ -54,6 +58,7 @@ base: VLLM_ROCM_USE_AITER: '1' VLLM_ROCM_USE_AITER_MOE: '1' AITER_TRITON_LOG_LEVEL: ERROR + VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA: 'True' # DeepseekV41ForCausalLM is not torch-compiled upstream; breakable # graphs keep FULL_AND_PIECEWISE capture working. VLLM_USE_BREAKABLE_CUDAGRAPH: '1' diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index 74c709949e..a2a2b3eaa2 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -1386,8 +1386,9 @@ dsv4-fp4-mi355x-sglang-agentic-mtp: # Two GPUs per server doubles the servers per node and is the layout that # decides whether DSv4.1-Flash is throughput- or interactivity-bound here. dsv41flash-fp4-mi355x-vllm-agentic-dspark: - # ROCm 10.0 nightly channel, shared with kimik3-fp4-mi355x-vllm-agentic-mtp. - image: vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098 + # TODO: repin once vllm-project/vllm#58655, #53492 and #58208 are merged + # upstream and released in a nightly image (see agentic.yaml). + image: vllm/vllm-openai-rocm:nightly-rocm100-TBD model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:mi355x-amds diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index a0ecd2fc71..7c0a30f636 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9003,3 +9003,10 @@ description: - "Update B300 vLLM AgentX to DSpark6 on a new image with a sampled concurrency grid and per-mode --kv-cache-memory-bytes pins." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3477 + +- config-keys: + - dsv41flash-fp4-mi355x-vllm-agentic-dspark + description: + - "Force VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True for vllm-project/vllm#53492's Gluon sparse-MLA kernel. Unpin image to TBD pending #58208 (faster DSA candidate topk selection), #58655 (fused mHC Triton kernel) and #53492 (sparse MLA Gluon kernel) landing in a nightly release: better topk/mHC/MLA kernels with better compute/communication stream overlap. Does not include vllm-project/vllm#58671's MXFP4 sparse-logits indexer: it isn't merging in time, and reverting a dependency it exposed (#58208's DSA candidate-block topk, vllm-project/vllm#59125) showed the fp8 dense indexer path #58671 would otherwise replace costs roughly 15% interactivity and 10% tail e2e latency at TP2 c16, with no measurable throughput/GPU change." + - "设置 VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True 以启用 vllm-project/vllm#53492 的 Gluon sparse-MLA kernel。将镜像取消固定为 TBD,待 #58208(更快的 DSA 候选块 topk 选择)、#58655(融合 mHC Triton kernel)与 #53492(sparse MLA Gluon kernel)合并至上游并随 nightly 镜像发布:更优的 topk/mHC/MLA kernel,以及更好的计算/通信流重叠模式。不包含 vllm-project/vllm#58671 的 MXFP4 稀疏 logits indexer:该 PR 未能及时合并;回滚它所暴露的一个依赖问题(#58208 的 DSA 候选块 topk,vllm-project/vllm#59125)后测得,#58671 本应替换的 fp8 稠密 indexer 路径在 TP2 c16 下约有 15% 的 interactivity 与 10% 的尾部 e2e 延迟损失,而 throughput/GPU 无明显变化。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3555 From d305501b5b75adb158504de152c5dc2be0f1ae9f Mon Sep 17 00:00:00 2001 From: Fangzhou Ai <31551580+Fangzhou-Ai@users.noreply.github.com> Date: Tue, 29 Sep 2026 05:51:58 +0000 Subject: [PATCH 2/5] Repin DSv4.1-Flash MI355X vLLM recipe to nightly-rocm100-36768d1b vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89 is the first published ROCm nightly build whose underlying vLLM commit (36768d1bfd39094681cdbc8cb37d4b31c0729c89) includes all three PRs this recipe was waiting on: vllm-project/vllm#58655, #53492 and #58208. Signed-off-by: Fangzhou Ai <31551580+Fangzhou-Ai@users.noreply.github.com> Co-authored-by: Cursor --- .../dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml | 9 ++++----- inferencex-e2e/configs/amd-master.yaml | 6 +++--- 2 files changed, 7 insertions(+), 8 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml index 2d09591d2b..7dc2964bd5 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml @@ -6,11 +6,10 @@ base: name: dsv41flash-fp4-mi355x-vllm-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - # TODO: repin once vllm-project/vllm#58655 (mHC fused Triton seams), - # #53492 (sparse MLA Gluon kernel) and #58208 (DSA candidate - # block-selection topk) are merged upstream and released in a nightly - # image. - container: vllm/vllm-openai-rocm:nightly-rocm100-TBD + # Repinned to a nightly build that includes vllm-project/vllm#58655 + # (mHC fused Triton seams), #53492 (sparse MLA Gluon kernel) and #58208 + # (DSA candidate block-selection topk). + container: vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89 precision: fp4 resources: gpu_type: mi355x diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index 1927ff7fcd..fee0ebfad0 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -1387,9 +1387,9 @@ dsv4-fp4-mi355x-sglang-agentic-mtp: # Two GPUs per server doubles the servers per node and is the layout that # decides whether DSv4.1-Flash is throughput- or interactivity-bound here. dsv41flash-fp4-mi355x-vllm-agentic-dspark: - # TODO: repin once vllm-project/vllm#58655, #53492 and #58208 are merged - # upstream and released in a nightly image (see agentic.yaml). - image: vllm/vllm-openai-rocm:nightly-rocm100-TBD + # Repinned to a nightly build that includes vllm-project/vllm#58655, + # #53492 and #58208 (see agentic.yaml). + image: vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89 model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:mi355x-amds From 46637d7c6bc6c37d49ff6e5add40c1529875d82f Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Wed, 30 Sep 2026 01:34:28 +0000 Subject: [PATCH 3/5] Fix changelog --- inferencex-e2e/perf-changelog.yaml | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index abe54a134c..926ba925e5 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9004,13 +9004,6 @@ - "Update B300 vLLM AgentX to DSpark6 on a new image with a sampled concurrency grid and per-mode --kv-cache-memory-bytes pins." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3477 -- config-keys: - - dsv41flash-fp4-mi355x-vllm-agentic-dspark - description: - - "Force VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True for vllm-project/vllm#53492's Gluon sparse-MLA kernel. Unpin image to TBD pending #58208 (faster DSA candidate topk selection), #58655 (fused mHC Triton kernel) and #53492 (sparse MLA Gluon kernel) landing in a nightly release: better topk/mHC/MLA kernels with better compute/communication stream overlap. Does not include vllm-project/vllm#58671's MXFP4 sparse-logits indexer: it isn't merging in time, and reverting a dependency it exposed (#58208's DSA candidate-block topk, vllm-project/vllm#59125) showed the fp8 dense indexer path #58671 would otherwise replace costs roughly 15% interactivity and 10% tail e2e latency at TP2 c16, with no measurable throughput/GPU change." - - "设置 VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True 以启用 vllm-project/vllm#53492 的 Gluon sparse-MLA kernel。将镜像取消固定为 TBD,待 #58208(更快的 DSA 候选块 topk 选择)、#58655(融合 mHC Triton kernel)与 #53492(sparse MLA Gluon kernel)合并至上游并随 nightly 镜像发布:更优的 topk/mHC/MLA kernel,以及更好的计算/通信流重叠模式。不包含 vllm-project/vllm#58671 的 MXFP4 稀疏 logits indexer:该 PR 未能及时合并;回滚它所暴露的一个依赖问题(#58208 的 DSA 候选块 topk,vllm-project/vllm#59125)后测得,#58671 本应替换的 fp8 稠密 indexer 路径在 TP2 c16 下约有 15% 的 interactivity 与 10% 的尾部 e2e 延迟损失,而 throughput/GPU 无明显变化。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3555 - - config-keys: - glm5.2-fp4-mi355x-sglang-agentic-mtp scenario-type: @@ -9111,3 +9104,10 @@ - "Restore the DCP8 LMCache bands on the native srt-slurm recipe: concurrency 14 and 16 (DSpark 3, ReplaySSM) and 48, 56 and 72 (no draft) run ATOM's in-process lmcache_offload connector through roles.agg.args.extra-kv-connectors (srt-slurm patch 507), with 128 GB/rank up to 48 and 192 GB/rank at 56 and 72. Concurrency 1 and 4 stay GPU-resident." - "No change to the Inferact/Kimi-K3-DSpark draft's precision: online_quant_config still excludes every draft linear (layers.*, context_proj), so its weights and activations stay BF16, and it keeps the target's FP8 KV cache (kv_cache_dtype fp8). FlyDSL FP8 prefill attention applies only to the target, since the draft runs its block pass as decode attention." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3407 + +- config-keys: + - dsv41flash-fp4-mi355x-vllm-agentic-dspark + description: + - "Force VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True for vllm-project/vllm#53492's Gluon sparse-MLA kernel. Unpin image to TBD pending #58208 (faster DSA candidate topk selection), #58655 (fused mHC Triton kernel) and #53492 (sparse MLA Gluon kernel) landing in a nightly release: better topk/mHC/MLA kernels with better compute/communication stream overlap. Does not include vllm-project/vllm#58671's MXFP4 sparse-logits indexer: it isn't merging in time, and reverting a dependency it exposed (#58208's DSA candidate-block topk, vllm-project/vllm#59125) showed the fp8 dense indexer path #58671 would otherwise replace costs roughly 15% interactivity and 10% tail e2e latency at TP2 c16, with no measurable throughput/GPU change." + - "设置 VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True 以启用 vllm-project/vllm#53492 的 Gluon sparse-MLA kernel。将镜像取消固定为 TBD,待 #58208(更快的 DSA 候选块 topk 选择)、#58655(融合 mHC Triton kernel)与 #53492(sparse MLA Gluon kernel)合并至上游并随 nightly 镜像发布:更优的 topk/mHC/MLA kernel,以及更好的计算/通信流重叠模式。不包含 vllm-project/vllm#58671 的 MXFP4 稀疏 logits indexer:该 PR 未能及时合并;回滚它所暴露的一个依赖问题(#58208 的 DSA 候选块 topk,vllm-project/vllm#59125)后测得,#58671 本应替换的 fp8 稠密 indexer 路径在 TP2 c16 下约有 15% 的 interactivity 与 10% 的尾部 e2e 延迟损失,而 throughput/GPU 无明显变化。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3555 From dc909d7c2192dea1a82e9df6f2ce1f2f201c9381 Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Wed, 30 Sep 2026 02:06:21 +0000 Subject: [PATCH 4/5] Fix changelog so that it accurately describe the changes in the PR --- inferencex-e2e/perf-changelog.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 853d65d26a..66d74c84a9 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9116,6 +9116,6 @@ - config-keys: - dsv41flash-fp4-mi355x-vllm-agentic-dspark description: - - "Force VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True for vllm-project/vllm#53492's Gluon sparse-MLA kernel. Unpin image to TBD pending #58208 (faster DSA candidate topk selection), #58655 (fused mHC Triton kernel) and #53492 (sparse MLA Gluon kernel) landing in a nightly release: better topk/mHC/MLA kernels with better compute/communication stream overlap. Does not include vllm-project/vllm#58671's MXFP4 sparse-logits indexer: it isn't merging in time, and reverting a dependency it exposed (#58208's DSA candidate-block topk, vllm-project/vllm#59125) showed the fp8 dense indexer path #58671 would otherwise replace costs roughly 15% interactivity and 10% tail e2e latency at TP2 c16, with no measurable throughput/GPU change." - - "设置 VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True 以启用 vllm-project/vllm#53492 的 Gluon sparse-MLA kernel。将镜像取消固定为 TBD,待 #58208(更快的 DSA 候选块 topk 选择)、#58655(融合 mHC Triton kernel)与 #53492(sparse MLA Gluon kernel)合并至上游并随 nightly 镜像发布:更优的 topk/mHC/MLA kernel,以及更好的计算/通信流重叠模式。不包含 vllm-project/vllm#58671 的 MXFP4 稀疏 logits indexer:该 PR 未能及时合并;回滚它所暴露的一个依赖问题(#58208 的 DSA 候选块 topk,vllm-project/vllm#59125)后测得,#58671 本应替换的 fp8 稠密 indexer 路径在 TP2 c16 下约有 15% 的 interactivity 与 10% 的尾部 e2e 延迟损失,而 throughput/GPU 无明显变化。" + - "Force VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True to enable vllm-project/vllm#53492's gfx950 Gluon sparse-MLA kernel. Repin the recipe and master config to vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89, which includes #53492 and vllm-project/vllm#58655 (fused mHC Triton kernel). vllm-project/vllm#58208 (DSA candidate topk) was reverted by vllm-project/vllm#59125 before this build, and vllm-project/vllm#58671's MXFP4 sparse indexer landed after it; the indexer is left to #3571." + - "设置 VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True,启用 vllm-project/vllm#53492 的 gfx950 Gluon sparse-MLA kernel。将配方与主配置固定到 vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89,该镜像包含 #53492 与 vllm-project/vllm#58655(融合 mHC Triton kernel)。vllm-project/vllm#58208(DSA 候选块 topk)在该构建之前已被 vllm-project/vllm#59125 回滚;vllm-project/vllm#58671 的 MXFP4 稀疏 indexer 在该构建之后才合入,留待 #3571。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3555 From bde38631988ee63804d99999862369cd40b1355a Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Wed, 30 Sep 2026 02:13:57 +0000 Subject: [PATCH 5/5] Correct image comments: nightly-rocm100-36768d1b does not include vllm#58208 vllm-project/vllm#58208 was reverted by vllm-project/vllm#59125 before the pinned build, so the image only carries #58655 and #53492. Comment-only change. Co-authored-by: Cursor --- .../dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml | 3 +-- inferencex-e2e/configs/amd-master.yaml | 4 ++-- 2 files changed, 3 insertions(+), 4 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml index 7dc2964bd5..94789ce3be 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml @@ -7,8 +7,7 @@ base: model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash # Repinned to a nightly build that includes vllm-project/vllm#58655 - # (mHC fused Triton seams), #53492 (sparse MLA Gluon kernel) and #58208 - # (DSA candidate block-selection topk). + # (mHC fused Triton seams) and #53492 (sparse MLA Gluon kernel). container: vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89 precision: fp4 resources: diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index b0365dbcc2..c6106d813d 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -1391,8 +1391,8 @@ dsv4-fp4-mi355x-sglang-agentic-mtp: # Two GPUs per server doubles the servers per node and is the layout that # decides whether DSv4.1-Flash is throughput- or interactivity-bound here. dsv41flash-fp4-mi355x-vllm-agentic-dspark: - # Repinned to a nightly build that includes vllm-project/vllm#58655, - # #53492 and #58208 (see agentic.yaml). + # Repinned to a nightly build that includes vllm-project/vllm#58655 and + # #53492 (see agentic.yaml). image: vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89 model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash