From d090548163f7ecfdde496ecf122198b5678ae3d8 Mon Sep 17 00:00:00 2001 From: "claude[bot]" <41898282+claude[bot]@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:56:57 +0000 Subject: [PATCH 1/2] chore(dsr1-fp4-b200-sglang): update SGLang image to v0.5.21-cu130 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pin lmsysorg/sglang:v0.5.21-cu130 by digest and replace the --cuda-graph-max-bs and --disable-piecewise-cuda-graph aliases, which v0.5.21 removed, with --cuda-graph-max-bs-decode and --cuda-graph-backend-prefill=disabled, their v0.5.19 targets. 将 dsr1-fp4-b200-sglang 的 SGLang 镜像更新为按 digest 固定的 lmsysorg/sglang:v0.5.21-cu130,并将 v0.5.21 已移除的 --cuda-graph-max-bs 和 --disable-piecewise-cuda-graph 别名替换为其在 v0.5.19 中对应的 --cuda-graph-max-bs-decode 和 --cuda-graph-backend-prefill=disabled。 Co-Authored-By: Claude Opus 5.5 --- .../srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml | 6 +++--- inferencex-e2e/configs/nvidia-master.yaml | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml index 98888e3f91..2ced513765 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml @@ -3,7 +3,7 @@ base: name: dsr1-fp4-b200-sglang-8k1k model: path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 - container: lmsysorg/sglang:v0.5.19-cu130 + container: lmsysorg/sglang:v0.5.21-cu130@sha256:b1259f3ea3275f66237c498ea388919729018bc9f01c3d638391e06e2cf3f469 precision: fp4 resources: gpu_type: b200 @@ -25,7 +25,7 @@ base: trust-remote-code: true tensor-parallel-size: 4 data-parallel-size: 1 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 max-running-requests: 256 mem-fraction-static: 0.85 kv-cache-dtype: fp8_e4m3 @@ -35,7 +35,7 @@ base: enable-flashinfer-allreduce-fusion: true scheduler-recv-interval: 10 enable-symm-mem: true - disable-piecewise-cuda-graph: true + cuda-graph-backend-prefill: disabled attention-backend: trtllm_mla moe-runner-backend: flashinfer_trtllm stream-interval: 10 diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 052eb90726..661de1c094 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -860,7 +860,7 @@ dsr1-fp8-b300-dynamo-trt: ep: 8 dp-attn: true dsr1-fp4-b200-sglang: - image: lmsysorg/sglang:v0.5.19-cu130 + image: lmsysorg/sglang:v0.5.21-cu130@sha256:b1259f3ea3275f66237c498ea388919729018bc9f01c3d638391e06e2cf3f469 model: nvidia/DeepSeek-R1-0528-FP4-V2 model-prefix: dsr1 runner: cluster:b200-nscale From 54f73cf27b553d31d8618296d4f344b3dcf2eddd Mon Sep 17 00:00:00 2001 From: "claude[bot]" <41898282+claude[bot]@users.noreply.github.com> Date: Fri, 2 Oct 2026 13:29:04 +0000 Subject: [PATCH 2/2] chore(perf-changelog): record dsr1-fp4-b200-sglang SGLang v0.5.21 update MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 dsr1-fp4-b200-sglang 的 SGLang v0.5.21 更新添加 perf-changelog 记录。 Co-Authored-By: Claude Opus 5.5 --- inferencex-e2e/perf-changelog.yaml | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index db5eaa6965..77feae7203 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9211,3 +9211,9 @@ - "Add --attention-config '{\"indexer_kv_dtype\":\"mxfp4\",\"indexer_sparse_logits\":true}' and --block-size 128 to enable the vllm-project/vllm#58671 ROCm paged MXFP4 sparse-logits indexer, replacing the dense fp8 indexer path. A live A/B test (TP2 c16, matched 900s window, vllm-project/vllm#58208 reverted via vllm-project/vllm#59125 so the dense fallback doesn't crash) measured +14.7/+14.9% p50/p90 interactivity and -9.5/-10.8% p50/p90 e2e latency over the dense path, with throughput/GPU unchanged." - "Drop c128 from both TP2 and TP4. Neither c128 point was on the Pareto frontier in #3555's run 36528242520: TP2 c64 dominated both (P90 E2EL 58 s against 185 s and 85 s, at 111k against 85k and 79k total tok/s/GPU)." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3571 + +- config-keys: + - dsr1-fp4-b200-sglang + description: + - "Update the SGLang image from v0.5.19-cu130 to v0.5.21-cu130 and rename the removed CUDA graph flags." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3673