From 98548a11e614556cdbec812a09829d682d051bca Mon Sep 17 00:00:00 2001 From: "claude[bot]" <41898282+claude[bot]@users.noreply.github.com> Date: Sat, 26 Sep 2026 06:51:02 +0000 Subject: [PATCH] feat(qwen3.5): update B200 FP4 SGLang MTP image to v0.5.20-cu130 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Update qwen3.5-fp4-b200-sglang-mtp from lmsysorg/sglang:v0.5.14-cu130 to lmsysorg/sglang:v0.5.20-cu130 (sglang 0.5.20, commit 94602c9c). SGLang v0.5.20 removed the deprecated --cuda-graph-max-bs alias, so the recipe passes the same values through --cuda-graph-max-bs-decode, the field the alias already set. Model, topology, MTP settings and all points are unchanged. 将 qwen3.5-fp4-b200-sglang-mtp 的镜像从 lmsysorg/sglang:v0.5.14-cu130 更新为 lmsysorg/sglang:v0.5.20-cu130(sglang 0.5.20,提交 94602c9c)。SGLang v0.5.20 移除了已弃用的 --cuda-graph-max-bs 别名,因此配方改用 --cuda-graph-max-bs-decode 传入相同数值,该别名原本即设置此字段。模型、拓扑、MTP 设置及所有测试点保持不变。 Co-Authored-By: Claude Opus 5.5 --- .../qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml | 8 ++++---- configs/nvidia-master.yaml | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml index 8604043522..34d6a85962 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml @@ -3,7 +3,7 @@ base: name: qwen3.5-fp4-b200-sglang-mtp-8k1k model: path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 - container: lmsysorg/sglang:v0.5.14-cu130 + container: lmsysorg/sglang:v0.5.20-cu130 precision: fp4 resources: gpu_type: b200 @@ -33,7 +33,7 @@ base: mamba-ssm-dtype: bfloat16 attention-backend: trtllm_mha moe-runner-backend: flashinfer_trtllm - cuda-graph-max-bs: 4 + cuda-graph-max-bs-decode: 4 max-running-requests: 4 max-prefill-tokens: 16384 chunked-prefill-size: 16384 @@ -70,7 +70,7 @@ zip_override_tp2_ep1: agg: gpus: 2 args: - cuda-graph-max-bs: [4, 8, 16, 32, 64] + cuda-graph-max-bs-decode: [4, 8, 16, 32, 64] max-running-requests: [4, 8, 16, 32, 64] scheduler-recv-interval: [10, 30, 30, 30, 30] tensor-parallel-size: 2 @@ -82,7 +82,7 @@ zip_override_tp2_ep2: agg: gpus: 2 args: - cuda-graph-max-bs: [16, 32, 64] + cuda-graph-max-bs-decode: [16, 32, 64] expert-parallel-size: 2 linear-attn-backend: triton linear-attn-decode-backend: flashinfer diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 30ce89afcd..b91e8641e3 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -1218,7 +1218,7 @@ qwen3.5-fp4-b200-sglang: srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml qwen3.5-fp4-b200-sglang-mtp: - image: lmsysorg/sglang:v0.5.14-cu130 + image: lmsysorg/sglang:v0.5.20-cu130 model: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 model-prefix: qwen3.5 runner: cluster:b200-nscale