From fc03e75c75030fdb44e7f5b0fbf62d8b141abb52 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sun, 27 Sep 2026 00:57:15 -0700 Subject: [PATCH] fix: disable GLM-5.2 GB300 draft MoE FP8 conversion MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Set SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE=0 so the NextN/MTP draft keeps its shipped precision, and use the 1800 s scheduler watchdog of the GLM-5.2 B200/GB200 recipes for disaggregated prefill and decode. 设置 SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE=0,使 NextN/MTP draft 保持原始发布精度; 分离式 prefill 和 decode 采用 GLM-5.2 B200/GB200 配方的 1800 秒调度 watchdog。 Co-authored-by: Wenyao Gao --- .../glm5.2/sglang/gb300-fp4/agentx/agg.yaml | 2 +- .../sglang/gb300-fp4/agentx/disagg-mtp-variants.yaml | 4 +++- inferencex-e2e/perf-changelog.yaml | 12 ++++++++++++ 3 files changed, 16 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/agg.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/agg.yaml index 72d65baae7..f79c92e72e 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/agg.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/agg.yaml @@ -48,7 +48,7 @@ roles: SGLANG_HICACHE_DEBUG_LOG: '1' SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' PIP_BREAK_SYSTEM_PACKAGES: '1' SGLANG_DG_CACHE_DIR: /deepgemm_cache diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/disagg-mtp-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/disagg-mtp-variants.yaml index 36ef2c32ff..9bebf23134 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/disagg-mtp-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/disagg-mtp-variants.yaml @@ -87,6 +87,7 @@ base: speculative-num-steps: 1 speculative-eagle-topk: 1 speculative-num-draft-tokens: 2 + watchdog-timeout: 1800 enable-metrics: true enable-cache-report: true decode: @@ -106,7 +107,7 @@ base: SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_ENABLE_THINKING: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' SGLANG_REASONING_EFFORT: max @@ -147,6 +148,7 @@ base: model-loader-extra-config: '{"enable_multithread_load": true}' mem-fraction-static: 0.9 disaggregation-decode-extra-slots: 0 + watchdog-timeout: 1800 enable-metrics: true enable-cache-report: true health_check: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 00f1838d6b..c58c9ca492 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9053,3 +9053,15 @@ description: - "Update the GB300 DeepSeek-V4.1-Flash SGLang AgentX curve to lmsysorg/sglang:dev-cu13-nightly-0924 (digest pinned): pure TP4 (EP1) at C1/C2 with Engram in HBM, and TP4/EP4 at C4+ with per-rank host Engram, 16K prefill chunks, no prefill-decode interval, 4096 SWA prefix tails and a 128 decode graph batch; all TP4 points use static ragged verify; TP2 is unchanged." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3421 + +- config-keys: + - glm5.2-fp4-gb300-dynamo-sglang-agentic-agg + - glm5.2-fp4-gb300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Disable FP8 conversion of the GLM-5.2 NextN/MTP draft MoE on GB300 AgentX so the draft runs at its shipped precision; images, topology, workload and acceptance are unchanged." + - "Disaggregated prefill and decode use the 1800 s scheduler watchdog of the GLM-5.2 B200/GB200 recipes instead of the 300 s default." + - "关闭 GB300 GLM-5.2 AgentX 对 NextN/MTP draft MoE 的 FP8 转换,使 draft 保持原始发布精度;镜像、拓扑、工作负载和验收不变。" + - "分离式 prefill 和 decode 采用 GLM-5.2 B200/GB200 配方的 1800 秒调度 watchdog,替代默认 300 秒。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3402