diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/agg.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/agg.yaml index 72d65baae7..f79c92e72e 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/agg.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/agg.yaml @@ -48,7 +48,7 @@ roles: SGLANG_HICACHE_DEBUG_LOG: '1' SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' PIP_BREAK_SYSTEM_PACKAGES: '1' SGLANG_DG_CACHE_DIR: /deepgemm_cache diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/disagg-mtp-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/disagg-mtp-variants.yaml index 36ef2c32ff..9bebf23134 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/disagg-mtp-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb300-fp4/agentx/disagg-mtp-variants.yaml @@ -87,6 +87,7 @@ base: speculative-num-steps: 1 speculative-eagle-topk: 1 speculative-num-draft-tokens: 2 + watchdog-timeout: 1800 enable-metrics: true enable-cache-report: true decode: @@ -106,7 +107,7 @@ base: SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_ENABLE_THINKING: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' SGLANG_REASONING_EFFORT: max @@ -147,6 +148,7 @@ base: model-loader-extra-config: '{"enable_multithread_load": true}' mem-fraction-static: 0.9 disaggregation-decode-extra-slots: 0 + watchdog-timeout: 1800 enable-metrics: true enable-cache-report: true health_check: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index c2430c0b2e..88b9fbf6dd 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9245,3 +9245,15 @@ - "Disable FP8 conversion of the GLM-5.2 NextN/MTP draft MoE on B200 AgentX so the draft runs at its shipped precision; images, topology, workload and acceptance are unchanged." - "关闭 B200 GLM-5.2 AgentX 对 NextN/MTP draft MoE 的 FP8 转换,使 draft 保持原始发布精度;镜像、拓扑、工作负载和验收不变。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3400 + +- config-keys: + - glm5.2-fp4-gb300-dynamo-sglang-agentic-agg + - glm5.2-fp4-gb300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Disable FP8 conversion of the GLM-5.2 NextN/MTP draft MoE on GB300 AgentX so the draft runs at its shipped precision; images, topology, workload and acceptance are unchanged." + - "Disaggregated prefill and decode use the 1800 s scheduler watchdog of the GLM-5.2 B200/GB200 recipes instead of the 300 s default." + - "关闭 GB300 GLM-5.2 AgentX 对 NextN/MTP draft MoE 的 FP8 转换,使 draft 保持原始发布精度;镜像、拓扑、工作负载和验收不变。" + - "分离式 prefill 和 decode 采用 GLM-5.2 B200/GB200 配方的 1800 秒调度 watchdog,替代默认 300 秒。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3402