From 9fbe9bf0c211f9e650b4783d7da5f55e86508cae Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sun, 27 Sep 2026 00:54:31 -0700 Subject: [PATCH] fix: preserve GLM-5.2 GB200 draft precision and use synchronous UCX prefill MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Keep the draft at shipped precision and disable only the two background UCX progress controls for disaggregated prefill. Preserve images, workload, strict synchronization and golden acceptance. Request the scoped engine-patch waiver. 保留 GLM-5.2 GB200 draft 的原始精度,仅关闭分离式 prefill 的两项 UCX 后台进展控制;保留镜像、工作负载、严格同步及 golden acceptance,并申请对应补丁例外。 --- .../configs/glm52-gb200-nixl-prefill.sh | 37 ++++++++++++++ .../gb200-fp4/agentx/agg-mtp-variants.yaml | 2 +- .../glm5.2/sglang/gb200-fp4/agentx/agg.yaml | 2 +- .../agentx/disagg-dep8-mtp-variants.yaml | 2 +- .../gb200-fp4/agentx/disagg-mtp-variants.yaml | 2 +- inferencex-e2e/docs/waiver/3401.md | 48 +++++++++++++++++++ inferencex-e2e/perf-changelog.yaml | 12 +++++ inferencex-e2e/runners/launch_gb200-nv.sh | 9 +++- 8 files changed, 109 insertions(+), 5 deletions(-) create mode 100755 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm52-gb200-nixl-prefill.sh create mode 100644 inferencex-e2e/docs/waiver/3401.md diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm52-gb200-nixl-prefill.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm52-gb200-nixl-prefill.sh new file mode 100755 index 0000000000..00f5d7948a --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm52-gb200-nixl-prefill.sh @@ -0,0 +1,37 @@ +#!/usr/bin/env bash +set -eo pipefail +bash /configs/install-torchao.sh + +# Avoid the shared UCX worker's event-arm path in the unchanged GLM5.2 images. +# This is a scoped workaround, not the UCX event fix: openucx/ucx#11499. +python3 - <<'PY' +import hashlib +from pathlib import Path + +path = Path("/sgl-workspace/sglang/python/sglang/srt/disaggregation/nixl/conn.py") +original = "28da79ad06baa8c1b725bc98983ca570f82a73b49439873831505ab7f1eef601" +patched = "00f857bc02cd54a1e869e44cd54d04bfb0dc8b2ada5e62a780e30dc6a113cf66" +source = path.read_text() +digest = hashlib.sha256(source.encode()).hexdigest() +if digest == patched: + print(f"GLM5.2 UCX prefill setup already applied: {digest}") + raise SystemExit(0) +if digest != original: + raise SystemExit(f"Unexpected GLM5.2 NIXL conn.py: {digest}") +source = source.replace( + ' num_threads = 8 if disaggregation_mode == DisaggregationMode.PREFILL else 0\n', + ' num_threads = 8 if disaggregation_mode == DisaggregationMode.PREFILL else 0\n' + ' synchronous_ucx = backend == "UCX" and disaggregation_mode == DisaggregationMode.PREFILL\n', +).replace( + ' backends=[],\n', + ' backends=[],\n enable_prog_thread=not synchronous_ucx,\n', +).replace( + ' self.agent.create_backend(backend, backend_params)\n', + ' if synchronous_ucx:\n backend_params["num_threads"] = "0"\n' + ' self.agent.create_backend(backend, backend_params)\n', +) +if hashlib.sha256(source.encode()).hexdigest() != patched: + raise SystemExit("Unexpected GLM5.2 NIXL patch output") +path.write_text(source) +print(f"GLM5.2 UCX prefill setup: {original} -> {patched}") +PY diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-mtp-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-mtp-variants.yaml index a9c3b602b3..858502c910 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-mtp-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-mtp-variants.yaml @@ -61,7 +61,7 @@ base: SGLANG_HICACHE_DEBUG_LOG: '1' SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' PIP_BREAK_SYSTEM_PACKAGES: '1' args: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml index 3f2650c0f7..736aa68b43 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml @@ -51,7 +51,7 @@ roles: SGLANG_HICACHE_DEBUG_LOG: '1' SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' PIP_BREAK_SYSTEM_PACKAGES: '1' SGLANG_DG_CACHE_DIR: /deepgemm_cache diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml index 6411d5880e..1f7c68f9fa 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml @@ -124,7 +124,7 @@ base: SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_ENABLE_THINKING: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' SGLANG_REASONING_EFFORT: max diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml index ecf94a8d60..73015a3d9d 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml @@ -130,7 +130,7 @@ base: SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_ENABLE_THINKING: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' SGLANG_REASONING_EFFORT: max diff --git a/inferencex-e2e/docs/waiver/3401.md b/inferencex-e2e/docs/waiver/3401.md new file mode 100644 index 0000000000..354432b9c6 --- /dev/null +++ b/inferencex-e2e/docs/waiver/3401.md @@ -0,0 +1,48 @@ +# Proposed engine-patch waiver — PR #3401 + +**Pending reviewer approval.** This document requests an exception; it does not grant one. + +The GB200 GLM-5.2 `dynamo-sglang` disaggregated recipes use +`lmsysorg/sglang:v0.5.17-cu130` and +`lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642`. +The scoped launcher selects `configs/glm52-gb200-nixl-prefill.sh`, which preserves +`install-torchao.sh` and inserts four lines into SGLang NIXL `conn.py`: +for UCX prefill, set `enable_prog_thread=False` and explicit backend +`num_threads="0"`. Strict synchronization and decode behavior remain unchanged. +Exact input/output SHA-256 guards reject unexpected source; this is a workaround, +not a backport of the UCX event-descriptor fix. + +The unmodified nightly image hit the recurring prefill +`cuEventQuery → ucp_worker_arm → nixlUcxSharedThread::run` segmentation fault +[during C45 profiling](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36395459538/job/108913827202). +The same minimal workaround completed release-image +[C12](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36296742242) +with 131 warmup and 1,580 profile requests, but C45/nightly efficacy and final +full-sweep qualification remain unverified. Images, model precision, topology, +workload and golden acceptance are preserved; affected performance must be remeasured. + +Upstream references: [NIXL #2102](https://github.com/ai-dynamo/nixl/issues/2102) +and [UCX #11499](https://github.com/openucx/ucx/pull/11499). +Remove this script, its launcher selection and this waiver when an upstream image +with the relevant fix completes the affected unpatched sweep. The reviewer must +link this waiver in the PR checklist's additional details before approving the exception. + +
中文 + +**待审阅者批准。** 本文申请例外,不表示例外已经获批。 + +补丁仅覆盖上述两个固定镜像的 GB200 GLM-5.2 分离式配方,保留原有 +`install-torchao.sh`,并向 SGLang NIXL `conn.py` 加入四行:仅对 UCX prefill +关闭 agent progress thread,并把显式 backend 的 `num_threads` 设为 `"0"`。 +保留严格同步和 decode 行为;输入及输出 SHA-256 不符即失败。 + +原始 nightly 镜像在 C45 profiling 中发生了上述 UCX 后台线程崩溃。 +相同最小补丁曾完成 release 镜像 C12 的 131 次 warmup 和 1,580 次 profile +请求,但尚未验证 C45/nightly 修复效果,也未完成本 PR 最终全量资格验证。 +镜像、模型精度、拓扑、工作负载和 golden acceptance 不变;仍需重新测量受影响性能。 + +相关上游记录见 NIXL #2102 和 UCX #11499。本补丁是规避措施,不是上游修复的回移。 +待包含相关修复的上游镜像在移除补丁后完成受影响 sweep,再删除脚本、选择逻辑及本文。 +审阅者批准例外前,须在 PR checklist 的 additional details 中链接本文。 + +
diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 00f1838d6b..8e3fa4e03a 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9053,3 +9053,15 @@ description: - "Update the GB300 DeepSeek-V4.1-Flash SGLang AgentX curve to lmsysorg/sglang:dev-cu13-nightly-0924 (digest pinned): pure TP4 (EP1) at C1/C2 with Engram in HBM, and TP4/EP4 at C4+ with per-rank host Engram, 16K prefill chunks, no prefill-decode interval, 4096 SWA prefix tails and a 128 decode graph batch; all TP4 points use static ragged verify; TP2 is unchanged." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3421 + +- config-keys: + - glm5.2-fp4-gb200-dynamo-sglang-agentic-agg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-disagg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg + scenario-type: + - agentic-coding + description: + - "Keep the GLM-5.2 NextN/MTP draft at shipped precision and disable background UCX progress for GB200 disaggregated prefill to avoid the observed event-arm crash; images, topology, workload and golden acceptance are unchanged." + - "使 GLM-5.2 NextN/MTP draft 保持原始发布精度,并关闭 GB200 分离式 prefill 的 UCX 后台进展以规避已观察到的 event-arm 崩溃;镜像、拓扑、工作负载和 golden acceptance 不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3401 diff --git a/inferencex-e2e/runners/launch_gb200-nv.sh b/inferencex-e2e/runners/launch_gb200-nv.sh index c217d42f0a..e2aa5d236d 100755 --- a/inferencex-e2e/runners/launch_gb200-nv.sh +++ b/inferencex-e2e/runners/launch_gb200-nv.sh @@ -496,7 +496,14 @@ SRTCTL_APPLY_ARGS=( --tags "gb200,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" ) if [[ "$FRAMEWORK" == "dynamo-sglang" ]]; then - SRTCTL_APPLY_ARGS+=(--setup-script install-torchao.sh) + case "$CONFIG_PATH" in + recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml|\ + benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml|\ + recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml|\ + benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml) + SRTCTL_APPLY_ARGS+=(--setup-script glm52-gb200-nixl-prefill.sh) ;; + *) SRTCTL_APPLY_ARGS+=(--setup-script install-torchao.sh) ;; + esac fi # srtctl gives RUNNER_NAME precedence over config.name; override it for the # submission so the #SBATCH job name keeps the namespace used above.