diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm52-gb200-nixl-prefill.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm52-gb200-nixl-prefill.sh new file mode 100755 index 0000000000..00f5d7948a --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm52-gb200-nixl-prefill.sh @@ -0,0 +1,37 @@ +#!/usr/bin/env bash +set -eo pipefail +bash /configs/install-torchao.sh + +# Avoid the shared UCX worker's event-arm path in the unchanged GLM5.2 images. +# This is a scoped workaround, not the UCX event fix: openucx/ucx#11499. +python3 - <<'PY' +import hashlib +from pathlib import Path + +path = Path("/sgl-workspace/sglang/python/sglang/srt/disaggregation/nixl/conn.py") +original = "28da79ad06baa8c1b725bc98983ca570f82a73b49439873831505ab7f1eef601" +patched = "00f857bc02cd54a1e869e44cd54d04bfb0dc8b2ada5e62a780e30dc6a113cf66" +source = path.read_text() +digest = hashlib.sha256(source.encode()).hexdigest() +if digest == patched: + print(f"GLM5.2 UCX prefill setup already applied: {digest}") + raise SystemExit(0) +if digest != original: + raise SystemExit(f"Unexpected GLM5.2 NIXL conn.py: {digest}") +source = source.replace( + ' num_threads = 8 if disaggregation_mode == DisaggregationMode.PREFILL else 0\n', + ' num_threads = 8 if disaggregation_mode == DisaggregationMode.PREFILL else 0\n' + ' synchronous_ucx = backend == "UCX" and disaggregation_mode == DisaggregationMode.PREFILL\n', +).replace( + ' backends=[],\n', + ' backends=[],\n enable_prog_thread=not synchronous_ucx,\n', +).replace( + ' self.agent.create_backend(backend, backend_params)\n', + ' if synchronous_ucx:\n backend_params["num_threads"] = "0"\n' + ' self.agent.create_backend(backend, backend_params)\n', +) +if hashlib.sha256(source.encode()).hexdigest() != patched: + raise SystemExit("Unexpected GLM5.2 NIXL patch output") +path.write_text(source) +print(f"GLM5.2 UCX prefill setup: {original} -> {patched}") +PY diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-mtp-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-mtp-variants.yaml index a9c3b602b3..858502c910 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-mtp-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg-mtp-variants.yaml @@ -61,7 +61,7 @@ base: SGLANG_HICACHE_DEBUG_LOG: '1' SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' PIP_BREAK_SYSTEM_PACKAGES: '1' args: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml index 01e2442933..eae2807daf 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml @@ -51,7 +51,7 @@ roles: SGLANG_HICACHE_DEBUG_LOG: '1' SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' PIP_BREAK_SYSTEM_PACKAGES: '1' SGLANG_DG_CACHE_DIR: /deepgemm_cache diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml index 6411d5880e..1f7c68f9fa 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml @@ -124,7 +124,7 @@ base: SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_ENABLE_THINKING: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' SGLANG_REASONING_EFFORT: max diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml index ecf94a8d60..73015a3d9d 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml @@ -130,7 +130,7 @@ base: SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1' SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512' SGLANG_MOE_NVFP4_DISPATCH: '1' - SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' SGLANG_ENABLE_THINKING: '1' SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' SGLANG_REASONING_EFFORT: max diff --git a/inferencex-e2e/docs/waiver/3401.md b/inferencex-e2e/docs/waiver/3401.md new file mode 100644 index 0000000000..0897cf94f0 --- /dev/null +++ b/inferencex-e2e/docs/waiver/3401.md @@ -0,0 +1,67 @@ +# Proposed engine-patch waiver — PR #3401 + +**Pending reviewer approval.** This document requests an exception; it does not grant one. + +The GB200 GLM-5.2 `dynamo-sglang` disaggregated recipes use +`lmsysorg/sglang:v0.5.17-cu130` and +`lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642`. +The SRT lane policy selects `configs/glm52-gb200-nixl-prefill.sh`, which preserves +`install-torchao.sh` and inserts four lines into SGLang NIXL `conn.py`: +for UCX prefill, set `enable_prog_thread=False` and explicit backend +`num_threads="0"`. Strict synchronization and decode behavior remain unchanged. +Exact input/output SHA-256 guards reject unexpected source; this is a workaround, +not a backport of the UCX event-descriptor fix. + +The unmodified nightly image hit the recurring prefill +`cuEventQuery → ucp_worker_arm → nixlUcxSharedThread::run` segmentation fault +[during C45 profiling](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36395459538/job/108913827202). +The same minimal workaround completed release-image +[C12](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36296742242) +with 131 warmup and 1,580 profile requests. The subsequent +[formal sweep](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36538876933/attempts/3) +on `9fbe9bf0c211f9e650b4783d7da5f55e86508cae` completed 14 performance points and +four evals, with per-cell artifact/runtime review and retained cancellation, +counter, identity, telemetry and shutdown limitations. This does not establish a +causal fix or grant this waiver. + +Main synchronization ports the same exact recipe selection to the Python SRT +launcher; the guarded script is unchanged. Focused submission tests cover both +selected disaggregated recipes and exclusion of aggregated/other lanes. The new +launcher, SRT producer and inherited aggregate telemetry command have not been +GPU-qualified by that historical sweep. Runtime compatibility and reuse eligibility +require disposition before merge. The consolidated report also retains both old +and retried nightly C8 rows; reuse must select the final CI job `110329500299` +without counting the earlier result twice. + +Upstream references: [NIXL #2102](https://github.com/ai-dynamo/nixl/issues/2102) +and [UCX #11499](https://github.com/openucx/ucx/pull/11499). +Remove this script, its launcher selection and this waiver when an upstream image +with the relevant fix completes the affected unpatched sweep. The reviewer must +link this waiver in the PR checklist's additional details before approving the exception. + +
中文 + +**待审阅者批准。** 本文申请例外,不表示例外已经获批。 + +补丁仅覆盖上述两个固定镜像的 GB200 GLM-5.2 分离式配方,保留原有 +`install-torchao.sh`,并向 SGLang NIXL `conn.py` 加入四行:仅对 UCX prefill +关闭 agent progress thread,并把显式 backend 的 `num_threads` 设为 `"0"`。 +保留严格同步和 decode 行为;输入及输出 SHA-256 不符即失败。 + +原始 nightly 镜像在 C45 profiling 中发生了上述 UCX 后台线程崩溃。 +相同最小补丁曾完成 release 镜像 C12 的 131 次 warmup 和 1,580 次 profile +请求。随后在 `9fbe9bf0c211f9e650b4783d7da5f55e86508cae` 上完成的正式 sweep +覆盖 14 个性能点和 4 个 eval;逐格 artifact/runtime 审查保留取消、计数差异、 +身份比对、遥测及关闭过程等限制。这不证明因果修复,也不表示本文例外已获批准。 + +同步 main 时,将完全相同的配方选择规则迁移到 Python SRT launcher,受哈希保护的 +脚本保持不变。定向提交参数测试覆盖两个分离式配方及聚合/其他路径的排除规则。 +新 launcher、SRT producer 及从 main 继承的聚合遥测命令并未由历史 sweep 验证; +合并前仍需处理运行兼容性及结果复用资格。合并结果保留 nightly C8 的原始记录和 +重跑记录;复用时应选择最终 CI job `110329500299`,不能重复计数。 + +相关上游记录见 NIXL #2102 和 UCX #11499。本补丁是规避措施,不是上游修复的回移。 +待包含相关修复的上游镜像在移除补丁后完成受影响 sweep,再删除脚本、选择逻辑及本文。 +审阅者批准例外前,须在 PR checklist 的 additional details 中链接本文。 + +
diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index 57d31ce99d..f33a08109d 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -32,6 +32,7 @@ class SrtLane: frameworks: frozenset[str] | None = None rejects: tuple[tuple[Match, str], ...] = () setup_scripts: Mapping[str, str] = field(default_factory=dict) + recipe_setup_scripts: Mapping[tuple[str, str], str] = field(default_factory=dict) mounts: tuple[LaneMount, ...] = () shared_run_root: tuple[Match, ...] = () eval_unsets: tuple[str, ...] = () @@ -68,6 +69,24 @@ class SrtLane: ("gb200-nv", LaunchPath.SRT_MULTI): SrtLane( frameworks=_DYNAMO, setup_scripts={"dynamo-sglang": "install-torchao.sh"}, + recipe_setup_scripts={ + ( + "dynamo-sglang", + "benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml", + ): "glm52-gb200-nixl-prefill.sh", + ( + "dynamo-sglang", + "benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml", + ): "glm52-gb200-nixl-prefill.sh", + ( + "dynamo-sglang", + "recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml", + ): "glm52-gb200-nixl-prefill.sh", + ( + "dynamo-sglang", + "recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml", + ): "glm52-gb200-nixl-prefill.sh", + }, mounts=( *_AGENTIC_CACHES, LaneMount( @@ -121,6 +140,14 @@ def srt_lane(cluster_id: str, path: LaunchPath) -> SrtLane: raise LaunchError(f"cluster {cluster_id!r} has no {path} srt-slurm lane") from None +def setup_script(lane: SrtLane, request: SrtRequest, config_file: str) -> str | None: + """Select an exact recipe override before the lane's framework setup.""" + recipe = config_file.split(":", 1)[0] + return lane.recipe_setup_scripts.get( + (request.framework, recipe), lane.setup_scripts.get(request.framework) + ) + + def check_request(lane: SrtLane, request: SrtRequest) -> None: """Refuse requests the lane does not run, before any setup.""" framework = request.framework diff --git a/inferencex-e2e/infx/launch/drivers/srt/submit.py b/inferencex-e2e/infx/launch/drivers/srt/submit.py index e26b8da18f..58c5d49563 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/submit.py +++ b/inferencex-e2e/infx/launch/drivers/srt/submit.py @@ -16,6 +16,7 @@ from infx.launch import proc from infx.launch.backends.slurm import srtctl_job_name from infx.launch.context import LaunchError +from infx.launch.drivers.srt.lanes import setup_script from infx.srt_slurm.single_node import submission_fields if TYPE_CHECKING: @@ -188,6 +189,6 @@ def multinode_arguments( "--tags", f"{run.srt.job_tag},{request.model_prefix},{request.precision},{workload},infmax-{stamp}", ] - if setup_script := lane.setup_scripts.get(request.framework): - arguments += ["--setup-script", setup_script] + if script := setup_script(lane, request, config_file): + arguments += ["--setup-script", script] return arguments diff --git a/inferencex-e2e/infx/tests/launch/test_srt_driver.py b/inferencex-e2e/infx/tests/launch/test_srt_driver.py index 176f8cb71a..15b09a57b8 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_driver.py @@ -21,6 +21,8 @@ from infx.launch.drivers.srt import lanes, models from infx.launch.drivers.srt.lanes import LaneMount, SrtLane from infx.launch.drivers.srt.models import Override +from infx.launch.drivers.srt.submit import multinode_arguments +from infx.launch.request import SrtRequest from infx.launch.policy import LaunchPath, Match from infx.tests.launch.fake_slurm import ( base_env, @@ -560,3 +562,51 @@ def test_multinode_eval_overrides_image_offline_mode_and_host_model_path(harness "http://worker:8000", str(harness.workspace), ] + + +@pytest.mark.parametrize(("cluster_id", "framework", "config_file", "expected"), [ + ("gb200-nv", "dynamo-sglang", "recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml:zip_override_mtp_agentx_frontier[0]", "glm52-gb200-nixl-prefill.sh"), + ("gb200-nv", "dynamo-sglang", "recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml:override_c45", "glm52-gb200-nixl-prefill.sh"), + ("gb200-nv", "dynamo-sglang", "benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml:override_eval", "glm52-gb200-nixl-prefill.sh"), + ("gb200-nv", "dynamo-sglang", "benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-dep8-mtp-variants.yaml", "glm52-gb200-nixl-prefill.sh"), + ("gb200-nv", "dynamo-sglang", "recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml", "install-torchao.sh"), + ("gb200-nv", "dynamo-sglang", "recipes/glm5.2/sglang/gb200-fp4/agentx/agg-mtp-variants.yaml", "install-torchao.sh"), + ("gb200-nv", "dynamo-sglang", "recipes/kimik3/sglang/gb200-fp4/agentx/disagg.yaml", "install-torchao.sh"), + ("gb200-nv", "dynamo-sglang", "glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml", "install-torchao.sh"), + ("gb300-nv", "dynamo-sglang", "recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml", None), + ("gb200-nv", "dynamo-vllm", "recipes/glm5.2/sglang/gb200-fp4/agentx/disagg-mtp-variants.yaml", None), +]) # fmt: skip +@pytest.mark.parametrize("eval_only", ["false", "true"]) +def test_multinode_submission_scopes_glm52_ucx_setup( + cluster_id, framework, config_file, expected, eval_only +): + point = SrtRequest.from_env( + { + "RUNNER_NAME": "runner_0", + "GITHUB_WORKSPACE": "/ws", + "IMAGE": "test:tag", + "FRAMEWORK": framework, + "MODEL_PREFIX": "glm5.2", + "PRECISION": "fp4", + "SPEC_DECODING": "mtp", + "RESULT_FILENAME": "point", + "IS_AGENTIC": "1", + "RUN_EVAL": eval_only, + "EVAL_ONLY": eval_only, + "THINKING_MODE": "max", + } + ) + run = SimpleNamespace( + request=point, + cluster=SimpleNamespace(id=cluster_id), + srt=SimpleNamespace(job_tag=None), + env=point.env, + ) + lane = lanes.srt_lane(cluster_id, LaunchPath.SRT_MULTI) + argv = multinode_arguments(run, lane, config_file, [], preflight=False) + if expected is None: + assert "--setup-script" not in argv + else: + assert argv.count("--setup-script") == 1 + assert argv[argv.index("--setup-script") + 1] == expected + assert argv[argv.index("-f") + 1] == config_file diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index e0ed44048b..5bfb29b209 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9235,3 +9235,15 @@ - "Add DSV4-Pro GB200 Dynamo+SGLang AgentX: TP8 aggregate C1/C4, 1P1D DEP8/DEP16 C64/C128, 1P1D DEP16/DEP32 C256, and 2P1D DEP16/DEP32 C768/C1024/C1280." - "Use SGLang nightly-dev-20260916-c9a8fba9, DSpark block size 6, and HiCache; omit enable-w4a4-mxfp4-megamoe while retaining the MegaMoE all-to-all backend and FP4 indexer." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3629 + +- config-keys: + - glm5.2-fp4-gb200-dynamo-sglang-agentic-agg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-disagg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg + scenario-type: + - agentic-coding + description: + - "Keep the GLM-5.2 NextN/MTP draft at shipped precision and disable background UCX progress for GB200 disaggregated prefill to avoid the observed event-arm crash; images, topology, workload and golden acceptance are unchanged." + - "使 GLM-5.2 NextN/MTP draft 保持原始发布精度,并关闭 GB200 分离式 prefill 的 UCX 后台进展以规避已观察到的 event-arm 崩溃;镜像、拓扑、工作负载和 golden acceptance 不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3401