From eb52ae7986916ef7ce5787dacdc8c085bc8fd746 Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:58:20 +0800 Subject: [PATCH 1/2] feat(agentx): move GB300 DSV4 AgentX disagg to the Mooncake external linker Port of the change onto the native srt-slurm variants layout. The four per-point recipes it used to edit are now one disagg-variants.yaml, so the settings shared by every point live in base and only the per-point values stay in each override. - base: replace hierarchical cache with the Mooncake unified-cache external linker, add the Mooncake store environment, 2048-token per-request prefill chunks with deferred intermediate KV transfer, sparse SWA retention every 32768 tokens, and --mooncake-store-contributor on decode; bump the image to nightly-dev-cu13-20260916-c9a8fba9 and add the mooncake-master service. - overrides: rename cuda-graph-max-bs to cuda-graph-max-bs-decode. The pinned SGLang branch rejects the legacy key as ambiguous, and every override sets it explicitly, so the rename cannot live in base. - TEMPORARY: the runner clones the reviewed SGLang branch into /configs and a setup script aborts the job if that tree is missing or the wrong revision; a sitecustomize shim restores ServerArgs.get_model_config for the pinned Dynamo wheel. Remove all three once the change ships in the pinned image. - The runner passes this cluster's RDMA devices to the Mooncake store. Each variant resolves to exactly the configuration the previous per-point recipe produced, apart from main's move of the benchmark script to benchmarks/srt_agentic.sh. --- .../configs/dsv4-gb300-sglang-mooncake-opt.sh | 51 ++++++++++++++++ .../sitecustomize.py | 27 +++++++++ .../gb300-fp4/agentx/disagg-variants.yaml | 59 +++++++++++++++---- inferencex-e2e/configs/nvidia-master.yaml | 10 ++-- inferencex-e2e/perf-changelog.yaml | 33 +++++++++++ inferencex-e2e/runners/launch_gb300-nv.sh | 27 +++++++++ 6 files changed, 189 insertions(+), 18 deletions(-) create mode 100755 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-sglang-mooncake-opt.sh create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-sglang-mooncake-opt.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-sglang-mooncake-opt.sh new file mode 100755 index 0000000000..6f3fe6fd13 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/dsv4-gb300-sglang-mooncake-opt.sh @@ -0,0 +1,51 @@ +#!/usr/bin/env bash +# TEMPORARY: verify the SGLang source tree carrying the DSV4 Mooncake +# external-linker optimizations before any server starts. +# +# The recipes put ${SGLANG_SRC}/python on PYTHONPATH so the servers import the +# reviewed branch instead of the container's installed SGLang, while still +# using the container's compiled kernels. If the runner failed to clone the +# tree, PYTHONPATH silently resolves to the stock package and the job would +# benchmark the wrong code, so fail loudly here instead. +# +# Delete this script, its runner wiring, and the recipes' setup_script and +# environment keys once the optimizations ship in the pinned image. +set -euo pipefail + +SGLANG_SRC="${SGLANG_MOONCAKE_OPT_SRC:-/configs/sglang-mooncake-opt}" +INIT="${SGLANG_SRC}/python/sglang/__init__.py" + +if [ ! -f "${INIT}" ]; then + echo "ERROR: SGLang optimization source missing at ${INIT}." >&2 + echo "The runner must clone it before submitting; refusing to run against" >&2 + echo "the container's stock SGLang." >&2 + exit 1 +fi + +python3 - "${SGLANG_SRC}" <<'PYEOF' +import sys +from pathlib import Path + +root = Path(sys.argv[1]) / "python" +required = { + "SGLANG_EXTERNAL_LINKER_SWA_RETENTION_INTERVAL": "sglang/srt/environ.py", + "mooncake_store_contributor": "sglang/srt/arg_groups/fields/memory.py", +} +for token, relative in sorted(required.items()): + path = root / relative + if token not in path.read_text(): + raise SystemExit(f"ERROR: {token} not found in {path}; wrong revision") +print(f"[sglang-mooncake-opt] verified source tree at {root}") +PYEOF + +COMPAT_SHIM="${SGLANG_MOONCAKE_OPT_COMPAT:-/configs/sglang-server-args-compat}/sitecustomize.py" +if [ ! -f "${COMPAT_SHIM}" ]; then + echo "ERROR: Dynamo compatibility shim missing at ${COMPAT_SHIM}." >&2 + exit 1 +fi + +resolved=$(PYTHONPATH="${SGLANG_SRC}/python:${PYTHONPATH:-}" python3 -c 'import sglang; print(sglang.__file__)') +case "${resolved}" in + "${SGLANG_SRC}"/*) echo "[sglang-mooncake-opt] sglang resolves to ${resolved}" ;; + *) echo "ERROR: sglang resolves to ${resolved}, not ${SGLANG_SRC}" >&2; exit 1 ;; +esac diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py new file mode 100644 index 0000000000..4ad5d3c77b --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/sglang-server-args-compat/sitecustomize.py @@ -0,0 +1,27 @@ +"""TEMPORARY: restore ServerArgs.get_model_config for the pinned Dynamo wheel. + +The reviewed SGLang branch these recipes import moved model-config resolution +out of ServerArgs, but the pinned Dynamo wheel still calls +``server_args.get_model_config()`` while parsing worker arguments. Re-attach +the accessor so the wheel keeps working against the newer source tree. + +This file is picked up because its directory is on PYTHONPATH, so it is +imported by every interpreter in the job, including ones that never import +SGLang. Failing to import SGLang there is expected and must stay silent. + +Remove this directory, its PYTHONPATH entry, and the rest of the temporary +source override once the optimizations ship in the pinned image. +""" + +try: + from sglang.srt.arg_groups.model_override_base import model_config_of + from sglang.srt.server_args import ServerArgs +except Exception: # noqa: BLE001 - non-SGLang interpreters legitimately fail here + pass +else: + if not hasattr(ServerArgs, "get_model_config"): + + def get_model_config(self): + return model_config_of(self) + + ServerArgs.get_model_config = get_model_config diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml index ad2af4be37..8af1e09ca1 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -13,7 +13,7 @@ base: model: repo: "deepseek-ai/DeepSeek-V4-Pro-0813" container: - image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21" + image: "lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9" dynamo: install: true source: @@ -23,6 +23,17 @@ base: health_check: max_attempts: 1440 interval_seconds: 10 + + # TEMPORARY: the Mooncake external-linker optimizations this Pareto point + # depends on are not in a released SGLang image yet. The gb300-nv runner clones + # the reviewed branch into the srt-slurm configs directory and the servers + # import it from there. The setup script aborts the job when that tree is + # missing, so a run can never silently fall back to the container's SGLang. + # Remove both keys once the optimizations ship in the pinned image. + setup_script: dsv4-gb300-sglang-mooncake-opt.sh + + environment: + PYTHONPATH: /configs/sglang-server-args-compat:/configs/sglang-mooncake-opt/python resources: gpu_type: gb300 gpus_per_node: 4 @@ -37,6 +48,10 @@ base: node: dedicated options: max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + args: + - --eviction_high_watermark_ratio=0.90 frontend: type: dynamo nginx_session_affinity: true @@ -86,6 +101,15 @@ base: SGLANG_LOG_MS: '1' SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + MOONCAKE_PROTOCOL: rdma + MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb + MOONCAKE_STANDALONE_STORAGE: '0' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' + SGLANG_EXTERNAL_LINKER_SWA_RETENTION_INTERVAL: '32768' + SGLANG_EXTERNAL_LINKER_WRITE_THROUGH_THRESHOLD: '1' args: host: 0.0.0.0 served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 @@ -119,10 +143,11 @@ base: speculative-num-steps: 1 speculative-eagle-topk: 1 speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 1 - hicache-io-backend: direct + prefill-chunk-size-per-request: 2048 + disaggregation-defer-partial-kv-transfer: true + enable-unified-cache-external-linker: true + unified-cache-external-linker-backend: mooncake + hicache-storage-backend-extra-config: '{"enable_group_semantics":true}' decode: nodes: 4 workers: 1 @@ -155,6 +180,13 @@ base: SGLANG_LOG_FORWARD_ITERS: '1' SGLANG_LOG_MS: '1' SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + MOONCAKE_PROTOCOL: rdma + MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb + MOONCAKE_STANDALONE_STORAGE: '0' + MC_STORE_CLIENT_METRIC: '1' + MC_STORE_CLIENT_METRIC_INTERVAL: '5' + MC_ENABLE_DEST_DEVICE_AFFINITY: '1' + WITH_NVIDIA_PEERMEM: '0' args: host: 0.0.0.0 served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 @@ -188,6 +220,7 @@ base: speculative-num-steps: 1 speculative-eagle-topk: 1 speculative-num-draft-tokens: 7 + mooncake-store-contributor: true sbatch_directives: mem: "0" cpus-per-task: "144" @@ -224,11 +257,11 @@ override_1p1d_c480: gpus: 8 args: max-running-requests: 256 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 decode: gpus: 16 args: - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 # Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300 # (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960. @@ -247,13 +280,13 @@ override_2p1d_c960: OMP_NUM_THREADS: '1' args: max-running-requests: 256 - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 decode: env: OMP_NUM_THREADS: '1' SGLANG_DSV4_MHC_PREWARM: '1' args: - cuda-graph-max-bs: 256 + cuda-graph-max-bs-decode: 256 # Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 # (6P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1440. @@ -271,11 +304,11 @@ override_3p1d_c1440: gpus: 8 args: max-running-requests: 512 - cuda-graph-max-bs: 512 + cuda-graph-max-bs-decode: 512 decode: gpus: 16 args: - cuda-graph-max-bs: 512 + cuda-graph-max-bs-decode: 512 # Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300 # (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920. @@ -301,13 +334,13 @@ override_4p1d_c1920: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" args: max-running-requests: 1024 - cuda-graph-max-bs: 1024 + cuda-graph-max-bs-decode: 1024 decode: gpus: 16 env: SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600" args: - cuda-graph-max-bs: 192 + cuda-graph-max-bs-decode: 192 benchmark: env: AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index b0fc9ef451..020ec59ed7 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -6661,7 +6661,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg: additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4" dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21 + image: lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9 model: deepseek-ai/DeepSeek-V4-Pro-0813 model-prefix: dsv4 runner: cluster:gb300-nv @@ -6678,7 +6678,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [480] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake } prefill: num-worker: 1 tp: 8 @@ -6694,7 +6694,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [960] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake } prefill: num-worker: 2 tp: 8 @@ -6710,7 +6710,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [1440] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake } prefill: num-worker: 3 tp: 8 @@ -6726,7 +6726,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg: - spec-decoding: draft_model conc-list: [1920] kv-offloading: dram - kv-offload-backend: { name: hicache } + kv-offload-backend: { name: mooncake } prefill: num-worker: 4 tp: 8 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index e280ad359f..4d6125ccf9 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -8995,3 +8995,36 @@ - "Validate MiniMax-M3 AgentX on H200 with vLLM FP8 after moving the end-to-end project into inferencex-e2e/. Preserve the existing image, recipe, and benchmark settings." - "将端到端项目迁入 inferencex-e2e/ 后,验证 H200 上的 vLLM FP8 MiniMax-M3 AgentX 运行路径,沿用现有镜像、配方和基准测试设置。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3525 + +- config-keys: + - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Move the GB300 DeepSeek-V4-Pro AgentX disaggregated ladder from hierarchical cache to the Mooncake unified-cache external linker, let decode ranks contribute host DRAM to the store, and keep only a sparse SWA checkpoint per 32768-token interval so the same DRAM holds about twice as many prefixes." + - "Prefill now splits a request into 2048-token chunks and defers intermediate chunk KV transfer. Image bumped to lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9, the first pinned image whose sglang-kernel satisfies the minimum the reviewed branch asserts." + - "Measured against the previous configuration at the same golden acceptance length: total tok/s/GPU +13.2% at 1P1D, +18.4% at 2P1D, +17.5% at 4P1D, with mean TTFT -21.6%, -30.7%, -37.3%. Steady-state prefix cache hit rate is 96.0-96.2% across all three points." + - "The runner also pins this cluster's RDMA devices for the Mooncake store: without an explicit device list the client routes store transfers over NVLink for same-domain peers, whose address lookup fails and aborts every worker during the store warmup put. Temporary: the serving change is still in review as sgl-project/sglang#39694, so the runner clones that branch into the srt-slurm configs directory and the servers import it through PYTHONPATH while using the container's compiled kernels. A setup script aborts the job when that tree is missing or is the wrong revision, so a run cannot silently fall back to the container's SGLang. Remove the clone, the setup script, and the recipes' PYTHONPATH once the change ships in the pinned image." + - "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到 Mooncake unified-cache external linker,让 decode rank 也贡献主机 DRAM 容量,并按 32768 token 间隔只保留稀疏 SWA checkpoint,使同样的 DRAM 可容纳约两倍前缀。" + - "Prefill 改为按 2048 token 切分单个请求并延迟中间 chunk 的 KV 传输,镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,这是首个 sglang-kernel 版本满足该分支最低要求的固定镜像。" + - "在相同 golden acceptance length 下相比原配置实测:1P1D、2P1D、4P1D 的单卡总吞吐分别提升 13.2%、18.4%、17.5%,平均 TTFT 分别下降 21.6%、30.7%、37.3%,三个点的稳态前缀命中率均为 96.0%-96.2%。" + - "runner 同时为 Mooncake store 指定该集群的 RDMA 设备:缺少显式设备列表时,store 的传输会对同 NVLink 域的对端走 NVLink,地址查找失败并在 warmup put 阶段让所有 worker 退出。临时方案:服务端改动仍在 sgl-project/sglang#39694 评审中,因此 runner 将该分支克隆到 srt-slurm 的 configs 目录,服务进程通过 PYTHONPATH 导入,同时继续使用容器内编译好的 kernel。配套 setup 脚本在该源码树缺失或版本不符时直接让任务失败,避免静默回退到容器自带的 SGLang。待改动进入固定镜像后应移除克隆、setup 脚本和 recipe 中的 PYTHONPATH。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187 + +- config-keys: + - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Replace the removed cuda-graph-max-bs recipe key with cuda-graph-max-bs-decode in every prefill and decode role across the GB300 DSV4 AgentX Mooncake C480-C1920 ladder, preserving each role's existing graph batch-size value. Run 35101789051 failed before server readiness because the reviewed SGLang branch split the option into decode/prefill variants and rejected the legacy spelling as ambiguous." + - "在 GB300 DSV4 AgentX Mooncake C480-C1920 阶梯的所有 prefill 与 decode role 中,将已移除的 cuda-graph-max-bs 配方键替换为 cuda-graph-max-bs-decode,并保留各 role 现有的 CUDA graph batch-size 数值。运行 35101789051 在服务就绪前失败,因为评审中的 SGLang 分支已将该选项拆分为 decode/prefill 两个变体,并因歧义拒绝旧写法。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187 + +- config-keys: + - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Drive the decode ranks' passive Mooncake capacity contribution from the server flag --mooncake-store-contributor instead of the SGLANG_MOONCAKE_STORE_CONTRIBUTOR environment variable, and repin the reviewed SGLang branch to the revision that builds the contributor where the KV pools are built. The environment variable was read by the Mooncake transfer manager, so the contribution silently disappeared whenever KV transfer used another backend, and setting it on a prefill rank did nothing without an error; the flag is rejected on non-decode roles and when combined with the external linker. Recipe behaviour at every ladder point is unchanged." + - "将 decode rank 向 Mooncake store 贡献主机 DRAM 的开关从环境变量 SGLANG_MOONCAKE_STORE_CONTRIBUTOR 改为服务端参数 --mooncake-store-contributor,并把评审中的 SGLang 分支重新固定到在构建 KV pool 处创建该 contributor 的版本。原环境变量由 Mooncake 传输管理器读取,因此只要 KV 传输改用其他后端,容量贡献就会静默消失;把它设在 prefill rank 上则毫无作用且不报错。新参数在非 decode 角色或与 external linker 同时开启时会直接报错。各并发点的配方行为不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187 diff --git a/inferencex-e2e/runners/launch_gb300-nv.sh b/inferencex-e2e/runners/launch_gb300-nv.sh index a09a238e40..e5c68042d5 100644 --- a/inferencex-e2e/runners/launch_gb300-nv.sh +++ b/inferencex-e2e/runners/launch_gb300-nv.sh @@ -180,6 +180,29 @@ rm -rf "$SRT_REPO_DIR" setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" "$USES_DCGM_POWER" || exit 1 +# TEMPORARY: the Mooncake external-linker optimizations the DSV4 AgentX +# disaggregated recipes rely on are not in a released SGLang image yet, so +# clone the reviewed branch into configs/ (mounted at /configs) and let the +# servers import it through PYTHONPATH while still using the container's +# compiled kernels. The recipes' setup script aborts the job if this tree is +# missing. Drop this block, the setup script, and the recipes' PYTHONPATH once +# the change ships in the pinned image. +if [[ "$IS_AGENTIC" == "1" && "$FRAMEWORK" == "dynamo-sglang" && "$MODEL_PREFIX" == "dsv4" ]]; then + SGLANG_MOONCAKE_OPT_URL="https://github.com/weireweire/sglang.git" + SGLANG_MOONCAKE_OPT_PIN="7b18eddd006a0166c5d092dfb399ca7d136494fc" + git init configs/sglang-mooncake-opt || exit 1 + git -C configs/sglang-mooncake-opt fetch --depth 1 \ + "$SGLANG_MOONCAKE_OPT_URL" "$SGLANG_MOONCAKE_OPT_PIN" || exit 1 + git -C configs/sglang-mooncake-opt checkout --detach FETCH_HEAD || exit 1 + + # The Mooncake store transfers from host buffers that only the RDMA + # transport registers on the fly. Without an explicit device list the + # client falls back to NVLink for same-domain peers, whose address lookup + # then fails and aborts every worker during the store warmup put. These + # are this cluster's RDMA devices; other clusters name theirs differently. + MOONCAKE_STORE_DEVICES="mlx5_0,mlx5_1,mlx5_2,mlx5_3" +fi + if [[ "$FRAMEWORK" == "dynamo-trt" && "$MODEL_PREFIX" == "dsv4" ]]; then SRT_SLURM_MODEL_PREFIX="deepseek-ai/DeepSeek-V4-Pro" fi @@ -262,6 +285,10 @@ SRTCTL_APPLY_ARGS=( -f "$CONFIG_FILE" --tags "gb300,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" ) +if [[ -n "${MOONCAKE_STORE_DEVICES:-}" ]]; then + SRTCTL_APPLY_ARGS+=(--set "roles.prefill.env.MOONCAKE_DEVICE=$MOONCAKE_STORE_DEVICES") + SRTCTL_APPLY_ARGS+=(--set "roles.decode.env.MOONCAKE_DEVICE=$MOONCAKE_STORE_DEVICES") +fi if [[ "$IS_AGENTIC" == "1" || ( "$MODEL_PREFIX" == "qwen3.5" && "$PRECISION" == "fp8" ) || ( "$MODEL_PREFIX" == "qwen3.5" && "$PRECISION" == "fp4" && ( "$FRAMEWORK" == "dynamo-trt" || "$USES_DCGM_POWER" == "1" ) ) || ( "$USES_DCGM_POWER" == "1" && "$MODEL_PREFIX" == "dsv4" && "$FRAMEWORK" == "dynamo-sglang" ) ]]; then SRTCTL_APPLY_ARGS+=(--no-preflight) fi From 6317a45f239d9991b160a94068b5e0d8c6e6530e Mon Sep 17 00:00:00 2001 From: weireweire <20922698+weireweire@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:59:11 +0800 Subject: [PATCH 2/2] fix(agentx): limit the Mooncake runner setup to the disagg recipe The block that clones the reviewed SGLang branch and forwards the Mooncake RDMA device list matched every dsv4 dynamo-sglang AgentX run, including the aggregated recipe, which has no prefill/decode roles for the device list to target and never imports the reviewed branch. Key it on the disaggregated variants file instead. --- inferencex-e2e/runners/launch_gb300-nv.sh | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/inferencex-e2e/runners/launch_gb300-nv.sh b/inferencex-e2e/runners/launch_gb300-nv.sh index e5c68042d5..bd0113d0d8 100644 --- a/inferencex-e2e/runners/launch_gb300-nv.sh +++ b/inferencex-e2e/runners/launch_gb300-nv.sh @@ -187,7 +187,11 @@ setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" "$USES_DCGM_POWER" || exit 1 # compiled kernels. The recipes' setup script aborts the job if this tree is # missing. Drop this block, the setup script, and the recipes' PYTHONPATH once # the change ships in the pinned image. -if [[ "$IS_AGENTIC" == "1" && "$FRAMEWORK" == "dynamo-sglang" && "$MODEL_PREFIX" == "dsv4" ]]; then +# Only the disaggregated recipe imports the reviewed branch and uses the +# Mooncake store; the aggregated dsv4 recipe has no prefill/decode roles for the +# device list to target. +if [[ "$IS_AGENTIC" == "1" && "$FRAMEWORK" == "dynamo-sglang" && + "${CONFIG_FILE%%:*}" == */dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml ]]; then SGLANG_MOONCAKE_OPT_URL="https://github.com/weireweire/sglang.git" SGLANG_MOONCAKE_OPT_PIN="7b18eddd006a0166c5d092dfb399ca7d136494fc" git init configs/sglang-mooncake-opt || exit 1