Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
#!/usr/bin/env bash
# TEMPORARY: verify the SGLang source tree carrying the DSV4 Mooncake
# external-linker optimizations before any server starts.
#
# The recipes put ${SGLANG_SRC}/python on PYTHONPATH so the servers import the
# reviewed branch instead of the container's installed SGLang, while still
# using the container's compiled kernels. If the runner failed to clone the
# tree, PYTHONPATH silently resolves to the stock package and the job would
# benchmark the wrong code, so fail loudly here instead.
#
# Delete this script, its runner wiring, and the recipes' setup_script and
# environment keys once the optimizations ship in the pinned image.
set -euo pipefail

SGLANG_SRC="${SGLANG_MOONCAKE_OPT_SRC:-/configs/sglang-mooncake-opt}"
INIT="${SGLANG_SRC}/python/sglang/__init__.py"

if [ ! -f "${INIT}" ]; then
echo "ERROR: SGLang optimization source missing at ${INIT}." >&2
echo "The runner must clone it before submitting; refusing to run against" >&2
echo "the container's stock SGLang." >&2
exit 1
fi

python3 - "${SGLANG_SRC}" <<'PYEOF'
import sys
from pathlib import Path

root = Path(sys.argv[1]) / "python"
required = {
"SGLANG_EXTERNAL_LINKER_SWA_RETENTION_INTERVAL": "sglang/srt/environ.py",
"mooncake_store_contributor": "sglang/srt/arg_groups/fields/memory.py",
}
for token, relative in sorted(required.items()):
path = root / relative
if token not in path.read_text():
raise SystemExit(f"ERROR: {token} not found in {path}; wrong revision")
print(f"[sglang-mooncake-opt] verified source tree at {root}")
PYEOF

COMPAT_SHIM="${SGLANG_MOONCAKE_OPT_COMPAT:-/configs/sglang-server-args-compat}/sitecustomize.py"
if [ ! -f "${COMPAT_SHIM}" ]; then
echo "ERROR: Dynamo compatibility shim missing at ${COMPAT_SHIM}." >&2
exit 1
fi

resolved=$(PYTHONPATH="${SGLANG_SRC}/python:${PYTHONPATH:-}" python3 -c 'import sglang; print(sglang.__file__)')
case "${resolved}" in
"${SGLANG_SRC}"/*) echo "[sglang-mooncake-opt] sglang resolves to ${resolved}" ;;
*) echo "ERROR: sglang resolves to ${resolved}, not ${SGLANG_SRC}" >&2; exit 1 ;;
esac
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
"""TEMPORARY: restore ServerArgs.get_model_config for the pinned Dynamo wheel.

The reviewed SGLang branch these recipes import moved model-config resolution
out of ServerArgs, but the pinned Dynamo wheel still calls
``server_args.get_model_config()`` while parsing worker arguments. Re-attach
the accessor so the wheel keeps working against the newer source tree.

This file is picked up because its directory is on PYTHONPATH, so it is
imported by every interpreter in the job, including ones that never import
SGLang. Failing to import SGLang there is expected and must stay silent.

Remove this directory, its PYTHONPATH entry, and the rest of the temporary
source override once the optimizations ship in the pinned image.
"""

try:
from sglang.srt.arg_groups.model_override_base import model_config_of
from sglang.srt.server_args import ServerArgs
except Exception: # noqa: BLE001 - non-SGLang interpreters legitimately fail here
pass
else:
if not hasattr(ServerArgs, "get_model_config"):

def get_model_config(self):
return model_config_of(self)

ServerArgs.get_model_config = get_model_config
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@ base:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro-0813"
container:
image: "lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21"
image: "lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9"
dynamo:
install: true
source:
Expand All @@ -23,6 +23,17 @@ base:
health_check:
max_attempts: 1440
interval_seconds: 10

# TEMPORARY: the Mooncake external-linker optimizations this Pareto point
# depends on are not in a released SGLang image yet. The gb300-nv runner clones
# the reviewed branch into the srt-slurm configs directory and the servers
# import it from there. The setup script aborts the job when that tree is
# missing, so a run can never silently fall back to the container's SGLang.
# Remove both keys once the optimizations ship in the pinned image.
setup_script: dsv4-gb300-sglang-mooncake-opt.sh

environment:
PYTHONPATH: /configs/sglang-server-args-compat:/configs/sglang-mooncake-opt/python
resources:
gpu_type: gb300
gpus_per_node: 4
Expand All @@ -37,6 +48,10 @@ base:
node: dedicated
options:
max_payload_mb: 32
- name: mooncake-master
type: mooncake-master
args:
- --eviction_high_watermark_ratio=0.90
frontend:
type: dynamo
nginx_session_affinity: true
Expand Down Expand Up @@ -86,6 +101,15 @@ base:
SGLANG_LOG_MS: '1'
SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60'
SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1'
MOONCAKE_PROTOCOL: rdma
MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb
MOONCAKE_STANDALONE_STORAGE: '0'
MC_STORE_CLIENT_METRIC: '1'
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
SGLANG_EXTERNAL_LINKER_SWA_RETENTION_INTERVAL: '32768'
SGLANG_EXTERNAL_LINKER_WRITE_THROUGH_THRESHOLD: '1'
args:
host: 0.0.0.0
served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813
Expand Down Expand Up @@ -119,10 +143,11 @@ base:
speculative-num-steps: 1
speculative-eagle-topk: 1
speculative-num-draft-tokens: 7
enable-hierarchical-cache: true
hicache-write-policy: write_back
hicache-ratio: 1
hicache-io-backend: direct
prefill-chunk-size-per-request: 2048
disaggregation-defer-partial-kv-transfer: true
enable-unified-cache-external-linker: true
unified-cache-external-linker-backend: mooncake
hicache-storage-backend-extra-config: '{"enable_group_semantics":true}'
decode:
nodes: 4
workers: 1
Expand Down Expand Up @@ -155,6 +180,13 @@ base:
SGLANG_LOG_FORWARD_ITERS: '1'
SGLANG_LOG_MS: '1'
SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60'
MOONCAKE_PROTOCOL: rdma
MOONCAKE_GLOBAL_SEGMENT_SIZE: 140gb
MOONCAKE_STANDALONE_STORAGE: '0'
MC_STORE_CLIENT_METRIC: '1'
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
args:
host: 0.0.0.0
served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813
Expand Down Expand Up @@ -188,6 +220,7 @@ base:
speculative-num-steps: 1
speculative-eagle-topk: 1
speculative-num-draft-tokens: 7
mooncake-store-contributor: true
sbatch_directives:
mem: "0"
cpus-per-task: "144"
Expand Down Expand Up @@ -224,11 +257,11 @@ override_1p1d_c480:
gpus: 8
args:
max-running-requests: 256
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256
decode:
gpus: 16
args:
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256

# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro on GB300
# (4P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 960.
Expand All @@ -247,13 +280,13 @@ override_2p1d_c960:
OMP_NUM_THREADS: '1'
args:
max-running-requests: 256
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256
decode:
env:
OMP_NUM_THREADS: '1'
SGLANG_DSV4_MHC_PREWARM: '1'
args:
cuda-graph-max-bs: 256
cuda-graph-max-bs-decode: 256

# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300
# (6P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1440.
Expand All @@ -271,11 +304,11 @@ override_3p1d_c1440:
gpus: 8
args:
max-running-requests: 512
cuda-graph-max-bs: 512
cuda-graph-max-bs-decode: 512
decode:
gpus: 16
args:
cuda-graph-max-bs: 512
cuda-graph-max-bs-decode: 512

# Agentic-coding SGLang disaggregated Pareto recipe for DeepSeek-V4-Pro on GB300
# (8P x DEP8 / 4D x DEP16, DSpark K=6 + hierarchical-cache KV offload), tuned for concurrency 1920.
Expand All @@ -301,13 +334,13 @@ override_4p1d_c1920:
SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600"
args:
max-running-requests: 1024
cuda-graph-max-bs: 1024
cuda-graph-max-bs-decode: 1024
decode:
gpus: 16
env:
SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "3600"
args:
cuda-graph-max-bs: 192
cuda-graph-max-bs-decode: 192
benchmark:
env:
AIPERF_HTTP_TCP_USER_TIMEOUT: "900000"
10 changes: 5 additions & 5 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6661,7 +6661,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-agg:
additional-settings:
- "CONFIG_FILE=recipes/dsv4/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4"
dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
image: lmsysorg/sglang:nightly-dev-cu13-20260829-89816a21
image: lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9
model: deepseek-ai/DeepSeek-V4-Pro-0813
model-prefix: dsv4
runner: cluster:gb300-nv
Expand All @@ -6678,7 +6678,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
- spec-decoding: draft_model
conc-list: [480]
kv-offloading: dram
kv-offload-backend: { name: hicache }
kv-offload-backend: { name: mooncake }

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 (optional) Operators reading nvidia-master.yaml lose the Mooncake component version that CONFIGS.md requires, making sweep provenance for this backend unrecoverable if the pinned Mooncake build ever needs to be identified after the fact. All four new entries set kv-offload-backend: { name: mooncake } with no version, but configs/CONFIGS.md:136-137 explicitly names Mooncake as an example of an independently-versioned backend that should supply version (unlike framework-native vLLM/HiCache). The Pydantic validator (KVOffloadBackendMetadata in infx/matrix/validation.py:127-132) doesn't enforce this, so nothing blocks the omission. …

Extended reasoning...

…Fix: add a non-image version (e.g. the Mooncake release/commit) for the mooncake kv-offload-backend entries, covering all four sites (configs/nvidia-master.yaml:9066,9084,9102,9120), or update CONFIGS.md if this backend is now treated as framework-native.

configs/CONFIGS.md:133-138 says kv-offload-backend requires non-empty name, version optional only for framework-native backends (vLLM built-in, SGLang HiCache), and explicitly lists Mooncake as needing version since it's independently versioned. The new entries at lines 9066, 9084, 9102, 9120 all set { name: mooncake } with no version key. infx/matrix/validation.py's KVOffloadBackendMetadata makes version fully Optional with no name-based branching (confirmed by reading lines 127-143, and by utils/matrix_logic/test_validation.py tests that never require version for any specific backend name). So CI will not fail, but the documented provenance convention is silently violated: nobody can tell from the master config which Mooncake build/commit these Pareto points were validated against, unlike the equivalent LMCache examples in…

Verification: nit. The deviation is real but the consequence is mild/overstated. CONFIGS.md:135-138 states version is optional only for framework-native backends and explicitly directs "Supply version for independently versioned backends such as LMCache or Mooncake." All four new search-space entries in configs/nvidia-master.yaml (lines 9066, 9084, 9102, 9120) set kv-offload-backend: { name: mooncake }…

prefill:
num-worker: 1
tp: 8
Expand All @@ -6694,7 +6694,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
- spec-decoding: draft_model
conc-list: [960]
kv-offloading: dram
kv-offload-backend: { name: hicache }
kv-offload-backend: { name: mooncake }
prefill:
num-worker: 2
tp: 8
Expand All @@ -6710,7 +6710,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
- spec-decoding: draft_model
conc-list: [1440]
kv-offloading: dram
kv-offload-backend: { name: hicache }
kv-offload-backend: { name: mooncake }
prefill:
num-worker: 3
tp: 8
Expand All @@ -6726,7 +6726,7 @@ dsv4-fp4-gb300-dynamo-sglang-agentic-disagg:
- spec-decoding: draft_model
conc-list: [1920]
kv-offloading: dram
kv-offload-backend: { name: hicache }
kv-offload-backend: { name: mooncake }
prefill:
num-worker: 4
tp: 8
Expand Down
33 changes: 33 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8995,3 +8995,36 @@
- "Validate MiniMax-M3 AgentX on H200 with vLLM FP8 after moving the end-to-end project into inferencex-e2e/. Preserve the existing image, recipe, and benchmark settings."
- "将端到端项目迁入 inferencex-e2e/ 后,验证 H200 上的 vLLM FP8 MiniMax-M3 AgentX 运行路径,沿用现有镜像、配方和基准测试设置。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3525

- config-keys:
- dsv4-fp4-gb300-dynamo-sglang-agentic-disagg
scenario-type:
- agentic-coding
description:
- "Move the GB300 DeepSeek-V4-Pro AgentX disaggregated ladder from hierarchical cache to the Mooncake unified-cache external linker, let decode ranks contribute host DRAM to the store, and keep only a sparse SWA checkpoint per 32768-token interval so the same DRAM holds about twice as many prefixes."
- "Prefill now splits a request into 2048-token chunks and defers intermediate chunk KV transfer. Image bumped to lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9, the first pinned image whose sglang-kernel satisfies the minimum the reviewed branch asserts."
- "Measured against the previous configuration at the same golden acceptance length: total tok/s/GPU +13.2% at 1P1D, +18.4% at 2P1D, +17.5% at 4P1D, with mean TTFT -21.6%, -30.7%, -37.3%. Steady-state prefix cache hit rate is 96.0-96.2% across all three points."
- "The runner also pins this cluster's RDMA devices for the Mooncake store: without an explicit device list the client routes store transfers over NVLink for same-domain peers, whose address lookup fails and aborts every worker during the store warmup put. Temporary: the serving change is still in review as sgl-project/sglang#39694, so the runner clones that branch into the srt-slurm configs directory and the servers import it through PYTHONPATH while using the container's compiled kernels. A setup script aborts the job when that tree is missing or is the wrong revision, so a run cannot silently fall back to the container's SGLang. Remove the clone, the setup script, and the recipes' PYTHONPATH once the change ships in the pinned image."
- "将 GB300 DeepSeek-V4-Pro AgentX 分离式配置从 hierarchical cache 切换到 Mooncake unified-cache external linker,让 decode rank 也贡献主机 DRAM 容量,并按 32768 token 间隔只保留稀疏 SWA checkpoint,使同样的 DRAM 可容纳约两倍前缀。"
- "Prefill 改为按 2048 token 切分单个请求并延迟中间 chunk 的 KV 传输,镜像升级到 lmsysorg/sglang:nightly-dev-cu13-20260916-c9a8fba9,这是首个 sglang-kernel 版本满足该分支最低要求的固定镜像。"
- "在相同 golden acceptance length 下相比原配置实测:1P1D、2P1D、4P1D 的单卡总吞吐分别提升 13.2%、18.4%、17.5%,平均 TTFT 分别下降 21.6%、30.7%、37.3%,三个点的稳态前缀命中率均为 96.0%-96.2%。"
- "runner 同时为 Mooncake store 指定该集群的 RDMA 设备:缺少显式设备列表时,store 的传输会对同 NVLink 域的对端走 NVLink,地址查找失败并在 warmup put 阶段让所有 worker 退出。临时方案:服务端改动仍在 sgl-project/sglang#39694 评审中,因此 runner 将该分支克隆到 srt-slurm 的 configs 目录,服务进程通过 PYTHONPATH 导入,同时继续使用容器内编译好的 kernel。配套 setup 脚本在该源码树缺失或版本不符时直接让任务失败,避免静默回退到容器自带的 SGLang。待改动进入固定镜像后应移除克隆、setup 脚本和 recipe 中的 PYTHONPATH。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187

- config-keys:
- dsv4-fp4-gb300-dynamo-sglang-agentic-disagg
scenario-type:
- agentic-coding
description:
- "Replace the removed cuda-graph-max-bs recipe key with cuda-graph-max-bs-decode in every prefill and decode role across the GB300 DSV4 AgentX Mooncake C480-C1920 ladder, preserving each role's existing graph batch-size value. Run 35101789051 failed before server readiness because the reviewed SGLang branch split the option into decode/prefill variants and rejected the legacy spelling as ambiguous."
- "在 GB300 DSV4 AgentX Mooncake C480-C1920 阶梯的所有 prefill 与 decode role 中,将已移除的 cuda-graph-max-bs 配方键替换为 cuda-graph-max-bs-decode,并保留各 role 现有的 CUDA graph batch-size 数值。运行 35101789051 在服务就绪前失败,因为评审中的 SGLang 分支已将该选项拆分为 decode/prefill 两个变体,并因歧义拒绝旧写法。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187

- config-keys:
- dsv4-fp4-gb300-dynamo-sglang-agentic-disagg
scenario-type:
- agentic-coding
description:
- "Drive the decode ranks' passive Mooncake capacity contribution from the server flag --mooncake-store-contributor instead of the SGLANG_MOONCAKE_STORE_CONTRIBUTOR environment variable, and repin the reviewed SGLang branch to the revision that builds the contributor where the KV pools are built. The environment variable was read by the Mooncake transfer manager, so the contribution silently disappeared whenever KV transfer used another backend, and setting it on a prefill rank did nothing without an error; the flag is rejected on non-decode roles and when combined with the external linker. Recipe behaviour at every ladder point is unchanged."
- "将 decode rank 向 Mooncake store 贡献主机 DRAM 的开关从环境变量 SGLANG_MOONCAKE_STORE_CONTRIBUTOR 改为服务端参数 --mooncake-store-contributor,并把评审中的 SGLang 分支重新固定到在构建 KV pool 处创建该 contributor 的版本。原环境变量由 Mooncake 传输管理器读取,因此只要 KV 传输改用其他后端,容量贡献就会静默消失;把它设在 prefill rank 上则毫无作用且不报错。新参数在非 decode 角色或与 external linker 同时开启时会直接报错。各并发点的配方行为不变。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3187
31 changes: 31 additions & 0 deletions inferencex-e2e/runners/launch_gb300-nv.sh
Original file line number Diff line number Diff line change
Expand Up @@ -180,6 +180,33 @@ rm -rf "$SRT_REPO_DIR"

setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" "$USES_DCGM_POWER" || exit 1

# TEMPORARY: the Mooncake external-linker optimizations the DSV4 AgentX
# disaggregated recipes rely on are not in a released SGLang image yet, so
# clone the reviewed branch into configs/ (mounted at /configs) and let the
# servers import it through PYTHONPATH while still using the container's
# compiled kernels. The recipes' setup script aborts the job if this tree is
# missing. Drop this block, the setup script, and the recipes' PYTHONPATH once
# the change ships in the pinned image.
# Only the disaggregated recipe imports the reviewed branch and uses the
# Mooncake store; the aggregated dsv4 recipe has no prefill/decode roles for the
# device list to target.
if [[ "$IS_AGENTIC" == "1" && "$FRAMEWORK" == "dynamo-sglang" &&
"${CONFIG_FILE%%:*}" == */dsv4/sglang/gb300-fp4/agentx/disagg-variants.yaml ]]; then
SGLANG_MOONCAKE_OPT_URL="https://github.com/weireweire/sglang.git"
SGLANG_MOONCAKE_OPT_PIN="7b18eddd006a0166c5d092dfb399ca7d136494fc"
git init configs/sglang-mooncake-opt || exit 1
git -C configs/sglang-mooncake-opt fetch --depth 1 \
"$SGLANG_MOONCAKE_OPT_URL" "$SGLANG_MOONCAKE_OPT_PIN" || exit 1
git -C configs/sglang-mooncake-opt checkout --detach FETCH_HEAD || exit 1

# The Mooncake store transfers from host buffers that only the RDMA
# transport registers on the fly. Without an explicit device list the
# client falls back to NVLink for same-domain peers, whose address lookup
# then fails and aborts every worker during the store warmup put. These
# are this cluster's RDMA devices; other clusters name theirs differently.
MOONCAKE_STORE_DEVICES="mlx5_0,mlx5_1,mlx5_2,mlx5_3"
fi

if [[ "$FRAMEWORK" == "dynamo-trt" && "$MODEL_PREFIX" == "dsv4" ]]; then
SRT_SLURM_MODEL_PREFIX="deepseek-ai/DeepSeek-V4-Pro"
fi
Expand Down Expand Up @@ -262,6 +289,10 @@ SRTCTL_APPLY_ARGS=(
-f "$CONFIG_FILE"
--tags "gb300,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)"
)
if [[ -n "${MOONCAKE_STORE_DEVICES:-}" ]]; then
SRTCTL_APPLY_ARGS+=(--set "roles.prefill.env.MOONCAKE_DEVICE=$MOONCAKE_STORE_DEVICES")
SRTCTL_APPLY_ARGS+=(--set "roles.decode.env.MOONCAKE_DEVICE=$MOONCAKE_STORE_DEVICES")
fi
if [[ "$IS_AGENTIC" == "1" || ( "$MODEL_PREFIX" == "qwen3.5" && "$PRECISION" == "fp8" ) || ( "$MODEL_PREFIX" == "qwen3.5" && "$PRECISION" == "fp4" && ( "$FRAMEWORK" == "dynamo-trt" || "$USES_DCGM_POWER" == "1" ) ) || ( "$USES_DCGM_POWER" == "1" && "$MODEL_PREFIX" == "dsv4" && "$FRAMEWORK" == "dynamo-sglang" ) ]]; then
SRTCTL_APPLY_ARGS+=(--no-preflight)
fi
Expand Down
Loading