Skip to content
Open
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
#!/usr/bin/env bash
set -eo pipefail

python3 -m pip uninstall --break-system-packages -y mooncake-transfer-engine-cuda13 mooncake-transfer-engine-efa-cuda13
python3 -m pip install --break-system-packages --no-deps mooncake-transfer-engine-efa-cuda13==0.3.13.post1
Original file line number Diff line number Diff line change
@@ -0,0 +1,163 @@
schema: 2
name: "agg-b300-dep8-c384-mtp-kvoffload"

# AgentX aggregate topology: one DEP8 worker (attention DP8 + EP8 Mega-MoE +
# FP4 indexer) occupies one eight-GPU B300 node and serves both prefill and
# decode with DSpark K=6 and a HiCache DRAM tier, tuned for concurrency 384.
# Engine settings follow the single-node SGLang DEP8 c384 point
# (benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-mtp/agentic.yaml,
# override_dep8_c384). The Dynamo KV router replaces the SGLang router: each
# DP rank publishes KV events, and sessions stay sticky via X-Dynamo-Session-ID.
# Concurrency is exported by the master config.

model:
path: "deepseek-v4-pro-0813"
container: "dynamo-sglang"
precision: "fp4"

identity:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro-0813"
container:
image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9"

dynamo:
install: false

health_check:
max_attempts: 1440
interval_seconds: 10

resources:
gpu_type: "b300"
gpus_per_node: 8
services:
- name: etcd
type: etcd
placement:
node: infra
- name: nats
type: nats
placement:
node: infra
options:
max_payload_mb: 32
frontend:
type: dynamo
nginx_session_affinity: true
nginx_session_affinity_header: X-Dynamo-Session-ID
enable_multiple_frontends: false
env:
PIP_BREAK_SYSTEM_PACKAGES: "1"
DYN_NATS_REQUEST_TIMEOUT_SECS: "1800"
# The 5 s default request-plane ack timeout expires while the worker
# ingests the c384 warmup burst; match the disaggregated recipes.
DYN_TCP_REQUEST_TIMEOUT: "60"
args:
router-mode: "kv"
router-session-affinity-ttl-secs: "3600"
active-decode-blocks-threshold: "None"
active-prefill-tokens-threshold: "None"
active-prefill-tokens-threshold-frac: "None"

engine: sglang
roles:
agg:
nodes: 1
workers: 1
gpus: 8

env:
SGLANG_RAGGED_VERIFY_MODE: "static"
SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1"
SGLANG_DEFAULT_THINKING: "1"
SGLANG_DSV4_REASONING_EFFORT: high
PIP_BREAK_SYSTEM_PACKAGES: "1"
PYTHONNOUSERSITE: "1"
TORCH_CUDA_ARCH_LIST: "10.0"
# Triton compiles with the image's CUDA ptxas.
TRITON_PTXAS_PATH: /usr/local/cuda/bin/ptxas
SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1"
SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1"
SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1"
SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1"
SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1"
SGLANG_OPT_USE_ONLINE_COMPRESS: "0"
SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1"
SGLANG_OPT_USE_JIT_NORM: "1"
SGLANG_OPT_USE_TOPK_V2: "True"
# Covers the 8192-token per-rank prefill budget (65536 / 8 DP ranks).
SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: "8320"

args:
served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813"
enable-metrics: true
enable-cache-report: true
trust-remote-code: true
weight-loader-prefetch-checkpoints: true
watchdog-timeout: 1000000
allow-auto-truncate: true
attention-backend: dsv4
page-size: 256
disable-shared-experts-fusion: true
disable-flashinfer-autotune: true
tp-size: 8
dp-size: 8
ep-size: 8
enable-dp-attention: true
enable-dp-lm-head: true
enable-dp-attention-local-control-broadcast: true
moe-a2a-backend: megamoe
enable-deepseek-v4-fp4-indexer: true
enable-prefill-delayer: true
prefill-decode-interval: 20
incremental-streaming-output: true
stream-interval: 20
# Mega-MoE's transient workspace sits outside the static pool.
mem-fraction-static: 0.88
swa-full-tokens-ratio: 0.075
chunked-prefill-size: 65536
max-running-requests: 768
# Per-rank graph limit: max-running-requests / dp-size = 768 / 8.
cuda-graph-max-bs-decode: 96
kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}'
speculative-algorithm: DSPARK
speculative-dspark-block-size: 6
speculative-num-steps: 1
speculative-eagle-topk: 1
speculative-num-draft-tokens: 7
# HiCache capacity is a host/device ratio; 3 keeps the tier near 2 TB.
enable-hierarchical-cache: true
hicache-ratio: 3
hicache-write-policy: write_back
hicache-io-backend: direct
hicache-mem-layout: page_first_direct

sbatch_directives:
mem: "0"
exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16"

srun_options:
mem: "0"
container-remap-root: ""

benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace"
RESULT_DIR: "/logs/agentic"
PORT: "8000"
IS_MULTINODE: "false"
TP: "8"
EP_SIZE: "8"
DP_ATTENTION: "true"
PP_SIZE: "1"
PCP_SIZE: "1"
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:"
# Warmup can leave bytes unacknowledged past AIPerf's 30 s default.
AIPERF_HTTP_TCP_USER_TIMEOUT: "900000"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
Original file line number Diff line number Diff line change
@@ -0,0 +1,128 @@
schema: 2
name: "agg-b300-tp4-c4-mtp"

# AgentX aggregate topology: one TP4 worker occupies half of one eight-GPU
# B300 node and serves both prefill and decode with DSpark K=6.

model:
path: "deepseek-v4-pro-0813"
container: "dynamo-sglang"
precision: "fp4"

identity:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro-0813"
container:
image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9"

dynamo:
install: false

health_check:
max_attempts: 1440
interval_seconds: 10

resources:
gpu_type: "b300"
gpus_per_node: 8
services:
- name: etcd
type: etcd
placement:
node: infra
- name: nats
type: nats
placement:
node: infra
options:
max_payload_mb: 32
frontend:
type: dynamo
nginx_session_affinity: true
nginx_session_affinity_header: X-Dynamo-Session-ID
enable_multiple_frontends: false
env:
PIP_BREAK_SYSTEM_PACKAGES: "1"
DYN_NATS_REQUEST_TIMEOUT_SECS: "1800"
args:
router-mode: "kv"
router-session-affinity-ttl-secs: "3600"
active-decode-blocks-threshold: "None"
active-prefill-tokens-threshold: "None"
active-prefill-tokens-threshold-frac: "None"

engine: sglang
roles:
agg:
nodes: 1
workers: 1
gpus: 4

env:
SGLANG_RAGGED_VERIFY_MODE: "static"
SGLANG_DEFAULT_THINKING: "1"
SGLANG_DSV4_REASONING_EFFORT: high
PIP_BREAK_SYSTEM_PACKAGES: "1"
SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1"
SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1"
SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1"
SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1"
SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1"
SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1"
SGLANG_OPT_USE_ONLINE_COMPRESS: "0"
SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1"
SGLANG_OPT_USE_JIT_NORM: "1"
SGLANG_OPT_USE_TOPK_V2: "True"

args:
served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813"
enable-metrics: true
enable-cache-report: true
trust-remote-code: true
weight-loader-prefetch-checkpoints: true
stream-interval: 10
watchdog-timeout: 1000000
mem-fraction-static: 0.90
page-size: 256
chunked-prefill-size: 8192
max-prefill-tokens: 8192
moe-runner-backend: "flashinfer_mxfp4"
enable-deepseek-v4-fp4-indexer: true
disable-flashinfer-autotune: true
swa-full-tokens-ratio: 0.1
max-running-requests: 8
cuda-graph-max-bs-decode: 8
scheduler-recv-interval: 30
dp-size: 1
tp-size: 4
ep-size: 1
speculative-algorithm: DSPARK
speculative-dspark-block-size: 6
speculative-num-steps: 1
speculative-eagle-topk: 1
speculative-num-draft-tokens: 7

sbatch_directives:
mem: "0"
exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16"

srun_options:
mem: "0"
container-remap-root: ""

benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace"
RESULT_DIR: "/logs/agentic"
PORT: "8000"
IS_MULTINODE: "false"
TP: "4"
PP_SIZE: "1"
PCP_SIZE: "1"
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
Loading
Loading