From 8c34042cb23e19457e84f3e4a63c0a0277578891 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 19:25:07 -0700 Subject: [PATCH 01/10] feat(dsxe): add Mooncake B300 AgentX recipes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 新增基于 Mooncake 的 B300 DeepSeek V4 AgentX 配方,并适配 Python 启动器。 --- .../configs/install-mooncake-efa-cu13.sh | 5 + .../b300-fp4/agentx/agg-tp4-c4-mtp.yaml | 128 ++++++++++ .../b300-fp4/agentx/agg-tp8-c1-mtp.yaml | 128 ++++++++++ ...agg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 227 +++++++++++++++++ ...sagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml | 227 +++++++++++++++++ ...agg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 229 ++++++++++++++++++ inferencex-e2e/configs/nvidia-master.yaml | 98 ++++++++ inferencex-e2e/configs/runners.yaml | 1 + inferencex-e2e/perf-changelog.yaml | 14 ++ 9 files changed, 1057 insertions(+) create mode 100755 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh new file mode 100755 index 0000000000..252168b822 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/install-mooncake-efa-cu13.sh @@ -0,0 +1,5 @@ +#!/usr/bin/env bash +set -eo pipefail + +python3 -m pip uninstall --break-system-packages -y mooncake-transfer-engine-cuda13 mooncake-transfer-engine-efa-cuda13 +python3 -m pip install --break-system-packages --no-deps mooncake-transfer-engine-efa-cuda13==0.3.13.post1 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml new file mode 100644 index 0000000000..4a0de7e8b3 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml @@ -0,0 +1,128 @@ +schema: 2 +name: "agg-b300-tp4-c4-mtp" + +# AgentX aggregate topology: one TP4 worker occupies half of one eight-GPU +# B300 node and serves both prefill and decode with DSpark K=6. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: "b300" + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: "1" + SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" + SGLANG_OPT_USE_ONLINE_COMPRESS: "0" + SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" + SGLANG_OPT_USE_JIT_NORM: "1" + SGLANG_OPT_USE_TOPK_V2: "True" + + args: + served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.90 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: "flashinfer_mxfp4" + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 8 + cuda-graph-max-bs-decode: 8 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 4 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + IS_MULTINODE: "false" + TP: "4" + PP_SIZE: "1" + PCP_SIZE: "1" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml new file mode 100644 index 0000000000..a7383fcac7 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml @@ -0,0 +1,128 @@ +schema: 2 +name: "agg-b300-tp8-c1-mtp" + +# AgentX aggregate topology: one TP8 worker occupies one eight-GPU B300 node +# and serves both prefill and decode with DSpark K=6. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: "b300" + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: "1" + SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" + SGLANG_OPT_USE_ONLINE_COMPRESS: "0" + SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" + SGLANG_OPT_USE_JIT_NORM: "1" + SGLANG_OPT_USE_TOPK_V2: "True" + + args: + served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.90 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: "flashinfer_mxfp4" + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 2 + cuda-graph-max-bs-decode: 2 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 8 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + IS_MULTINODE: "false" + TP: "8" + PP_SIZE: "1" + PCP_SIZE: "1" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml new file mode 100644 index 0000000000..dc11039597 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -0,0 +1,227 @@ +schema: 2 +name: "disagg-b300-1p1d-dep8-dep8-c240-mtp-kvoffload" + +# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 on B300 +# (1P x DEP8 / 1D x DEP8, DSpark K=6 + hierarchical-cache KV offload), +# tuned for concurrency 240. Concurrency is exported by the master config. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +setup_script: install-mooncake-efa-cu13.sh + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: b300 + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_TCP_REQUEST_TIMEOUT: "60" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs-prefill: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 3 + hicache-io-backend: direct + + decode: + nodes: 1 + workers: 1 + gpus: 8 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 480 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "true" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml new file mode 100644 index 0000000000..17d83b38ce --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml @@ -0,0 +1,227 @@ +schema: 2 +name: "disagg-b300-1p1d-dep8-dep8-c64-mtp-kvoffload" + +# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 on B300 +# (1P x DEP8 / 1D x DEP8, DSpark K=6 + hierarchical-cache KV offload), +# tuned for concurrency 64. Concurrency is exported by the master config. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +setup_script: install-mooncake-efa-cu13.sh + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: b300 + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_TCP_REQUEST_TIMEOUT: "60" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs-prefill: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 3 + hicache-io-backend: direct + + decode: + nodes: 1 + workers: 1 + gpus: 8 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 128 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "true" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml new file mode 100644 index 0000000000..0c5f4a59bc --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -0,0 +1,229 @@ +schema: 2 +name: "disagg-b300-2p1d-dep8-dep8-c480-mtp-kvoffload" + +# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 on B300 +# (2P x DEP8 / 1D x DEP8, DSpark K=6 + hierarchical-cache KV offload), +# tuned for concurrency 480. Concurrency is exported by the master config. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +setup_script: install-mooncake-efa-cu13.sh + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: b300 + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_TCP_REQUEST_TIMEOUT: "60" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + prefill: + nodes: 2 + workers: 2 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + OMP_NUM_THREADS: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + SGLANG_DSV4_MHC_PREWARM: '1' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs-prefill: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 3 + hicache-io-backend: direct + + decode: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + OMP_NUM_THREADS: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_DSV4_MHC_PREWARM: '1' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-w4a4-mxfp4-megamoe: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 960 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + + +sbatch_directives: + mem: "0" + cpus-per-task: "192" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "true" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 63dcf8fff8..c08e5cdd78 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1161,6 +1161,104 @@ dsv4-fp4-b300-sglang-agentic-hicache-eagle: - { tp: 8, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml } - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [64, 128, 256, 384, 512], router: { name: sglang-router, version: "0.3.2" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml } +dsv4-fp4-b300-dynamo-sglang-agentic-agg: + image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 + model: deepseek-ai/DeepSeek-V4-Pro-0813 + model-prefix: dsv4 + runner: cluster:b300-dsxe + precision: fp4 + framework: dynamo-sglang + router: { name: dynamo-router, version: "1.5.0.dev20260914" } + multinode: true + disagg: false + scenarios: + agentic-coding: + - search-space: + - spec-decoding: draft_model + conc-list: [1] + num-nodes: 1 + worker: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml" + - spec-decoding: draft_model + conc-list: [4] + num-nodes: 1 + worker: + num-worker: 1 + tp: 4 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml" + +dsv4-fp4-b300-dynamo-sglang-agentic-disagg: + image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 + model: deepseek-ai/DeepSeek-V4-Pro-0813 + model-prefix: dsv4 + runner: cluster:b300-dsxe + precision: fp4 + framework: dynamo-sglang + router: { name: dynamo-router, version: "1.5.0.dev20260914" } + kv-p2p-transfer: mooncake + multinode: true + disagg: true + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - spec-decoding: draft_model + conc-list: [64] + kv-offloading: dram + kv-offload-backend: { name: hicache } + prefill: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml" + decode: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + - spec-decoding: draft_model + conc-list: [240] + kv-offloading: dram + kv-offload-backend: { name: hicache } + prefill: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml" + decode: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + - spec-decoding: draft_model + conc-list: [480] + kv-offloading: dram + kv-offload-backend: { name: hicache } + prefill: + num-worker: 2 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml" + decode: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + qwen3.5-fp8-b200-sglang: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 model: Qwen/Qwen3.5-397B-A17B-FP8 diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index 8036d3de69..6c7e1381a0 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -578,6 +578,7 @@ clusters: cpus-per-gpu: 24 salloc-args: [--mem=0] volumes: + aiperf-cache: {path: /data/home/sa-gha-runner/aiperf-cache} hf-home: {path: ~/.cache/huggingface} hf-hub-cache: {path: ~/.cache/huggingface/hub} scratch: {path: /scratch/models, visibility: node-local} diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9b9c48a0ef..1c3435d45e 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9168,3 +9168,17 @@ - "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid." - "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605 + +- config-keys: + - dsv4-fp4-b300-dynamo-sglang-agentic-agg + - dsv4-fp4-b300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4 recipes, plus disaggregated 1P1D DEP8 c64/c240 and 2P1D DEP8 c480 recipes." + - "Use Mooncake for disaggregated KV transfer and install mooncake-transfer-engine-efa-cuda13 0.3.13.post1 during container setup; rely on the DSXE-managed fabric injection without explicit host EFA/OFI mounts." + - "Use 192 logical CPUs per task, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE node." + - "新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4,以及分离式 1P1D DEP8 c64/c240 和 2P1D DEP8 c480。" + - "分离式 KV 传输使用 Mooncake,并在容器初始化时安装 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 DSXE 管理的网络栈注入,不显式挂载主机 EFA/OFI 目录。" + - "每任务使用 192 个逻辑 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 From e202ea342864dd211d4e2525f57148aa39b63d11 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 20:55:51 -0700 Subject: [PATCH 02/10] fix(dsxe): use cluster CPU directives MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 移除配方中与集群级每 GPU CPU 配置冲突的任务级 CPU 参数。 --- .../dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml | 1 - .../dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml | 1 - .../agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 1 - .../agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml | 1 - .../agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 1 - inferencex-e2e/perf-changelog.yaml | 4 ++-- 6 files changed, 2 insertions(+), 7 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml index 4a0de7e8b3..a33048ebc5 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml @@ -103,7 +103,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml index a7383fcac7..335d1082a4 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml @@ -103,7 +103,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml index dc11039597..4a0db2f37b 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -205,7 +205,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml index 17d83b38ce..23f215cf11 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml @@ -205,7 +205,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml index 0c5f4a59bc..fc9ba21e43 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml @@ -207,7 +207,6 @@ roles: sbatch_directives: mem: "0" - cpus-per-task: "192" exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" srun_options: diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 1c3435d45e..ff532c6b87 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9177,8 +9177,8 @@ description: - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4 recipes, plus disaggregated 1P1D DEP8 c64/c240 and 2P1D DEP8 c480 recipes." - "Use Mooncake for disaggregated KV transfer and install mooncake-transfer-engine-efa-cuda13 0.3.13.post1 during container setup; rely on the DSXE-managed fabric injection without explicit host EFA/OFI mounts." - - "Use 192 logical CPUs per task, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE node." + - "Use the DSXE cluster's 24 CPUs per GPU, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE node." - "新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4,以及分离式 1P1D DEP8 c64/c240 和 2P1D DEP8 c480。" - "分离式 KV 传输使用 Mooncake,并在容器初始化时安装 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 DSXE 管理的网络栈注入,不显式挂载主机 EFA/OFI 目录。" - - "每任务使用 192 个逻辑 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" + - "使用 DSXE 集群的每 GPU 24 个 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 From c7a7ded46b68d52dafd9a67da195cc41d485dda0 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Wed, 30 Sep 2026 04:40:03 -0700 Subject: [PATCH 03/10] fix(dsxe): enable DeepGEMM fast warmup for B300 agg recipes Co-Authored-By: Claude Opus 5.5 --- .../dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml | 1 + .../dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml | 1 + 2 files changed, 2 insertions(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml index a33048ebc5..d4d6a4b6ec 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml @@ -64,6 +64,7 @@ roles: SGLANG_DSV4_REASONING_EFFORT: high PIP_BREAK_SYSTEM_PACKAGES: "1" SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml index 335d1082a4..a5f30a1bff 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml @@ -65,6 +65,7 @@ roles: SGLANG_DSV4_REASONING_EFFORT: high PIP_BREAK_SYSTEM_PACKAGES: "1" SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" From e511ba3c01575f89eabaf68cdc4f38f3003b7262 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Wed, 30 Sep 2026 09:07:37 -0700 Subject: [PATCH 04/10] fix(dsxe): read DSV4-Pro-0813 weights from node-local NVMe for B300 AgentX The pluggable launcher resolves DeepSeek-V4-Pro-0813 on b300-dsxe to the shared /data/models copy for every non-vLLM framework, including multi-node Dynamo+SGLang, which previously read the node-local /scratch/models copy. With several points loading concurrently from shared storage, the TP8 c1 agg worker spent ~2.9h in weight loading and lost its etcd lease; the TP4 c4 agg worker never became healthy within the 4h window on the previous head. Pin every point of both configs to /scratch/models/DeepSeek-V4-Pro-0813 via the MODEL_PATH additional-setting, restoring the storage path under which this recipe set last passed. Co-Authored-By: Claude Opus 5.5 --- inferencex-e2e/configs/nvidia-master.yaml | 5 +++++ inferencex-e2e/perf-changelog.yaml | 2 ++ 2 files changed, 7 insertions(+) diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index c08e5cdd78..088807abe2 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1184,6 +1184,7 @@ dsv4-fp4-b300-dynamo-sglang-agentic-agg: dp-attn: false additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml" + - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" - spec-decoding: draft_model conc-list: [4] num-nodes: 1 @@ -1194,6 +1195,7 @@ dsv4-fp4-b300-dynamo-sglang-agentic-agg: dp-attn: false additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml" + - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" dsv4-fp4-b300-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 @@ -1221,6 +1223,7 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: dp-attn: true additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml" + - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" decode: num-worker: 1 tp: 8 @@ -1237,6 +1240,7 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: dp-attn: true additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml" + - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" decode: num-worker: 1 tp: 8 @@ -1253,6 +1257,7 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: dp-attn: true additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml" + - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" decode: num-worker: 1 tp: 8 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index ff532c6b87..63f236375f 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9178,7 +9178,9 @@ - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4 recipes, plus disaggregated 1P1D DEP8 c64/c240 and 2P1D DEP8 c480 recipes." - "Use Mooncake for disaggregated KV transfer and install mooncake-transfer-engine-efa-cuda13 0.3.13.post1 during container setup; rely on the DSXE-managed fabric injection without explicit host EFA/OFI mounts." - "Use the DSXE cluster's 24 CPUs per GPU, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE node." + - "Serve weights from the node-local NVMe copy (/scratch/models/DeepSeek-V4-Pro-0813) via MODEL_PATH instead of the shared /data/models copy, which slowed startup by hours when several points loaded concurrently." - "新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4,以及分离式 1P1D DEP8 c64/c240 和 2P1D DEP8 c480。" - "分离式 KV 传输使用 Mooncake,并在容器初始化时安装 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 DSXE 管理的网络栈注入,不显式挂载主机 EFA/OFI 目录。" - "使用 DSXE 集群的每 GPU 24 个 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" + - "通过 MODEL_PATH 从节点本地 NVMe 副本(/scratch/models/DeepSeek-V4-Pro-0813)加载权重,而非共享的 /data/models 副本;后者在多个测试点并发加载时会使启动慢数小时。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 From a62f872509c2094571eb7ff40c1c60cac0e467b3 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Wed, 30 Sep 2026 09:26:16 -0700 Subject: [PATCH 05/10] fix(dsxe): route Dynamo+SGLang DSV4-Pro-0813 to the node-local checkpoint A point-level MODEL_PATH is opaque to the launcher, so srtctl's model preflight (srt-slurm v2.39.1) ran on the runner host and rejected the node-local /scratch/models path. Instead, add dynamo-sglang to the b300-dsxe override that already routes vLLM to DeepSeek-V4-Pro-0813@scratch: the checkpoint is then known to be node-local, model_paths maps the recipe alias to /scratch/models/DeepSeek-V4-Pro-0813, and preflight is skipped as it is for vLLM. Single-node SGLang keeps the shared /data/models copy. Drop the MODEL_PATH additional-settings added in the previous commit. Co-Authored-By: Claude Opus 5.5 --- inferencex-e2e/configs/nvidia-master.yaml | 5 ----- inferencex-e2e/infx/launch/drivers/srt/models.py | 2 +- inferencex-e2e/perf-changelog.yaml | 4 ++-- 3 files changed, 3 insertions(+), 8 deletions(-) diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 088807abe2..c08e5cdd78 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1184,7 +1184,6 @@ dsv4-fp4-b300-dynamo-sglang-agentic-agg: dp-attn: false additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp8-c1-mtp.yaml" - - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" - spec-decoding: draft_model conc-list: [4] num-nodes: 1 @@ -1195,7 +1194,6 @@ dsv4-fp4-b300-dynamo-sglang-agentic-agg: dp-attn: false additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml" - - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" dsv4-fp4-b300-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 @@ -1223,7 +1221,6 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: dp-attn: true additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml" - - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" decode: num-worker: 1 tp: 8 @@ -1240,7 +1237,6 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: dp-attn: true additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml" - - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" decode: num-worker: 1 tp: 8 @@ -1257,7 +1253,6 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: dp-attn: true additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml" - - "MODEL_PATH=/scratch/models/DeepSeek-V4-Pro-0813" decode: num-worker: 1 tp: 8 diff --git a/inferencex-e2e/infx/launch/drivers/srt/models.py b/inferencex-e2e/infx/launch/drivers/srt/models.py index bb7f701534..69e71da09b 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/models.py +++ b/inferencex-e2e/infx/launch/drivers/srt/models.py @@ -46,7 +46,7 @@ class Override: ), "b300-dsxe": ( Override( - Match(frameworks=any_of("vllm"), model_glob="*/DeepSeek-V4-Pro-0813"), + Match(frameworks=any_of("vllm", "dynamo-sglang"), model_glob="*/DeepSeek-V4-Pro-0813"), entry="DeepSeek-V4-Pro-0813@scratch", ), Override(Match(model_glob="*/DeepSeek-V4-Pro-0813"), entry="DeepSeek-V4-Pro-0813"), diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 63f236375f..93b29e536b 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9178,9 +9178,9 @@ - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4 recipes, plus disaggregated 1P1D DEP8 c64/c240 and 2P1D DEP8 c480 recipes." - "Use Mooncake for disaggregated KV transfer and install mooncake-transfer-engine-efa-cuda13 0.3.13.post1 during container setup; rely on the DSXE-managed fabric injection without explicit host EFA/OFI mounts." - "Use the DSXE cluster's 24 CPUs per GPU, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE node." - - "Serve weights from the node-local NVMe copy (/scratch/models/DeepSeek-V4-Pro-0813) via MODEL_PATH instead of the shared /data/models copy, which slowed startup by hours when several points loaded concurrently." + - "Serve B300 DSXE Dynamo+SGLang DeepSeek-V4-Pro-0813 weights from the node-local NVMe copy (DeepSeek-V4-Pro-0813@scratch, as for vLLM) instead of the shared /data/models copy, which slowed startup by hours when several points loaded concurrently." - "新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4,以及分离式 1P1D DEP8 c64/c240 和 2P1D DEP8 c480。" - "分离式 KV 传输使用 Mooncake,并在容器初始化时安装 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 DSXE 管理的网络栈注入,不显式挂载主机 EFA/OFI 目录。" - "使用 DSXE 集群的每 GPU 24 个 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" - - "通过 MODEL_PATH 从节点本地 NVMe 副本(/scratch/models/DeepSeek-V4-Pro-0813)加载权重,而非共享的 /data/models 副本;后者在多个测试点并发加载时会使启动慢数小时。" + - "B300 DSXE Dynamo+SGLang 的 DeepSeek-V4-Pro-0813 权重改为从节点本地 NVMe 副本(DeepSeek-V4-Pro-0813@scratch,与 vLLM 相同)加载,而非共享的 /data/models 副本;后者在多个测试点并发加载时会使启动慢数小时。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 From 27e204c5ca7caa1f754a59bb6302110b1d2c7898 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Thu, 1 Oct 2026 01:20:10 -0700 Subject: [PATCH 06/10] feat(dsxe): replace B300 2P1D with an aggregate DEP8 c384 recipe MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace the disaggregated 2P1D DEP8 c480 point with a one-node aggregate DEP8 c384 Dynamo+SGLang recipe that follows the single-node SGLang DEP8 c384 point: attention DP8 + EP8 Mega-MoE, FP4 indexer, prefill delayer (interval 20), chunked prefill 65536, max-running-requests 768, cuda-graph-max-bs-decode 544, mem-fraction-static 0.88, swa-full-tokens-ratio 0.075 and a HiCache ratio 3 write_back DRAM tier with the page_first_direct layout. The recipe uses this PR's lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 image and its arg names (tp-size/dp-size/ep-size, attention-backend dsv4 instead of the deprecated compressed alias, no deprecated SGLANG_ENABLE_UNIFIED_RADIX_TREE). The Dynamo frontend tokenizes and routes, so SGLang-server-only settings (tokenizer workers, parsers, chat template, server warmup, keep-alive) are dropped; per-DP-rank KV events feed the Dynamo KV router. Drop enable-w4a4-mxfp4-megamoe from every recipe, so Mega-MoE runs its default FP8xFP4 kernels. 将 B300 分离式 2P1D DEP8 c480 测试点替换为单节点聚合式 DEP8 c384 Dynamo+SGLang 配方,沿用单节点 SGLang DEP8 c384 测试点:注意力 DP8 + EP8 Mega-MoE、FP4 indexer、prefill delayer(间隔 20)、chunked prefill 65536、max-running-requests 768、cuda-graph-max-bs-decode 544、 mem-fraction-static 0.88、swa-full-tokens-ratio 0.075,以及 HiCache 比例 3、write_back、page_first_direct 布局的 DRAM 层。 配方使用本 PR 的 lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 镜像及其参数 命名(tp-size/dp-size/ep-size、以 attention-backend dsv4 替代已弃用的 compressed 别名、不再设置已弃用的 SGLANG_ENABLE_UNIFIED_RADIX_TREE)。由于 Dynamo 前端负责分词与路由,移除仅适用于 SGLang 服务端的设置(tokenizer worker、解析器、聊天模板、服务端预热、keep-alive);各 DP rank 的 KV 事件 供 Dynamo KV 路由器使用。 所有配方移除 enable-w4a4-mxfp4-megamoe,Mega-MoE 使用默认的 FP8xFP4 kernel。 Co-Authored-By: Claude Opus 5.5 --- .../agentx/agg-dep8-c384-mtp-kvoffload.yaml | 159 ++++++++++++ ...agg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml | 2 - ...sagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml | 2 - ...agg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml | 228 ------------------ inferencex-e2e/configs/nvidia-master.yaml | 30 ++- inferencex-e2e/perf-changelog.yaml | 10 +- 6 files changed, 180 insertions(+), 251 deletions(-) create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml delete mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml new file mode 100644 index 0000000000..023b54e889 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml @@ -0,0 +1,159 @@ +schema: 2 +name: "agg-b300-dep8-c384-mtp-kvoffload" + +# AgentX aggregate topology: one DEP8 worker (attention DP8 + EP8 Mega-MoE + +# FP4 indexer) occupies one eight-GPU B300 node and serves both prefill and +# decode with DSpark K=6 and a HiCache DRAM tier, tuned for concurrency 384. +# Engine settings follow the single-node SGLang DEP8 c384 point +# (benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-mtp/agentic.yaml, +# override_dep8_c384). The Dynamo KV router replaces the SGLang router: each +# DP rank publishes KV events, and sessions stay sticky via X-Dynamo-Session-ID. +# Concurrency is exported by the master config. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: "b300" + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: "1" + PYTHONNOUSERSITE: "1" + TORCH_CUDA_ARCH_LIST: "10.0" + # Triton compiles with the image's CUDA ptxas. + TRITON_PTXAS_PATH: /usr/local/cuda/bin/ptxas + SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" + SGLANG_OPT_USE_ONLINE_COMPRESS: "0" + SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" + SGLANG_OPT_USE_JIT_NORM: "1" + SGLANG_OPT_USE_TOPK_V2: "True" + # Covers the 8192-token per-rank prefill budget (65536 / 8 DP ranks). + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: "8320" + + args: + served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + watchdog-timeout: 1000000 + allow-auto-truncate: true + attention-backend: dsv4 + page-size: 256 + disable-shared-experts-fusion: true + disable-flashinfer-autotune: true + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + enable-dp-attention-local-control-broadcast: true + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + enable-prefill-delayer: true + prefill-decode-interval: 20 + incremental-streaming-output: true + stream-interval: 20 + # Mega-MoE's transient workspace sits outside the static pool. + mem-fraction-static: 0.88 + swa-full-tokens-ratio: 0.075 + chunked-prefill-size: 65536 + max-running-requests: 768 + cuda-graph-max-bs-decode: 544 + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + # HiCache capacity is a host/device ratio; 3 keeps the tier near 2 TB. + enable-hierarchical-cache: true + hicache-ratio: 3 + hicache-write-policy: write_back + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + +sbatch_directives: + mem: "0" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + IS_MULTINODE: "false" + TP: "8" + EP_SIZE: "8" + DP_ATTENTION: "true" + PP_SIZE: "1" + PCP_SIZE: "1" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + # Warmup can leave bytes unacknowledged past AIPerf's 30 s default. + AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml index 4a0db2f37b..58cdbe513b 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c240-mtp-kvoffload.yaml @@ -110,7 +110,6 @@ roles: moe-dense-tp-size: 1 moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true disaggregation-transfer-backend: mooncake disaggregation-mode: prefill load-balance-method: total_tokens @@ -184,7 +183,6 @@ roles: moe-dense-tp-size: 1 moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true disaggregation-transfer-backend: mooncake disaggregation-mode: decode load-balance-method: total_tokens diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml index 23f215cf11..667b6cf88f 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p1d-dep8-dep8-c64-mtp-kvoffload.yaml @@ -110,7 +110,6 @@ roles: moe-dense-tp-size: 1 moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true disaggregation-transfer-backend: mooncake disaggregation-mode: prefill load-balance-method: total_tokens @@ -184,7 +183,6 @@ roles: moe-dense-tp-size: 1 moe-a2a-backend: megamoe enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true disaggregation-transfer-backend: mooncake disaggregation-mode: decode load-balance-method: total_tokens diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml deleted file mode 100644 index fc9ba21e43..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml +++ /dev/null @@ -1,228 +0,0 @@ -schema: 2 -name: "disagg-b300-2p1d-dep8-dep8-c480-mtp-kvoffload" - -# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 on B300 -# (2P x DEP8 / 1D x DEP8, DSpark K=6 + hierarchical-cache KV offload), -# tuned for concurrency 480. Concurrency is exported by the master config. - -model: - path: "deepseek-v4-pro-0813" - container: "dynamo-sglang" - precision: "fp4" - -identity: - model: - repo: "deepseek-ai/DeepSeek-V4-Pro-0813" - container: - image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" - -dynamo: - install: false - -setup_script: install-mooncake-efa-cu13.sh - -health_check: - max_attempts: 1440 - interval_seconds: 10 - -resources: - gpu_type: b300 - gpus_per_node: 8 -services: - - name: etcd - type: etcd - placement: - node: dedicated - - name: nats - type: nats - placement: - node: dedicated - options: - max_payload_mb: 32 -frontend: - type: dynamo - nginx_session_affinity: true - nginx_session_affinity_header: X-Dynamo-Session-ID - enable_multiple_frontends: true - num_additional_frontends: 4 - env: - PIP_BREAK_SYSTEM_PACKAGES: "1" - DYN_TCP_REQUEST_TIMEOUT: "60" - args: - router-mode: "kv" - router-session-affinity-ttl-secs: "3600" - active-decode-blocks-threshold: "None" - active-prefill-tokens-threshold: "None" - active-prefill-tokens-threshold-frac: "None" - -engine: sglang -roles: - prefill: - nodes: 2 - workers: 2 - gpus: 8 - env: - SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - OMP_NUM_THREADS: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' - SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_CUMEM_ENABLE: '1' - MOONCAKE_PROTOCOL: efa - MC_MS_FILTERS: rdmap191s0 - WITH_NVIDIA_PEERMEM: '0' - NCCL_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' - SGLANG_DSV4_MHC_PREWARM: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: prefill - load-balance-method: total_tokens - mem-fraction-static: 0.85 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 256 - cuda-graph-max-bs-prefill: 256 - chunked-prefill-size: 65536 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - enable-hierarchical-cache: true - hicache-write-policy: write_back - hicache-ratio: 3 - hicache-io-backend: direct - - decode: - nodes: 1 - workers: 1 - gpus: 8 - env: - SGLANG_RAGGED_VERIFY_MODE: "static" - SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" - PIP_BREAK_SYSTEM_PACKAGES: '1' - PYTHONUNBUFFERED: '1' - OMP_NUM_THREADS: '1' - SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' - SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache - SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' - SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' - SGLANG_DEFAULT_THINKING: '1' - SGLANG_DSV4_REASONING_EFFORT: high - SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' - SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" - SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' - SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' - SGLANG_OPT_USE_ONLINE_COMPRESS: '0' - NCCL_CUMEM_ENABLE: '1' - MOONCAKE_PROTOCOL: efa - MC_MS_FILTERS: rdmap191s0 - WITH_NVIDIA_PEERMEM: '0' - NCCL_TIMEOUT: '100000' - SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' - SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' - DYN_SKIP_SGLANG_LOG_FORMATTING: '1' - DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn - SGLANG_LOG_FORWARD_ITERS: '1' - SGLANG_LOG_MS: '1' - SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' - SGLANG_DSV4_MHC_PREWARM: '1' - - args: - host: 0.0.0.0 - served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 - enable-metrics: true - enable-cache-report: true - model-path: /model/ - trust-remote-code: true - watchdog-timeout: 86400 - stream-interval: 60 - tp-size: 8 - dp-size: 8 - ep-size: 8 - enable-dp-attention: true - enable-dp-lm-head: true - moe-dense-tp-size: 1 - moe-a2a-backend: megamoe - enable-deepseek-v4-fp4-indexer: true - enable-w4a4-mxfp4-megamoe: true - disaggregation-transfer-backend: mooncake - disaggregation-mode: decode - load-balance-method: total_tokens - mem-fraction-static: 0.9 - page-size: 256 - swa-full-tokens-ratio: 0.02 - max-running-requests: 960 - cuda-graph-max-bs-decode: 256 - disable-flashinfer-autotune: true - model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' - kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' - speculative-algorithm: DSPARK - speculative-dspark-block-size: 6 - speculative-num-steps: 1 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 7 - - -sbatch_directives: - mem: "0" - exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" - -srun_options: - mem: "0" - container-remap-root: "" - -benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/srt_agentic.sh - env: - INFMAX_CONTAINER_WORKSPACE: /infmax-workspace - RESULT_DIR: /logs/agentic - PORT: "8000" - IS_MULTINODE: "true" - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" - AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" - AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" - HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index c08e5cdd78..c59c3663d6 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1194,6 +1194,20 @@ dsv4-fp4-b300-dynamo-sglang-agentic-agg: dp-attn: false additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml" + - dram-utilization: 0.80 + search-space: + - spec-decoding: draft_model + conc-list: [384] + kv-offloading: dram + kv-offload-backend: { name: hicache } + num-nodes: 1 + worker: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml" dsv4-fp4-b300-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-20260916-c9a8fba9 @@ -1242,22 +1256,6 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: tp: 8 ep: 8 dp-attn: true - - spec-decoding: draft_model - conc-list: [480] - kv-offloading: dram - kv-offload-backend: { name: hicache } - prefill: - num-worker: 2 - tp: 8 - ep: 8 - dp-attn: true - additional-settings: - - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-2p1d-dep8-dep8-c480-mtp-kvoffload.yaml" - decode: - num-worker: 1 - tp: 8 - ep: 8 - dp-attn: true qwen3.5-fp8-b200-sglang: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 93b29e536b..59b179913a 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9175,12 +9175,16 @@ scenario-type: - agentic-coding description: - - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1 and TP4 c4 recipes, plus disaggregated 1P1D DEP8 c64/c240 and 2P1D DEP8 c480 recipes." + - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1, TP4 c4 and DEP8 c384 recipes, plus disaggregated 1P1D DEP8 c64/c240 recipes." + - "The aggregate DEP8 c384 recipe follows the single-node SGLang DEP8 c384 point (attention DP8 + EP8 Mega-MoE, FP4 indexer, prefill delayer with interval 20, chunked prefill 65536, max-running-requests 768, mem-fraction-static 0.88, HiCache ratio 3 write_back DRAM tier) on the newer image, behind the Dynamo KV router with per-DP-rank KV events and session affinity." + - "Mega-MoE uses the default FP8xFP4 kernels in every DEP8 recipe; the W4A4 MXFP4 Mega-MoE path is not enabled." - "Use Mooncake for disaggregated KV transfer and install mooncake-transfer-engine-efa-cuda13 0.3.13.post1 during container setup; rely on the DSXE-managed fabric injection without explicit host EFA/OFI mounts." - "Use the DSXE cluster's 24 CPUs per GPU, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE node." - "Serve B300 DSXE Dynamo+SGLang DeepSeek-V4-Pro-0813 weights from the node-local NVMe copy (DeepSeek-V4-Pro-0813@scratch, as for vLLM) instead of the shared /data/models copy, which slowed startup by hours when several points loaded concurrently." - - "新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4,以及分离式 1P1D DEP8 c64/c240 和 2P1D DEP8 c480。" + - "新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4 和 DEP8 c384,以及分离式 1P1D DEP8 c64/c240。" + - "聚合式 DEP8 c384 配方沿用单节点 SGLang DEP8 c384 测试点(注意力 DP8 + EP8 Mega-MoE、FP4 indexer、间隔为 20 的 prefill delayer、chunked prefill 65536、max-running-requests 768、mem-fraction-static 0.88、HiCache 比例 3 的 write_back DRAM 层),改用更新的镜像,并由 Dynamo KV 路由器基于各 DP rank 的 KV 事件和会话亲和性进行路由。" + - "所有 DEP8 配方的 Mega-MoE 均使用默认的 FP8xFP4 kernel,不启用 W4A4 MXFP4 Mega-MoE 路径。" - "分离式 KV 传输使用 Mooncake,并在容器初始化时安装 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 DSXE 管理的网络栈注入,不显式挂载主机 EFA/OFI 目录。" - "使用 DSXE 集群的每 GPU 24 个 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" - "B300 DSXE Dynamo+SGLang 的 DeepSeek-V4-Pro-0813 权重改为从节点本地 NVMe 副本(DeepSeek-V4-Pro-0813@scratch,与 vLLM 相同)加载,而非共享的 /data/models 副本;后者在多个测试点并发加载时会使启动慢数小时。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3190 + pr-link: TBD From c34a19f256e1d20970f50bc8ce19cb68ca85705f Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Thu, 1 Oct 2026 01:21:48 -0700 Subject: [PATCH 07/10] chore(changelog): link the B300 Dynamo+SGLang AgentX entry to #3631 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 B300 Dynamo+SGLang AgentX 变更日志条目的 pr-link 指向 #3631。 Co-Authored-By: Claude Opus 5.5 --- inferencex-e2e/perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 59b179913a..9da71bfa38 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9187,4 +9187,4 @@ - "分离式 KV 传输使用 Mooncake,并在容器初始化时安装 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 DSXE 管理的网络栈注入,不显式挂载主机 EFA/OFI 目录。" - "使用 DSXE 集群的每 GPU 24 个 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" - "B300 DSXE Dynamo+SGLang 的 DeepSeek-V4-Pro-0813 权重改为从节点本地 NVMe 副本(DeepSeek-V4-Pro-0813@scratch,与 vLLM 相同)加载,而非共享的 /data/models 副本;后者在多个测试点并发加载时会使启动慢数小时。" - pr-link: TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3631 From 24c55ff7afa2f997850f60e5fd18d5b4ed20740e Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Thu, 1 Oct 2026 02:11:44 -0700 Subject: [PATCH 08/10] fix(dsxe): size the B300 agg DEP8 c384 decode graph to the per-rank cap MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Set cuda-graph-max-bs-decode to 96, the per-DP-rank request cap (max-running-requests 768 / dp-size 8). SGLang already clamps captured decode batch sizes to the per-rank request pool, so 544 never captured more than this; the explicit value states the intended limit. 将 B300 聚合式 DEP8 c384 配方的 cuda-graph-max-bs-decode 设为 96,即每个 DP rank 的请求上限(max-running-requests 768 / dp-size 8)。SGLang 本就会把 捕获的 decode batch size 限制在每个 rank 的请求池以内,544 实际上从未捕获 超过该值;显式设置可明确预期的上限。 Co-Authored-By: Claude Opus 5.5 --- .../sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml | 3 ++- inferencex-e2e/perf-changelog.yaml | 4 ++-- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml index 023b54e889..25a92d7fcf 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml @@ -115,7 +115,8 @@ roles: swa-full-tokens-ratio: 0.075 chunked-prefill-size: 65536 max-running-requests: 768 - cuda-graph-max-bs-decode: 544 + # Per-rank graph limit: max-running-requests / dp-size = 768 / 8. + cuda-graph-max-bs-decode: 96 kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' speculative-algorithm: DSPARK speculative-dspark-block-size: 6 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9da71bfa38..91b61cba84 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9176,13 +9176,13 @@ - agentic-coding description: - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1, TP4 c4 and DEP8 c384 recipes, plus disaggregated 1P1D DEP8 c64/c240 recipes." - - "The aggregate DEP8 c384 recipe follows the single-node SGLang DEP8 c384 point (attention DP8 + EP8 Mega-MoE, FP4 indexer, prefill delayer with interval 20, chunked prefill 65536, max-running-requests 768, mem-fraction-static 0.88, HiCache ratio 3 write_back DRAM tier) on the newer image, behind the Dynamo KV router with per-DP-rank KV events and session affinity." + - "The aggregate DEP8 c384 recipe follows the single-node SGLang DEP8 c384 point (attention DP8 + EP8 Mega-MoE, FP4 indexer, prefill delayer with interval 20, chunked prefill 65536, max-running-requests 768, mem-fraction-static 0.88, HiCache ratio 3 write_back DRAM tier) on the newer image, with cuda-graph-max-bs-decode set to the 96-request per-rank cap (768 / 8), behind the Dynamo KV router with per-DP-rank KV events and session affinity." - "Mega-MoE uses the default FP8xFP4 kernels in every DEP8 recipe; the W4A4 MXFP4 Mega-MoE path is not enabled." - "Use Mooncake for disaggregated KV transfer and install mooncake-transfer-engine-efa-cuda13 0.3.13.post1 during container setup; rely on the DSXE-managed fabric injection without explicit host EFA/OFI mounts." - "Use the DSXE cluster's 24 CPUs per GPU, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE node." - "Serve B300 DSXE Dynamo+SGLang DeepSeek-V4-Pro-0813 weights from the node-local NVMe copy (DeepSeek-V4-Pro-0813@scratch, as for vLLM) instead of the shared /data/models copy, which slowed startup by hours when several points loaded concurrently." - "新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4 和 DEP8 c384,以及分离式 1P1D DEP8 c64/c240。" - - "聚合式 DEP8 c384 配方沿用单节点 SGLang DEP8 c384 测试点(注意力 DP8 + EP8 Mega-MoE、FP4 indexer、间隔为 20 的 prefill delayer、chunked prefill 65536、max-running-requests 768、mem-fraction-static 0.88、HiCache 比例 3 的 write_back DRAM 层),改用更新的镜像,并由 Dynamo KV 路由器基于各 DP rank 的 KV 事件和会话亲和性进行路由。" + - "聚合式 DEP8 c384 配方沿用单节点 SGLang DEP8 c384 测试点(注意力 DP8 + EP8 Mega-MoE、FP4 indexer、间隔为 20 的 prefill delayer、chunked prefill 65536、max-running-requests 768、mem-fraction-static 0.88、HiCache 比例 3 的 write_back DRAM 层),改用更新的镜像,并将 cuda-graph-max-bs-decode 设为每个 rank 96 个请求的上限(768 / 8),并由 Dynamo KV 路由器基于各 DP rank 的 KV 事件和会话亲和性进行路由。" - "所有 DEP8 配方的 Mega-MoE 均使用默认的 FP8xFP4 kernel,不启用 W4A4 MXFP4 Mega-MoE 路径。" - "分离式 KV 传输使用 Mooncake,并在容器初始化时安装 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 DSXE 管理的网络栈注入,不显式挂载主机 EFA/OFI 目录。" - "使用 DSXE 集群的每 GPU 24 个 CPU、HiCache 比例 3,decode 的 max-running-requests 为并发数两倍,挂载持久化 AgentX/Hugging Face 缓存,并排除已知异常 DSXE 节点。" From ea31827ab6ae7d6e3300e7f05d933bade1374429 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Thu, 1 Oct 2026 02:47:22 -0700 Subject: [PATCH 09/10] fix(dsxe): raise the Dynamo TCP request timeout for B300 agg DEP8 c384 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The c384 warmup burst hit Dynamo's 5 s default request-plane ack timeout while the worker ingested the large AgentX requests; the frontend then marked the only worker unreachable and returned 500/503 for every warmup request. Set DYN_TCP_REQUEST_TIMEOUT to 60 s, as in the disaggregated recipes of this PR. c384 预热突发请求期间,worker 接收大量 AgentX 大请求时触发了 Dynamo 请求平面 默认 5 秒的确认超时;前端随后将唯一的 worker 标记为不可达,所有预热请求均返回 500/503。与本 PR 的分离式配方一致,将 DYN_TCP_REQUEST_TIMEOUT 设为 60 秒。 Co-Authored-By: Claude Opus 5.5 --- .../sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml index 25a92d7fcf..724333814f 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-dep8-c384-mtp-kvoffload.yaml @@ -50,6 +50,9 @@ frontend: env: PIP_BREAK_SYSTEM_PACKAGES: "1" DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + # The 5 s default request-plane ack timeout expires while the worker + # ingests the c384 warmup burst; match the disaggregated recipes. + DYN_TCP_REQUEST_TIMEOUT: "60" args: router-mode: "kv" router-session-affinity-ttl-secs: "3600" From 8a406d3d83f7855ce3d1cb05934f7e922f2a274b Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Thu, 1 Oct 2026 23:29:39 -0700 Subject: [PATCH 10/10] feat(dsxe): add B300 agg TP4 c8 and disagg 1P2D DEP8 c64 AgentX points MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fill the interactivity gap between agg TP4 c4 (~244 tok/s/user) and disagg 1P1D DEP8 c64 (~128 tok/s/user) with two points: - agg TP4 c8: the TP4 c4 recipe at max-running-requests 16 and cuda-graph-max-bs-decode 16, plus a HiCache DRAM tier (ratio 2.75, write_through, direct I/O, page_first_direct, as in the B200 TP8 AgentX recipe on the same image) because TP4 c4 already fills about 60% of the GPU KV pool. - disagg 1P2D DEP8 c64: the 1P1D c64 recipe with a second decode worker, which halves each decode rank's load. 在聚合式 TP4 c4(约 244 tok/s/user)与分离式 1P1D DEP8 c64(约 128 tok/s/user)之间增加两个测试点: - 聚合式 TP4 c8:基于 TP4 c4 配方,max-running-requests 16、 cuda-graph-max-bs-decode 16;由于 TP4 c4 已占用约 60% 的 GPU KV 池, 增加 HiCache DRAM 层(比例 2.75、write_through、direct I/O、 page_first_direct,与同一镜像上的 B200 TP8 AgentX 配方相同)。 - 分离式 1P2D DEP8 c64:即 1P1D c64 配方增加第二个 decode worker,使每个 decode rank 的负载减半。 Co-Authored-By: Claude Opus 5.5 --- .../agentx/agg-tp4-c8-mtp-kvoffload.yaml | 136 +++++++++++ ...sagg-1p2d-dep8-dep8-c64-mtp-kvoffload.yaml | 226 ++++++++++++++++++ inferencex-e2e/configs/nvidia-master.yaml | 28 +++ inferencex-e2e/perf-changelog.yaml | 6 +- 4 files changed, 394 insertions(+), 2 deletions(-) create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp-kvoffload.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p2d-dep8-dep8-c64-mtp-kvoffload.yaml diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp-kvoffload.yaml new file mode 100644 index 0000000000..60166007f8 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp-kvoffload.yaml @@ -0,0 +1,136 @@ +schema: 2 +name: "agg-b300-tp4-c8-mtp-kvoffload" + +# AgentX aggregate topology: one TP4 worker occupies half of one eight-GPU +# B300 node and serves both prefill and decode with DSpark K=6, tuned for +# concurrency 8. The TP4 c4 recipe already fills ~60% of the GPU KV pool, so +# this point adds a HiCache DRAM tier (settings from the B200 TP8 HiCache +# AgentX recipe on the same image). + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: "b300" + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV4_REASONING_EFFORT: high + PIP_BREAK_SYSTEM_PACKAGES: "1" + SGLANG_JIT_DEEPGEMM_PRECOMPILE: "1" + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: "1" + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: "1" + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: "1" + SGLANG_OPT_USE_ONLINE_COMPRESS: "0" + SGLANG_OPT_USE_JIT_INDEXER_METADATA: "1" + SGLANG_OPT_USE_JIT_NORM: "1" + SGLANG_OPT_USE_TOPK_V2: "True" + + args: + served-model-name: "deepseek-ai/DeepSeek-V4-Pro-0813" + enable-metrics: true + enable-cache-report: true + trust-remote-code: true + weight-loader-prefetch-checkpoints: true + stream-interval: 10 + watchdog-timeout: 1000000 + mem-fraction-static: 0.90 + page-size: 256 + chunked-prefill-size: 8192 + max-prefill-tokens: 8192 + moe-runner-backend: "flashinfer_mxfp4" + enable-deepseek-v4-fp4-indexer: true + disable-flashinfer-autotune: true + swa-full-tokens-ratio: 0.1 + max-running-requests: 16 + cuda-graph-max-bs-decode: 16 + scheduler-recv-interval: 30 + dp-size: 1 + tp-size: 4 + ep-size: 1 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-ratio: 2.75 + hicache-write-policy: write_through + hicache-io-backend: direct + hicache-mem-layout: page_first_direct + +sbatch_directives: + mem: "0" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + IS_MULTINODE: "false" + TP: "4" + PP_SIZE: "1" + PCP_SIZE: "1" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p2d-dep8-dep8-c64-mtp-kvoffload.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p2d-dep8-dep8-c64-mtp-kvoffload.yaml new file mode 100644 index 0000000000..16a3c42ad9 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p2d-dep8-dep8-c64-mtp-kvoffload.yaml @@ -0,0 +1,226 @@ +schema: 2 +name: "disagg-b300-1p2d-dep8-dep8-c64-mtp-kvoffload" + +# Agentic-coding SGLang disaggregated recipe for DeepSeek-V4-Pro-0813 on B300 +# (1P x DEP8 / 2D x DEP8, DSpark K=6 + hierarchical-cache KV offload), +# tuned for concurrency 64. Identical to the 1P1D c64 recipe except for a +# second decode worker, which halves each decode rank's load. Concurrency is +# exported by the master config. + +model: + path: "deepseek-v4-pro-0813" + container: "dynamo-sglang" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro-0813" + container: + image: "lmsysorg/sglang:nightly-dev-20260916-c9a8fba9" + +dynamo: + install: false + +setup_script: install-mooncake-efa-cu13.sh + +health_check: + max_attempts: 1440 + interval_seconds: 10 + +resources: + gpu_type: b300 + gpus_per_node: 8 +services: + - name: etcd + type: etcd + placement: + node: dedicated + - name: nats + type: nats + placement: + node: dedicated + options: + max_payload_mb: 32 +frontend: + type: dynamo + nginx_session_affinity: true + nginx_session_affinity_header: X-Dynamo-Session-ID + enable_multiple_frontends: true + num_additional_frontends: 4 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_TCP_REQUEST_TIMEOUT: "60" + args: + router-mode: "kv" + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + +engine: sglang +roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_DSV4_MHC_PREWARM: '1' + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1' + SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '9216' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + SGLANG_NCCL_ALL_GATHER_IN_OVERLAP_SCHEDULER_SYNC_BATCH: '1' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: prefill + load-balance-method: total_tokens + mem-fraction-static: 0.85 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 256 + cuda-graph-max-bs-prefill: 256 + chunked-prefill-size: 65536 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5557"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + enable-hierarchical-cache: true + hicache-write-policy: write_back + hicache-ratio: 3 + hicache-io-backend: direct + + decode: + nodes: 2 + workers: 2 + gpus: 8 + + env: + SGLANG_RAGGED_VERIFY_MODE: "static" + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + PIP_BREAK_SYSTEM_PACKAGES: '1' + PYTHONUNBUFFERED: '1' + SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: '0' + SGLANG_DG_CACHE_DIR: /configs/deepgemm_cache + SGLANG_JIT_DEEPGEMM_PRECOMPILE: '1' + SGLANG_JIT_DEEPGEMM_FAST_WARMUP: '1' + SGLANG_DEFAULT_THINKING: '1' + SGLANG_DSV4_REASONING_EFFORT: high + SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT: '1' + SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS: "1" + SGLANG_OPT_DEEPGEMM_MEGA_MOE: '1' + SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK: '4096' + SGLANG_OPT_USE_ONLINE_COMPRESS: '0' + NCCL_CUMEM_ENABLE: '1' + MOONCAKE_PROTOCOL: efa + MC_MS_FILTERS: rdmap191s0 + WITH_NVIDIA_PEERMEM: '0' + NCCL_TIMEOUT: '100000' + SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW: '1' + SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8' + DYN_SKIP_SGLANG_LOG_FORMATTING: '1' + DYN_LOG: info,dynamo_runtime::pipeline::network::ingress::push_handler=warn + SGLANG_LOG_FORWARD_ITERS: '1' + SGLANG_LOG_MS: '1' + SGLANG_REQUEST_STATE_WAIT_TIMEOUT: '60' + + args: + host: 0.0.0.0 + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + enable-metrics: true + enable-cache-report: true + model-path: /model/ + trust-remote-code: true + watchdog-timeout: 86400 + stream-interval: 60 + tp-size: 8 + dp-size: 8 + ep-size: 8 + enable-dp-attention: true + enable-dp-lm-head: true + moe-dense-tp-size: 1 + moe-a2a-backend: megamoe + enable-deepseek-v4-fp4-indexer: true + disaggregation-transfer-backend: mooncake + disaggregation-mode: decode + load-balance-method: total_tokens + mem-fraction-static: 0.9 + page-size: 256 + swa-full-tokens-ratio: 0.02 + max-running-requests: 128 + cuda-graph-max-bs-decode: 256 + disable-flashinfer-autotune: true + model-loader-extra-config: '{"enable_multithread_load":true,"num_threads":8}' + kv-events-config: '{"publisher":"zmq","topic":"kv-events","endpoint":"tcp://*:5567"}' + speculative-algorithm: DSPARK + speculative-dspark-block-size: 6 + speculative-num-steps: 1 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 7 + + +sbatch_directives: + mem: "0" + exclude: "dsxe-sa-b300-prd0-gpu-00,dsxe-sa-b300-prd0-gpu-01,dsxe-sa-b300-prd0-gpu-04,dsxe-sa-b300-prd0-gpu-06,dsxe-sa-b300-prd0-gpu-11,dsxe-sa-b300-prd0-gpu-16" + +srun_options: + mem: "0" + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "true" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index c59c3663d6..51399e783b 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1196,6 +1196,18 @@ dsv4-fp4-b300-dynamo-sglang-agentic-agg: - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c4-mtp.yaml" - dram-utilization: 0.80 search-space: + - spec-decoding: draft_model + conc-list: [8] + kv-offloading: dram + kv-offload-backend: { name: hicache } + num-nodes: 1 + worker: + num-worker: 1 + tp: 4 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/agg-tp4-c8-mtp-kvoffload.yaml" - spec-decoding: draft_model conc-list: [384] kv-offloading: dram @@ -1240,6 +1252,22 @@ dsv4-fp4-b300-dynamo-sglang-agentic-disagg: tp: 8 ep: 8 dp-attn: true + - spec-decoding: draft_model + conc-list: [64] + kv-offloading: dram + kv-offload-backend: { name: hicache } + prefill: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + additional-settings: + - "CONFIG_FILE=recipes/dsv4/sglang/b300-fp4/agentx/disagg-1p2d-dep8-dep8-c64-mtp-kvoffload.yaml" + decode: + num-worker: 2 + tp: 8 + ep: 8 + dp-attn: true - spec-decoding: draft_model conc-list: [240] kv-offloading: dram diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 91b61cba84..8b54bd06e2 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9175,13 +9175,15 @@ scenario-type: - agentic-coding description: - - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1, TP4 c4 and DEP8 c384 recipes, plus disaggregated 1P1D DEP8 c64/c240 recipes." + - "Add DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX aggregate TP8 c1, TP4 c4, TP4 c8 and DEP8 c384 recipes, plus disaggregated 1P1D DEP8 c64/c240 and 1P2D DEP8 c64 recipes." + - "TP4 c8 adds a HiCache DRAM tier (ratio 2.75, write_through, direct I/O, page_first_direct, as in the B200 TP8 AgentX recipe on the same image) because TP4 c4 already fills about 60% of the GPU KV pool. 1P2D DEP8 c64 is the 1P1D c64 recipe with a second decode worker, which halves each decode rank's load." - "The aggregate DEP8 c384 recipe follows the single-node SGLang DEP8 c384 point (attention DP8 + EP8 Mega-MoE, FP4 indexer, prefill delayer with interval 20, chunked prefill 65536, max-running-requests 768, mem-fraction-static 0.88, HiCache ratio 3 write_back DRAM tier) on the newer image, with cuda-graph-max-bs-decode set to the 96-request per-rank cap (768 / 8), behind the Dynamo KV router with per-DP-rank KV events and session affinity." - "Mega-MoE uses the default FP8xFP4 kernels in every DEP8 recipe; the W4A4 MXFP4 Mega-MoE path is not enabled." - "Use Mooncake for disaggregated KV transfer and install mooncake-transfer-engine-efa-cuda13 0.3.13.post1 during container setup; rely on the DSXE-managed fabric injection without explicit host EFA/OFI mounts." - "Use the DSXE cluster's 24 CPUs per GPU, HiCache ratio 3, decode max-running-requests at twice concurrency, persistent AgentX/Hugging Face caches, and exclude the known degraded DSXE node." - "Serve B300 DSXE Dynamo+SGLang DeepSeek-V4-Pro-0813 weights from the node-local NVMe copy (DeepSeek-V4-Pro-0813@scratch, as for vLLM) instead of the shared /data/models copy, which slowed startup by hours when several points loaded concurrently." - - "新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4 和 DEP8 c384,以及分离式 1P1D DEP8 c64/c240。" + - "新增 DeepSeek-V4-Pro-0813 FP4 B300 Dynamo+SGLang AgentX 配方:聚合式 TP8 c1、TP4 c4、TP4 c8 和 DEP8 c384,以及分离式 1P1D DEP8 c64/c240 和 1P2D DEP8 c64。" + - "由于 TP4 c4 已占用约 60% 的 GPU KV 池,TP4 c8 增加 HiCache DRAM 层(比例 2.75、write_through、direct I/O、page_first_direct,与同一镜像上的 B200 TP8 AgentX 配方相同)。1P2D DEP8 c64 即 1P1D c64 配方增加第二个 decode worker,使每个 decode rank 的负载减半。" - "聚合式 DEP8 c384 配方沿用单节点 SGLang DEP8 c384 测试点(注意力 DP8 + EP8 Mega-MoE、FP4 indexer、间隔为 20 的 prefill delayer、chunked prefill 65536、max-running-requests 768、mem-fraction-static 0.88、HiCache 比例 3 的 write_back DRAM 层),改用更新的镜像,并将 cuda-graph-max-bs-decode 设为每个 rank 96 个请求的上限(768 / 8),并由 Dynamo KV 路由器基于各 DP rank 的 KV 事件和会话亲和性进行路由。" - "所有 DEP8 配方的 Mega-MoE 均使用默认的 FP8xFP4 kernel,不启用 W4A4 MXFP4 Mega-MoE 路径。" - "分离式 KV 传输使用 Mooncake,并在容器初始化时安装 mooncake-transfer-engine-efa-cuda13 0.3.13.post1;依赖 DSXE 管理的网络栈注入,不显式挂载主机 EFA/OFI 目录。"