Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ base:
name: dsv41flash-fp4-gb300-sglang-agentic
model:
path: hf:deepseek-ai/DeepSeek-V4.1-Flash
container: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:0e1b14e302619a42ef5581b87db6804651d4f946301cf543d74a3a1eb1c33b40
container: lmsysorg/sglang:dev-cu13-nightly-0924@sha256:d2c9929f8cc9889326203f6cbc46da48b69de42891b167592c9c76cc803d246e
precision: fp4
resources:
gpu_type: gb300
Expand Down Expand Up @@ -52,6 +52,9 @@ base:
health_check:
max_attempts: 1440
interval_seconds: 5
# Temporary: remove once im-gb300-r01-c001's corrupt container cache is fixed.
sbatch_directives:
exclude: im-gb300-r01-c001
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
Expand All @@ -63,8 +66,11 @@ base:

# One variant per point. Admission is 2x CONC, capped at the graph batch. SWA
# prefix tails are 64x CONC up to 4096 from c2. TP2 keeps row-sharded Engram
# tables in host DRAM (TP4 fits them in HBM) and doubles the prefill chunk from
# c64. Saturation points get a longer warmup drain.
# tables in host DRAM and doubles the prefill chunk from c64. TP4 uses static
# ragged verify; pure TP4 c1/c2 keep Engram in HBM, while TP4/EP4 from c4 keeps
# it in host DRAM with 16K chunks, no prefill-decode interval, 4096 SWA tails
# and a 128 graph batch, as matched one-hour c64 runs favored. Saturation
# points get a longer warmup drain.
override_tp2_c1:
roles:
agg:
Expand Down Expand Up @@ -214,11 +220,12 @@ override_tp4_c1:
gpus: 4
args:
tensor-parallel-size: 4
expert-parallel-size: 4
expert-parallel-size: 1
chunked-prefill-size: 4096
max-running-requests: 2
env:
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0'
SGLANG_RAGGED_VERIFY_MODE: static
benchmark:
env:
CONC: '1'
Expand All @@ -230,12 +237,13 @@ override_tp4_c2:
gpus: 4
args:
tensor-parallel-size: 4
expert-parallel-size: 4
expert-parallel-size: 1
chunked-prefill-size: 4096
swa-prefix-tails: 128
max-running-requests: 4
env:
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0'
SGLANG_RAGGED_VERIFY_MODE: static
benchmark:
env:
CONC: '2'
Expand All @@ -248,11 +256,15 @@ override_tp4_c4:
args:
tensor-parallel-size: 4
expert-parallel-size: 4
chunked-prefill-size: 4096
swa-prefix-tails: 256
max-running-requests: 8
chunked-prefill-size: 16384
prefill-decode-interval: 0
swa-prefix-tails: 4096
max-running-requests: 128
cuda-graph-max-bs-decode: 128
env:
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0'
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1'
SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank
SGLANG_RAGGED_VERIFY_MODE: static
benchmark:
env:
CONC: '4'
Expand All @@ -265,11 +277,15 @@ override_tp4_c8:
args:
tensor-parallel-size: 4
expert-parallel-size: 4
chunked-prefill-size: 4096
swa-prefix-tails: 512
max-running-requests: 16
chunked-prefill-size: 16384
prefill-decode-interval: 0
swa-prefix-tails: 4096
max-running-requests: 128
cuda-graph-max-bs-decode: 128
env:
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0'
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1'
SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank
SGLANG_RAGGED_VERIFY_MODE: static
benchmark:
env:
CONC: '8'
Expand All @@ -282,11 +298,15 @@ override_tp4_c16:
args:
tensor-parallel-size: 4
expert-parallel-size: 4
chunked-prefill-size: 4096
swa-prefix-tails: 1024
max-running-requests: 32
chunked-prefill-size: 16384
prefill-decode-interval: 0
swa-prefix-tails: 4096
max-running-requests: 128
cuda-graph-max-bs-decode: 128
env:
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0'
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1'
SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank
SGLANG_RAGGED_VERIFY_MODE: static
benchmark:
env:
CONC: '16'
Expand All @@ -299,11 +319,15 @@ override_tp4_c32:
args:
tensor-parallel-size: 4
expert-parallel-size: 4
chunked-prefill-size: 4096
swa-prefix-tails: 2048
max-running-requests: 64
chunked-prefill-size: 16384
prefill-decode-interval: 0
swa-prefix-tails: 4096
max-running-requests: 128
cuda-graph-max-bs-decode: 128
env:
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0'
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1'
SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank
SGLANG_RAGGED_VERIFY_MODE: static
benchmark:
env:
CONC: '32'
Expand All @@ -316,11 +340,15 @@ override_tp4_c64:
args:
tensor-parallel-size: 4
expert-parallel-size: 4
chunked-prefill-size: 4096
chunked-prefill-size: 16384
prefill-decode-interval: 0
swa-prefix-tails: 4096
max-running-requests: 64
max-running-requests: 128
cuda-graph-max-bs-decode: 128
env:
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0'
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1'
SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank
SGLANG_RAGGED_VERIFY_MODE: static
benchmark:
env:
CONC: '64'
Expand All @@ -333,11 +361,15 @@ override_tp4_c128:
args:
tensor-parallel-size: 4
expert-parallel-size: 4
chunked-prefill-size: 4096
chunked-prefill-size: 16384
prefill-decode-interval: 0
swa-prefix-tails: 4096
max-running-requests: 64
max-running-requests: 128
cuda-graph-max-bs-decode: 128
env:
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0'
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1'
SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank
SGLANG_RAGGED_VERIFY_MODE: static
benchmark:
env:
CONC: '128'
Expand Down
7 changes: 4 additions & 3 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8925,9 +8925,9 @@ dsv41flash-fp4-b300-sglang-agentic-dspark:
- { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/agentic.yaml }
- { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/agentic.yaml }

# Official SGLang nightly; Engram weights in host DRAM, native DSpark draft.
# Official SGLang nightly; TP4 C1/C2 use GPU Engram, other points use host DRAM.
dsv41flash-fp4-gb300-sglang-agentic-dspark:
image: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:0e1b14e302619a42ef5581b87db6804651d4f946301cf543d74a3a1eb1c33b40
image: lmsysorg/sglang:dev-cu13-nightly-0924@sha256:d2c9929f8cc9889326203f6cbc46da48b69de42891b167592c9c76cc803d246e
model: deepseek-ai/DeepSeek-V4.1-Flash
model-prefix: dsv41flash
runner: cluster:gb300-nv
Expand All @@ -8939,4 +8939,5 @@ dsv41flash-fp4-gb300-sglang-agentic-dspark:
- dram-utilization: 0.80
search-space:
- { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml }
- { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml }
- { tp: 4, ep: 1, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml }
- { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml }
8 changes: 6 additions & 2 deletions inferencex-e2e/docs/configuration-procedures.md
Original file line number Diff line number Diff line change
Expand Up @@ -435,6 +435,10 @@ Source: [upstream recipe](https://github.com/vllm-project/recipes/blob/main/mode

### DeepSeek-V4.1-Flash DSpark on SGLang

The GB300 DeepSeek-V4.1-Flash SGLang curve uses `dev-cu13-nightly-0924`:
TP4/EP1 at C1/C2 with GPU Engram and 4K prefill chunks, and TP4/EP4 at C4+
with host Engram and 16K chunks. TP2 is unchanged. Full sweep validation is pending.

The H100 SGLang candidate sweeps DSpark at concurrency 1/2/4/8/16/20. It retains 8 SWA prefix tails per concurrency at C1/C2 and 32 at C4 and above. A matched one-hour comparison rejected a blanket 128-tail floor: C2 throughput improved only 1.7% while interactivity fell 44.5%. Completed STP comparisons did not contribute a measured frontier point, so STP is excluded from the selected sweep. The recipe interleaves 16 decode steps between prefill chunks, preserving trace content and context limits.

The same sweep also qualifies supported TP8/EP8/DP8 attention at C4/C8/C16/C20. DP uses a stock consistent-hash router with stable session keys, DP LM-head execution, and 64 SWA prefix tails per rank. Full C16 GSM8K passed on all 1,319 examples; its performance contribution remains under measurement. The native 1M context and the AgentX subagent/session semantics are preserved.
Expand All @@ -444,8 +448,8 @@ The nightly candidate uses `nightly-dev-cu13-20260922-582389ce`, native MXFP4 Ma
`dsv41flash-fp4-<sku>-sglang-agentic-dspark` are the SGLang counterparts of the vLLM
arms, one PR per SKU across h100, h200, b200, b300, gb200, gb300 and mi355x. They follow the
[SGLang cookbook](https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4_1),
which has no released SGLang version for this model yet. B200, B300, GB300 and H100 pin the CUDA 13 nightly
`lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce` by digest; GB200 and H200 use
which has no released SGLang version for this model yet. B200, B300 and H100 pin the CUDA 13 nightly
`lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce` by digest; GB300 pins `lmsysorg/sglang:dev-cu13-nightly-0924` by digest; GB200 and H200 use
`lmsysorg/sglang:nightly-dev-cu13-20260923-06008c17` (GB200 by digest), and MI355X pins
`lmsysorg/sglang:dev-dsv41-mi35x` by digest. Each master entry's `image` is authoritative.

Expand Down
8 changes: 6 additions & 2 deletions inferencex-e2e/docs/configuration-procedures_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -356,6 +356,10 @@ Maximum concurrency for 1,048,576 tokens per request: 6.70x

### SGLang 上的 DeepSeek-V4.1-Flash DSpark

GB300 DeepSeek-V4.1-Flash SGLang 曲线使用 `dev-cu13-nightly-0924`:
C1/C2 使用 TP4/EP1、GPU Engram 和 4K prefill chunk;C4+ 使用 TP4/EP4、
主机 Engram 和 16K chunk。TP2 不变,仍待全量 sweep 验证。

H100 SGLang 候选配方在并发 1/2/4/8/16/20 下测试 DSpark。C1/C2 按并发数的 8 倍保留 SWA 前缀尾部,C4 及以上按 32 倍保留。相同条件下的一小时对比否决了统一的 128 尾部下限:C2 吞吐量仅提高 1.7%,交互性能却下降 44.5%。已完成的 STP 对比没有贡献实测性能前沿点,因此所选 sweep 不包含 STP。配方在预填充分块之间插入 16 步解码,轨迹内容和上下文限制保持不变。

同一 sweep 还会在 C4/C8/C16/C20 下验证受支持的 TP8/EP8/DP8 attention。DP 使用原生一致性哈希路由器与稳定会话键、DP LM-head,以及每 rank 64 个 SWA 前缀尾部。C16 的完整 GSM8K 已通过全部 1,319 个样本;其性能贡献仍在测量中。原生 1M 上下文与 AgentX 子代理/会话语义保持不变。
Expand All @@ -366,8 +370,8 @@ nightly 候选配方使用 `nightly-dev-cu13-20260922-582389ce`、原生 MXFP4 M
`dsv41flash-fp4-<sku>-sglang-agentic-dspark` 是 vLLM 配方在 h100、h200、b200、b300、gb200、gb300
与 mi355x 上的 SGLang 对应版本(每个 SKU 一个 PR),遵循
[SGLang cookbook](https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4_1)。
该模型尚无正式发布的 SGLang 版本。B200、B300、GB300 与 H100 通过 digest 固定 CUDA 13 nightly 镜像
`lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce`;GB200 与 H200 使用
该模型尚无正式发布的 SGLang 版本。B200、B300 与 H100 通过 digest 固定 CUDA 13 nightly 镜像
`lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce`;GB300 通过 digest 固定 `lmsysorg/sglang:dev-cu13-nightly-0924`;GB200 与 H200 使用
`lmsysorg/sglang:nightly-dev-cu13-20260923-06008c17`(GB200 通过 digest 固定),MI355X 通过 digest 固定
`lmsysorg/sglang:dev-dsv41-mi35x`。以各主配置条目的 `image` 为准。

Expand Down
8 changes: 8 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9045,3 +9045,11 @@
- "DEP8 uses the dsv4 attention backend, mem-fraction-static 0.84, prefill-decode-interval 32, mixed chunking, shortest-prefill-first scheduling, DP speculative prefill coordination and a router with the circuit breaker disabled, without the DP LM head."
- "Runs on the native srt-slurm recipe benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3426

- config-keys:
- dsv41flash-fp4-gb300-sglang-agentic-dspark
scenario-type:
- agentic-coding
description:
- "Update the GB300 DeepSeek-V4.1-Flash SGLang AgentX curve to lmsysorg/sglang:dev-cu13-nightly-0924 (digest pinned): pure TP4 (EP1) at C1/C2 with Engram in HBM, and TP4/EP4 at C4+ with per-rank host Engram, 16K prefill chunks, no prefill-decode interval, 4096 SWA prefix tails and a 128 decode graph batch; all TP4 points use static ragged verify; TP2 is unchanged."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3421
9 changes: 9 additions & 0 deletions inferencex-e2e/runners/launch_gb300-nv.sh
Original file line number Diff line number Diff line change
Expand Up @@ -91,6 +91,15 @@ NGINX_SQUASH_FILE="/data/home/sa-shared/gharunners/squash/$(echo "$NGINX_IMAGE"
# The login node is x86_64 and the compute nodes aarch64, so import on a compute node.
import_squash() {
local squash="$1" image="$2"
# Enroot uses the digest as the manifest tag, not Docker's @ syntax.
if [[ "$image" == *@sha256:* ]]; then
local image_digest="${image##*@}"
image="${image%@*}"
if [[ "${image##*/}" == *:* ]]; then
image="${image%:*}"
fi
image="${image}:${image_digest}"
fi
local lock="${squash}.lock"
srun --account="$SLURM_ACCOUNT" --partition="$SLURM_PARTITION" --exclusive --time=180 bash -c "
exec 9>\"$lock\"
Expand Down
Loading