diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml index e002f52cf1..73259bf231 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml @@ -5,7 +5,7 @@ base: name: dsv41flash-fp4-gb300-sglang-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - container: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:0e1b14e302619a42ef5581b87db6804651d4f946301cf543d74a3a1eb1c33b40 + container: lmsysorg/sglang:dev-cu13-nightly-0924@sha256:d2c9929f8cc9889326203f6cbc46da48b69de42891b167592c9c76cc803d246e precision: fp4 resources: gpu_type: gb300 @@ -52,6 +52,9 @@ base: health_check: max_attempts: 1440 interval_seconds: 5 + # Temporary: remove once im-gb300-r01-c001's corrupt container cache is fixed. + sbatch_directives: + exclude: im-gb300-r01-c001 benchmark: type: custom command: bash /infmax-workspace/benchmarks/srt_agentic.sh @@ -63,8 +66,11 @@ base: # One variant per point. Admission is 2x CONC, capped at the graph batch. SWA # prefix tails are 64x CONC up to 4096 from c2. TP2 keeps row-sharded Engram -# tables in host DRAM (TP4 fits them in HBM) and doubles the prefill chunk from -# c64. Saturation points get a longer warmup drain. +# tables in host DRAM and doubles the prefill chunk from c64. TP4 uses static +# ragged verify; pure TP4 c1/c2 keep Engram in HBM, while TP4/EP4 from c4 keeps +# it in host DRAM with 16K chunks, no prefill-decode interval, 4096 SWA tails +# and a 128 graph batch, as matched one-hour c64 runs favored. Saturation +# points get a longer warmup drain. override_tp2_c1: roles: agg: @@ -214,11 +220,12 @@ override_tp4_c1: gpus: 4 args: tensor-parallel-size: 4 - expert-parallel-size: 4 + expert-parallel-size: 1 chunked-prefill-size: 4096 max-running-requests: 2 env: SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0' + SGLANG_RAGGED_VERIFY_MODE: static benchmark: env: CONC: '1' @@ -230,12 +237,13 @@ override_tp4_c2: gpus: 4 args: tensor-parallel-size: 4 - expert-parallel-size: 4 + expert-parallel-size: 1 chunked-prefill-size: 4096 swa-prefix-tails: 128 max-running-requests: 4 env: SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0' + SGLANG_RAGGED_VERIFY_MODE: static benchmark: env: CONC: '2' @@ -248,11 +256,15 @@ override_tp4_c4: args: tensor-parallel-size: 4 expert-parallel-size: 4 - chunked-prefill-size: 4096 - swa-prefix-tails: 256 - max-running-requests: 8 + chunked-prefill-size: 16384 + prefill-decode-interval: 0 + swa-prefix-tails: 4096 + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 env: - SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0' + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1' + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_RAGGED_VERIFY_MODE: static benchmark: env: CONC: '4' @@ -265,11 +277,15 @@ override_tp4_c8: args: tensor-parallel-size: 4 expert-parallel-size: 4 - chunked-prefill-size: 4096 - swa-prefix-tails: 512 - max-running-requests: 16 + chunked-prefill-size: 16384 + prefill-decode-interval: 0 + swa-prefix-tails: 4096 + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 env: - SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0' + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1' + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_RAGGED_VERIFY_MODE: static benchmark: env: CONC: '8' @@ -282,11 +298,15 @@ override_tp4_c16: args: tensor-parallel-size: 4 expert-parallel-size: 4 - chunked-prefill-size: 4096 - swa-prefix-tails: 1024 - max-running-requests: 32 + chunked-prefill-size: 16384 + prefill-decode-interval: 0 + swa-prefix-tails: 4096 + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 env: - SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0' + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1' + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_RAGGED_VERIFY_MODE: static benchmark: env: CONC: '16' @@ -299,11 +319,15 @@ override_tp4_c32: args: tensor-parallel-size: 4 expert-parallel-size: 4 - chunked-prefill-size: 4096 - swa-prefix-tails: 2048 - max-running-requests: 64 + chunked-prefill-size: 16384 + prefill-decode-interval: 0 + swa-prefix-tails: 4096 + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 env: - SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0' + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1' + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_RAGGED_VERIFY_MODE: static benchmark: env: CONC: '32' @@ -316,11 +340,15 @@ override_tp4_c64: args: tensor-parallel-size: 4 expert-parallel-size: 4 - chunked-prefill-size: 4096 + chunked-prefill-size: 16384 + prefill-decode-interval: 0 swa-prefix-tails: 4096 - max-running-requests: 64 + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 env: - SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0' + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1' + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_RAGGED_VERIFY_MODE: static benchmark: env: CONC: '64' @@ -333,11 +361,15 @@ override_tp4_c128: args: tensor-parallel-size: 4 expert-parallel-size: 4 - chunked-prefill-size: 4096 + chunked-prefill-size: 16384 + prefill-decode-interval: 0 swa-prefix-tails: 4096 - max-running-requests: 64 + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 env: - SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '0' + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: '1' + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_RAGGED_VERIFY_MODE: static benchmark: env: CONC: '128' diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index a606a90555..7d256bd73b 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8925,9 +8925,9 @@ dsv41flash-fp4-b300-sglang-agentic-dspark: - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/agentic.yaml } - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b300-fp4-mtp/agentic.yaml } -# Official SGLang nightly; Engram weights in host DRAM, native DSpark draft. +# Official SGLang nightly; TP4 C1/C2 use GPU Engram, other points use host DRAM. dsv41flash-fp4-gb300-sglang-agentic-dspark: - image: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:0e1b14e302619a42ef5581b87db6804651d4f946301cf543d74a3a1eb1c33b40 + image: lmsysorg/sglang:dev-cu13-nightly-0924@sha256:d2c9929f8cc9889326203f6cbc46da48b69de42891b167592c9c76cc803d246e model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:gb300-nv @@ -8939,4 +8939,5 @@ dsv41flash-fp4-gb300-sglang-agentic-dspark: - dram-utilization: 0.80 search-space: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } + - { tp: 4, ep: 1, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } + - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 50986d4c72..701b54cef1 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -435,6 +435,10 @@ Source: [upstream recipe](https://github.com/vllm-project/recipes/blob/main/mode ### DeepSeek-V4.1-Flash DSpark on SGLang +The GB300 DeepSeek-V4.1-Flash SGLang curve uses `dev-cu13-nightly-0924`: +TP4/EP1 at C1/C2 with GPU Engram and 4K prefill chunks, and TP4/EP4 at C4+ +with host Engram and 16K chunks. TP2 is unchanged. Full sweep validation is pending. + The H100 SGLang candidate sweeps DSpark at concurrency 1/2/4/8/16/20. It retains 8 SWA prefix tails per concurrency at C1/C2 and 32 at C4 and above. A matched one-hour comparison rejected a blanket 128-tail floor: C2 throughput improved only 1.7% while interactivity fell 44.5%. Completed STP comparisons did not contribute a measured frontier point, so STP is excluded from the selected sweep. The recipe interleaves 16 decode steps between prefill chunks, preserving trace content and context limits. The same sweep also qualifies supported TP8/EP8/DP8 attention at C4/C8/C16/C20. DP uses a stock consistent-hash router with stable session keys, DP LM-head execution, and 64 SWA prefix tails per rank. Full C16 GSM8K passed on all 1,319 examples; its performance contribution remains under measurement. The native 1M context and the AgentX subagent/session semantics are preserved. @@ -444,8 +448,8 @@ The nightly candidate uses `nightly-dev-cu13-20260922-582389ce`, native MXFP4 Ma `dsv41flash-fp4--sglang-agentic-dspark` are the SGLang counterparts of the vLLM arms, one PR per SKU across h100, h200, b200, b300, gb200, gb300 and mi355x. They follow the [SGLang cookbook](https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4_1), -which has no released SGLang version for this model yet. B200, B300, GB300 and H100 pin the CUDA 13 nightly -`lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce` by digest; GB200 and H200 use +which has no released SGLang version for this model yet. B200, B300 and H100 pin the CUDA 13 nightly +`lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce` by digest; GB300 pins `lmsysorg/sglang:dev-cu13-nightly-0924` by digest; GB200 and H200 use `lmsysorg/sglang:nightly-dev-cu13-20260923-06008c17` (GB200 by digest), and MI355X pins `lmsysorg/sglang:dev-dsv41-mi35x` by digest. Each master entry's `image` is authoritative. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 089763c4e0..bc29939872 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -356,6 +356,10 @@ Maximum concurrency for 1,048,576 tokens per request: 6.70x ### SGLang 上的 DeepSeek-V4.1-Flash DSpark +GB300 DeepSeek-V4.1-Flash SGLang 曲线使用 `dev-cu13-nightly-0924`: +C1/C2 使用 TP4/EP1、GPU Engram 和 4K prefill chunk;C4+ 使用 TP4/EP4、 +主机 Engram 和 16K chunk。TP2 不变,仍待全量 sweep 验证。 + H100 SGLang 候选配方在并发 1/2/4/8/16/20 下测试 DSpark。C1/C2 按并发数的 8 倍保留 SWA 前缀尾部,C4 及以上按 32 倍保留。相同条件下的一小时对比否决了统一的 128 尾部下限:C2 吞吐量仅提高 1.7%,交互性能却下降 44.5%。已完成的 STP 对比没有贡献实测性能前沿点,因此所选 sweep 不包含 STP。配方在预填充分块之间插入 16 步解码,轨迹内容和上下文限制保持不变。 同一 sweep 还会在 C4/C8/C16/C20 下验证受支持的 TP8/EP8/DP8 attention。DP 使用原生一致性哈希路由器与稳定会话键、DP LM-head,以及每 rank 64 个 SWA 前缀尾部。C16 的完整 GSM8K 已通过全部 1,319 个样本;其性能贡献仍在测量中。原生 1M 上下文与 AgentX 子代理/会话语义保持不变。 @@ -366,8 +370,8 @@ nightly 候选配方使用 `nightly-dev-cu13-20260922-582389ce`、原生 MXFP4 M `dsv41flash-fp4--sglang-agentic-dspark` 是 vLLM 配方在 h100、h200、b200、b300、gb200、gb300 与 mi355x 上的 SGLang 对应版本(每个 SKU 一个 PR),遵循 [SGLang cookbook](https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4_1)。 -该模型尚无正式发布的 SGLang 版本。B200、B300、GB300 与 H100 通过 digest 固定 CUDA 13 nightly 镜像 -`lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce`;GB200 与 H200 使用 +该模型尚无正式发布的 SGLang 版本。B200、B300 与 H100 通过 digest 固定 CUDA 13 nightly 镜像 +`lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce`;GB300 通过 digest 固定 `lmsysorg/sglang:dev-cu13-nightly-0924`;GB200 与 H200 使用 `lmsysorg/sglang:nightly-dev-cu13-20260923-06008c17`(GB200 通过 digest 固定),MI355X 通过 digest 固定 `lmsysorg/sglang:dev-dsv41-mi35x`。以各主配置条目的 `image` 为准。 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 0040ed1701..00f1838d6b 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9045,3 +9045,11 @@ - "DEP8 uses the dsv4 attention backend, mem-fraction-static 0.84, prefill-decode-interval 32, mixed chunking, shortest-prefill-first scheduling, DP speculative prefill coordination and a router with the circuit breaker disabled, without the DP LM head." - "Runs on the native srt-slurm recipe benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b300-fp4-eagle/agentic.yaml." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3426 + +- config-keys: + - dsv41flash-fp4-gb300-sglang-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Update the GB300 DeepSeek-V4.1-Flash SGLang AgentX curve to lmsysorg/sglang:dev-cu13-nightly-0924 (digest pinned): pure TP4 (EP1) at C1/C2 with Engram in HBM, and TP4/EP4 at C4+ with per-rank host Engram, 16K prefill chunks, no prefill-decode interval, 4096 SWA prefix tails and a 128 decode graph batch; all TP4 points use static ragged verify; TP2 is unchanged." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3421 diff --git a/inferencex-e2e/runners/launch_gb300-nv.sh b/inferencex-e2e/runners/launch_gb300-nv.sh index a09a238e40..88509712c5 100644 --- a/inferencex-e2e/runners/launch_gb300-nv.sh +++ b/inferencex-e2e/runners/launch_gb300-nv.sh @@ -91,6 +91,15 @@ NGINX_SQUASH_FILE="/data/home/sa-shared/gharunners/squash/$(echo "$NGINX_IMAGE" # The login node is x86_64 and the compute nodes aarch64, so import on a compute node. import_squash() { local squash="$1" image="$2" + # Enroot uses the digest as the manifest tag, not Docker's @ syntax. + if [[ "$image" == *@sha256:* ]]; then + local image_digest="${image##*@}" + image="${image%@*}" + if [[ "${image##*/}" == *:* ]]; then + image="${image%:*}" + fi + image="${image}:${image_digest}" + fi local lock="${squash}.lock" srun --account="$SLURM_ACCOUNT" --partition="$SLURM_PARTITION" --exclusive --time=180 bash -c " exec 9>\"$lock\"