From 69acc81f12032a1206007947da80a5e2508e54ca Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 13:49:14 -0500 Subject: [PATCH 01/18] feat(b300): add DeepSeek V4.1 SGLang nightly AgentX recipe --- .../agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 134 ++++++++++++++++++ configs/nvidia-master.yaml | 17 +++ perf-changelog.yaml | 9 ++ runners/launch_b300-dsxe.sh | 2 +- 4 files changed, 161 insertions(+), 1 deletion(-) create mode 100644 benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh new file mode 100644 index 0000000000..44665f8cd9 --- /dev/null +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -0,0 +1,134 @@ +#!/usr/bin/env bash +set -eo pipefail + +# DeepSeek-V4.1-Flash AgentX on B300 with SGLang native DSpark, following the +# cookbook's verified Blackwell TP4/EP4 low-latency cell. The KV cache is GPU-resident. +# https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4_1 +source "$(dirname "$0")/../../benchmark_lib.sh" +check_env_vars MODEL TP EP_SIZE CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION +check_env_vars EVAL_ONLY +require_agentic_kv_offload_none +export GPU_COUNT="$TP" + +if [[ -n "${SLURM_JOB_ID:-}" ]]; then + echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}" +fi + +# Complete/resume partial downloads instead of trusting nonempty directories. +if [[ -n "${MODEL_PATH:-}" && "$MODEL_PATH" != "$MODEL" ]]; then + hf download "$MODEL" --local-dir "$MODEL_PATH" +else + hf download "$MODEL" + export MODEL_PATH="$MODEL" +fi + +nvidia-smi +resolve_trace_source +install_agentic_deps +mkdir -p "$RESULT_DIR" +SERVER_LOG="$RESULT_DIR/server.log" +export PYTHONNOUSERSITE=1 +export PYTHONUNBUFFERED=1 + +# Agentic warmup dispatches hundreds of large prompts at once and SGLang's +# tokenizer can leave bytes unacknowledged past AIPerf's default 30 s +# TCP_USER_TIMEOUT, so Linux aborts live localhost connections. +export AIPERF_HTTP_TCP_USER_TIMEOUT=900000 +# Outlast AIPerf's pooled connections so an inter-turn idle gap cannot race +# Uvicorn's five-second keep-alive closure. +export SGLANG_TIMEOUT_KEEP_ALIVE=900 + +# AgentX measures the thinking-on regime, which is also the committed golden-AL +# curve. SGLang ships thinking off by default for this model. +export SGLANG_DEFAULT_THINKING=1 +export SGLANG_DSV41_REASONING_EFFORT=high + +# One shared host copy of the two fp8 Engram tables instead of a row-sharded +# copy per rank: the SGLang analogue of the vLLM arm's Engram CPU offload. It +# frees ~46 GiB of HBM per GPU for the 1M-context prefill working set and the +# KV pool, and output is bitwise unchanged (cookbook). The first sweep ran +# with the tables on GPU and the server died on the first long AgentX prompts +# (run 35304517453: c4 came up, then the server exited on the first two warmup prompts). +export SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1 + +# AgentX concurrency counts live session trees, not individual requests. +# Allow subagent fan-out to exceed CONC without clipping request bursts, but +# never let the pool exceed the decode graph batch: a DSpark verify step for a +# batch above the captured 64 runs eagerly and allocates its attention +# workspace on the fly, which OOMed the H200 eval at 128 running requests +# (6.4 GiB allocation with 2 GiB free, run 35306704553). Batches within the +# graph tier reuse the capture-time workspace instead. +CUDA_GRAPH_MAX_BS=64 +MAX_RUNNING_REQUESTS=$((2 * CONC)) +if (( MAX_RUNNING_REQUESTS > CUDA_GRAPH_MAX_BS )); then + MAX_RUNNING_REQUESTS=$CUDA_GRAPH_MAX_BS +fi + +# Saturation arms carry a larger in-flight working set than the 30-minute +# default warmup drain allows. +if (( CONC >= 32 )); then + export AGENTIC_WARMUP_GRACE_PERIOD=3600 +fi + +# Pyxis shares the host network; port 8888 can already belong to a host service. +select_available_server_port +export AIPERF_SERVER_URL="http://localhost:${PORT}" +export AIPERF_SERVER_METRICS_URLS="${AIPERF_SERVER_URL}/metrics" +export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="sglang:" +echo "Using SGLang endpoint ${AIPERF_SERVER_URL}" + +# DSpark is the checkpoint's own bundled draft: no EAGLE/MTP path and no +# --speculative-num-steps knob; the block size is the only tunable. Golden AL: +# golden_al_distribution/dsv41flash_dspark.yaml, thinking_on, five draft tokens. +# Throughput fixes acceptance to AL 3.51; accuracy evals keep real verification. +DSPARK_BLOCK_SIZE=5 +DSV41_GOLDEN_AL=3.51 +if [[ "${EVAL_ONLY}" != true ]]; then + export SGLANG_SIMULATE_ACC_LEN="$DSV41_GOLDEN_AL" + export SGLANG_SIMULATE_ACC_METHOD=match-expected + export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token +fi +echo "DSpark block size: $DSPARK_BLOCK_SIZE, golden AL=$DSV41_GOLDEN_AL" + +SGLANG_CMD=( + python3 -m sglang.launch_server + --model-path "$MODEL_PATH" --served-model-name "$MODEL" + --host 0.0.0.0 --port "$PORT" + --trust-remote-code + --tp "$TP" --ep-size "$EP_SIZE" + # Backends resolve automatically (dsv4 / flashinfer_mxfp4 / flashinfer_cutedsl + # on Blackwell); the cookbook warns that overriding them costs decode speed. + # 0.70 rather than the cookbook's 0.8, and a bounded prefill chunk: the + # sparse-attention indexer and DSpark prefill buffers scale with the chunk + # times the 1M context, and the default 16384 chunk exhausted HBM on the + # first 66k-99k-token AgentX prompts. + --mem-fraction-static 0.70 + --chunked-prefill-size 4096 + --speculative-algorithm DSPARK + --speculative-dspark-block-size "$DSPARK_BLOCK_SIZE" + --max-running-requests "$MAX_RUNNING_REQUESTS" + --cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS" + --reasoning-parser auto + --tool-call-parser auto + # Draft-token forward passes under long-context agentic load block the + # scheduler long enough to trip the 1800 s default watchdog mid-warmup. + --watchdog-timeout 3600 + --enable-metrics +) +write_command "$RESULT_DIR/sglang_command.txt" "${SGLANG_CMD[@]}" +{ + echo "=== SGLANG_* env vars at launch ===" + env | grep -E '^SGLANG_' | sort + echo "===================================" +} | tee "$SERVER_LOG" +"${SGLANG_CMD[@]}" >> "$SERVER_LOG" 2>&1 & +SERVER_PID=$! +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +if [[ "${EVAL_ONLY}" == true ]]; then + run_eval --port "$PORT" +else + build_replay_cmd "$RESULT_DIR" + REPLAY_CMD+=" --server-metrics ${AIPERF_SERVER_METRICS_URLS}" + run_agentic_replay_and_write_outputs "$RESULT_DIR" +fi diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 1215a1a5f7..cefe990429 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8394,3 +8394,20 @@ qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp: dp-attn: false kv-offload-backend: name: hicache + +# Official 2026-09-21 CUDA 13 nightly, amd64 manifest pinned for B300. +# Latest multiarch index: sha256:987c7e4bd26918647211a5dcad72a2bdf2a5f394ac1469eff730e7517fc139be. +# Uses the cookbook's automatic backends and the existing B200 AgentX memory bounds. +dsv41flash-fp4-b300-sglang-agentic-dspark: + image: lmsysorg/sglang:nightly-dev-cu13-20260921-0f6761b5@sha256:18aed2cd75ed45932a9d409e167e75191e0596a47925de1f176d1cc03036cf8c + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:b300-dsxe + precision: fp4 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 9419a68f0b..795cae837c 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8531,3 +8531,12 @@ - "为 GB300 vLLM DeepSeek-V4.1-Flash AgentX 配方在现有 TP4 臂旁新增 TP2 臂,Engram 表继续通过 --engram-config cpu_offload 放在固定页主机 DRAM;每张 277 GiB GPU 的权重升至约 175 GiB" - "在 dsv41flash_fp4_vllm_mtp.sh 中将 B200 TP2 的上限推广到所有 TP2 臂:--max-num-batched-tokens 4096(上游 16384 时 indexer 的 [batched-tokens, 1M] fp8 缓冲区达 32 GiB)、--max-num-seqs 为并发的两倍(16-256)、CUDA graph 捕获上限 512,为每张 GPU 留出约 36 GiB KV;TP4 与 TP8 臂沿用上游默认值" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3321 + +- config-keys: + - dsv41flash-fp4-b300-sglang-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260921-0f6761b5), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals." + - "Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/TBD diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index d49b28b417..87581af4ec 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -336,7 +336,7 @@ else fi # Keep all new AgentX runtime directories outside /workspace. - if [[ "$MODEL_PREFIX" == "dsv41flash" && "$FRAMEWORK" == "vllm" ]]; then + if [[ "$MODEL_PREFIX" == "dsv41flash" && ( "$FRAMEWORK" == "vllm" || "$FRAMEWORK" == "sglang" ) ]]; then CONTAINER_MOUNT_DIR=/ix export INFMAX_CONTAINER_WORKSPACE=/ix export RESULT_DIR=/ix/results From 7b29802813cc7a6e894ad4da6c157a91f458d27c Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 13:49:48 -0500 Subject: [PATCH 02/18] fix(b300): record recipe pull request provenance --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 795cae837c..bf3bede6be 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8539,4 +8539,4 @@ description: - "Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260921-0f6761b5), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals." - "Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix." - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From 96daee34ce9343791884ccab56ecb3810eeabec2 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 14:11:56 -0500 Subject: [PATCH 03/18] feat(b300): add non-speculative SGLang AgentX arm --- .../agentic/dsv41flash_fp4_b300_sglang.sh | 1 + .../agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 52 +++++++++++-------- configs/nvidia-master.yaml | 15 ++++++ perf-changelog.yaml | 9 ++++ 4 files changed, 54 insertions(+), 23 deletions(-) create mode 120000 benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang.sh diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang.sh new file mode 120000 index 0000000000..03357261bc --- /dev/null +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang.sh @@ -0,0 +1 @@ +dsv41flash_fp4_b300_sglang_mtp.sh \ No newline at end of file diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh index 44665f8cd9..c341b74672 100644 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -1,12 +1,12 @@ #!/usr/bin/env bash set -eo pipefail -# DeepSeek-V4.1-Flash AgentX on B300 with SGLang native DSpark, following the -# cookbook's verified Blackwell TP4/EP4 low-latency cell. The KV cache is GPU-resident. +# DeepSeek-V4.1-Flash AgentX on B300 with SGLang, supporting STP and DSpark. +# The KV cache is GPU-resident; caller SPEC_DECODING selects the serving mode. # https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4_1 source "$(dirname "$0")/../../benchmark_lib.sh" check_env_vars MODEL TP EP_SIZE CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION -check_env_vars EVAL_ONLY +check_env_vars EVAL_ONLY SPEC_DECODING require_agentic_kv_offload_none export GPU_COUNT="$TP" @@ -43,13 +43,11 @@ export SGLANG_TIMEOUT_KEEP_ALIVE=900 export SGLANG_DEFAULT_THINKING=1 export SGLANG_DSV41_REASONING_EFFORT=high -# One shared host copy of the two fp8 Engram tables instead of a row-sharded -# copy per rank: the SGLang analogue of the vLLM arm's Engram CPU offload. It -# frees ~46 GiB of HBM per GPU for the 1M-context prefill working set and the -# KV pool, and output is bitwise unchanged (cookbook). The first sweep ran -# with the tables on GPU and the server died on the first long AgentX prompts -# (run 35304517453: c4 came up, then the server exited on the first two warmup prompts). +# Keep Engram tables in host RAM to make room for long-context AgentX KV. +# Per-rank anonymous mappings can use THP without requiring shared-memory THP +# or host sysctl changes. The table payload remains native FP8. export SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1 +export SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT=per_rank # AgentX concurrency counts live session trees, not individual requests. # Allow subagent fan-out to exceed CONC without clipping request bursts, but @@ -77,18 +75,27 @@ export AIPERF_SERVER_METRICS_URLS="${AIPERF_SERVER_URL}/metrics" export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="sglang:" echo "Using SGLang endpoint ${AIPERF_SERVER_URL}" -# DSpark is the checkpoint's own bundled draft: no EAGLE/MTP path and no -# --speculative-num-steps knob; the block size is the only tunable. Golden AL: -# golden_al_distribution/dsv41flash_dspark.yaml, thinking_on, five draft tokens. -# Throughput fixes acceptance to AL 3.51; accuracy evals keep real verification. -DSPARK_BLOCK_SIZE=5 -DSV41_GOLDEN_AL=3.51 -if [[ "${EVAL_ONLY}" != true ]]; then - export SGLANG_SIMULATE_ACC_LEN="$DSV41_GOLDEN_AL" - export SGLANG_SIMULATE_ACC_METHOD=match-expected - export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token -fi -echo "DSpark block size: $DSPARK_BLOCK_SIZE, golden AL=$DSV41_GOLDEN_AL" +# STP and accuracy evaluations must never inherit synthetic acceptance. +unset SGLANG_SIMULATE_ACC_LEN SGLANG_SIMULATE_ACC_METHOD SGLANG_SIMULATE_ACC_TOKEN_MODE +SPECULATIVE_ARGS=() +case "$SPEC_DECODING" in + none) + echo "Non-speculative decoding; synthetic acceptance disabled" + ;; + mtp) + # Existing measured curve: dsv41flash_dspark.yaml, thinking_on, K5. + SPECULATIVE_ARGS=(--speculative-algorithm DSPARK --speculative-dspark-block-size 5) + if [[ "$EVAL_ONLY" != true ]]; then + export SGLANG_SIMULATE_ACC_LEN=3.51 + export SGLANG_SIMULATE_ACC_METHOD=match-expected + export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token + fi + ;; + *) + echo "Unsupported SPEC_DECODING: $SPEC_DECODING" >&2 + exit 1 + ;; +esac SGLANG_CMD=( python3 -m sglang.launch_server @@ -104,8 +111,7 @@ SGLANG_CMD=( # first 66k-99k-token AgentX prompts. --mem-fraction-static 0.70 --chunked-prefill-size 4096 - --speculative-algorithm DSPARK - --speculative-dspark-block-size "$DSPARK_BLOCK_SIZE" + "${SPECULATIVE_ARGS[@]}" --max-running-requests "$MAX_RUNNING_REQUESTS" --cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS" --reasoning-parser auto diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index cefe990429..cf0e84fcb8 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8411,3 +8411,18 @@ dsv41flash-fp4-b300-sglang-agentic-dspark: - dram-utilization: 0.80 search-space: - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } + +# Non-speculative cookbook high-throughput path; no draft or synthetic acceptance. +dsv41flash-fp4-b300-sglang-agentic: + image: lmsysorg/sglang:nightly-dev-cu13-20260921-0f6761b5@sha256:18aed2cd75ed45932a9d409e167e75191e0596a47925de1f176d1cc03036cf8c + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:b300-dsxe + precision: fp4 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: none, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 2614f69540..e7829e42fc 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8551,3 +8551,12 @@ - "Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260921-0f6761b5), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals." - "Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 + +- config-keys: + - dsv41flash-fp4-b300-sglang-agentic + scenario-type: + - agentic-coding + description: + - "Add non-speculative B300 DeepSeek-V4.1-Flash SGLang AgentX TP4/EP4 on the pinned latest nightly, with no draft and no synthetic acceptance, while native DSpark precision qualification remains blocked." + - "Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From b69b0641242400abef4027e478b3ea6a2e5a186d Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 14:19:46 -0500 Subject: [PATCH 04/18] docs: consolidate performance changelog into one PR entry --- perf-changelog.yaml | 20 +++++++------------- 1 file changed, 7 insertions(+), 13 deletions(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index e7829e42fc..a990734188 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8544,19 +8544,13 @@ pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3321 - config-keys: - - dsv41flash-fp4-b300-sglang-agentic-dspark + - dsv41flash-fp4-b300-sglang-agentic-dspark + - dsv41flash-fp4-b300-sglang-agentic scenario-type: - - agentic-coding - description: - - "Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260921-0f6761b5), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals." - - "Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix." - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 - -- config-keys: - - dsv41flash-fp4-b300-sglang-agentic - scenario-type: - - agentic-coding + - agentic-coding description: - - "Add non-speculative B300 DeepSeek-V4.1-Flash SGLang AgentX TP4/EP4 on the pinned latest nightly, with no draft and no synthetic acceptance, while native DSpark precision qualification remains blocked." - - "Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection." + - Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260921-0f6761b5), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals. + - Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix. + - Add non-speculative B300 DeepSeek-V4.1-Flash SGLang AgentX TP4/EP4 on the pinned latest nightly, with no draft and no synthetic acceptance, while native DSpark precision qualification remains blocked. + - Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection. pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From 0661c240faaa2f9a2fd00048fe42164c87860d07 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 14:40:05 -0500 Subject: [PATCH 05/18] fix(b300): preserve native DSpark FP8 draft projections --- .../agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 12 ++- .../agentic/patch_sglang_dsv41_native_wo_a.py | 66 ++++++++++++++++ .../patches/sglang_dsv41_native_wo_a.patch | 76 +++++++++++++++++++ docs/configuration-procedures.md | 8 ++ docs/configuration-procedures_zh.md | 7 ++ perf-changelog.yaml | 1 + 6 files changed, 169 insertions(+), 1 deletion(-) create mode 100644 benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py create mode 100644 benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh index c341b74672..1631e5539a 100644 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -19,7 +19,8 @@ if [[ -n "${MODEL_PATH:-}" && "$MODEL_PATH" != "$MODEL" ]]; then hf download "$MODEL" --local-dir "$MODEL_PATH" else hf download "$MODEL" - export MODEL_PATH="$MODEL" + MODEL_PATH=$(python3 -c 'from huggingface_hub import snapshot_download; import sys; print(snapshot_download(repo_id=sys.argv[1], local_files_only=True))' "$MODEL") + export MODEL_PATH fi nvidia-smi @@ -30,6 +31,12 @@ SERVER_LOG="$RESULT_DIR/server.log" export PYTHONNOUSERSITE=1 export PYTHONUNBUFFERED=1 +# Preserve native draft FP8 weights with the exact-nightly, hash-guarded fix. +if [[ "$SPEC_DECODING" == mtp ]]; then + python3 "$(dirname "$0")/patch_sglang_dsv41_native_wo_a.py" \ + | tee "$RESULT_DIR/native_wo_a_patch.txt" +fi + # Agentic warmup dispatches hundreds of large prompts at once and SGLang's # tokenizer can leave bytes unacknowledged past AIPerf's default 30 s # TCP_USER_TIMEOUT, so Linux aborts live localhost connections. @@ -123,6 +130,9 @@ SGLANG_CMD=( ) write_command "$RESULT_DIR/sglang_command.txt" "${SGLANG_CMD[@]}" { + if [[ "$SPEC_DECODING" == mtp ]]; then + cat "$RESULT_DIR/native_wo_a_patch.txt" + fi echo "=== SGLANG_* env vars at launch ===" env | grep -E '^SGLANG_' | sort echo "===================================" diff --git a/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py b/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py new file mode 100644 index 0000000000..b462590fcd --- /dev/null +++ b/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py @@ -0,0 +1,66 @@ +"""Preserve native V4.1 DSpark FP8 WO_A weights in the official SGLang nightly. + +Applies only to the exact upstream source shipped at 0f6761b54facebb47f2068f87ecccd8f14da3a0e. +The dense FP8 linear path supports the checkpoint's 32x32 scales; the upstream +grouped projection instead dequantizes these draft weights permanently to BF16. +""" + +import hashlib +import importlib.util +import subprocess +from pathlib import Path + +BASE_SHA256 = "61dc79f075c9e1e5a68a466de5eb95a85ff1fa2e5a9f4d66c2fa69e956fd7b61" +PATCHED_SHA256 = "f33b370d9e49f87c5909e23109c1f22968166690765a7df169144d149f0f3d87" +HELPER_SHA256 = "9e0d5e2cb5afe0d729ffffa160f1b44ddcca163e44a675a393154a6148a610de" + + +def digest(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def apply_native_wo_a_patch(package_root: Path) -> None: + model = package_root / "srt/models/deepseek_v4_dspark.py" + helper = package_root / "srt/models/dspark_projection.py" + actual = digest(model) + if ( + actual == PATCHED_SHA256 + and helper.is_file() + and digest(helper) == HELPER_SHA256 + ): + print(f"Native DSpark WO_A patch already applied: sha256:{actual}") + return + if actual != BASE_SHA256 or helper.exists(): + raise RuntimeError( + "Refusing to patch unexpected SGLang DSpark source: " + f"model sha256:{actual}; expected sha256:{BASE_SHA256}. " + "Revalidate the native FP8 fix when changing the serving image." + ) + patch = Path(__file__).parent / "patches/sglang_dsv41_native_wo_a.patch" + command = [ + "patch", + "--batch", + "--forward", + "--fuzz=0", + "-p3", + "-i", + str(patch.resolve()), + ] + subprocess.run([*command, "--dry-run"], cwd=package_root, check=True) + subprocess.run(command, cwd=package_root, check=True) + if digest(model) != PATCHED_SHA256 or digest(helper) != HELPER_SHA256: + raise RuntimeError( + "Patched DSpark source does not match the validated native FP8 fix" + ) + print(f"Applied native DSpark WO_A patch: sha256:{PATCHED_SHA256}") + + +def main() -> None: + spec = importlib.util.find_spec("sglang") + if spec is None or spec.origin is None: + raise RuntimeError("SGLang package is unavailable in the serving container") + apply_native_wo_a_patch(Path(spec.origin).parent) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch b/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch new file mode 100644 index 0000000000..21c94e0d58 --- /dev/null +++ b/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch @@ -0,0 +1,76 @@ +diff --git a/python/sglang/srt/models/deepseek_v4_dspark.py b/python/sglang/srt/models/deepseek_v4_dspark.py +index baebc2d..e2cd9f8 100644 +--- a/python/sglang/srt/models/deepseek_v4_dspark.py ++++ b/python/sglang/srt/models/deepseek_v4_dspark.py +@@ -55,0 +56 @@ from sglang.srt.models.dspark import ( ++from sglang.srt.models.dspark_projection import apply_grouped_wo_a +@@ -106,0 +108,3 @@ class DSparkAttention(MqaAttentionBase): ++ native_wo_a = ( ++ getattr(config, "model_type", None) == "deepseek_v41" and not _is_npu ++ ) +@@ -117 +121 @@ class DSparkAttention(MqaAttentionBase): +- wo_a_keeps_quant_config=False, ++ wo_a_keeps_quant_config=native_wo_a, +@@ -123,0 +128 @@ class DSparkAttention(MqaAttentionBase): ++ self.native_wo_a = native_wo_a +@@ -355,2 +360,8 @@ class DSparkAttention(MqaAttentionBase): +- wo_a = self.wo_a.weight.view(self.n_local_groups, self.o_lora_rank, -1) +- if self._use_fast_kernel: ++ if self.native_wo_a: ++ # V4.1 uses 32x32-scaled FP8 weights. The grouped absorb GEMM only ++ # supports 128x128 scales, so use the existing dense quant method ++ # and retain each group's own output. At TP8 there is one local ++ # group and this is a single ordinary FP8 linear projection. ++ o = apply_grouped_wo_a(self.wo_a, o, self.o_lora_rank) ++ elif self._use_fast_kernel: ++ wo_a = self.wo_a.weight.view(self.n_local_groups, self.o_lora_rank, -1) +@@ -364,0 +376 @@ class DSparkAttention(MqaAttentionBase): ++ wo_a = self.wo_a.weight.view(self.n_local_groups, self.o_lora_rank, -1) +@@ -1074 +1086,2 @@ class DeepseekV4ForCausalLMDSpark(nn.Module): +- weights = _dequant_fp8_wo_a_streaming(weights) ++ if getattr(self.config, "model_type", None) != "deepseek_v41" or _is_npu: ++ weights = _dequant_fp8_wo_a_streaming(weights) +@@ -1145,0 +1159,17 @@ class DeepseekV4ForCausalLMDSpark(nn.Module): ++ if getattr(self.config, "model_type", None) == "deepseek_v41" and not _is_npu: ++ for stage_id, stage in enumerate(self.stages): ++ prefix = f"stages.{stage_id}.self_attn.wo_a" ++ required = {f"{prefix}.weight", f"{prefix}.weight_scale_inv"} ++ if not required.issubset(loaded_params): ++ raise ValueError( ++ f"Native FP8 DSpark projection missing {required - loaded_params}" ++ ) ++ if stage.self_attn.wo_a.weight.dtype != torch.float8_e4m3fn: ++ raise ValueError( ++ f"{prefix} must retain native FP8 checkpoint weights" ++ ) ++ logger.info( ++ "DSpark native FP8 WO_A weights and block scales loaded for %d stages", ++ len(self.stages), ++ ) ++ +diff --git a/python/sglang/srt/models/dspark_projection.py b/python/sglang/srt/models/dspark_projection.py +new file mode 100644 +index 0000000..89df415 +--- /dev/null ++++ b/python/sglang/srt/models/dspark_projection.py +@@ -0,0 +1,20 @@ ++from typing import Callable, Optional, Tuple ++ ++import torch ++ ++ ++def apply_grouped_wo_a( ++ projection: Callable[[torch.Tensor], Tuple[torch.Tensor, Optional[torch.Tensor]]], ++ hidden_states: torch.Tensor, ++ output_rank: int, ++) -> torch.Tensor: ++ """Apply each local group's projection through its native linear method. ++ ++ The linear owns all local groups' weight rows. Projecting each input group ++ through it and retaining the diagonal supports its existing quantized ++ kernels without changing weight storage. One local group needs one GEMM. ++ """ ++ tokens, groups, width = hidden_states.shape ++ projected, _ = projection(hidden_states.reshape(tokens * groups, width)) ++ projected = projected.view(tokens, groups, groups, output_rank) ++ return projected.diagonal(dim1=1, dim2=2).transpose(1, 2).contiguous() diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 06906129e9..70a7a0a3f4 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -405,6 +405,14 @@ multi-arch preview build `lmsysorg/sglang:dev-dsv41` and MI355X uses `lmsysorg/sglang:dev-dsv41-mi35x`. Both tags are mutable, so the master configs and the changelog record the digests they were validated against. +B300 registers STP and DSpark on the pinned latest CUDA 13 nightly. Its DSpark +path applies the hash-verified `patch_sglang_dsv41_native_wo_a.py` fix to preserve +all three native FP8 WO_A weights and block scales; Markov weights remain native +BF16. STP loads no draft. Both STP and evals clear synthetic acceptance. B300 +uses per-rank anonymous Engram host tables without host sysctl changes, and +resolves downloaded Hugging Face snapshots locally before serving. Verify +actual huge-page backing, native draft startup checks and full accuracy results. + DSpark is the checkpoint's own bundled draft. SGLang exposes no EAGLE or MTP path and no `--speculative-num-steps` knob for it; the recipes pass `--speculative-algorithm DSPARK --speculative-dspark-block-size 5`. Throughput uses the same diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index f97f1737ce..117d4e416e 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -352,6 +352,13 @@ Maximum concurrency for 1,048,576 tokens per request: 6.70x `lmsysorg/sglang:dev-dsv41`,MI355X 使用 `lmsysorg/sglang:dev-dsv41-mi35x`。两个标签均可变, 因此 master 配置与 changelog 记录了验证时的 digest。 +B300 在固定到摘要的最新 CUDA 13 nightly 上注册 STP 和 DSpark。DSpark 路径应用 +经过哈希校验的 `patch_sglang_dsv41_native_wo_a.py` 修复,保留三个原生 FP8 WO_A +权重及分块 scale;Markov 权重保持原生 BF16。STP 不加载草稿模型,STP 与评估均清除 +合成接受率设置。B300 使用每个 rank 的匿名 Engram 主机表,无需修改主机 sysctl, +并在服务启动前从本地解析已下载的 Hugging Face 快照。必须核实实际大页覆盖率、 +草稿原生精度启动检查及完整准确率结果。 + DSpark 是检查点自带的草稿模型。SGLang 对它不提供 EAGLE 或 MTP 路径,也没有 `--speculative-num-steps` 参数;配方传入 `--speculative-algorithm DSPARK --speculative-dspark-block-size 5`。吞吐测试通过 `SGLANG_SIMULATE_ACC_LEN`(`match-expected`、 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index a990734188..f0266b7eba 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8553,4 +8553,5 @@ - Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix. - Add non-speculative B300 DeepSeek-V4.1-Flash SGLang AgentX TP4/EP4 on the pinned latest nightly, with no draft and no synthetic acceptance, while native DSpark precision qualification remains blocked. - Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection. + - Apply the hash-verified native FP8 DSpark WO_A preservation fix on the pinned nightly, retain native BF16 Markov weights, and resolve downloaded checkpoints locally before serving. pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From d09f8a66c9720e9f515d94b4e87bb0197617c396 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 14:46:19 -0500 Subject: [PATCH 06/18] fix(b300): normalize strided native FP8 projection inputs --- .../single_node/agentic/patch_sglang_dsv41_native_wo_a.py | 2 +- .../single_node/agentic/patches/sglang_dsv41_native_wo_a.patch | 2 +- perf-changelog.yaml | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py b/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py index b462590fcd..75a8f9cd28 100644 --- a/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py +++ b/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py @@ -12,7 +12,7 @@ BASE_SHA256 = "61dc79f075c9e1e5a68a466de5eb95a85ff1fa2e5a9f4d66c2fa69e956fd7b61" PATCHED_SHA256 = "f33b370d9e49f87c5909e23109c1f22968166690765a7df169144d149f0f3d87" -HELPER_SHA256 = "9e0d5e2cb5afe0d729ffffa160f1b44ddcca163e44a675a393154a6148a610de" +HELPER_SHA256 = "4d1eb84af8cd9813c17417d31bdcf1c01fff96e5eb281e8ea2282aa2bae41a0e" def digest(path: Path) -> str: diff --git a/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch b/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch index 21c94e0d58..2e2265755f 100644 --- a/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch +++ b/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch @@ -71,6 +71,6 @@ index 0000000..89df415 + kernels without changing weight storage. One local group needs one GEMM. + """ + tokens, groups, width = hidden_states.shape -+ projected, _ = projection(hidden_states.reshape(tokens * groups, width)) ++ projected, _ = projection(hidden_states.reshape(tokens * groups, width).contiguous()) + projected = projected.view(tokens, groups, groups, output_rank) + return projected.diagonal(dim1=1, dim2=2).transpose(1, 2).contiguous() diff --git a/perf-changelog.yaml b/perf-changelog.yaml index f0266b7eba..db8d50e589 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8553,5 +8553,5 @@ - Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix. - Add non-speculative B300 DeepSeek-V4.1-Flash SGLang AgentX TP4/EP4 on the pinned latest nightly, with no draft and no synthetic acceptance, while native DSpark precision qualification remains blocked. - Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection. - - Apply the hash-verified native FP8 DSpark WO_A preservation fix on the pinned nightly, retain native BF16 Markov weights, and resolve downloaded checkpoints locally before serving. + - Apply the hash-verified native FP8 DSpark WO_A preservation fix on the pinned nightly, retain native BF16 Markov weights, normalize strided FP8 projection inputs for eager and CUDA-graph execution, and resolve downloaded checkpoints locally before serving. pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From 17e7d2539043e9f136bbb9569b63babd931c19dc Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 15:11:48 -0500 Subject: [PATCH 07/18] fix: restore default SGLang DSpark precision --- .../agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 9 +-- .../agentic/patch_sglang_dsv41_native_wo_a.py | 66 ---------------- .../patches/sglang_dsv41_native_wo_a.patch | 76 ------------------- docs/configuration-procedures.md | 8 +- docs/configuration-procedures_zh.md | 7 +- perf-changelog.yaml | 4 +- 6 files changed, 5 insertions(+), 165 deletions(-) delete mode 100644 benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py delete mode 100644 benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh index 1631e5539a..7324d76d5f 100644 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -31,11 +31,7 @@ SERVER_LOG="$RESULT_DIR/server.log" export PYTHONNOUSERSITE=1 export PYTHONUNBUFFERED=1 -# Preserve native draft FP8 weights with the exact-nightly, hash-guarded fix. -if [[ "$SPEC_DECODING" == mtp ]]; then - python3 "$(dirname "$0")/patch_sglang_dsv41_native_wo_a.py" \ - | tee "$RESULT_DIR/native_wo_a_patch.txt" -fi +# Use the default DSpark precision shipped by the pinned SGLang nightly. # Agentic warmup dispatches hundreds of large prompts at once and SGLang's # tokenizer can leave bytes unacknowledged past AIPerf's default 30 s @@ -130,9 +126,6 @@ SGLANG_CMD=( ) write_command "$RESULT_DIR/sglang_command.txt" "${SGLANG_CMD[@]}" { - if [[ "$SPEC_DECODING" == mtp ]]; then - cat "$RESULT_DIR/native_wo_a_patch.txt" - fi echo "=== SGLANG_* env vars at launch ===" env | grep -E '^SGLANG_' | sort echo "===================================" diff --git a/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py b/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py deleted file mode 100644 index 75a8f9cd28..0000000000 --- a/benchmarks/single_node/agentic/patch_sglang_dsv41_native_wo_a.py +++ /dev/null @@ -1,66 +0,0 @@ -"""Preserve native V4.1 DSpark FP8 WO_A weights in the official SGLang nightly. - -Applies only to the exact upstream source shipped at 0f6761b54facebb47f2068f87ecccd8f14da3a0e. -The dense FP8 linear path supports the checkpoint's 32x32 scales; the upstream -grouped projection instead dequantizes these draft weights permanently to BF16. -""" - -import hashlib -import importlib.util -import subprocess -from pathlib import Path - -BASE_SHA256 = "61dc79f075c9e1e5a68a466de5eb95a85ff1fa2e5a9f4d66c2fa69e956fd7b61" -PATCHED_SHA256 = "f33b370d9e49f87c5909e23109c1f22968166690765a7df169144d149f0f3d87" -HELPER_SHA256 = "4d1eb84af8cd9813c17417d31bdcf1c01fff96e5eb281e8ea2282aa2bae41a0e" - - -def digest(path: Path) -> str: - return hashlib.sha256(path.read_bytes()).hexdigest() - - -def apply_native_wo_a_patch(package_root: Path) -> None: - model = package_root / "srt/models/deepseek_v4_dspark.py" - helper = package_root / "srt/models/dspark_projection.py" - actual = digest(model) - if ( - actual == PATCHED_SHA256 - and helper.is_file() - and digest(helper) == HELPER_SHA256 - ): - print(f"Native DSpark WO_A patch already applied: sha256:{actual}") - return - if actual != BASE_SHA256 or helper.exists(): - raise RuntimeError( - "Refusing to patch unexpected SGLang DSpark source: " - f"model sha256:{actual}; expected sha256:{BASE_SHA256}. " - "Revalidate the native FP8 fix when changing the serving image." - ) - patch = Path(__file__).parent / "patches/sglang_dsv41_native_wo_a.patch" - command = [ - "patch", - "--batch", - "--forward", - "--fuzz=0", - "-p3", - "-i", - str(patch.resolve()), - ] - subprocess.run([*command, "--dry-run"], cwd=package_root, check=True) - subprocess.run(command, cwd=package_root, check=True) - if digest(model) != PATCHED_SHA256 or digest(helper) != HELPER_SHA256: - raise RuntimeError( - "Patched DSpark source does not match the validated native FP8 fix" - ) - print(f"Applied native DSpark WO_A patch: sha256:{PATCHED_SHA256}") - - -def main() -> None: - spec = importlib.util.find_spec("sglang") - if spec is None or spec.origin is None: - raise RuntimeError("SGLang package is unavailable in the serving container") - apply_native_wo_a_patch(Path(spec.origin).parent) - - -if __name__ == "__main__": - main() diff --git a/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch b/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch deleted file mode 100644 index 2e2265755f..0000000000 --- a/benchmarks/single_node/agentic/patches/sglang_dsv41_native_wo_a.patch +++ /dev/null @@ -1,76 +0,0 @@ -diff --git a/python/sglang/srt/models/deepseek_v4_dspark.py b/python/sglang/srt/models/deepseek_v4_dspark.py -index baebc2d..e2cd9f8 100644 ---- a/python/sglang/srt/models/deepseek_v4_dspark.py -+++ b/python/sglang/srt/models/deepseek_v4_dspark.py -@@ -55,0 +56 @@ from sglang.srt.models.dspark import ( -+from sglang.srt.models.dspark_projection import apply_grouped_wo_a -@@ -106,0 +108,3 @@ class DSparkAttention(MqaAttentionBase): -+ native_wo_a = ( -+ getattr(config, "model_type", None) == "deepseek_v41" and not _is_npu -+ ) -@@ -117 +121 @@ class DSparkAttention(MqaAttentionBase): -- wo_a_keeps_quant_config=False, -+ wo_a_keeps_quant_config=native_wo_a, -@@ -123,0 +128 @@ class DSparkAttention(MqaAttentionBase): -+ self.native_wo_a = native_wo_a -@@ -355,2 +360,8 @@ class DSparkAttention(MqaAttentionBase): -- wo_a = self.wo_a.weight.view(self.n_local_groups, self.o_lora_rank, -1) -- if self._use_fast_kernel: -+ if self.native_wo_a: -+ # V4.1 uses 32x32-scaled FP8 weights. The grouped absorb GEMM only -+ # supports 128x128 scales, so use the existing dense quant method -+ # and retain each group's own output. At TP8 there is one local -+ # group and this is a single ordinary FP8 linear projection. -+ o = apply_grouped_wo_a(self.wo_a, o, self.o_lora_rank) -+ elif self._use_fast_kernel: -+ wo_a = self.wo_a.weight.view(self.n_local_groups, self.o_lora_rank, -1) -@@ -364,0 +376 @@ class DSparkAttention(MqaAttentionBase): -+ wo_a = self.wo_a.weight.view(self.n_local_groups, self.o_lora_rank, -1) -@@ -1074 +1086,2 @@ class DeepseekV4ForCausalLMDSpark(nn.Module): -- weights = _dequant_fp8_wo_a_streaming(weights) -+ if getattr(self.config, "model_type", None) != "deepseek_v41" or _is_npu: -+ weights = _dequant_fp8_wo_a_streaming(weights) -@@ -1145,0 +1159,17 @@ class DeepseekV4ForCausalLMDSpark(nn.Module): -+ if getattr(self.config, "model_type", None) == "deepseek_v41" and not _is_npu: -+ for stage_id, stage in enumerate(self.stages): -+ prefix = f"stages.{stage_id}.self_attn.wo_a" -+ required = {f"{prefix}.weight", f"{prefix}.weight_scale_inv"} -+ if not required.issubset(loaded_params): -+ raise ValueError( -+ f"Native FP8 DSpark projection missing {required - loaded_params}" -+ ) -+ if stage.self_attn.wo_a.weight.dtype != torch.float8_e4m3fn: -+ raise ValueError( -+ f"{prefix} must retain native FP8 checkpoint weights" -+ ) -+ logger.info( -+ "DSpark native FP8 WO_A weights and block scales loaded for %d stages", -+ len(self.stages), -+ ) -+ -diff --git a/python/sglang/srt/models/dspark_projection.py b/python/sglang/srt/models/dspark_projection.py -new file mode 100644 -index 0000000..89df415 ---- /dev/null -+++ b/python/sglang/srt/models/dspark_projection.py -@@ -0,0 +1,20 @@ -+from typing import Callable, Optional, Tuple -+ -+import torch -+ -+ -+def apply_grouped_wo_a( -+ projection: Callable[[torch.Tensor], Tuple[torch.Tensor, Optional[torch.Tensor]]], -+ hidden_states: torch.Tensor, -+ output_rank: int, -+) -> torch.Tensor: -+ """Apply each local group's projection through its native linear method. -+ -+ The linear owns all local groups' weight rows. Projecting each input group -+ through it and retaining the diagonal supports its existing quantized -+ kernels without changing weight storage. One local group needs one GEMM. -+ """ -+ tokens, groups, width = hidden_states.shape -+ projected, _ = projection(hidden_states.reshape(tokens * groups, width).contiguous()) -+ projected = projected.view(tokens, groups, groups, output_rank) -+ return projected.diagonal(dim1=1, dim2=2).transpose(1, 2).contiguous() diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 70a7a0a3f4..6967f21d58 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -405,13 +405,7 @@ multi-arch preview build `lmsysorg/sglang:dev-dsv41` and MI355X uses `lmsysorg/sglang:dev-dsv41-mi35x`. Both tags are mutable, so the master configs and the changelog record the digests they were validated against. -B300 registers STP and DSpark on the pinned latest CUDA 13 nightly. Its DSpark -path applies the hash-verified `patch_sglang_dsv41_native_wo_a.py` fix to preserve -all three native FP8 WO_A weights and block scales; Markov weights remain native -BF16. STP loads no draft. Both STP and evals clear synthetic acceptance. B300 -uses per-rank anonymous Engram host tables without host sysctl changes, and -resolves downloaded Hugging Face snapshots locally before serving. Verify -actual huge-page backing, native draft startup checks and full accuracy results. +DSpark uses the default precision shipped by the pinned official nightly, without custom draft quantization or precision patches. STP loads no draft; full accuracy and performance validation are still required. DSpark is the checkpoint's own bundled draft. SGLang exposes no EAGLE or MTP path and no `--speculative-num-steps` knob for it; the recipes pass `--speculative-algorithm DSPARK diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 117d4e416e..7bdf140336 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -352,12 +352,7 @@ Maximum concurrency for 1,048,576 tokens per request: 6.70x `lmsysorg/sglang:dev-dsv41`,MI355X 使用 `lmsysorg/sglang:dev-dsv41-mi35x`。两个标签均可变, 因此 master 配置与 changelog 记录了验证时的 digest。 -B300 在固定到摘要的最新 CUDA 13 nightly 上注册 STP 和 DSpark。DSpark 路径应用 -经过哈希校验的 `patch_sglang_dsv41_native_wo_a.py` 修复,保留三个原生 FP8 WO_A -权重及分块 scale;Markov 权重保持原生 BF16。STP 不加载草稿模型,STP 与评估均清除 -合成接受率设置。B300 使用每个 rank 的匿名 Engram 主机表,无需修改主机 sysctl, -并在服务启动前从本地解析已下载的 Hugging Face 快照。必须核实实际大页覆盖率、 -草稿原生精度启动检查及完整准确率结果。 +DSpark 使用固定官方 nightly 默认提供的精度,不应用自定义草稿量化或精度补丁。STP 不加载草稿模型;完整准确率和性能验证仍然必需。 DSpark 是检查点自带的草稿模型。SGLang 对它不提供 EAGLE 或 MTP 路径,也没有 `--speculative-num-steps` 参数;配方传入 `--speculative-algorithm DSPARK diff --git a/perf-changelog.yaml b/perf-changelog.yaml index db8d50e589..87b897117c 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8551,7 +8551,7 @@ description: - Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260921-0f6761b5), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals. - Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix. - - Add non-speculative B300 DeepSeek-V4.1-Flash SGLang AgentX TP4/EP4 on the pinned latest nightly, with no draft and no synthetic acceptance, while native DSpark precision qualification remains blocked. + - Add non-speculative B300 DeepSeek-V4.1-Flash SGLang AgentX TP4/EP4 on the pinned latest nightly, with no draft and no synthetic acceptance, alongside the default-precision DSpark arm. - Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection. - - Apply the hash-verified native FP8 DSpark WO_A preservation fix on the pinned nightly, retain native BF16 Markov weights, normalize strided FP8 projection inputs for eager and CUDA-graph execution, and resolve downloaded checkpoints locally before serving. + - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From b06b4f832b99944e50b1038e5ddc3403aa8c3978 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 16:41:42 -0500 Subject: [PATCH 08/18] perf(b300): expand DSpark parallelism and interleave decode --- .../single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 2 ++ configs/nvidia-master.yaml | 1 + perf-changelog.yaml | 1 + 3 files changed, 4 insertions(+) diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh index 7324d76d5f..28358f9edf 100644 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -114,6 +114,8 @@ SGLANG_CMD=( # first 66k-99k-token AgentX prompts. --mem-fraction-static 0.70 --chunked-prefill-size 4096 + # Long AgentX prefills otherwise starve ready decode requests. + --prefill-decode-interval 16 "${SPECULATIVE_ARGS[@]}" --max-running-requests "$MAX_RUNNING_REQUESTS" --cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS" diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index cf0e84fcb8..e99d78a67d 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8411,6 +8411,7 @@ dsv41flash-fp4-b300-sglang-agentic-dspark: - dram-utilization: 0.80 search-space: - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } + - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } # Non-speculative cookbook high-throughput path; no draft or synthetic acceptance. dsv41flash-fp4-b300-sglang-agentic: diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 87b897117c..00d3a2245c 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8554,4 +8554,5 @@ - Add non-speculative B300 DeepSeek-V4.1-Flash SGLang AgentX TP4/EP4 on the pinned latest nightly, with no draft and no synthetic acceptance, alongside the default-precision DSpark arm. - Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection. - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." + - "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From 0a68cc0e2aaee191b37bdcf9a08693422b73afdd Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 17:31:13 -0500 Subject: [PATCH 09/18] Launch B300 SGLang AgentX through an owned Slurm batch job --- perf-changelog.yaml | 1 + runners/launch_b300-dsxe.sh | 53 ++++++++++++++++++++++++++++++++----- 2 files changed, 48 insertions(+), 6 deletions(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 00d3a2245c..c9da37e86b 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8555,4 +8555,5 @@ - Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection. - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." - "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness." + - "Submit this SGLang AgentX lane through a Slurm batch allocation and launch its steps from the compute node, avoiding failed interactive allocation notifications on the runner login host while preserving GPU reservations and terminal-status checks." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index f49ccec597..76c81630ed 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -13,6 +13,38 @@ source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 SLURM_PARTITION="batch_1" SLURM_ACCOUNT="benchmark" +# This lane's interactive allocation notifications fail on login-02, while +# batch submission and steps launched from the allocated node work. Keep the +# workaround scoped to this recipe and use normal Slurm resource accounting. +if [[ "$IS_MULTINODE" != true && "${MODEL_PREFIX:-}" == dsv41flash && + "${FRAMEWORK:-}" == sglang && "${IS_AGENTIC:-}" == 1 && + "${B300_AGENTX_BATCH:-}" != 1 ]]; then + check_env_vars GITHUB_WORKSPACE GPU_COUNT RUNNER_NAME + BATCH_SCRIPT=$(mktemp "${RUNNER_TEMP:-$GITHUB_WORKSPACE}/b300-agentx.XXXXXX.sh") || exit 1 + BATCH_LOG="${BATCH_SCRIPT%.sh}.log" + { + printf '#!/usr/bin/env bash\nexport B300_AGENTX_BATCH=1\nexec bash ' + printf '%q\n' "$GITHUB_WORKSPACE/runners/launch_b300-dsxe.sh" + } > "$BATCH_SCRIPT" + BATCH_ARGS=(--parsable --partition="$SLURM_PARTITION" --account="$SLURM_ACCOUNT" + --nodes=1 --ntasks=1 --gres="gpu:$GPU_COUNT" --exclusive --mem=0 + --time="$SALLOC_TIME_LIMIT" --job-name="$RUNNER_NAME" --export=ALL + --chdir="$GITHUB_WORKSPACE" --output="$BATCH_LOG") + if [[ -n "${SALLOC_EXCLUDE:-}" ]]; then + BATCH_ARGS+=(--exclude="$SALLOC_EXCLUDE") + fi + JOB_ID=$(sbatch "${BATCH_ARGS[@]}" "$BATCH_SCRIPT") || { rm -f "$BATCH_SCRIPT"; exit 1; } + JOB_ID="${JOB_ID%%;*}" + [[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo 'ERROR: B300 batch allocation unavailable' >&2; rm -f "$BATCH_SCRIPT"; exit 1; } + trap 'rc=$?; scancel "$JOB_ID" 2>/dev/null || true; rm -f "$BATCH_SCRIPT"; exit "$rc"' EXIT + trap 'exit 130' INT + trap 'exit 143' TERM + echo "B300 AgentX batch job $JOB_ID; log: $BATCH_LOG" + stream_slurm_job_log "$JOB_ID" "$BATCH_LOG" || exit 1 + verify_slurm_job_status "$JOB_ID" + exit $? +fi + # enroot squash images. Must be on storage every compute node mounts and writable # by the runner user (/data/squash is root-owned, hence the per-user default). SQUASH_DIR="/data/home/sa-gha-runner/squash" @@ -67,8 +99,13 @@ import_squash_image() { return 0 fi - srun -N 1 -A "$SLURM_ACCOUNT" -p "$SLURM_PARTITION" \ - --time="${ENROOT_IMPORT_TIME_LIMIT}" bash -c " + local import_launcher=(srun -N 1 -A "$SLURM_ACCOUNT" -p "$SLURM_PARTITION" + --time="${ENROOT_IMPORT_TIME_LIMIT}") + if [[ "${B300_AGENTX_BATCH:-}" == 1 ]]; then + # Already inside our exclusive compute-node allocation. + import_launcher=() + fi + "${import_launcher[@]}" bash -c " set -eo pipefail exec 9>\"$lock\" flock -w 3600 9 @@ -364,13 +401,17 @@ else SALLOC_ARGS+=(--exclude="$SALLOC_EXCLUDE") fi # Capture this allocation's ID; a runner name can also match an older job. - JOB_ID=$( + if [[ "${B300_AGENTX_BATCH:-}" == 1 ]]; then + JOB_ID="${SLURM_JOB_ID:?B300 batch execution requires a Slurm allocation}" + else + JOB_ID=$( set -o pipefail LC_ALL=C salloc "${SALLOC_ARGS[@]}" 2>&1 | tee /dev/stderr | sed -n 's/.*Granted job allocation \([0-9][0-9]*\)$/\1/p' - ) || exit 1 - [[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo 'ERROR: B300 allocation unavailable' >&2; exit 1; } - trap 'rc=$?; scancel "$JOB_ID" 2>/dev/null || true; exit "$rc"' EXIT + ) || exit 1 + [[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo 'ERROR: B300 allocation unavailable' >&2; exit 1; } + trap 'rc=$?; scancel "$JOB_ID" 2>/dev/null || true; exit "$rc"' EXIT + fi if [[ "$MODEL_MOUNT_DIR" == "$MODEL_ROOT" ]]; then # MODEL_ROOT is node-local: probe the allocated compute node, not the login host. srun --jobid="$JOB_ID" test -r "$MODEL_PATH/config.json" || { From e2b357a12ab90784b1b3e6e1e8baf12552ab4b4e Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 17:44:41 -0500 Subject: [PATCH 10/18] Match B300 batch PMI environment to non-MPI container step --- runners/launch_b300-dsxe.sh | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index 76c81630ed..59f7d12b8c 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -23,7 +23,9 @@ if [[ "$IS_MULTINODE" != true && "${MODEL_PREFIX:-}" == dsv41flash && BATCH_SCRIPT=$(mktemp "${RUNNER_TEMP:-$GITHUB_WORKSPACE}/b300-agentx.XXXXXX.sh") || exit 1 BATCH_LOG="${BATCH_SCRIPT%.sh}.log" { - printf '#!/usr/bin/env bash\nexport B300_AGENTX_BATCH=1\nexec bash ' + # Enroot's PMI hook also inspects this variable. Without it, inherited + # batch PMIX variables trigger mounts absent from our --mpi=none step. + printf '#!/usr/bin/env bash\nexport B300_AGENTX_BATCH=1\nexport SLURM_MPI_TYPE=none\nexec bash ' printf '%q\n' "$GITHUB_WORKSPACE/runners/launch_b300-dsxe.sh" } > "$BATCH_SCRIPT" BATCH_ARGS=(--parsable --partition="$SLURM_PARTITION" --account="$SLURM_ACCOUNT" From e3803f030c504626c7ba7d32e1fb5e3f7de5a5cb Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 18:04:27 -0500 Subject: [PATCH 11/18] Use supported PMIx container steps for B300 AgentX batch jobs --- runners/launch_b300-dsxe.sh | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index 59f7d12b8c..da289a2f2e 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -23,9 +23,7 @@ if [[ "$IS_MULTINODE" != true && "${MODEL_PREFIX:-}" == dsv41flash && BATCH_SCRIPT=$(mktemp "${RUNNER_TEMP:-$GITHUB_WORKSPACE}/b300-agentx.XXXXXX.sh") || exit 1 BATCH_LOG="${BATCH_SCRIPT%.sh}.log" { - # Enroot's PMI hook also inspects this variable. Without it, inherited - # batch PMIX variables trigger mounts absent from our --mpi=none step. - printf '#!/usr/bin/env bash\nexport B300_AGENTX_BATCH=1\nexport SLURM_MPI_TYPE=none\nexec bash ' + printf '#!/usr/bin/env bash\nexport B300_AGENTX_BATCH=1\nexec bash ' printf '%q\n' "$GITHUB_WORKSPACE/runners/launch_b300-dsxe.sh" } > "$BATCH_SCRIPT" BATCH_ARGS=(--parsable --partition="$SLURM_PARTITION" --account="$SLURM_ACCOUNT" @@ -438,8 +436,14 @@ else fi CONTAINER_MOUNTS_ARG=$(IFS=,; printf '%s' "${CONTAINER_MOUNTS[*]}") + B300_CONTAINER_MPI=none + if [[ "${B300_AGENTX_BATCH:-}" == 1 ]]; then + # The installed Enroot hook sees PMIx variables in batch jobs. Use the + # supported plugin so its required per-step mount directories exist. + B300_CONTAINER_MPI=pmix + fi srun --jobid="$JOB_ID" \ - --mpi=none \ + --mpi="$B300_CONTAINER_MPI" \ --container-image="$SQUASH_FILE" \ --container-mounts="$CONTAINER_MOUNTS_ARG" \ --no-container-mount-home \ From a753192d9a9f2c341a6966d91ee5fb5491e33263 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 18:28:08 -0500 Subject: [PATCH 12/18] perf(b300): scope final AgentX sweep to DSpark candidates --- .../agentic/dsv41flash_fp4_b300_sglang.sh | 1 - configs/nvidia-master.yaml | 14 -------------- perf-changelog.yaml | 2 -- 3 files changed, 17 deletions(-) delete mode 120000 benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang.sh diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang.sh deleted file mode 120000 index 03357261bc..0000000000 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang.sh +++ /dev/null @@ -1 +0,0 @@ -dsv41flash_fp4_b300_sglang_mtp.sh \ No newline at end of file diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index fd81aedb75..e7c5743a40 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8411,17 +8411,3 @@ dsv41flash-fp4-b300-sglang-agentic-dspark: - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } -# Non-speculative cookbook high-throughput path; no draft or synthetic acceptance. -dsv41flash-fp4-b300-sglang-agentic: - image: lmsysorg/sglang:nightly-dev-cu13-20260921-0f6761b5@sha256:18aed2cd75ed45932a9d409e167e75191e0596a47925de1f176d1cc03036cf8c - model: deepseek-ai/DeepSeek-V4.1-Flash - model-prefix: dsv41flash - runner: cluster:b300-dsxe - precision: fp4 - framework: sglang - multinode: false - scenarios: - agentic-coding: - - dram-utilization: 0.80 - search-space: - - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: none, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } diff --git a/perf-changelog.yaml b/perf-changelog.yaml index c9da37e86b..401dc6cc29 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8545,13 +8545,11 @@ - config-keys: - dsv41flash-fp4-b300-sglang-agentic-dspark - - dsv41flash-fp4-b300-sglang-agentic scenario-type: - agentic-coding description: - Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260921-0f6761b5), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals. - Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix. - - Add non-speculative B300 DeepSeek-V4.1-Flash SGLang AgentX TP4/EP4 on the pinned latest nightly, with no draft and no synthetic acceptance, alongside the default-precision DSpark arm. - Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection. - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." - "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness." From 72c41b899fb5b12f732f1690da55730b05cb1b67 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 20:16:02 -0500 Subject: [PATCH 13/18] Test B300 low-concurrency prefix retention and checkpoint prefetch --- .../agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 11 +++++++++++ perf-changelog.yaml | 1 + 2 files changed, 12 insertions(+) diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh index 28358f9edf..907d73653f 100644 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -65,6 +65,14 @@ if (( MAX_RUNNING_REQUESTS > CUDA_GRAPH_MAX_BS )); then MAX_RUNNING_REQUESTS=$CUDA_GRAPH_MAX_BS fi +# AgentX reuses long prefixes across turns even at low session concurrency. +# Floor the nightly's four-tails-per-request default while preserving its +# existing larger-request budget until the high-concurrency sweep is measured. +SWA_PREFIX_TAILS=$((4 * MAX_RUNNING_REQUESTS)) +if (( SWA_PREFIX_TAILS < 128 )); then + SWA_PREFIX_TAILS=128 +fi + # Saturation arms carry a larger in-flight working set than the 30-minute # default warmup drain allows. if (( CONC >= 32 )); then @@ -105,6 +113,8 @@ SGLANG_CMD=( --model-path "$MODEL_PATH" --served-model-name "$MODEL" --host 0.0.0.0 --port "$PORT" --trust-remote-code + # Feed mmap weight copies sequentially from shared Lustre storage. + --weight-loader-prefetch-checkpoints --tp "$TP" --ep-size "$EP_SIZE" # Backends resolve automatically (dsv4 / flashinfer_mxfp4 / flashinfer_cutedsl # on Blackwell); the cookbook warns that overriding them costs decode speed. @@ -118,6 +128,7 @@ SGLANG_CMD=( --prefill-decode-interval 16 "${SPECULATIVE_ARGS[@]}" --max-running-requests "$MAX_RUNNING_REQUESTS" + --swa-prefix-tails "$SWA_PREFIX_TAILS" --cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS" --reasoning-parser auto --tool-call-parser auto diff --git a/perf-changelog.yaml b/perf-changelog.yaml index b1eead9f5d..830b059e60 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8565,4 +8565,5 @@ - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." - "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness." - "Submit this SGLang AgentX lane through a Slurm batch allocation and launch its steps from the compute node, avoiding failed interactive allocation notifications on the runner login host while preserving GPU reservations and terminal-status checks." + - "Test a 128-tail minimum for low-concurrency prefix reuse while preserving the larger-concurrency budget and 0.70 static memory fraction; prefetch checkpoints for shared Lustre loading." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From 72a8341750f3a1562c7eb0efc86ebf6b648ac032 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 20:32:03 -0500 Subject: [PATCH 14/18] Scale B300 TP4 prefix retention within measured KV budget --- .../agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 10 ++++++++-- perf-changelog.yaml | 2 +- 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh index 907d73653f..cd23f1eba3 100644 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -66,9 +66,15 @@ if (( MAX_RUNNING_REQUESTS > CUDA_GRAPH_MAX_BS )); then fi # AgentX reuses long prefixes across turns even at low session concurrency. -# Floor the nightly's four-tails-per-request default while preserving its -# existing larger-request budget until the high-concurrency sweep is measured. +# TP4's measured 110.55 GiB KV budget leaves room to retain session prefixes +# across subagent turns. TP2 keeps its conservative budget until measured. SWA_PREFIX_TAILS=$((4 * MAX_RUNNING_REQUESTS)) +if (( TP == 4 )); then + SWA_PREFIX_TAILS=$((64 * CONC)) + if (( SWA_PREFIX_TAILS > 4096 )); then + SWA_PREFIX_TAILS=4096 + fi +fi if (( SWA_PREFIX_TAILS < 128 )); then SWA_PREFIX_TAILS=128 fi diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 830b059e60..debf876524 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8565,5 +8565,5 @@ - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." - "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness." - "Submit this SGLang AgentX lane through a Slurm batch allocation and launch its steps from the compute node, avoiding failed interactive allocation notifications on the runner login host while preserving GPU reservations and terminal-status checks." - - "Test a 128-tail minimum for low-concurrency prefix reuse while preserving the larger-concurrency budget and 0.70 static memory fraction; prefetch checkpoints for shared Lustre loading." + - "Test a 128-tail minimum for low-concurrency prefix reuse and scale TP4 tails with concurrency up to 4096, keeping TP2's larger-concurrency budget and the 0.70 static memory fraction; prefetch checkpoints for shared Lustre loading." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From efd7f6621441ebcfb89922eb59179bc78c93dd4d Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 20:36:43 -0500 Subject: [PATCH 15/18] Pin B300 to the September 22 SGLang nightly --- configs/nvidia-master.yaml | 2 +- perf-changelog.yaml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 83f54c5797..87f2c24658 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8579,7 +8579,7 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: # Latest multiarch index: sha256:987c7e4bd26918647211a5dcad72a2bdf2a5f394ac1469eff730e7517fc139be. # Uses the cookbook's automatic backends and the existing B200 AgentX memory bounds. dsv41flash-fp4-b300-sglang-agentic-dspark: - image: lmsysorg/sglang:nightly-dev-cu13-20260921-0f6761b5@sha256:18aed2cd75ed45932a9d409e167e75191e0596a47925de1f176d1cc03036cf8c + image: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:35ea4d321b0735051dcce2599fc495853ad1ef30f6a1908f0a75e74204362d14 model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:b300-dsxe diff --git a/perf-changelog.yaml b/perf-changelog.yaml index debf876524..4f13f665b7 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8559,7 +8559,7 @@ scenario-type: - agentic-coding description: - - Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260921-0f6761b5), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals. + - Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260922-582389ce), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals. - Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix. - Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection. - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." From c684a14148ab8ff4224531054bcdd86e7068f75d Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 20:54:49 -0500 Subject: [PATCH 16/18] Test larger B300 TP2 KV budget for high concurrency --- .../agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 27 ++++++++++--------- perf-changelog.yaml | 2 +- 2 files changed, 15 insertions(+), 14 deletions(-) diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh index cd23f1eba3..0a114c9d6a 100644 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -66,14 +66,16 @@ if (( MAX_RUNNING_REQUESTS > CUDA_GRAPH_MAX_BS )); then fi # AgentX reuses long prefixes across turns even at low session concurrency. -# TP4's measured 110.55 GiB KV budget leaves room to retain session prefixes -# across subagent turns. TP2 keeps its conservative budget until measured. -SWA_PREFIX_TAILS=$((4 * MAX_RUNNING_REQUESTS)) -if (( TP == 4 )); then - SWA_PREFIX_TAILS=$((64 * CONC)) - if (( SWA_PREFIX_TAILS > 4096 )); then - SWA_PREFIX_TAILS=4096 - fi +# At memory fraction 0.70, TP4 has 110.55 GiB of KV budget but TP2 has only +# 38.83 GiB. Give TP2's larger working sets more KV space before retaining +# additional SWA tails; keep the conservative low-concurrency allocation. +MEM_FRACTION_STATIC=0.70 +if (( TP == 2 && CONC >= 16 )); then + MEM_FRACTION_STATIC=0.80 +fi +SWA_PREFIX_TAILS=$((64 * CONC)) +if (( SWA_PREFIX_TAILS > 4096 )); then + SWA_PREFIX_TAILS=4096 fi if (( SWA_PREFIX_TAILS < 128 )); then SWA_PREFIX_TAILS=128 @@ -124,11 +126,10 @@ SGLANG_CMD=( --tp "$TP" --ep-size "$EP_SIZE" # Backends resolve automatically (dsv4 / flashinfer_mxfp4 / flashinfer_cutedsl # on Blackwell); the cookbook warns that overriding them costs decode speed. - # 0.70 rather than the cookbook's 0.8, and a bounded prefill chunk: the - # sparse-attention indexer and DSpark prefill buffers scale with the chunk - # times the 1M context, and the default 16384 chunk exhausted HBM on the - # first 66k-99k-token AgentX prompts. - --mem-fraction-static 0.70 + # Bound transient prefill allocations: the sparse-attention indexer and + # DSpark buffers scale with the chunk times the 1M context. Static KV + # memory is selected above from the measured TP2/TP4 weight footprints. + --mem-fraction-static "$MEM_FRACTION_STATIC" --chunked-prefill-size 4096 # Long AgentX prefills otherwise starve ready decode requests. --prefill-decode-interval 16 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index dafcd96283..fc835d4a68 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8599,5 +8599,5 @@ - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." - "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness." - "Submit this SGLang AgentX lane through a Slurm batch allocation and launch its steps from the compute node, avoiding failed interactive allocation notifications on the runner login host while preserving GPU reservations and terminal-status checks." - - "Test a 128-tail minimum for low-concurrency prefix reuse and scale TP4 tails with concurrency up to 4096, keeping TP2's larger-concurrency budget and the 0.70 static memory fraction; prefetch checkpoints for shared Lustre loading." + - "Test retained SWA tails at 64 times concurrency, bounded to 128–4096; use static memory 0.80 for TP2 at concurrency 16 and above after measuring its smaller KV budget, otherwise retain 0.70. Prefetch checkpoints for shared Lustre loading." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From 252105c44688a9144c4bc7d342953f7113e74225 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 22:27:10 -0500 Subject: [PATCH 17/18] Increase high-concurrency TP2 cache memory on B300 --- .../single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 6 +++++- perf-changelog.yaml | 2 +- 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh index 0a114c9d6a..e00d9df036 100644 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -70,7 +70,11 @@ fi # 38.83 GiB. Give TP2's larger working sets more KV space before retaining # additional SWA tails; keep the conservative low-concurrency allocation. MEM_FRACTION_STATIC=0.70 -if (( TP == 2 && CONC >= 16 )); then +if (( TP == 2 && CONC >= 32 )); then + # At C64, 0.80 left 43.59 GiB after graphs but only 18.29M full tokens. + # Retain more long prefixes while leaving room for transient prefills. + MEM_FRACTION_STATIC=0.85 +elif (( TP == 2 && CONC >= 16 )); then MEM_FRACTION_STATIC=0.80 fi SWA_PREFIX_TAILS=$((64 * CONC)) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index fc835d4a68..d1c2e6a3cd 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8599,5 +8599,5 @@ - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." - "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness." - "Submit this SGLang AgentX lane through a Slurm batch allocation and launch its steps from the compute node, avoiding failed interactive allocation notifications on the runner login host while preserving GPU reservations and terminal-status checks." - - "Test retained SWA tails at 64 times concurrency, bounded to 128–4096; use static memory 0.80 for TP2 at concurrency 16 and above after measuring its smaller KV budget, otherwise retain 0.70. Prefetch checkpoints for shared Lustre loading." + - "Test retained SWA tails at 64 times concurrency, bounded to 128–4096; use static memory 0.80 for TP2 C16 and 0.85 at C32 and above after measuring its smaller KV budget and 43.59 GiB of post-graph headroom at C64, otherwise retain 0.70. Prefetch checkpoints for shared Lustre loading." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 From cd702aefaa9e3732c16bf27fd51f02ffacdd3c82 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 21 Sep 2026 23:33:20 -0500 Subject: [PATCH 18/18] Screen higher prefill duty for B300 TP2 DSpark --- .../agentic/dsv41flash_fp4_b300_sglang_mtp.sh | 9 ++++++++- perf-changelog.yaml | 2 +- 2 files changed, 9 insertions(+), 2 deletions(-) diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh index e00d9df036..dfb7056385 100644 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -120,6 +120,13 @@ case "$SPEC_DECODING" in ;; esac +# At high TP2 concurrency, test more prefill duty against the matched C64 +# baseline. Keep the latency-oriented cadence on low-C and TP4 points. +PREFILL_DECODE_INTERVAL=16 +if (( TP == 2 && CONC >= 32 )); then + PREFILL_DECODE_INTERVAL=4 +fi + SGLANG_CMD=( python3 -m sglang.launch_server --model-path "$MODEL_PATH" --served-model-name "$MODEL" @@ -136,7 +143,7 @@ SGLANG_CMD=( --mem-fraction-static "$MEM_FRACTION_STATIC" --chunked-prefill-size 4096 # Long AgentX prefills otherwise starve ready decode requests. - --prefill-decode-interval 16 + --prefill-decode-interval "$PREFILL_DECODE_INTERVAL" "${SPECULATIVE_ARGS[@]}" --max-running-requests "$MAX_RUNNING_REQUESTS" --swa-prefix-tails "$SWA_PREFIX_TAILS" diff --git a/perf-changelog.yaml b/perf-changelog.yaml index d1c2e6a3cd..988294821a 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8599,5 +8599,5 @@ - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." - "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness." - "Submit this SGLang AgentX lane through a Slurm batch allocation and launch its steps from the compute node, avoiding failed interactive allocation notifications on the runner login host while preserving GPU reservations and terminal-status checks." - - "Test retained SWA tails at 64 times concurrency, bounded to 128–4096; use static memory 0.80 for TP2 C16 and 0.85 at C32 and above after measuring its smaller KV budget and 43.59 GiB of post-graph headroom at C64, otherwise retain 0.70. Prefetch checkpoints for shared Lustre loading." + - "Test retained SWA tails at 64 times concurrency, bounded to 128–4096; use static memory 0.80 for TP2 C16 and 0.85 at C32 and above after measuring its smaller KV budget and 43.59 GiB of post-graph headroom at C64, otherwise retain 0.70. Prefetch checkpoints for shared Lustre loading. Test prefill/decode interval 4 at TP2 C32 and above against the interval-16 baseline; keep other points at 16." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342