Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
24 commits
Select commit Hold shift + click to select a range
69acc81
feat(b300): add DeepSeek V4.1 SGLang nightly AgentX recipe
cquil11 Sep 21, 2026
7b29802
fix(b300): record recipe pull request provenance
cquil11 Sep 21, 2026
d63d393
Merge remote-tracking branch 'origin/main' into codex/b300-dsv41-sgla…
cquil11 Sep 21, 2026
96daee3
feat(b300): add non-speculative SGLang AgentX arm
cquil11 Sep 21, 2026
b69b064
docs: consolidate performance changelog into one PR entry
cquil11 Sep 21, 2026
0661c24
fix(b300): preserve native DSpark FP8 draft projections
cquil11 Sep 21, 2026
d09f8a6
fix(b300): normalize strided native FP8 projection inputs
cquil11 Sep 21, 2026
17e7d25
fix: restore default SGLang DSpark precision
cquil11 Sep 21, 2026
b06b4f8
perf(b300): expand DSpark parallelism and interleave decode
cquil11 Sep 21, 2026
fa2ca74
Merge remote-tracking branch 'origin/main' into codex/b300-dsv41-sgla…
cquil11 Sep 21, 2026
0a68cc0
Launch B300 SGLang AgentX through an owned Slurm batch job
cquil11 Sep 21, 2026
e2b357a
Match B300 batch PMI environment to non-MPI container step
cquil11 Sep 21, 2026
e3803f0
Use supported PMIx container steps for B300 AgentX batch jobs
cquil11 Sep 21, 2026
a753192
perf(b300): scope final AgentX sweep to DSpark candidates
cquil11 Sep 21, 2026
a6b5ff0
Merge remote-tracking branch 'origin/main' into codex/b300-dsv41-sgla…
cquil11 Sep 21, 2026
72c41b8
Test B300 low-concurrency prefix retention and checkpoint prefetch
cquil11 Sep 22, 2026
72a8341
Scale B300 TP4 prefix retention within measured KV budget
cquil11 Sep 22, 2026
efd7f66
Pin B300 to the September 22 SGLang nightly
cquil11 Sep 22, 2026
22032a1
Merge remote-tracking branch 'origin/main' into codex/b300-dsv41-sgla…
cquil11 Sep 22, 2026
c684a14
Test larger B300 TP2 KV budget for high concurrency
cquil11 Sep 22, 2026
252105c
Increase high-concurrency TP2 cache memory on B300
cquil11 Sep 22, 2026
cd702ae
Screen higher prefill duty for B300 TP2 DSpark
cquil11 Sep 22, 2026
d4c3881
Merge remote-tracking branch 'origin/main' into codex/b300-dsv41-sgla…
cquil11 Sep 22, 2026
0ee9a34
Merge remote-tracking branch 'origin/main' into pr-3342-reuse-94220
cquil11 Sep 22, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
174 changes: 174 additions & 0 deletions benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,174 @@
#!/usr/bin/env bash
set -eo pipefail

# DeepSeek-V4.1-Flash AgentX on B300 with SGLang, supporting STP and DSpark.
# The KV cache is GPU-resident; caller SPEC_DECODING selects the serving mode.
# https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4_1
source "$(dirname "$0")/../../benchmark_lib.sh"
check_env_vars MODEL TP EP_SIZE CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION
check_env_vars EVAL_ONLY SPEC_DECODING
require_agentic_kv_offload_none
export GPU_COUNT="$TP"

if [[ -n "${SLURM_JOB_ID:-}" ]]; then
echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}"
fi

# Complete/resume partial downloads instead of trusting nonempty directories.
if [[ -n "${MODEL_PATH:-}" && "$MODEL_PATH" != "$MODEL" ]]; then
hf download "$MODEL" --local-dir "$MODEL_PATH"
else
hf download "$MODEL"
MODEL_PATH=$(python3 -c 'from huggingface_hub import snapshot_download; import sys; print(snapshot_download(repo_id=sys.argv[1], local_files_only=True))' "$MODEL")
export MODEL_PATH
fi

nvidia-smi
resolve_trace_source
install_agentic_deps
mkdir -p "$RESULT_DIR"
SERVER_LOG="$RESULT_DIR/server.log"
export PYTHONNOUSERSITE=1
export PYTHONUNBUFFERED=1

# Use the default DSpark precision shipped by the pinned SGLang nightly.

# Agentic warmup dispatches hundreds of large prompts at once and SGLang's
# tokenizer can leave bytes unacknowledged past AIPerf's default 30 s
# TCP_USER_TIMEOUT, so Linux aborts live localhost connections.
export AIPERF_HTTP_TCP_USER_TIMEOUT=900000
# Outlast AIPerf's pooled connections so an inter-turn idle gap cannot race
# Uvicorn's five-second keep-alive closure.
export SGLANG_TIMEOUT_KEEP_ALIVE=900

# AgentX measures the thinking-on regime, which is also the committed golden-AL
# curve. SGLang ships thinking off by default for this model.
export SGLANG_DEFAULT_THINKING=1
export SGLANG_DSV41_REASONING_EFFORT=high

# Keep Engram tables in host RAM to make room for long-context AgentX KV.
# Per-rank anonymous mappings can use THP without requiring shared-memory THP
# or host sysctl changes. The table payload remains native FP8.
export SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1
export SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT=per_rank

# AgentX concurrency counts live session trees, not individual requests.
# Allow subagent fan-out to exceed CONC without clipping request bursts, but
# never let the pool exceed the decode graph batch: a DSpark verify step for a
# batch above the captured 64 runs eagerly and allocates its attention
# workspace on the fly, which OOMed the H200 eval at 128 running requests
# (6.4 GiB allocation with 2 GiB free, run 35306704553). Batches within the
# graph tier reuse the capture-time workspace instead.
CUDA_GRAPH_MAX_BS=64
MAX_RUNNING_REQUESTS=$((2 * CONC))
if (( MAX_RUNNING_REQUESTS > CUDA_GRAPH_MAX_BS )); then
MAX_RUNNING_REQUESTS=$CUDA_GRAPH_MAX_BS
fi

# AgentX reuses long prefixes across turns even at low session concurrency.
# At memory fraction 0.70, TP4 has 110.55 GiB of KV budget but TP2 has only
# 38.83 GiB. Give TP2's larger working sets more KV space before retaining
# additional SWA tails; keep the conservative low-concurrency allocation.
MEM_FRACTION_STATIC=0.70
if (( TP == 2 && CONC >= 32 )); then
# At C64, 0.80 left 43.59 GiB after graphs but only 18.29M full tokens.
# Retain more long prefixes while leaving room for transient prefills.
MEM_FRACTION_STATIC=0.85
elif (( TP == 2 && CONC >= 16 )); then
MEM_FRACTION_STATIC=0.80
fi
SWA_PREFIX_TAILS=$((64 * CONC))
if (( SWA_PREFIX_TAILS > 4096 )); then
SWA_PREFIX_TAILS=4096
fi
if (( SWA_PREFIX_TAILS < 128 )); then
SWA_PREFIX_TAILS=128
fi

# Saturation arms carry a larger in-flight working set than the 30-minute
# default warmup drain allows.
if (( CONC >= 32 )); then
export AGENTIC_WARMUP_GRACE_PERIOD=3600
fi

# Pyxis shares the host network; port 8888 can already belong to a host service.
select_available_server_port
export AIPERF_SERVER_URL="http://localhost:${PORT}"
export AIPERF_SERVER_METRICS_URLS="${AIPERF_SERVER_URL}/metrics"
export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="sglang:"
echo "Using SGLang endpoint ${AIPERF_SERVER_URL}"

# STP and accuracy evaluations must never inherit synthetic acceptance.
unset SGLANG_SIMULATE_ACC_LEN SGLANG_SIMULATE_ACC_METHOD SGLANG_SIMULATE_ACC_TOKEN_MODE
SPECULATIVE_ARGS=()
case "$SPEC_DECODING" in
none)
echo "Non-speculative decoding; synthetic acceptance disabled"
;;
mtp)
# Existing measured curve: dsv41flash_dspark.yaml, thinking_on, K5.
SPECULATIVE_ARGS=(--speculative-algorithm DSPARK --speculative-dspark-block-size 5)
if [[ "$EVAL_ONLY" != true ]]; then
export SGLANG_SIMULATE_ACC_LEN=3.51
export SGLANG_SIMULATE_ACC_METHOD=match-expected
export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token
fi
;;
*)
echo "Unsupported SPEC_DECODING: $SPEC_DECODING" >&2
exit 1
;;
esac

# At high TP2 concurrency, test more prefill duty against the matched C64
# baseline. Keep the latency-oriented cadence on low-C and TP4 points.
PREFILL_DECODE_INTERVAL=16
if (( TP == 2 && CONC >= 32 )); then
PREFILL_DECODE_INTERVAL=4
fi

SGLANG_CMD=(
python3 -m sglang.launch_server
--model-path "$MODEL_PATH" --served-model-name "$MODEL"
--host 0.0.0.0 --port "$PORT"
--trust-remote-code
# Feed mmap weight copies sequentially from shared Lustre storage.
--weight-loader-prefetch-checkpoints
--tp "$TP" --ep-size "$EP_SIZE"
# Backends resolve automatically (dsv4 / flashinfer_mxfp4 / flashinfer_cutedsl
# on Blackwell); the cookbook warns that overriding them costs decode speed.
# Bound transient prefill allocations: the sparse-attention indexer and
# DSpark buffers scale with the chunk times the 1M context. Static KV
# memory is selected above from the measured TP2/TP4 weight footprints.
--mem-fraction-static "$MEM_FRACTION_STATIC"
--chunked-prefill-size 4096
# Long AgentX prefills otherwise starve ready decode requests.
--prefill-decode-interval "$PREFILL_DECODE_INTERVAL"
"${SPECULATIVE_ARGS[@]}"
--max-running-requests "$MAX_RUNNING_REQUESTS"
--swa-prefix-tails "$SWA_PREFIX_TAILS"
--cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS"
--reasoning-parser auto
--tool-call-parser auto
# Draft-token forward passes under long-context agentic load block the
# scheduler long enough to trip the 1800 s default watchdog mid-warmup.
--watchdog-timeout 3600
--enable-metrics
)
write_command "$RESULT_DIR/sglang_command.txt" "${SGLANG_CMD[@]}"
{
echo "=== SGLANG_* env vars at launch ==="
env | grep -E '^SGLANG_' | sort
echo "==================================="
} | tee "$SERVER_LOG"
"${SGLANG_CMD[@]}" >> "$SERVER_LOG" 2>&1 &
SERVER_PID=$!
wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID"

if [[ "${EVAL_ONLY}" == true ]]; then
run_eval --port "$PORT"
else
build_replay_cmd "$RESULT_DIR"
REPLAY_CMD+=" --server-metrics ${AIPERF_SERVER_METRICS_URLS}"
run_agentic_replay_and_write_outputs "$RESULT_DIR"
fi
18 changes: 18 additions & 0 deletions configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8701,3 +8701,21 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg:
tp: 2
ep: 2
dp-attn: false

# Official 2026-09-21 CUDA 13 nightly, amd64 manifest pinned for B300.
# Latest multiarch index: sha256:987c7e4bd26918647211a5dcad72a2bdf2a5f394ac1469eff730e7517fc139be.
# Uses the cookbook's automatic backends and the existing B200 AgentX memory bounds.
dsv41flash-fp4-b300-sglang-agentic-dspark:
image: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:35ea4d321b0735051dcce2599fc495853ad1ef30f6a1908f0a75e74204362d14
model: deepseek-ai/DeepSeek-V4.1-Flash
model-prefix: dsv41flash
runner: cluster:b300-dsxe
precision: fp4
framework: sglang
multinode: false
scenarios:
agentic-coding:
- dram-utilization: 0.80
search-space:
- { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] }
- { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] }
2 changes: 2 additions & 0 deletions docs/configuration-procedures.md
Original file line number Diff line number Diff line change
Expand Up @@ -435,6 +435,8 @@ sweep remains required. Draft precision remains upstream default.
The B200 launcher also converts pinned Docker digests to the installed Enroot
manifest-reference syntax and stops immediately on import failure.

DSpark uses the default precision shipped by the pinned official nightly, without custom draft quantization or precision patches. STP loads no draft; full accuracy and performance validation are still required.

DSpark is the checkpoint's own bundled draft. SGLang exposes no EAGLE or MTP path and no
`--speculative-num-steps` knob for it; the recipes pass `--speculative-algorithm DSPARK
--speculative-dspark-block-size 5`. Throughput uses the same
Expand Down
2 changes: 2 additions & 0 deletions docs/configuration-procedures_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -378,6 +378,8 @@ TP2 每 GPU 加载约 147.76 GiB 的目标和草稿权重。配方校验固定
B200 启动器还将固定 Docker digest
转为已安装 Enroot 支持的 manifest 引用格式,并在导入失败时立即停止。

DSpark 使用固定官方 nightly 默认提供的精度,不应用自定义草稿量化或精度补丁。STP 不加载草稿模型;完整准确率和性能验证仍然必需。

DSpark 是检查点自带的草稿模型。SGLang 对它不提供 EAGLE 或 MTP 路径,也没有
`--speculative-num-steps` 参数;配方传入 `--speculative-algorithm DSPARK
--speculative-dspark-block-size 5`。吞吐测试通过 `SGLANG_SIMULATE_ACC_LEN`(`match-expected`、
Expand Down
14 changes: 14 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8633,3 +8633,17 @@
- "Test TP2 loading with the unmodified pinned stock loader and expandable CUDA allocator segments alone; preserve tensor payloads, layouts, and computation"
- "Resolve local model snapshots, retain robust AgentX HTTP timeouts, and normalize digest-pinned images for the installed Enroot parser with immediate import failure handling"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3346

- config-keys:
- dsv41flash-fp4-b300-sglang-agentic-dspark
scenario-type:
- agentic-coding
description:
- Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260922-582389ce), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals.
- Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix.
- Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection.
- "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches."
- "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness."
- "Submit this SGLang AgentX lane through a Slurm batch allocation and launch its steps from the compute node, avoiding failed interactive allocation notifications on the runner login host while preserving GPU reservations and terminal-status checks."
- "Test retained SWA tails at 64 times concurrency, bounded to 128–4096; use static memory 0.80 for TP2 C16 and 0.85 at C32 and above after measuring its smaller KV budget and 43.59 GiB of post-graph headroom at C64, otherwise retain 0.70. Prefetch checkpoints for shared Lustre loading. Test prefill/decode interval 4 at TP2 C32 and above against the interval-16 baseline; keep other points at 16."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342
63 changes: 55 additions & 8 deletions runners/launch_b300-dsxe.sh
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,38 @@ source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1
SLURM_PARTITION="batch_1"
SLURM_ACCOUNT="benchmark"

# This lane's interactive allocation notifications fail on login-02, while
# batch submission and steps launched from the allocated node work. Keep the
# workaround scoped to this recipe and use normal Slurm resource accounting.
if [[ "$IS_MULTINODE" != true && "${MODEL_PREFIX:-}" == dsv41flash &&
"${FRAMEWORK:-}" == sglang && "${IS_AGENTIC:-}" == 1 &&
"${B300_AGENTX_BATCH:-}" != 1 ]]; then
check_env_vars GITHUB_WORKSPACE GPU_COUNT RUNNER_NAME
BATCH_SCRIPT=$(mktemp "${RUNNER_TEMP:-$GITHUB_WORKSPACE}/b300-agentx.XXXXXX.sh") || exit 1
BATCH_LOG="${BATCH_SCRIPT%.sh}.log"
{
printf '#!/usr/bin/env bash\nexport B300_AGENTX_BATCH=1\nexec bash '
printf '%q\n' "$GITHUB_WORKSPACE/runners/launch_b300-dsxe.sh"
} > "$BATCH_SCRIPT"
BATCH_ARGS=(--parsable --partition="$SLURM_PARTITION" --account="$SLURM_ACCOUNT"
--nodes=1 --ntasks=1 --gres="gpu:$GPU_COUNT" --exclusive --mem=0
--time="$SALLOC_TIME_LIMIT" --job-name="$RUNNER_NAME" --export=ALL
--chdir="$GITHUB_WORKSPACE" --output="$BATCH_LOG")
if [[ -n "${SALLOC_EXCLUDE:-}" ]]; then
BATCH_ARGS+=(--exclude="$SALLOC_EXCLUDE")
fi
JOB_ID=$(sbatch "${BATCH_ARGS[@]}" "$BATCH_SCRIPT") || { rm -f "$BATCH_SCRIPT"; exit 1; }
JOB_ID="${JOB_ID%%;*}"
[[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo 'ERROR: B300 batch allocation unavailable' >&2; rm -f "$BATCH_SCRIPT"; exit 1; }
trap 'rc=$?; scancel "$JOB_ID" 2>/dev/null || true; rm -f "$BATCH_SCRIPT"; exit "$rc"' EXIT
trap 'exit 130' INT
trap 'exit 143' TERM
echo "B300 AgentX batch job $JOB_ID; log: $BATCH_LOG"
stream_slurm_job_log "$JOB_ID" "$BATCH_LOG" || exit 1
verify_slurm_job_status "$JOB_ID"
exit $?
fi

# enroot squash images. Must be on storage every compute node mounts and writable
# by the runner user (/data/squash is root-owned, hence the per-user default).
SQUASH_DIR="/data/home/sa-gha-runner/squash"
Expand Down Expand Up @@ -67,8 +99,13 @@ import_squash_image() {
return 0
fi

srun -N 1 -A "$SLURM_ACCOUNT" -p "$SLURM_PARTITION" \
--time="${ENROOT_IMPORT_TIME_LIMIT}" bash -c "
local import_launcher=(srun -N 1 -A "$SLURM_ACCOUNT" -p "$SLURM_PARTITION"
--time="${ENROOT_IMPORT_TIME_LIMIT}")
if [[ "${B300_AGENTX_BATCH:-}" == 1 ]]; then
# Already inside our exclusive compute-node allocation.
import_launcher=()
fi
"${import_launcher[@]}" bash -c "
set -eo pipefail
exec 9>\"$lock\"
flock -w 3600 9
Expand Down Expand Up @@ -318,7 +355,7 @@ else
fi

# Keep all new AgentX runtime directories outside /workspace.
if [[ "$MODEL_PREFIX" == "dsv41flash" && "$FRAMEWORK" == "vllm" ]]; then
if [[ "$MODEL_PREFIX" == "dsv41flash" && ( "$FRAMEWORK" == "vllm" || "$FRAMEWORK" == "sglang" ) ]]; then
CONTAINER_MOUNT_DIR=/ix
export INFMAX_CONTAINER_WORKSPACE=/ix
export RESULT_DIR=/ix/results
Expand Down Expand Up @@ -346,13 +383,17 @@ else
SALLOC_ARGS+=(--exclude="$SALLOC_EXCLUDE")
fi
# Capture this allocation's ID; a runner name can also match an older job.
JOB_ID=$(
if [[ "${B300_AGENTX_BATCH:-}" == 1 ]]; then
JOB_ID="${SLURM_JOB_ID:?B300 batch execution requires a Slurm allocation}"
else
JOB_ID=$(
set -o pipefail
LC_ALL=C salloc "${SALLOC_ARGS[@]}" 2>&1 | tee /dev/stderr |
sed -n 's/.*Granted job allocation \([0-9][0-9]*\)$/\1/p'
) || exit 1
[[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo 'ERROR: B300 allocation unavailable' >&2; exit 1; }
trap 'rc=$?; scancel "$JOB_ID" 2>/dev/null || true; exit "$rc"' EXIT
) || exit 1
[[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo 'ERROR: B300 allocation unavailable' >&2; exit 1; }
trap 'rc=$?; scancel "$JOB_ID" 2>/dev/null || true; exit "$rc"' EXIT
fi
if [[ "$MODEL_MOUNT_DIR" == "$MODEL_ROOT" ]]; then
# MODEL_ROOT is node-local: probe the allocated compute node, not the login host.
srun --jobid="$JOB_ID" test -r "$MODEL_PATH/config.json" || {
Expand All @@ -377,8 +418,14 @@ else
fi
CONTAINER_MOUNTS_ARG=$(IFS=,; printf '%s' "${CONTAINER_MOUNTS[*]}")

B300_CONTAINER_MPI=none
if [[ "${B300_AGENTX_BATCH:-}" == 1 ]]; then
# The installed Enroot hook sees PMIx variables in batch jobs. Use the
# supported plugin so its required per-step mount directories exist.
B300_CONTAINER_MPI=pmix
fi
srun --jobid="$JOB_ID" \
--mpi=none \
--mpi="$B300_CONTAINER_MPI" \
--container-image="$SQUASH_FILE" \
--container-mounts="$CONTAINER_MOUNTS_ARG" \
--no-container-mount-home \
Expand Down
Loading