diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh new file mode 100644 index 0000000000..dfb7056385 --- /dev/null +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_b300_sglang_mtp.sh @@ -0,0 +1,174 @@ +#!/usr/bin/env bash +set -eo pipefail + +# DeepSeek-V4.1-Flash AgentX on B300 with SGLang, supporting STP and DSpark. +# The KV cache is GPU-resident; caller SPEC_DECODING selects the serving mode. +# https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4_1 +source "$(dirname "$0")/../../benchmark_lib.sh" +check_env_vars MODEL TP EP_SIZE CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION +check_env_vars EVAL_ONLY SPEC_DECODING +require_agentic_kv_offload_none +export GPU_COUNT="$TP" + +if [[ -n "${SLURM_JOB_ID:-}" ]]; then + echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}" +fi + +# Complete/resume partial downloads instead of trusting nonempty directories. +if [[ -n "${MODEL_PATH:-}" && "$MODEL_PATH" != "$MODEL" ]]; then + hf download "$MODEL" --local-dir "$MODEL_PATH" +else + hf download "$MODEL" + MODEL_PATH=$(python3 -c 'from huggingface_hub import snapshot_download; import sys; print(snapshot_download(repo_id=sys.argv[1], local_files_only=True))' "$MODEL") + export MODEL_PATH +fi + +nvidia-smi +resolve_trace_source +install_agentic_deps +mkdir -p "$RESULT_DIR" +SERVER_LOG="$RESULT_DIR/server.log" +export PYTHONNOUSERSITE=1 +export PYTHONUNBUFFERED=1 + +# Use the default DSpark precision shipped by the pinned SGLang nightly. + +# Agentic warmup dispatches hundreds of large prompts at once and SGLang's +# tokenizer can leave bytes unacknowledged past AIPerf's default 30 s +# TCP_USER_TIMEOUT, so Linux aborts live localhost connections. +export AIPERF_HTTP_TCP_USER_TIMEOUT=900000 +# Outlast AIPerf's pooled connections so an inter-turn idle gap cannot race +# Uvicorn's five-second keep-alive closure. +export SGLANG_TIMEOUT_KEEP_ALIVE=900 + +# AgentX measures the thinking-on regime, which is also the committed golden-AL +# curve. SGLang ships thinking off by default for this model. +export SGLANG_DEFAULT_THINKING=1 +export SGLANG_DSV41_REASONING_EFFORT=high + +# Keep Engram tables in host RAM to make room for long-context AgentX KV. +# Per-rank anonymous mappings can use THP without requiring shared-memory THP +# or host sysctl changes. The table payload remains native FP8. +export SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1 +export SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT=per_rank + +# AgentX concurrency counts live session trees, not individual requests. +# Allow subagent fan-out to exceed CONC without clipping request bursts, but +# never let the pool exceed the decode graph batch: a DSpark verify step for a +# batch above the captured 64 runs eagerly and allocates its attention +# workspace on the fly, which OOMed the H200 eval at 128 running requests +# (6.4 GiB allocation with 2 GiB free, run 35306704553). Batches within the +# graph tier reuse the capture-time workspace instead. +CUDA_GRAPH_MAX_BS=64 +MAX_RUNNING_REQUESTS=$((2 * CONC)) +if (( MAX_RUNNING_REQUESTS > CUDA_GRAPH_MAX_BS )); then + MAX_RUNNING_REQUESTS=$CUDA_GRAPH_MAX_BS +fi + +# AgentX reuses long prefixes across turns even at low session concurrency. +# At memory fraction 0.70, TP4 has 110.55 GiB of KV budget but TP2 has only +# 38.83 GiB. Give TP2's larger working sets more KV space before retaining +# additional SWA tails; keep the conservative low-concurrency allocation. +MEM_FRACTION_STATIC=0.70 +if (( TP == 2 && CONC >= 32 )); then + # At C64, 0.80 left 43.59 GiB after graphs but only 18.29M full tokens. + # Retain more long prefixes while leaving room for transient prefills. + MEM_FRACTION_STATIC=0.85 +elif (( TP == 2 && CONC >= 16 )); then + MEM_FRACTION_STATIC=0.80 +fi +SWA_PREFIX_TAILS=$((64 * CONC)) +if (( SWA_PREFIX_TAILS > 4096 )); then + SWA_PREFIX_TAILS=4096 +fi +if (( SWA_PREFIX_TAILS < 128 )); then + SWA_PREFIX_TAILS=128 +fi + +# Saturation arms carry a larger in-flight working set than the 30-minute +# default warmup drain allows. +if (( CONC >= 32 )); then + export AGENTIC_WARMUP_GRACE_PERIOD=3600 +fi + +# Pyxis shares the host network; port 8888 can already belong to a host service. +select_available_server_port +export AIPERF_SERVER_URL="http://localhost:${PORT}" +export AIPERF_SERVER_METRICS_URLS="${AIPERF_SERVER_URL}/metrics" +export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="sglang:" +echo "Using SGLang endpoint ${AIPERF_SERVER_URL}" + +# STP and accuracy evaluations must never inherit synthetic acceptance. +unset SGLANG_SIMULATE_ACC_LEN SGLANG_SIMULATE_ACC_METHOD SGLANG_SIMULATE_ACC_TOKEN_MODE +SPECULATIVE_ARGS=() +case "$SPEC_DECODING" in + none) + echo "Non-speculative decoding; synthetic acceptance disabled" + ;; + mtp) + # Existing measured curve: dsv41flash_dspark.yaml, thinking_on, K5. + SPECULATIVE_ARGS=(--speculative-algorithm DSPARK --speculative-dspark-block-size 5) + if [[ "$EVAL_ONLY" != true ]]; then + export SGLANG_SIMULATE_ACC_LEN=3.51 + export SGLANG_SIMULATE_ACC_METHOD=match-expected + export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token + fi + ;; + *) + echo "Unsupported SPEC_DECODING: $SPEC_DECODING" >&2 + exit 1 + ;; +esac + +# At high TP2 concurrency, test more prefill duty against the matched C64 +# baseline. Keep the latency-oriented cadence on low-C and TP4 points. +PREFILL_DECODE_INTERVAL=16 +if (( TP == 2 && CONC >= 32 )); then + PREFILL_DECODE_INTERVAL=4 +fi + +SGLANG_CMD=( + python3 -m sglang.launch_server + --model-path "$MODEL_PATH" --served-model-name "$MODEL" + --host 0.0.0.0 --port "$PORT" + --trust-remote-code + # Feed mmap weight copies sequentially from shared Lustre storage. + --weight-loader-prefetch-checkpoints + --tp "$TP" --ep-size "$EP_SIZE" + # Backends resolve automatically (dsv4 / flashinfer_mxfp4 / flashinfer_cutedsl + # on Blackwell); the cookbook warns that overriding them costs decode speed. + # Bound transient prefill allocations: the sparse-attention indexer and + # DSpark buffers scale with the chunk times the 1M context. Static KV + # memory is selected above from the measured TP2/TP4 weight footprints. + --mem-fraction-static "$MEM_FRACTION_STATIC" + --chunked-prefill-size 4096 + # Long AgentX prefills otherwise starve ready decode requests. + --prefill-decode-interval "$PREFILL_DECODE_INTERVAL" + "${SPECULATIVE_ARGS[@]}" + --max-running-requests "$MAX_RUNNING_REQUESTS" + --swa-prefix-tails "$SWA_PREFIX_TAILS" + --cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS" + --reasoning-parser auto + --tool-call-parser auto + # Draft-token forward passes under long-context agentic load block the + # scheduler long enough to trip the 1800 s default watchdog mid-warmup. + --watchdog-timeout 3600 + --enable-metrics +) +write_command "$RESULT_DIR/sglang_command.txt" "${SGLANG_CMD[@]}" +{ + echo "=== SGLANG_* env vars at launch ===" + env | grep -E '^SGLANG_' | sort + echo "===================================" +} | tee "$SERVER_LOG" +"${SGLANG_CMD[@]}" >> "$SERVER_LOG" 2>&1 & +SERVER_PID=$! +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +if [[ "${EVAL_ONLY}" == true ]]; then + run_eval --port "$PORT" +else + build_replay_cmd "$RESULT_DIR" + REPLAY_CMD+=" --server-metrics ${AIPERF_SERVER_METRICS_URLS}" + run_agentic_replay_and_write_outputs "$RESULT_DIR" +fi diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 97d4aa3934..9988ca1dc3 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8701,3 +8701,21 @@ qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg: tp: 2 ep: 2 dp-attn: false + +# Official 2026-09-21 CUDA 13 nightly, amd64 manifest pinned for B300. +# Latest multiarch index: sha256:987c7e4bd26918647211a5dcad72a2bdf2a5f394ac1469eff730e7517fc139be. +# Uses the cookbook's automatic backends and the existing B200 AgentX memory bounds. +dsv41flash-fp4-b300-sglang-agentic-dspark: + image: lmsysorg/sglang:nightly-dev-cu13-20260922-582389ce@sha256:35ea4d321b0735051dcce2599fc495853ad1ef30f6a1908f0a75e74204362d14 + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:b300-dsxe + precision: fp4 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } + - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index f1e39af598..60fd748767 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -435,6 +435,8 @@ sweep remains required. Draft precision remains upstream default. The B200 launcher also converts pinned Docker digests to the installed Enroot manifest-reference syntax and stops immediately on import failure. +DSpark uses the default precision shipped by the pinned official nightly, without custom draft quantization or precision patches. STP loads no draft; full accuracy and performance validation are still required. + DSpark is the checkpoint's own bundled draft. SGLang exposes no EAGLE or MTP path and no `--speculative-num-steps` knob for it; the recipes pass `--speculative-algorithm DSPARK --speculative-dspark-block-size 5`. Throughput uses the same diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 2f4ed95c9f..dc4b4cfcbb 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -378,6 +378,8 @@ TP2 每 GPU 加载约 147.76 GiB 的目标和草稿权重。配方校验固定 B200 启动器还将固定 Docker digest 转为已安装 Enroot 支持的 manifest 引用格式,并在导入失败时立即停止。 +DSpark 使用固定官方 nightly 默认提供的精度,不应用自定义草稿量化或精度补丁。STP 不加载草稿模型;完整准确率和性能验证仍然必需。 + DSpark 是检查点自带的草稿模型。SGLang 对它不提供 EAGLE 或 MTP 路径,也没有 `--speculative-num-steps` 参数;配方传入 `--speculative-algorithm DSPARK --speculative-dspark-block-size 5`。吞吐测试通过 `SGLANG_SIMULATE_ACC_LEN`(`match-expected`、 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 1941379e55..bf1d5652d0 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8633,3 +8633,17 @@ - "Test TP2 loading with the unmodified pinned stock loader and expandable CUDA allocator segments alone; preserve tensor payloads, layouts, and computation" - "Resolve local model snapshots, retain robust AgentX HTTP timeouts, and normalize digest-pinned images for the installed Enroot parser with immediate import failure handling" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3346 + +- config-keys: + - dsv41flash-fp4-b300-sglang-agentic-dspark + scenario-type: + - agentic-coding + description: + - Add B300 SGLang DeepSeek-V4.1-Flash AgentX TP4/EP4 with the newest official CUDA 13 nightly (20260922-582389ce), pinned amd64 digest, native DSpark5, measured thinking-on AL 3.51 for throughput and real verification for evals. + - Keep automatic backends, host Engram tables, GPU-resident KV, bounded prefill and decode graphs; route the B300 SGLang runtime through /ix. + - Use per-rank anonymous host Engram tables to allow transparent huge pages without changing host settings; preserve native FP8 table payload and automatic backend selection. + - "Use the default MTP/DSpark precision shipped by the official nightly, without custom draft quantization or precision patches." + - "Add DSpark TP2/EP2 alongside TP4/EP4 to explore the vLLM parallelism frontier, keeping native host Engram tables and GPU KV; interleave 16 decode batches between prefill chunks to test long-context responsiveness." + - "Submit this SGLang AgentX lane through a Slurm batch allocation and launch its steps from the compute node, avoiding failed interactive allocation notifications on the runner login host while preserving GPU reservations and terminal-status checks." + - "Test retained SWA tails at 64 times concurrency, bounded to 128–4096; use static memory 0.80 for TP2 C16 and 0.85 at C32 and above after measuring its smaller KV budget and 43.59 GiB of post-graph headroom at C64, otherwise retain 0.70. Prefetch checkpoints for shared Lustre loading. Test prefill/decode interval 4 at TP2 C32 and above against the interval-16 baseline; keep other points at 16." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3342 diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index 6d563ddf45..a9c21d65f9 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -13,6 +13,38 @@ source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 SLURM_PARTITION="batch_1" SLURM_ACCOUNT="benchmark" +# This lane's interactive allocation notifications fail on login-02, while +# batch submission and steps launched from the allocated node work. Keep the +# workaround scoped to this recipe and use normal Slurm resource accounting. +if [[ "$IS_MULTINODE" != true && "${MODEL_PREFIX:-}" == dsv41flash && + "${FRAMEWORK:-}" == sglang && "${IS_AGENTIC:-}" == 1 && + "${B300_AGENTX_BATCH:-}" != 1 ]]; then + check_env_vars GITHUB_WORKSPACE GPU_COUNT RUNNER_NAME + BATCH_SCRIPT=$(mktemp "${RUNNER_TEMP:-$GITHUB_WORKSPACE}/b300-agentx.XXXXXX.sh") || exit 1 + BATCH_LOG="${BATCH_SCRIPT%.sh}.log" + { + printf '#!/usr/bin/env bash\nexport B300_AGENTX_BATCH=1\nexec bash ' + printf '%q\n' "$GITHUB_WORKSPACE/runners/launch_b300-dsxe.sh" + } > "$BATCH_SCRIPT" + BATCH_ARGS=(--parsable --partition="$SLURM_PARTITION" --account="$SLURM_ACCOUNT" + --nodes=1 --ntasks=1 --gres="gpu:$GPU_COUNT" --exclusive --mem=0 + --time="$SALLOC_TIME_LIMIT" --job-name="$RUNNER_NAME" --export=ALL + --chdir="$GITHUB_WORKSPACE" --output="$BATCH_LOG") + if [[ -n "${SALLOC_EXCLUDE:-}" ]]; then + BATCH_ARGS+=(--exclude="$SALLOC_EXCLUDE") + fi + JOB_ID=$(sbatch "${BATCH_ARGS[@]}" "$BATCH_SCRIPT") || { rm -f "$BATCH_SCRIPT"; exit 1; } + JOB_ID="${JOB_ID%%;*}" + [[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo 'ERROR: B300 batch allocation unavailable' >&2; rm -f "$BATCH_SCRIPT"; exit 1; } + trap 'rc=$?; scancel "$JOB_ID" 2>/dev/null || true; rm -f "$BATCH_SCRIPT"; exit "$rc"' EXIT + trap 'exit 130' INT + trap 'exit 143' TERM + echo "B300 AgentX batch job $JOB_ID; log: $BATCH_LOG" + stream_slurm_job_log "$JOB_ID" "$BATCH_LOG" || exit 1 + verify_slurm_job_status "$JOB_ID" + exit $? +fi + # enroot squash images. Must be on storage every compute node mounts and writable # by the runner user (/data/squash is root-owned, hence the per-user default). SQUASH_DIR="/data/home/sa-gha-runner/squash" @@ -67,8 +99,13 @@ import_squash_image() { return 0 fi - srun -N 1 -A "$SLURM_ACCOUNT" -p "$SLURM_PARTITION" \ - --time="${ENROOT_IMPORT_TIME_LIMIT}" bash -c " + local import_launcher=(srun -N 1 -A "$SLURM_ACCOUNT" -p "$SLURM_PARTITION" + --time="${ENROOT_IMPORT_TIME_LIMIT}") + if [[ "${B300_AGENTX_BATCH:-}" == 1 ]]; then + # Already inside our exclusive compute-node allocation. + import_launcher=() + fi + "${import_launcher[@]}" bash -c " set -eo pipefail exec 9>\"$lock\" flock -w 3600 9 @@ -318,7 +355,7 @@ else fi # Keep all new AgentX runtime directories outside /workspace. - if [[ "$MODEL_PREFIX" == "dsv41flash" && "$FRAMEWORK" == "vllm" ]]; then + if [[ "$MODEL_PREFIX" == "dsv41flash" && ( "$FRAMEWORK" == "vllm" || "$FRAMEWORK" == "sglang" ) ]]; then CONTAINER_MOUNT_DIR=/ix export INFMAX_CONTAINER_WORKSPACE=/ix export RESULT_DIR=/ix/results @@ -346,13 +383,17 @@ else SALLOC_ARGS+=(--exclude="$SALLOC_EXCLUDE") fi # Capture this allocation's ID; a runner name can also match an older job. - JOB_ID=$( + if [[ "${B300_AGENTX_BATCH:-}" == 1 ]]; then + JOB_ID="${SLURM_JOB_ID:?B300 batch execution requires a Slurm allocation}" + else + JOB_ID=$( set -o pipefail LC_ALL=C salloc "${SALLOC_ARGS[@]}" 2>&1 | tee /dev/stderr | sed -n 's/.*Granted job allocation \([0-9][0-9]*\)$/\1/p' - ) || exit 1 - [[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo 'ERROR: B300 allocation unavailable' >&2; exit 1; } - trap 'rc=$?; scancel "$JOB_ID" 2>/dev/null || true; exit "$rc"' EXIT + ) || exit 1 + [[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo 'ERROR: B300 allocation unavailable' >&2; exit 1; } + trap 'rc=$?; scancel "$JOB_ID" 2>/dev/null || true; exit "$rc"' EXIT + fi if [[ "$MODEL_MOUNT_DIR" == "$MODEL_ROOT" ]]; then # MODEL_ROOT is node-local: probe the allocated compute node, not the login host. srun --jobid="$JOB_ID" test -r "$MODEL_PATH/config.json" || { @@ -377,8 +418,14 @@ else fi CONTAINER_MOUNTS_ARG=$(IFS=,; printf '%s' "${CONTAINER_MOUNTS[*]}") + B300_CONTAINER_MPI=none + if [[ "${B300_AGENTX_BATCH:-}" == 1 ]]; then + # The installed Enroot hook sees PMIx variables in batch jobs. Use the + # supported plugin so its required per-step mount directories exist. + B300_CONTAINER_MPI=pmix + fi srun --jobid="$JOB_ID" \ - --mpi=none \ + --mpi="$B300_CONTAINER_MPI" \ --container-image="$SQUASH_FILE" \ --container-mounts="$CONTAINER_MOUNTS_ARG" \ --no-container-mount-home \