From b81a9e64b33e71b6335a48a5958568bd6883158c Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 03:46:40 -0400 Subject: [PATCH 01/12] Migrate 6 SPEED-Bench AL collectors to native srt-slurm single-node path Replace the legacy bash collector scripts (dsr1, dsv4, glm5, glm52, qwen3.5, qwen3.8next) with per-model srt-slurm recipe YAMLs and a shared srt_speedbench.sh client. Each recipe uses zip_override groups to express the full (thinking mode x MTP 1-8) matrix, and single_node.py now detects collector mode to skip ISL/OSL/ RANDOM_RANGE_RATIO/USE_CHAT_TEMPLATE validation while binding CATEGORY, SPEEDBENCH_OUTPUT_LEN, and optional CHAT_TEMPLATE_KWARGS_ON into the per-cell benchmark environment. The workflow routes migrated model-prefixes through the srt-slurm path (one submission per cell) and keeps draft-model collectors (dsv4dspark, dsv4dsparkprob, kimik3, minimaxm3) on the legacy BENCH_SCRIPT_OVERRIDE path. An aggregation module (speedbench_matrix.py) reads per-cell result JSONs and produces the golden YAML matrix. Co-Authored-By: Claude Opus 4.6 --- .github/workflows/speedbench-al.yml | 78 ++++- .../speedbench/dsr1_fp4_b300_vllm.sh | 212 ------------- .../speedbench/dsv4_fp4_b300_vllm.sh | 236 -------------- .../speedbench/glm52_fp4_b300_vllm.sh | 228 ------------- .../speedbench/glm5_fp4_b300_vllm.sh | 243 -------------- .../speedbench/qwen3.5_fp4_b300_vllm.sh | 254 --------------- .../speedbench/qwen3.8next_fp4_b300_vllm.sh | 300 ------------------ .../vllm/b300-fp4-speedbench/speedbench.yaml | 94 ++++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 131 ++++++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 132 ++++++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 129 ++++++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 139 ++++++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 150 +++++++++ benchmarks/single_node/srt_speedbench.sh | 203 ++++++++++++ golden_al_distribution/README.md | 8 +- golden_al_distribution/README_zh.md | 8 +- infx/srt_slurm/single_node.py | 33 +- infx/workflows/speedbench_matrix.py | 71 +++++ runners/launch_b300-dsxe.sh | 4 +- utils/test_speedbench_matrix.py | 47 +++ utils/test_srt_single_node.py | 153 ++++++++- 21 files changed, 1363 insertions(+), 1490 deletions(-) delete mode 100755 benchmarks/single_node/speedbench/dsr1_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/dsv4_fp4_b300_vllm.sh delete mode 100644 benchmarks/single_node/speedbench/glm52_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/glm5_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/qwen3.5_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/qwen3.8next_fp4_b300_vllm.sh create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt_speedbench.sh create mode 100644 infx/workflows/speedbench_matrix.py create mode 100644 utils/test_speedbench_matrix.py diff --git a/.github/workflows/speedbench-al.yml b/.github/workflows/speedbench-al.yml index de6aaf6b88..a077c1d8b2 100644 --- a/.github/workflows/speedbench-al.yml +++ b/.github/workflows/speedbench-al.yml @@ -105,8 +105,6 @@ env: EP_SIZE: '1' DP_ATTENTION: 'false' SPEC_DECODING: mtp - # Run the AL-matrix collector instead of the auto-selected throughput script. - BENCH_SCRIPT_OVERRIDE: benchmarks/single_node/speedbench/${{ inputs.model-prefix }}_fp4_b300_vllm.sh SALLOC_TIME_LIMIT: ${{ inputs.salloc-time }} # Matrix-collector tunables (propagated into the container via srun --export=ALL). MTP_LIST: ${{ inputs.mtp-list }} @@ -239,7 +237,77 @@ jobs: if [[ -f runners/runtime_settings.sh ]]; then source runners/runtime_settings.sh fi - bash ./runners/launch_${RUNNER_NAME%%_*}.sh + + # Migrated native-MTP prefixes use the srt-slurm single-node path. + # Draft-model collectors (dsv4dspark*, kimik3, minimaxm3) keep the + # legacy BENCH_SCRIPT_OVERRIDE path. + _is_srt_slurm_prefix() { + case "$1" in + dsr1|dsv4|glm5|glm52|qwen3.5|qwen3.8next) return 0 ;; + *) return 1 ;; + esac + } + + if _is_srt_slurm_prefix "$MODEL_PREFIX"; then + # --- srt-slurm per-cell collection --- + RECIPE_PATH="benchmarks/single_node/srt-slurm-recipes/${MODEL_PREFIX}/vllm/b300-fp4-speedbench/speedbench.yaml" + if [[ ! -f "$RECIPE_PATH" ]]; then + echo "ERROR: SPEED-Bench recipe not found: $RECIPE_PATH" >&2 + exit 1 + fi + mkdir -p speedbench_results + ALL_FAILED=true + # Harmless values for launch_srt_single_node env checks; the collector + # recipe and client do not read them. + export ISL=256 + export OSL=256 + export RANDOM_RANGE_RATIO=0.0 + export CONC=1 + export PP_SIZE=1 + export DCP_SIZE=1 + export PCP_SIZE=1 + for mode in $THINKING_MODES; do + for mtp in $MTP_LIST; do + IDX=$((mtp - 1)) + export SRT_RECIPE="${RECIPE_PATH}:zip_override_thinking_${mode}[${IDX}]" + export RESULT_FILENAME="speedbench_${mode}_mtp${mtp}" + echo "" + echo "==========================================" + echo " srt-slurm cell: thinking=${mode} MTP=${mtp}" + echo " SRT_RECIPE=${SRT_RECIPE}" + echo "==========================================" + CELL_RC=0 + bash ./runners/launch_"${RUNNER_NAME%%_*}".sh || CELL_RC=$? + # Collect per-cell artifacts into the results directory. + if [[ -f "${RESULT_FILENAME}.json" ]]; then + mv "${RESULT_FILENAME}.json" "speedbench_results/${RESULT_FILENAME}.json" + ALL_FAILED=false + fi + if [[ -f srt-single-node-logs.tar.gz ]]; then + mv srt-single-node-logs.tar.gz "speedbench_results/srt-logs_${mode}_mtp${mtp}.tar.gz" + fi + if [[ "$CELL_RC" -ne 0 ]]; then + echo " -> cell failed (rc=$CELL_RC), recording N/A" + fi + done + done + if [[ "$ALL_FAILED" == true ]]; then + echo "ERROR: every SPEED-Bench cell failed" >&2 + exit 1 + fi + # Aggregate per-cell result JSONs into the reference YAML. + PYTHONPATH="${GITHUB_WORKSPACE}${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.workflows.speedbench_matrix \ + --result-dir speedbench_results \ + --model-key "$(basename "$(echo "$MODEL" | tr '[:upper:]' '[:lower:]')")" \ + --thinking-modes "$THINKING_MODES" \ + --mtp-list "$MTP_LIST" \ + > speedbench-reference-al.yaml + else + # --- Legacy collector script --- + export BENCH_SCRIPT_OVERRIDE="benchmarks/single_node/speedbench/${MODEL_PREFIX}_fp4_b300_vllm.sh" + bash ./runners/launch_"${RUNNER_NAME%%_*}".sh + fi if [ ! -f "speedbench-reference-al.yaml" ]; then echo "AL collection failed: speedbench-reference-al.yaml not produced." >&2 @@ -292,7 +360,9 @@ jobs: uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: speedbench_server_logs-${{ inputs.model-prefix }} - path: speedbench_results/server_*.log + path: | + speedbench_results/server_*.log + speedbench_results/srt-logs_*.tar.gz if-no-files-found: ignore # Per-request benchmark detail (vllm bench serve --save-detailed): includes diff --git a/benchmarks/single_node/speedbench/dsr1_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/dsr1_fp4_b300_vllm.sh deleted file mode 100755 index a4cb56935b..0000000000 --- a/benchmarks/single_node/speedbench/dsr1_fp4_b300_vllm.sh +++ /dev/null @@ -1,212 +0,0 @@ -#!/usr/bin/env bash - -# DeepSeek-R1 B300 vLLM SPEED-Bench AL matrix collector. -# -# For each MTP level (num_speculative_tokens), measure the REAL acceptance length (AL) -# on one SPEED-Bench category and emit a YAML matrix in the golden_al_distribution -# shape. The synthetic value is injected downstream by the throughput recipe, not here. -# -# R1 is DeepSeek-V3 architecture (plain MLA), not V4: no deepseek_v4 tokenizer/parsers -# and no --attention_config.use_fp4_indexer_cache (dsv32/MLA-indexer only). AL is read -# from /metrics, so a reasoning parser is irrelevant. --kv-cache-dtype fp8 is kept so -# all golden AL values share one kv-cache numeric regime. R1 always emits (no -# enable_thinking toggle), so only thinking_on is measured and no chat-template-kwargs -# shim is needed. Checkpoint: nvidia/DeepSeek-R1-0528-NVFP4-v2, basename dsr1-fp4 on -# the runner. -# -# Dispatch this collector through speedbench-al.yml. -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY DP_ATTENTION MODEL MODEL_PATH MTP_LIST OUT_YAML \ - PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -CONCURRENCY="1" -# Sampling from the DeepSeek-R1 generation_config (temperature 0.6, top_p 0.95; no -# top_k). vLLM's default top_p is 1.0, so it MUST be passed or the AL is measured at -# the wrong settings. -TEMPERATURE="0.6" -TOP_P="0.95" - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -# Flat results dir to match the speedbench-al.yml artifact glob -# (speedbench_results/server_*.log) and its pre-run `rm -rf speedbench_results`. -RESULTS_DIR="/workspace/speedbench_results" - -# FP4 MoE on Blackwell needs FlashInfer (vLLM R1 docs). -export VLLM_USE_FLASHINFER_MOE_FP4="1" -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi -if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode MTP=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --trust-remote-code - --enable-expert-parallel - --kv-cache-dtype fp8 - --no-enable-prefix-caching - --max-model-len 16384 - --speculative-config "{\"method\": \"mtp\", \"num_speculative_tokens\": $mtp}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$TEMPERATURE" \ - --top-p "$TOP_P" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode MTP=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | temperature: $TEMPERATURE | top_p: $TOP_P | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# DeepSeek-R1 always reasons (no thinking-off mode), so only thinking_on is emitted." - echo "# Measured on $MODEL_KEY (B300, vLLM MTP), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/dsr1_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (MTP level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" diff --git a/benchmarks/single_node/speedbench/dsv4_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/dsv4_fp4_b300_vllm.sh deleted file mode 100755 index 27c87b3401..0000000000 --- a/benchmarks/single_node/speedbench/dsv4_fp4_b300_vllm.sh +++ /dev/null @@ -1,236 +0,0 @@ -#!/usr/bin/env bash - -# DSV4-Pro B300 vLLM SPEED-Bench AL matrix collector. -# -# For each thinking mode (on/off) and MTP level (num_speculative_tokens), measure the -# acceptance length (AL) on one SPEED-Bench category and emit a YAML matrix in the -# golden_al_distribution shape. Wired into the speedbench-al.yml GitHub Action. -# -# Dispatch this collector through speedbench-al.yml. -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -# MODEL_PATH is the launcher-resolved weights dir (e.g. /scratch/models/DeepSeek-V4-Pro). -# A leading "/" makes the download guard below a no-op. -SERVE_MODEL="${MODEL_PATH}" - -# Top-level key in the emitted YAML matrix comes from the model basename. -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -CONCURRENCY="1" -TEMPERATURE="1.0" -# MUST match the golden config: benchmarks/speedbench-reference-al.yaml was measured -# with reasoning_effort=high. - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi -if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -if [[ " $THINKING_MODES " == *" on "* ]]; then - if ! apply_chat_template_kwargs_shim; then - echo "CRITICAL: --chat-template-kwargs shim failed — aborting" - exit 1 - fi -fi - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi -EP_ARGS=() -if [ "${EP_SIZE}" -gt 1 ]; then - EP_ARGS=(--enable-expert-parallel) -fi -MOE_ARGS=() -if [ "${DP_ATTENTION}" = "true" ]; then - MOE_ARGS=(--moe-backend deep_gemm_mega_moe) -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -# Descendant PIDs of $1 by PARENT pid. This can never include this script (an -# ancestor of the server), unlike a name-based `pkill -f vllm`, which self-killed -# because the script filename contains "vllm". -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - # Snapshot the worker/EngineCore subprocesses BEFORE killing the parent: once it - # dies the children reparent to init and the tree link is lost. An orphaned - # worker holds GPU memory and OOMs the next server start. - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - if [[ "$mode" == "on" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode MTP=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --kv-cache-dtype fp8 - --trust-remote-code - --block-size 256 - --no-enable-prefix-caching - "${EP_ARGS[@]}" - "${MOE_ARGS[@]}" - --compilation-config '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' - --attention_config.use_fp4_indexer_cache True - --tokenizer-mode deepseek_v4 - --tool-call-parser deepseek_v4 - --enable-auto-tool-choice - --reasoning-parser deepseek_v4 - --max-cudagraph-capture-size 2048 - --max-model-len 16384 - --speculative-config "{\"method\": \"mtp\", \"num_speculative_tokens\": $mtp}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --tokenizer-mode deepseek_v4 \ - --temperature "$TEMPERATURE" \ - "${think_args[@]}" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode MTP=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | temperature: $TEMPERATURE | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - echo "# Measured on $MODEL_KEY (B300, vLLM MTP), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/dsv4_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (MTP level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" diff --git a/benchmarks/single_node/speedbench/glm52_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/glm52_fp4_b300_vllm.sh deleted file mode 100644 index 07f85ddcce..0000000000 --- a/benchmarks/single_node/speedbench/glm52_fp4_b300_vllm.sh +++ /dev/null @@ -1,228 +0,0 @@ -#!/usr/bin/env bash - -# GLM-5.2 B300 vLLM SPEED-Bench AL matrix collector. -# -# Same serve parameters, sampling and thinking kwargs as glm5_fp4_b300_vllm.sh (GLM-5.2 -# shares the glm_moe_dsa architecture, MTP head and chat template), plus a download -# guard for the not-yet-staged checkpoint. -# -# Dispatch this collector through speedbench-al.yml. -# -# Tunables (env): same as glm5_fp4_b300_vllm.sh - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" -GPU_MEM_UTIL="0.80" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -CONCURRENCY="1" -TEMPERATURE="1.0" -TOP_P="0.95" -CHAT_TEMPLATE_KWARGS_OFF='{"enable_thinking": false}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - echo "=== MODEL_PATH ($MODEL_PATH) is empty, downloading $MODEL ===" - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -# --chat-template-kwargs is consumed natively by `vllm bench serve` here: GLM-5.2 only -# loads on the dedicated vLLM image (>=0.23), which carries vllm-project/vllm#44244, -# so no client-side shim is needed. - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi -EP_ARGS=() -if [ "${EP_SIZE}" -gt 1 ]; then - EP_ARGS=(--enable-expert-parallel) -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - if [[ "$mode" == "on" && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - elif [[ "$mode" == "off" && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode MTP=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --kv-cache-dtype fp8 - --trust-remote-code - --no-enable-prefix-caching - "${EP_ARGS[@]}" - --reasoning-parser glm45 - --tool-call-parser glm47 - --enable-auto-tool-choice - --chat-template-content-format=string - --gpu-memory-utilization "$GPU_MEM_UTIL" - --max-model-len 16384 - --speculative-config "{\"method\": \"mtp\", \"num_speculative_tokens\": $mtp}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$TEMPERATURE" \ - --top-p "$TOP_P" \ - "${think_args[@]}" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode MTP=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | temperature: $TEMPERATURE | top_p: $TOP_P | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - echo "# thinking_off chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - echo "# Measured on $MODEL_KEY (B300, vLLM MTP), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/glm52_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (MTP level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" diff --git a/benchmarks/single_node/speedbench/glm5_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/glm5_fp4_b300_vllm.sh deleted file mode 100755 index 13c1f0eb24..0000000000 --- a/benchmarks/single_node/speedbench/glm5_fp4_b300_vllm.sh +++ /dev/null @@ -1,243 +0,0 @@ -#!/usr/bin/env bash - -# GLM-5 B300 vLLM SPEED-Bench AL matrix collector. -# -# For each thinking mode (on/off) and MTP level (num_speculative_tokens), measure the -# REAL acceptance length (AL) on one SPEED-Bench category and emit a YAML matrix in -# the golden_al_distribution shape. The synthetic value is injected downstream by the -# throughput recipe, not here. Serves the NVFP4 build (GLM-5-NVFP4), like every model -# in this matrix. -# -# GLM requires --chat-template-content-format=string (vLLM docs). Do NOT pass -# --attention_config.use_fp4_indexer_cache: despite GLM-5 also being DSA sparse -# attention, that knob is read only by vllm/models/deepseek_v4/attention.py and the -# MLA indexer backend; GLM's DSA (GlmMoeDsaForCausalLM) never reads it. Thinking is ON -# by default for GLM, so the OFF cell MUST pass enable_thinking:false explicitly. -# -# Usage (inside the GLM vLLM container, on a B300 node): -# export MODEL=/scratch/models/GLM-5-NVFP4 -# bash benchmarks/single_node/speedbench/glm5_fp4_b300_vllm.sh -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" -# NVIDIA's GLM-5-NVFP4 model card serves with 0.80; NVFP4 + DSA + MTP draft -# layers leave less headroom than DSV4, so match it to avoid startup OOM. -GPU_MEM_UTIL="0.80" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -CONCURRENCY="1" -# Sampling from the GLM-5 generation_config.json (temperature 1.0, top_p 0.95). vLLM's -# default top_p is 1.0, so it MUST be passed or the AL is measured at the wrong settings. -TEMPERATURE="1.0" -TOP_P="0.95" -# GLM thinking toggles via the enable_thinking chat_template key (default ON). -CHAT_TEMPLATE_KWARGS_OFF='{"enable_thinking": false}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -# Flat results dir to match the speedbench-al.yml artifact glob -# (speedbench_results/server_*.log) and its pre-run `rm -rf speedbench_results`. -RESULTS_DIR="/workspace/speedbench_results" - -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi -if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -NEED_SHIM=0 -if [[ " $THINKING_MODES " == *" on "* && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then NEED_SHIM=1; fi -if [[ " $THINKING_MODES " == *" off "* && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then NEED_SHIM=1; fi -if [[ "$NEED_SHIM" == "1" ]]; then - if ! apply_chat_template_kwargs_shim; then - echo "CRITICAL: --chat-template-kwargs shim failed — aborting" - exit 1 - fi -fi - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi -EP_ARGS=() -if [ "${EP_SIZE}" -gt 1 ]; then - EP_ARGS=(--enable-expert-parallel) -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - if [[ "$mode" == "on" && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - elif [[ "$mode" == "off" && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode MTP=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --kv-cache-dtype fp8 - --trust-remote-code - --no-enable-prefix-caching - "${EP_ARGS[@]}" - --reasoning-parser glm45 - --tool-call-parser glm47 - --enable-auto-tool-choice - --chat-template-content-format=string - --gpu-memory-utilization "$GPU_MEM_UTIL" - --max-model-len 16384 - --speculative-config "{\"method\": \"mtp\", \"num_speculative_tokens\": $mtp}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$TEMPERATURE" \ - --top-p "$TOP_P" \ - "${think_args[@]}" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode MTP=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | temperature: $TEMPERATURE | top_p: $TOP_P | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - echo "# thinking_off chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - echo "# Measured on $MODEL_KEY (B300, vLLM MTP), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/glm5_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (MTP level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" diff --git a/benchmarks/single_node/speedbench/qwen3.5_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/qwen3.5_fp4_b300_vllm.sh deleted file mode 100755 index 6395477576..0000000000 --- a/benchmarks/single_node/speedbench/qwen3.5_fp4_b300_vllm.sh +++ /dev/null @@ -1,254 +0,0 @@ -#!/usr/bin/env bash - -# Qwen3.5-397B-A17B B300 vLLM SPEED-Bench AL matrix collector. -# -# For each thinking mode (on/off) and MTP level (num_speculative_tokens), measure the -# REAL acceptance length (AL) on one SPEED-Bench category and emit a YAML matrix in -# the golden_al_distribution shape. The synthetic value is injected downstream by the -# throughput recipe, not here. Serves the NVFP4 build (Qwen3.5-397B-A17B-NVFP4). -# -# --max-cudagraph-capture-size 512: Qwen3.5 is a mamba hybrid and a large capture -# size trips the causal_conv1d assert (vLLM PR #34571). Thinking OFF must be passed -# explicitly via enable_thinking (Qwen does not treat "no kwargs" as off). -# -# Dispatch this collector through speedbench-al.yml. -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -CONCURRENCY="1" -# Model-card sampling DIFFERS by mode and MUST be passed per-mode or the AL is -# measured at the wrong settings: -# thinking : temperature 0.6, top_p 0.95, top_k 20, presence_penalty 0.0 -# instruct : temperature 0.7, top_p 0.8, top_k 20, presence_penalty 1.5 -TEMPERATURE_ON="0.6"; TOP_P_ON="0.95"; TOP_K_ON="20"; PRESENCE_PENALTY_ON="0.0" -TEMPERATURE_OFF="0.7"; TOP_P_OFF="0.8"; TOP_K_OFF="20"; PRESENCE_PENALTY_OFF="1.5" -# Unset -> vLLM default (deterministic seed=0); vary it to measure temperature>0 -# variance. -SEED="${SEED:-}" -# --save-detailed keeps per-request completions to eyeball that thinking_on emits -# and thinking_off does not; off by default (bloats the result JSON). -SAVE_DETAILED="${SAVE_DETAILED:-}" -# Qwen thinking toggles via the enable_thinking chat_template key. -CHAT_TEMPLATE_KWARGS_OFF='{"enable_thinking": false}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -# Flat results dir to match the speedbench-al.yml artifact glob -# (speedbench_results/server_*.log) and its pre-run `rm -rf speedbench_results`. -RESULTS_DIR="/workspace/speedbench_results" - -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi -if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -NEED_SHIM=0 -if [[ " $THINKING_MODES " == *" on "* && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then NEED_SHIM=1; fi -if [[ " $THINKING_MODES " == *" off "* && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then NEED_SHIM=1; fi -if [[ "$NEED_SHIM" == "1" ]]; then - if ! apply_chat_template_kwargs_shim; then - echo "CRITICAL: --chat-template-kwargs shim failed — aborting" - exit 1 - fi -fi - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi -EP_ARGS=() -if [ "${EP_SIZE}" -gt 1 ]; then - EP_ARGS=(--enable-expert-parallel) -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - local temp top_p top_k pp - if [[ "$mode" == "on" ]]; then - [[ -n "$CHAT_TEMPLATE_KWARGS_ON" ]] && think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - temp="$TEMPERATURE_ON"; top_p="$TOP_P_ON"; top_k="$TOP_K_ON"; pp="$PRESENCE_PENALTY_ON" - else - [[ -n "$CHAT_TEMPLATE_KWARGS_OFF" ]] && think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - temp="$TEMPERATURE_OFF"; top_p="$TOP_P_OFF"; top_k="$TOP_K_OFF"; pp="$PRESENCE_PENALTY_OFF" - fi - local seed_args=() - [[ -n "$SEED" ]] && seed_args=(--seed "$SEED") - local detail_args=() - [[ -n "$SAVE_DETAILED" ]] && detail_args=(--save-detailed) - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode MTP=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --kv-cache-dtype fp8 - --trust-remote-code - --no-enable-prefix-caching - "${EP_ARGS[@]}" - --reasoning-parser qwen3 - --tool-call-parser qwen3_coder - --enable-auto-tool-choice - --language-model-only - --max-cudagraph-capture-size 512 - --max-model-len 16384 - --speculative-config "{\"method\": \"mtp\", \"num_speculative_tokens\": $mtp}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$temp" \ - --top-p "$top_p" \ - --top-k "$top_k" \ - --presence-penalty "$pp" \ - "${seed_args[@]}" \ - "${detail_args[@]}" \ - "${think_args[@]}" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode MTP=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on : temp $TEMPERATURE_ON top_p $TOP_P_ON top_k $TOP_K_ON presence_penalty $PRESENCE_PENALTY_ON | chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - echo "# thinking_off: temp $TEMPERATURE_OFF top_p $TOP_P_OFF top_k $TOP_K_OFF presence_penalty $PRESENCE_PENALTY_OFF | chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - echo "# Measured on $MODEL_KEY (B300, vLLM MTP), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/qwen3.5_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (MTP level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" diff --git a/benchmarks/single_node/speedbench/qwen3.8next_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/qwen3.8next_fp4_b300_vllm.sh deleted file mode 100755 index 8c0e8ad865..0000000000 --- a/benchmarks/single_node/speedbench/qwen3.8next_fp4_b300_vllm.sh +++ /dev/null @@ -1,300 +0,0 @@ -#!/usr/bin/env bash - -# Qwen3.8-Flash-Next B300 vLLM SPEED-Bench AL matrix collector (native MTP). -# -# For each thinking mode (on/off) and MTP level (num_speculative_tokens), measure the -# REAL acceptance length (AL) on one SPEED-Bench category and emit a YAML matrix in -# the golden_al_distribution shape. The synthetic value is injected downstream by the -# throughput recipe, not here. Qwen3.8-Flash-Next ships a built-in MTP module, so no -# separate draft model: https://recipes.vllm.ai/Qwen/Qwen3.8-Flash-Next -# -# Recipe-required serve flags: --no-enable-flashinfer-autotune; --max-num-seqs 256 -# (avoids a Mamba-cache capacity error at startup); kv-cache dtype left at default -# for the hybrid GDN + Qwen Sparse Attention architecture (do not force fp8); -# --max-cudagraph-capture-size 512 (mamba-hybrid causal_conv1d capture-size assert); -# --language-model-only (the checkpoint is multimodal, AL is collected on text only). -# The 51B n-gram table fits in HBM at TP4, so VLLM_PLE_CPU_OFFLOAD is not needed. -# -# Dispatch (speedbench-al.yml): the image and thinking-kwargs defaults are DSV4's, -# so override both: -# gh workflow run speedbench-al.yml \ -# --repo SemiAnalysisAI/InferenceX \ -# --ref BRANCH \ -# -f runner=b300 \ -# -f model=Qwen/Qwen3.8-Flash-Next-FP8 \ -# -f model-prefix=qwen3.8next \ -# -f image=vllm/vllm-openai:qwen38-flash-next \ -# -f 'mtp-list=1 2 3 4 5 6 7 8' \ -# -f 'thinking-modes=off on' \ -# -f 'thinking-kwargs={"enable_thinking": true}' \ -# -f category=coding \ -# -f output-len=4096 \ -# -f open-pr=false -# -# Dispatch this collector through speedbench-al.yml. -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" -GPU_MEM_UTIL="0.90" -MAX_NUM_SEQS="256" - -# Plain TP8 is incompatible with the official FP8 checkpoint (128-wide quantization -# blocks, per the vLLM recipe). speedbench-al.yml exports TP=8 unconditionally; fold -# it back to the recipe-validated TP4 unless the caller runs TEP (EP_SIZE>1). AL is -# GPU-count-independent, so collecting on 4 of 8 GPUs does not affect the curve. -if [[ "$TP" == "8" && "${EP_SIZE}" -le 1 ]]; then - echo "NOTE: TP=8 without expert parallelism is incompatible with the FP8 checkpoint; using recipe-validated TP=4." - TP=4 -fi - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -# AL is concurrency-independent (per-token accept/reject; no spec-disable-by-batch is -# set), so batch the SPEED-Bench pass to stay under the CI wall-time limit; conc=1 -# blew the 8h budget on Kimi-K3. -CONCURRENCY="64" -# Model-card sampling DIFFERS by mode and MUST be passed per-mode or the AL is -# measured at the wrong settings: -# thinking : temperature 1.0, top_p 0.95, top_k 20, presence_penalty 0.0 -# instruct : temperature 0.7, top_p 0.80, top_k 20, presence_penalty 1.5 -TEMPERATURE_ON="1.0"; TOP_P_ON="0.95"; TOP_K_ON="20"; PRESENCE_PENALTY_ON="0.0" -TEMPERATURE_OFF="0.7"; TOP_P_OFF="0.8"; TOP_K_OFF="20"; PRESENCE_PENALTY_OFF="1.5" -# Unset -> vLLM default (deterministic seed=0); vary it to measure temperature>0 -# variance. -SEED="${SEED:-}" -# --save-detailed keeps per-request completions to eyeball that thinking_on emits -# and thinking_off does not; off by default (bloats the result JSON). -SAVE_DETAILED="${SAVE_DETAILED:-}" -# Qwen thinking toggles via enable_thinking (default ON; reasoning_effort -# stays at its xhigh default). -CHAT_TEMPLATE_KWARGS_OFF='{"enable_thinking": false}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -# Flat results dir to match the speedbench-al.yml artifact glob -# (speedbench_results/server_*.log) and its pre-run `rm -rf speedbench_results`. -RESULTS_DIR="/workspace/speedbench_results" - -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi - -# Qwen3.8-Flash-Next-FP8 is not in the launcher's STAGED_MODELS, so MODEL_PATH -# resolves into the writable models dir; download only when it is an empty dir. -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -NEED_SHIM=0 -if [[ " $THINKING_MODES " == *" on "* && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then NEED_SHIM=1; fi -if [[ " $THINKING_MODES " == *" off "* && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then NEED_SHIM=1; fi -if [[ "$NEED_SHIM" == "1" ]]; then - if ! apply_chat_template_kwargs_shim; then - echo "CRITICAL: --chat-template-kwargs support is missing and the shim failed — aborting" - exit 1 - fi -fi - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi -EP_ARGS=() -if [ "${EP_SIZE}" -gt 1 ]; then - EP_ARGS=(--enable-expert-parallel) -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - local temp top_p top_k pp - if [[ "$mode" == "on" ]]; then - [[ -n "$CHAT_TEMPLATE_KWARGS_ON" ]] && think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - temp="$TEMPERATURE_ON"; top_p="$TOP_P_ON"; top_k="$TOP_K_ON"; pp="$PRESENCE_PENALTY_ON" - else - [[ -n "$CHAT_TEMPLATE_KWARGS_OFF" ]] && think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - temp="$TEMPERATURE_OFF"; top_p="$TOP_P_OFF"; top_k="$TOP_K_OFF"; pp="$PRESENCE_PENALTY_OFF" - fi - local seed_args=() - [[ -n "$SEED" ]] && seed_args=(--seed "$SEED") - local detail_args=() - [[ -n "$SAVE_DETAILED" ]] && detail_args=(--save-detailed) - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode MTP=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --trust-remote-code - --no-enable-prefix-caching - "${EP_ARGS[@]}" - --reasoning-parser qwen3 - --tool-call-parser qwen3_coder - --enable-auto-tool-choice - --language-model-only - --no-enable-flashinfer-autotune - --gpu-memory-utilization "$GPU_MEM_UTIL" - --max-num-seqs "$MAX_NUM_SEQS" - --max-cudagraph-capture-size 512 - --max-model-len 16384 - --speculative-config "{\"method\": \"mtp\", \"num_speculative_tokens\": $mtp}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$temp" \ - --top-p "$top_p" \ - --top-k "$top_k" \ - --presence-penalty "$pp" \ - "${seed_args[@]}" \ - "${detail_args[@]}" \ - "${think_args[@]}" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode MTP=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on : temp $TEMPERATURE_ON top_p $TOP_P_ON top_k $TOP_K_ON presence_penalty $PRESENCE_PENALTY_ON | chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - echo "# thinking_off: temp $TEMPERATURE_OFF top_p $TOP_P_OFF top_k $TOP_K_OFF presence_penalty $PRESENCE_PENALTY_OFF | chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - echo "# Measured on $MODEL_KEY (B300, vLLM native MTP), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/qwen3.8next_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (MTP level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..fd53d4e7e2 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,94 @@ +# DeepSeek-R1 B300 vLLM SPEED-Bench AL matrix (native MTP). +# +# R1 always reasons (no thinking-off mode), so only zip_override_thinking_on is +# provided. Each entry varies num_speculative_tokens 1..8. +# +# FP4 MoE on Blackwell needs FlashInfer (VLLM_USE_FLASHINFER_MOE_FP4). +# R1 is DeepSeek-V3 architecture (plain MLA): no --attention_config.use_fp4_indexer_cache, +# no deepseek_v4 tokenizer/parsers. --kv-cache-dtype fp8 keeps all golden AL values +# under one KV-cache numeric regime. Sampling: temperature 0.6, top_p 0.95 +# (DeepSeek-R1 generation_config). +base: + schema: 2 + name: dsr1-fp4-b300-vllm-speedbench + model: + path: hf:nvidia/DeepSeek-R1-0528-NVFP4-v2 + container: vllm/vllm-openai:v0.21.0 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: nvidia/DeepSeek-R1-0528-NVFP4-v2 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + enable-expert-parallel: true + kv-cache-dtype: fp8 + no-enable-prefix-caching: true + max-model-len: 16384 + env: + VLLM_USE_FLASHINFER_MOE_FP4: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-NVFP4-v2 + SPEC_DECODING: mtp + TEMPERATURE: '0.6' + TOP_P: '0.95' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + +zip_override_thinking_on: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..4e3fb594f5 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,131 @@ +# DeepSeek-V4-Pro B300 vLLM SPEED-Bench AL matrix (native MTP). +# +# DSV4 tokenizer, parsers, compilation, and FP4 indexer cache match the AgentX +# recipe. Temperature 1.0 (generation_config); thinking-on kwargs from the +# workflow (CHAT_TEMPLATE_KWARGS_ON). The client-side --chat-template-kwargs +# shim is applied when APPLY_CHAT_TEMPLATE_KWARGS_SHIM=1. +base: + schema: 2 + name: dsv4-fp4-b300-vllm-speedbench + model: + path: hf:deepseek-ai/DeepSeek-V4-Pro + container: vllm/vllm-openai:v0.21.0 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + kv-cache-dtype: fp8 + block-size: 256 + no-enable-prefix-caching: true + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' + attention-config: '{"use_fp4_indexer_cache":true}' + tokenizer-mode: deepseek_v4 + tool-call-parser: deepseek_v4 + enable-auto-tool-choice: true + reasoning-parser: deepseek_v4 + max-cudagraph-capture-size: 2048 + max-model-len: 16384 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro + SPEC_DECODING: mtp + TEMPERATURE: '1.0' + SPEEDBENCH_TOKENIZER_MODE: deepseek_v4 + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + +zip_override_thinking_off: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' + +zip_override_thinking_on: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..ac51991360 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,132 @@ +# GLM-5 B300 vLLM SPEED-Bench AL matrix (native MTP). +# +# GLM requires --chat-template-content-format=string. Do NOT pass +# --attention_config.use_fp4_indexer_cache: despite GLM-5 also being DSA sparse +# attention, that knob is read only by vllm/models/deepseek_v4/attention.py. +# Thinking ON is default; OFF requires explicit enable_thinking:false kwargs. +# GPU memory utilization 0.80 matches the NVIDIA model-card recommendation for +# NVFP4 + DSA + MTP draft layers. +# Sampling: temperature 1.0, top_p 0.95 (GLM-5 generation_config). +base: + schema: 2 + name: glm5-fp4-b300-vllm-speedbench + model: + path: hf:nvidia/GLM-5-NVFP4 + container: vllm/vllm-openai:v0.21.0 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: nvidia/GLM-5-NVFP4 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + kv-cache-dtype: fp8 + no-enable-prefix-caching: true + reasoning-parser: glm45 + tool-call-parser: glm47 + enable-auto-tool-choice: true + chat-template-content-format: string + gpu-memory-utilization: 0.80 + max-model-len: 16384 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: nvidia/GLM-5-NVFP4 + SPEC_DECODING: mtp + TEMPERATURE: '1.0' + TOP_P: '0.95' + CHAT_TEMPLATE_KWARGS_OFF: '{"enable_thinking": false}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + +zip_override_thinking_off: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' + +zip_override_thinking_on: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..68c687ed09 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,129 @@ +# GLM-5.2 B300 vLLM SPEED-Bench AL matrix (native MTP). +# +# Same serve parameters, sampling and thinking kwargs as GLM-5 (GLM-5.2 shares +# the glm_moe_dsa architecture, MTP head and chat template). +# GLM-5.2 loads on vLLM >=0.23 which carries native --chat-template-kwargs +# support (vllm-project/vllm#44244), so APPLY_CHAT_TEMPLATE_KWARGS_SHIM is NOT +# set. +base: + schema: 2 + name: glm52-fp4-b300-vllm-speedbench + model: + path: hf:nvidia/GLM-5.2-NVFP4 + container: vllm/vllm-openai:v0.21.0 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: nvidia/GLM-5.2-NVFP4 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + kv-cache-dtype: fp8 + no-enable-prefix-caching: true + reasoning-parser: glm45 + tool-call-parser: glm47 + enable-auto-tool-choice: true + chat-template-content-format: string + gpu-memory-utilization: 0.80 + max-model-len: 16384 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: nvidia/GLM-5.2-NVFP4 + SPEC_DECODING: mtp + TEMPERATURE: '1.0' + TOP_P: '0.95' + CHAT_TEMPLATE_KWARGS_OFF: '{"enable_thinking": false}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + +zip_override_thinking_off: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' + +zip_override_thinking_on: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..28625c8769 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,139 @@ +# Qwen3.5-397B-A17B B300 vLLM SPEED-Bench AL matrix (native MTP). +# +# --max-cudagraph-capture-size 512: Qwen3.5 is a mamba hybrid and a large capture +# size trips the causal_conv1d assert (vLLM PR #34571). --language-model-only: +# the checkpoint is multimodal, AL is collected on text only. +# Thinking OFF must be passed explicitly via enable_thinking. +# +# Per-mode sampling (model-card): +# thinking : temperature 0.6, top_p 0.95, top_k 20, presence_penalty 0.0 +# instruct : temperature 0.7, top_p 0.8, top_k 20, presence_penalty 1.5 +base: + schema: 2 + name: qwen35-fp4-b300-vllm-speedbench + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: vllm/vllm-openai:v0.21.0 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + kv-cache-dtype: fp8 + no-enable-prefix-caching: true + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-auto-tool-choice: true + language-model-only: true + max-cudagraph-capture-size: 512 + max-model-len: 16384 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + SPEC_DECODING: mtp + TEMPERATURE_ON: '0.6' + TOP_P_ON: '0.95' + TOP_K_ON: '20' + PRESENCE_PENALTY_ON: '0.0' + TEMPERATURE_OFF: '0.7' + TOP_P_OFF: '0.8' + TOP_K_OFF: '20' + PRESENCE_PENALTY_OFF: '1.5' + CHAT_TEMPLATE_KWARGS_OFF: '{"enable_thinking": false}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + +zip_override_thinking_off: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' + +zip_override_thinking_on: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..b7abcfa03d --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,150 @@ +# Qwen3.8-Flash-Next B300 vLLM SPEED-Bench AL matrix (native MTP). +# +# TP4 (not 8): plain TP8 is incompatible with the FP8 checkpoint's 128-wide +# quantization blocks. AL is GPU-count-independent, so collecting on 4 of 8 +# GPUs does not affect the curve. +# +# --no-enable-flashinfer-autotune; --max-num-seqs 256 (avoids a Mamba-cache +# capacity error at startup); kv-cache dtype left at default for the hybrid +# GDN + Qwen Sparse Attention architecture (do not force fp8); +# --max-cudagraph-capture-size 512 (mamba-hybrid causal_conv1d capture-size +# assert); --language-model-only (the checkpoint is multimodal, AL is collected +# on text only). +# +# SPEEDBENCH_CONCURRENCY=64 keeps collection under the CI wall-time limit. +# +# Per-mode sampling (model-card): +# thinking : temperature 1.0, top_p 0.95, top_k 20, presence_penalty 0.0 +# instruct : temperature 0.7, top_p 0.8, top_k 20, presence_penalty 1.5 +base: + schema: 2 + name: qwen38next-fp4-b300-vllm-speedbench + model: + path: hf:Qwen/Qwen3.8-Flash-Next-FP8 + container: vllm/vllm-openai:qwen38-flash-next + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: Qwen/Qwen3.8-Flash-Next-FP8 + tensor-parallel-size: 4 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + no-enable-prefix-caching: true + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-auto-tool-choice: true + language-model-only: true + no-enable-flashinfer-autotune: true + gpu-memory-utilization: 0.90 + max-num-seqs: 256 + max-cudagraph-capture-size: 512 + max-model-len: 16384 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: Qwen/Qwen3.8-Flash-Next-FP8 + SPEC_DECODING: mtp + TEMPERATURE_ON: '1.0' + TOP_P_ON: '0.95' + TOP_K_ON: '20' + PRESENCE_PENALTY_ON: '0.0' + TEMPERATURE_OFF: '0.7' + TOP_P_OFF: '0.8' + TOP_K_OFF: '20' + PRESENCE_PENALTY_OFF: '1.5' + CHAT_TEMPLATE_KWARGS_OFF: '{"enable_thinking": false}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + SPEEDBENCH_CONCURRENCY: '64' + +zip_override_thinking_off: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + - 'off' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' + +zip_override_thinking_on: + roles: + agg: + args: + speculative-config: + - '{"method":"mtp","num_speculative_tokens":1}' + - '{"method":"mtp","num_speculative_tokens":2}' + - '{"method":"mtp","num_speculative_tokens":3}' + - '{"method":"mtp","num_speculative_tokens":4}' + - '{"method":"mtp","num_speculative_tokens":5}' + - '{"method":"mtp","num_speculative_tokens":6}' + - '{"method":"mtp","num_speculative_tokens":7}' + - '{"method":"mtp","num_speculative_tokens":8}' + benchmark: + env: + THINKING: + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + - 'on' + MTP: + - '1' + - '2' + - '3' + - '4' + - '5' + - '6' + - '7' + - '8' diff --git a/benchmarks/single_node/srt_speedbench.sh b/benchmarks/single_node/srt_speedbench.sh new file mode 100644 index 0000000000..32b0191d38 --- /dev/null +++ b/benchmarks/single_node/srt_speedbench.sh @@ -0,0 +1,203 @@ +#!/usr/bin/env bash + +# SRT-slurm client for one SPEED-Bench AL cell. +# +# SRT owns the server lifecycle; this script runs as benchmark.type=custom +# against the already-ready frontend. For each cell the recipe selects one +# (thinking mode, MTP level) pair. The client prepares the dataset, snapshots +# the speculative-decode metrics, runs vllm bench serve, computes the real +# acceptance length (AL), and writes a result JSON. +# +# Environment inputs come from the recipe (benchmark.env) and runtime bindings +# (single_node.py). Model-specific sampling differences are expressed as +# per-mode env vars (TEMPERATURE_ON/OFF, TOP_P_ON/OFF, ...) or uniform ones +# (TEMPERATURE, TOP_P). + +set -eo pipefail + +# Jobs inherit the legacy scripts' /workspace, which srt-slurm does not mount; +# fall back to the repo mount this client runs from. +if [[ ! -f "${INFMAX_CONTAINER_WORKSPACE:-}/benchmarks/benchmark_lib.sh" ]]; then + INFMAX_CONTAINER_WORKSPACE="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +fi +export INFMAX_CONTAINER_WORKSPACE + +source "$INFMAX_CONTAINER_WORKSPACE/benchmarks/benchmark_lib.sh" --validation-only +check_env_vars \ + MODEL CATEGORY SPEEDBENCH_OUTPUT_LEN THINKING MTP \ + RESULT_DIR RESULT_FILENAME SRT_FRONTEND_HOST SRT_FRONTEND_PORT + +SPEEDBENCH_DIR="/tmp/speed_bench_data" +MODEL_KEY="$(basename "$MODEL" | tr '[:upper:]' '[:lower:]')" +CONCURRENCY="${SPEEDBENCH_CONCURRENCY:-1}" + +# --- Resolve per-mode sampling parameters ------------------------------------ + +if [[ -n "${TEMPERATURE_ON:-}" && -n "${TEMPERATURE_OFF:-}" ]]; then + # Per-mode sampling (Qwen3.5, Qwen3.8-Flash-Next). + if [[ "$THINKING" == "on" ]]; then + TEMPERATURE="$TEMPERATURE_ON" + TOP_P="${TOP_P_ON:-}" + TOP_K="${TOP_K_ON:-}" + PRESENCE_PENALTY="${PRESENCE_PENALTY_ON:-}" + else + TEMPERATURE="$TEMPERATURE_OFF" + TOP_P="${TOP_P_OFF:-}" + TOP_K="${TOP_K_OFF:-}" + PRESENCE_PENALTY="${PRESENCE_PENALTY_OFF:-}" + fi +fi +check_env_vars TEMPERATURE + +# --- Chat-template kwargs ---------------------------------------------------- + +THINK_ARGS=() +if [[ "$THINKING" == "on" && -n "${CHAT_TEMPLATE_KWARGS_ON:-}" ]]; then + THINK_ARGS=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") +elif [[ "$THINKING" == "off" && -n "${CHAT_TEMPLATE_KWARGS_OFF:-}" ]]; then + THINK_ARGS=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") +fi + +# --- Full benchmark_lib initialization --------------------------------------- + +source "$INFMAX_CONTAINER_WORKSPACE/benchmarks/benchmark_lib.sh" + +# --- Dataset preparation ----------------------------------------------------- + +echo "=== Downloading SPEED-Bench dataset ===" +pip install -q datasets tiktoken +curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ + | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" + +if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then + echo "CRITICAL: SPEED-Bench download failed -- $SPEEDBENCH_DIR/qualitative.jsonl not found" >&2 + exit 1 +fi + +# --- Apply chat-template-kwargs shim if requested ---------------------------- + +if [[ "${APPLY_CHAT_TEMPLATE_KWARGS_SHIM:-}" == "1" ]]; then + NEED_SHIM=0 + if [[ "$THINKING" == "on" && -n "${CHAT_TEMPLATE_KWARGS_ON:-}" ]]; then NEED_SHIM=1; fi + if [[ "$THINKING" == "off" && -n "${CHAT_TEMPLATE_KWARGS_OFF:-}" ]]; then NEED_SHIM=1; fi + if [[ "$NEED_SHIM" == "1" ]]; then + if ! apply_chat_template_kwargs_shim; then + echo "CRITICAL: --chat-template-kwargs shim failed -- aborting" >&2 + exit 1 + fi + fi +fi + +# --- Build metrics endpoint list --------------------------------------------- +# A router frontend does not re-export engine metrics; read from each worker. + +METRICS_URLS="" +if [[ -n "${SRT_AGG_ENDPOINTS:-}" ]]; then + METRICS_URLS=$(sed -E 's#([^,]+)#http://\1/metrics#g' <<< "${SRT_AGG_ENDPOINTS%,}") +fi +if [[ -z "$METRICS_URLS" ]]; then + METRICS_URLS="http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}/metrics" +fi + +fetch_metric() { + local name="$1" + local total=0 url value + for url in ${METRICS_URLS//,/ }; do + value=$(curl -s "$url" \ + | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0") + total=$(awk "BEGIN {printf \"%d\", $total + $value}") + done + echo "$total" +} + +# --- Build client args ------------------------------------------------------- + +CLIENT_ARGS=() +if [[ "${SPEEDBENCH_TRUST_REMOTE_CODE:-}" == "1" ]]; then + CLIENT_ARGS+=(--trust-remote-code) +fi +if [[ -n "${SPEEDBENCH_TOKENIZER_MODE:-}" ]]; then + CLIENT_ARGS+=(--tokenizer-mode "$SPEEDBENCH_TOKENIZER_MODE") +fi +CLIENT_ARGS+=(--temperature "$TEMPERATURE") +if [[ -n "${TOP_P:-}" ]]; then + CLIENT_ARGS+=(--top-p "$TOP_P") +fi +if [[ -n "${TOP_K:-}" ]]; then + CLIENT_ARGS+=(--top-k "$TOP_K") +fi +if [[ -n "${PRESENCE_PENALTY:-}" ]]; then + CLIENT_ARGS+=(--presence-penalty "$PRESENCE_PENALTY") +fi + +# --- Run the benchmark ------------------------------------------------------- + +echo "" +echo "==========================================" +echo " Cell: thinking=$THINKING MTP=$MTP category=$CATEGORY" +echo "==========================================" + +acc_before=$(fetch_metric "vllm:spec_decode_num_accepted_tokens_total") +drf_before=$(fetch_metric "vllm:spec_decode_num_drafts_total") + +vllm bench serve \ + --model "$MODEL" \ + --port "$SRT_FRONTEND_PORT" \ + --base-url "http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" \ + --dataset-name speed_bench \ + --dataset-path "$SPEEDBENCH_DIR" \ + --speed-bench-category "$CATEGORY" \ + --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ + --num-prompts -1 \ + --max-concurrency "$CONCURRENCY" \ + --save-result \ + --save-detailed \ + --result-dir "$RESULT_DIR" \ + --result-filename "speedbench_${THINKING}_mtp${MTP}" \ + "${CLIENT_ARGS[@]}" \ + "${THINK_ARGS[@]}" + +acc_after=$(fetch_metric "vllm:spec_decode_num_accepted_tokens_total") +drf_after=$(fetch_metric "vllm:spec_decode_num_drafts_total") + +# --- Compute acceptance length ----------------------------------------------- + +delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") +delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") +if [[ "$delta_drf" -gt 0 ]]; then + al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") +else + al="N/A" +fi +echo " -> thinking=$THINKING MTP=$MTP AL=$al (accepted=$delta_acc drafts=$delta_drf)" + +# --- Write result JSON ------------------------------------------------------- + +python3 - "$RESULT_DIR" "$RESULT_FILENAME" "$MODEL_KEY" "$THINKING" "$MTP" \ + "$al" "$delta_acc" "$delta_drf" "$CATEGORY" "$SPEEDBENCH_OUTPUT_LEN" \ + "$TEMPERATURE" "${TOP_P:-}" <<'PYEOF' +import json, sys +result_dir, result_filename = sys.argv[1], sys.argv[2] +model_key, thinking, mtp = sys.argv[3], sys.argv[4], sys.argv[5] +al_str, accepted, drafts = sys.argv[6], sys.argv[7], sys.argv[8] +category, output_len = sys.argv[9], sys.argv[10] +temperature, top_p = sys.argv[11], sys.argv[12] +result = { + "model_key": model_key, + "thinking": thinking, + "mtp": int(mtp), + "al": al_str if al_str == "N/A" else float(al_str), + "accepted_delta": int(accepted), + "drafts_delta": int(drafts), + "category": category, + "output_len": int(output_len), + "temperature": float(temperature), +} +if top_p: + result["top_p"] = float(top_p) +path = f"{result_dir}/{result_filename}.json" +with open(path, "w") as f: + json.dump(result, f, indent=2) + f.write("\n") +print(f"Result JSON written to {path}") +PYEOF diff --git a/golden_al_distribution/README.md b/golden_al_distribution/README.md index e464fbba52..8c8af9b407 100644 --- a/golden_al_distribution/README.md +++ b/golden_al_distribution/README.md @@ -77,9 +77,9 @@ This policy follows the same broad principle as MLPerf Inference: prescribe the The push-button [`speedbench-al.yml`](../.github/workflows/speedbench-al.yml) workflow, introduced in [InferenceX#1650](https://github.com/SemiAnalysisAI/InferenceX/pull/1650) and extended to additional MTP and EAGLE3 models in [InferenceX#1706](https://github.com/SemiAnalysisAI/InferenceX/pull/1706), performs the following process. It superseded the early manually assembled reference in [InferenceX#1592](https://github.com/SemiAnalysisAI/InferenceX/pull/1592), making the exact commands, logs, outputs, and generated YAML auditable from one run. 1. A maintainer dispatches the workflow with a model, model prefix, vLLM image, draft lengths (normally 1–8), thinking modes, `category=coding`, and `output-len=4096`. -2. The workflow launches the model on a B300 runner and selects the matching collector under [`benchmarks/single_node/speedbench/`](../benchmarks/single_node/speedbench/). -3. For every `(thinking mode, draft length)` cell, the collector starts a clean vLLM server with real MTP or EAGLE3 decoding and the model's production sampling/chat-template settings. -4. The collector snapshots vLLM's cumulative accepted-token and verification-draft counters, runs every prompt in the SPEED-Bench Qualitative `coding` category through `vllm bench serve`, and snapshots the counters again. +2. The workflow launches the model on a B300 runner. Native-MTP models (dsr1, dsv4, glm5, glm52, qwen3.5, qwen3.8next) use the srt-slurm single-node path with per-model recipes under [`benchmarks/single_node/srt-slurm-recipes/`](../benchmarks/single_node/srt-slurm-recipes/) and the shared client [`srt_speedbench.sh`](../benchmarks/single_node/srt_speedbench.sh). Draft-model collectors (dsv4dspark*, kimik3, minimaxm3) keep dedicated scripts under [`benchmarks/single_node/speedbench/`](../benchmarks/single_node/speedbench/). +3. For every `(thinking mode, draft length)` cell, the server starts with real MTP or EAGLE3 decoding and the model's production sampling/chat-template settings. +4. The client snapshots vLLM's cumulative accepted-token and verification-draft counters, runs every prompt in the SPEED-Bench Qualitative `coding` category through `vllm bench serve`, and snapshots the counters again. 5. It computes the mean acceptance length as: ```text @@ -87,7 +87,7 @@ The push-button [`speedbench-al.yml`](../.github/workflows/speedbench-al.yml) wo ``` The `1` is the target model's guaranteed verification token. Values are rounded to two decimal places. -6. The collector emits a YAML matrix. The workflow publishes it in the GitHub Actions step summary and uploads it as a `speedbench-reference-al-` artifact. +6. The per-cell results are aggregated into a YAML matrix. The workflow publishes it in the GitHub Actions step summary and uploads it as a `speedbench-reference-al-` artifact. 7. Server logs and detailed per-request results are retained so reviewers can confirm sensible output, correct thinking mode, and the absence of silent server or chat-template failures. 8. After review, the matrix is committed here with its exact sampling metadata and source Actions run URL. diff --git a/golden_al_distribution/README_zh.md b/golden_al_distribution/README_zh.md index b59c4874f3..7729d67b52 100644 --- a/golden_al_distribution/README_zh.md +++ b/golden_al_distribution/README_zh.md @@ -77,9 +77,9 @@ python -m atom.entrypoints.openai_server \ 一键触发的 [`speedbench-al.yml`](../.github/workflows/speedbench-al.yml) 工作流最初由 [InferenceX#1650](https://github.com/SemiAnalysisAI/InferenceX/pull/1650) 引入,随后在 [InferenceX#1706](https://github.com/SemiAnalysisAI/InferenceX/pull/1706) 中扩展到更多 MTP 和 EAGLE3 模型。它取代了 [InferenceX#1592](https://github.com/SemiAnalysisAI/InferenceX/pull/1592) 中早期手工整理的参考值,使精确命令、日志、输出和生成的 YAML 都可以从同一次运行中审计。其流程如下: 1. 维护者触发工作流,指定模型、模型前缀、vLLM 镜像、草稿长度(通常为 1–8)、思考模式、`category=coding` 和 `output-len=4096`。 -2. 工作流在 B300 runner 上启动模型,并选择 [`benchmarks/single_node/speedbench/`](../benchmarks/single_node/speedbench/) 下对应的收集脚本。 -3. 对每个“思考模式 × 草稿长度”组合,收集脚本使用真实 MTP 或 EAGLE3 解码以及该模型的生产采样和聊天模板设置,启动一个干净的 vLLM 服务。 -4. 收集脚本读取 vLLM 累计的已接受 token 和验证草稿计数器,通过 `vllm bench serve` 运行 SPEED-Bench Qualitative `coding` 类别中的全部提示词,然后再次读取计数器。 +2. 工作流在 B300 runner 上启动模型。原生 MTP 模型(dsr1、dsv4、glm5、glm52、qwen3.5、qwen3.8next)使用 srt-slurm 单节点路径,通过 [`benchmarks/single_node/srt-slurm-recipes/`](../benchmarks/single_node/srt-slurm-recipes/) 下的配方和共享客户端 [`srt_speedbench.sh`](../benchmarks/single_node/srt_speedbench.sh)。草稿模型收集脚本(dsv4dspark*、kimik3、minimaxm3)保留在 [`benchmarks/single_node/speedbench/`](../benchmarks/single_node/speedbench/) 下。 +3. 对每个”思考模式 × 草稿长度”组合,服务使用真实 MTP 或 EAGLE3 解码以及该模型的生产采样和聊天模板设置启动。 +4. 客户端读取 vLLM 累计的已接受 token 和验证草稿计数器,通过 `vllm bench serve` 运行 SPEED-Bench Qualitative `coding` 类别中的全部提示词,然后再次读取计数器。 5. 按以下公式计算平均接受长度: ```text @@ -87,7 +87,7 @@ python -m atom.entrypoints.openai_server \ ``` 其中 `1` 是目标模型保证生成的验证 token。结果四舍五入到小数点后两位。 -6. 收集脚本生成 YAML 矩阵。工作流将其发布到 GitHub Actions step summary,并上传为 `speedbench-reference-al-` artifact。 +6. 逐单元格结果汇总为 YAML 矩阵。工作流将其发布到 GitHub Actions step summary,并上传为 `speedbench-reference-al-` artifact。 7. 工作流保留服务日志和逐请求详细结果,以便审阅者确认输出合理、思考模式正确,并且没有静默的服务或聊天模板故障。 8. 审阅完成后,将矩阵连同准确的采样元数据和源 Actions run URL 一并提交到本目录。 diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index 1e6baece4b..6f6060885f 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -1,4 +1,4 @@ -"""Bind a native single-node SRT recipe to one fixed-sequence or AgentX matrix point.""" +"""Bind a native single-node SRT recipe to one fixed-sequence, AgentX, or SPEED-Bench point.""" from __future__ import annotations @@ -16,6 +16,13 @@ SINGLE_NODE_ENGINES = {**ENGINES, "atom": "atom"} +SPEEDBENCH_CLIENT = "srt_speedbench.sh" + + +def _is_speedbench(benchmark: dict[str, Any]) -> bool: + """Detect a SPEED-Bench AL collector from its benchmark command.""" + return benchmark.get("command", "").endswith(SPEEDBENCH_CLIENT) + def parallelism_constraints( engine: str, args: Mapping[str, Any], environment: Mapping[str, str] @@ -95,6 +102,7 @@ def validate_recipe(recipe: dict[str, Any], environment: Mapping[str, str]) -> N # A point that stops drafting may keep its matrix label. speculation = "mtp" if spec else workload.get("SPEC_DECODING", "none") agentic = environment["IS_AGENTIC"] == "1" + collector = _is_speedbench(benchmark) expected = { "engine": (engine, SINGLE_NODE_ENGINES[environment["FRAMEWORK"]]), "model": (recipe["model"]["path"], f"hf:{environment['MODEL']}"), @@ -114,9 +122,13 @@ def validate_recipe(recipe: dict[str, Any], environment: Mapping[str, str]) -> N if environment["SPEC_DECODING"] == "draft_model" else environment["SPEC_DECODING"], ), - "AgentX client": (benchmark.get("command", "").endswith("srt_agentic.sh"), agentic), } - if not agentic: + if not collector: + expected["AgentX client"] = ( + benchmark.get("command", "").endswith("srt_agentic.sh"), + agentic, + ) + if not agentic and not collector: expected["USE_CHAT_TEMPLATE"] = (workload["USE_CHAT_TEMPLATE"], "true" if spec else "false") for name in ("ISL", "OSL", "RANDOM_RANGE_RATIO"): expected[name] = (str(workload[name]), environment[name]) @@ -160,6 +172,7 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: for key, value in options.items(): overrides += ["--set", f"srun_options.{key}={json.dumps(value)}"] agentic = environment["IS_AGENTIC"] == "1" + collector = _is_speedbench(recipe["benchmark"]) names = [ "CONC", "RESULT_FILENAME", @@ -179,6 +192,20 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: if name == "CONC" and name in recipe["benchmark"]["env"]: continue overrides += ["--set", f"benchmark.env.{name}={json.dumps(value)}"] + if collector: + # SPEED-Bench collector tunables: bind workflow-level settings into + # the per-cell benchmark environment so the client reads them. + for name in ("CATEGORY", "SPEEDBENCH_OUTPUT_LEN"): + value = environment.get(name, "") + if not value: + raise ValueError(f"Missing SPEED-Bench input: {name}") + overrides += ["--set", f"benchmark.env.{name}={json.dumps(value)}"] + # Chat-template kwargs are optional (dsr1 has no off mode). + for name in ("CHAT_TEMPLATE_KWARGS_ON",): + value = environment.get(name, "") + if value: + overrides += ["--set", f"benchmark.env.{name}={json.dumps(value)}"] + return [*overrides, "--set", 'benchmark.env.RESULT_DIR="/logs"'] if agentic: # The aggregated result lands where fixed-sequence results do. overrides += ["--set", 'benchmark.env.AGENTIC_OUTPUT_DIR="/logs"'] diff --git a/infx/workflows/speedbench_matrix.py b/infx/workflows/speedbench_matrix.py new file mode 100644 index 0000000000..6100eb5d8e --- /dev/null +++ b/infx/workflows/speedbench_matrix.py @@ -0,0 +1,71 @@ +"""Aggregate per-cell SPEED-Bench AL result JSONs into the golden YAML matrix.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + + +def aggregate_cells( + result_dir: Path, + model_key: str, + thinking_modes: list[str], + mtp_list: list[int], + header_lines: list[str], +) -> str: + """Read per-cell JSONs and produce the YAML matrix string. + + Each cell is expected at ``result_dir/speedbench_{mode}_mtp{mtp}.json`` with + an ``al`` field (float or ``"N/A"``). Missing or unreadable cells produce + ``N/A``. The output format matches the legacy collector scripts exactly: + same key ordering, 2-decimal ALs, ``N/A`` for failed/missing cells. + """ + cells: dict[str, dict[int, str]] = {} + for mode in thinking_modes: + cells[mode] = {} + for mtp in mtp_list: + path = result_dir / f"speedbench_{mode}_mtp{mtp}.json" + al = "N/A" + try: + data = json.loads(path.read_text()) + raw = data["al"] + if raw != "N/A": + al = f"{float(raw):.2f}" + except (OSError, KeyError, ValueError, TypeError): + pass + cells[mode][mtp] = al + + lines: list[str] = [] + for header in header_lines: + lines.append(f"# {header}") + lines.append(f"{model_key}:") + for mode in thinking_modes: + lines.append(f" thinking_{mode}:") + for mtp in mtp_list: + lines.append(f" {mtp}: {cells[mode][mtp]}") + return "\n".join(lines) + "\n" + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--result-dir", type=Path, required=True) + parser.add_argument("--model-key", required=True) + parser.add_argument("--thinking-modes", required=True, help="Space-separated modes") + parser.add_argument("--mtp-list", required=True, help="Space-separated MTP levels") + parser.add_argument("--header", action="append", default=[], help="Header comment line") + parser.add_argument("--output", type=Path, help="Write to file instead of stdout") + args = parser.parse_args() + + modes = args.thinking_modes.split() + mtps = [int(m) for m in args.mtp_list.split()] + result = aggregate_cells(args.result_dir, args.model_key, modes, mtps, args.header) + + if args.output: + args.output.write_text(result) + else: + print(result, end="") + + +if __name__ == "__main__": + main() diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index efd0506131..7964506032 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -124,7 +124,9 @@ EXECUTION_PATH=agentic if [[ "$IS_MULTINODE" == true ]]; then EXECUTION_PATH=multinode elif [[ -n "${BENCH_SCRIPT_OVERRIDE:-}" ]]; then - # SPEED-Bench collectors explicitly supply their script outside this migration. + # Legacy SPEED-Bench collectors (draft-model variants: dsv4dspark*, kimik3, + # minimaxm3) explicitly supply their script. Native-MTP collectors migrated + # to the srt-slurm path set SRT_RECIPE instead. EXECUTION_PATH=script elif [[ "$IS_AGENTIC" == 0 || -n "${SRT_RECIPE:-}" ]]; then check_env_vars SRT_RECIPE diff --git a/utils/test_speedbench_matrix.py b/utils/test_speedbench_matrix.py new file mode 100644 index 0000000000..f7851df5c1 --- /dev/null +++ b/utils/test_speedbench_matrix.py @@ -0,0 +1,47 @@ +"""Unit tests for infx.workflows.speedbench_matrix.""" + +import json + +import pytest + +from infx.workflows.speedbench_matrix import aggregate_cells + + +def test_aggregate_cells_reads_per_cell_jsons(tmp_path): + """Happy-path: all cells present produce the expected YAML matrix.""" + for mode in ("off", "on"): + for mtp in (1, 2): + al = 2.50 if mode == "off" else 3.10 + path = tmp_path / f"speedbench_{mode}_mtp{mtp}.json" + path.write_text(json.dumps({"al": al + mtp * 0.1})) + result = aggregate_cells(tmp_path, "dsv4", ["off", "on"], [1, 2], ["Test header"]) + assert "# Test header\n" in result + assert "dsv4:\n" in result + assert " thinking_off:\n" in result + assert " 1: 2.60\n" in result + assert " 2: 2.70\n" in result + assert " thinking_on:\n" in result + assert " 1: 3.20\n" in result + assert " 2: 3.30\n" in result + + +def test_aggregate_cells_missing_cell_produces_na(tmp_path): + """Missing or unreadable result JSONs produce N/A.""" + (tmp_path / "speedbench_on_mtp1.json").write_text(json.dumps({"al": 3.14})) + result = aggregate_cells(tmp_path, "dsr1", ["on"], [1, 2], []) + assert " 1: 3.14\n" in result + assert " 2: N/A\n" in result + + +def test_aggregate_cells_na_value_passthrough(tmp_path): + """A cell whose AL is already 'N/A' stays N/A.""" + (tmp_path / "speedbench_off_mtp1.json").write_text(json.dumps({"al": "N/A"})) + result = aggregate_cells(tmp_path, "glm5", ["off"], [1], []) + assert " 1: N/A\n" in result + + +def test_aggregate_cells_malformed_json_produces_na(tmp_path): + """Corrupt JSON produces N/A without crashing.""" + (tmp_path / "speedbench_off_mtp1.json").write_text("not json") + result = aggregate_cells(tmp_path, "glm52", ["off"], [1], []) + assert " 1: N/A\n" in result diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 2eb047749a..d25a4caf8c 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -10,7 +10,13 @@ import pytest import yaml -from infx.srt_slurm.single_node import runtime_arguments, select_recipe, submission_fields +from infx.srt_slurm.single_node import ( + SPEEDBENCH_CLIENT, + _is_speedbench, + runtime_arguments, + select_recipe, + submission_fields, +) from infx.srt_slurm.synthetic_acceptance import plan_commands, selected_recipes ROOT = Path(__file__).resolve().parents[1] @@ -453,3 +459,148 @@ def test_b300_keeps_agentic_and_explicit_collector_dispatch(tmp_path, collector) assert calls[-1][-2:] == ["bash", expected] assert "--jobid=42" in calls[-1] assert (tmp_path / "cancelled").read_text() == "42\n" + + +# --------------------------------------------------------------------------- +# SPEED-Bench collector tests +# --------------------------------------------------------------------------- + + +@pytest.fixture +def speedbench_point(tmp_path): + """A recipe that targets the SPEED-Bench collector client.""" + recipe = { + "engine": "vllm", + "resources": {"gpus_per_node": 8}, + "model": {"path": "hf:deepseek-ai/DeepSeek-V4-Pro", "container": "vllm/vllm-openai:v0.21.0", "precision": "fp4"}, + "roles": {"agg": { + "nodes": 1, "workers": 1, "gpus": 8, + "args": { + "tensor-parallel-size": 8, "data-parallel-size": 1, "pipeline-parallel-size": 1, + "trust-remote-code": True, "kv-cache-dtype": "fp8", + "speculative-config": '{"method":"mtp","num_speculative_tokens":3}', + }, + }}, + "benchmark": {"type": "custom", + "command": f"bash /infmax-workspace/benchmarks/single_node/{SPEEDBENCH_CLIENT}", + "env": { + "MODEL": "deepseek-ai/DeepSeek-V4-Pro", + "SPEC_DECODING": "mtp", + "TEMPERATURE": "1.0", + "THINKING": "off", + "MTP": "3", + }, + }, + } + path = tmp_path / "speedbench.yaml" + path.write_text(yaml.safe_dump({"base": recipe})) + env = { + "FRAMEWORK": "vllm", "MODEL": "deepseek-ai/DeepSeek-V4-Pro", + "IMAGE": "vllm/vllm-openai:v0.21.0", "PRECISION": "fp4", + "TP": "8", "GPU_COUNT": "8", "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", + "EP_SIZE": "1", "DP_ATTENTION": "false", "SPEC_DECODING": "mtp", "IS_AGENTIC": "0", + "RUN_EVAL": "false", "EVAL_ONLY": "false", + "CONC": "1", "RESULT_FILENAME": "speedbench_off_mtp3", + "GPU_MONITOR_INTERVAL": "3", "MODEL_PREFIX": "dsv4", + "CATEGORY": "qualitative", "SPEEDBENCH_OUTPUT_LEN": "512", + # ISL/OSL/RANDOM_RANGE_RATIO intentionally absent -- collector skips them + } + return path, recipe, env + + +def test_speedbench_detection(): + assert _is_speedbench({"command": f"bash /infmax-workspace/benchmarks/single_node/{SPEEDBENCH_CLIENT}"}) + assert not _is_speedbench({"command": "bash /infmax-workspace/benchmarks/single_node/srt_agentic.sh"}) + assert not _is_speedbench({}) + + +def test_speedbench_collector_skips_isl_osl_and_chat_template(speedbench_point): + """Collector recipes must validate without ISL, OSL, RANDOM_RANGE_RATIO, or USE_CHAT_TEMPLATE.""" + path, _, env = speedbench_point + # These keys would make validate_recipe fail for a normal (non-collector) recipe + assert "ISL" not in env + assert "OSL" not in env + argv = runtime_arguments(f"{path}:base", env) + assert any("CATEGORY" in arg for arg in argv) + assert any("SPEEDBENCH_OUTPUT_LEN" in arg for arg in argv) + assert any("RESULT_DIR" in arg and "/logs" in arg for arg in argv) + + +def test_speedbench_collector_binds_chat_template_kwargs_on(speedbench_point): + """CHAT_TEMPLATE_KWARGS_ON is bound when present in the environment.""" + path, _, env = speedbench_point + env_with_kwargs = {**env, "CHAT_TEMPLATE_KWARGS_ON": '{"enable_thinking": true}'} + argv = runtime_arguments(f"{path}:base", env_with_kwargs) + assert any("CHAT_TEMPLATE_KWARGS_ON" in arg for arg in argv) + + +def test_speedbench_collector_omits_chat_template_kwargs_when_empty(speedbench_point): + """CHAT_TEMPLATE_KWARGS_ON is omitted when not set.""" + path, _, env = speedbench_point + argv = runtime_arguments(f"{path}:base", env) + assert not any("CHAT_TEMPLATE_KWARGS_ON" in arg for arg in argv) + + +def test_speedbench_collector_rejects_missing_category(speedbench_point): + path, _, env = speedbench_point + with pytest.raises(ValueError, match="Missing SPEED-Bench input.*CATEGORY"): + runtime_arguments(f"{path}:base", {**env, "CATEGORY": ""}) + + +def test_speedbench_collector_rejects_missing_output_len(speedbench_point): + path, _, env = speedbench_point + with pytest.raises(ValueError, match="Missing SPEED-Bench input.*SPEEDBENCH_OUTPUT_LEN"): + runtime_arguments(f"{path}:base", {**env, "SPEEDBENCH_OUTPUT_LEN": ""}) + + +def test_speedbench_zip_override_selects_cell(tmp_path): + """Collector recipe with zip_override + selector resolves to one cell. + + The workflow constructs the exact selector (e.g. zip_override_thinking_off[1]) + for each cell, so select_recipe sees exactly one matching variant. + """ + recipe = { + "engine": "vllm", + "resources": {"gpus_per_node": 8}, + "model": {"path": "hf:deepseek-ai/DeepSeek-V4-Pro", "container": "vllm/vllm-openai:v0.21.0", "precision": "fp4"}, + "roles": {"agg": { + "nodes": 1, "workers": 1, "gpus": 8, + "args": { + "tensor-parallel-size": 8, "data-parallel-size": 1, "pipeline-parallel-size": 1, + "trust-remote-code": True, "kv-cache-dtype": "fp8", + }, + }}, + "benchmark": {"type": "custom", + "command": f"bash /infmax-workspace/benchmarks/single_node/{SPEEDBENCH_CLIENT}", + "env": {"MODEL": "deepseek-ai/DeepSeek-V4-Pro", "SPEC_DECODING": "mtp", "TEMPERATURE": "1.0"}, + }, + } + raw = {"base": recipe, "zip_override_thinking_off": { + "roles": {"agg": {"args": {"speculative-config": [ + '{"method":"mtp","num_speculative_tokens":1}', + '{"method":"mtp","num_speculative_tokens":2}', + ]}}}, + "benchmark": {"env": { + "THINKING": ["off", "off"], + "MTP": ["1", "2"], + }}, + }} + path = tmp_path / "speedbench.yaml" + path.write_text(yaml.safe_dump(raw)) + env = { + "FRAMEWORK": "vllm", "MODEL": "deepseek-ai/DeepSeek-V4-Pro", + "IMAGE": "vllm/vllm-openai:v0.21.0", "PRECISION": "fp4", + "TP": "8", "GPU_COUNT": "8", "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", + "EP_SIZE": "1", "DP_ATTENTION": "false", "SPEC_DECODING": "mtp", "IS_AGENTIC": "0", + "RUN_EVAL": "false", "EVAL_ONLY": "false", + "CONC": "1", "RESULT_FILENAME": "speedbench_off_mtp2", + "GPU_MONITOR_INTERVAL": "3", "MODEL_PREFIX": "dsv4", + "CATEGORY": "qualitative", "SPEEDBENCH_OUTPUT_LEN": "512", + } + # The workflow provides an explicit selector: select cell [1] (MTP=2). + config, selected = select_recipe(f"{path}:zip_override_thinking_off[1]", env) + assert "zip_override_thinking_off[1]" in config + assert selected["benchmark"]["env"]["MTP"] == "2" + # Verify runtime_arguments works on the selected cell. + argv = runtime_arguments(f"{path}:zip_override_thinking_off[1]", env) + assert any("CATEGORY" in arg for arg in argv) From 5aea3af34e31700727ca26b84fe2a87f24798981 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 03:48:34 -0400 Subject: [PATCH 02/12] Fix SPEED-Bench lint and bind qwen3.8next TP4 Replace the append loops flagged by PERF401 in speedbench_matrix.py, and set TP/GPU_COUNT to 4 for qwen3.8next cells so the matrix matches its TP4 recipe (plain TP8 is incompatible with the FP8 checkpoint, as in the legacy collector). Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/speedbench-al.yml | 6 ++++++ infx/workflows/speedbench_matrix.py | 7 ++----- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/.github/workflows/speedbench-al.yml b/.github/workflows/speedbench-al.yml index a077c1d8b2..563ac7dc27 100644 --- a/.github/workflows/speedbench-al.yml +++ b/.github/workflows/speedbench-al.yml @@ -255,6 +255,12 @@ jobs: echo "ERROR: SPEED-Bench recipe not found: $RECIPE_PATH" >&2 exit 1 fi + # Plain TP8 is incompatible with the Qwen3.8-Flash-Next FP8 checkpoint's + # 128-wide quantization blocks; its recipe runs TP4 (as the legacy + # collector did), so bind the matrix TP and GPU count to match. + if [[ "$MODEL_PREFIX" == qwen3.8next ]]; then + export TP=4 GPU_COUNT=4 + fi mkdir -p speedbench_results ALL_FAILED=true # Harmless values for launch_srt_single_node env checks; the collector diff --git a/infx/workflows/speedbench_matrix.py b/infx/workflows/speedbench_matrix.py index 6100eb5d8e..b5625fbfcb 100644 --- a/infx/workflows/speedbench_matrix.py +++ b/infx/workflows/speedbench_matrix.py @@ -36,14 +36,11 @@ def aggregate_cells( pass cells[mode][mtp] = al - lines: list[str] = [] - for header in header_lines: - lines.append(f"# {header}") + lines = [f"# {header}" for header in header_lines] lines.append(f"{model_key}:") for mode in thinking_modes: lines.append(f" thinking_{mode}:") - for mtp in mtp_list: - lines.append(f" {mtp}: {cells[mode][mtp]}") + lines.extend(f" {mtp}: {cells[mode][mtp]}" for mtp in mtp_list) return "\n".join(lines) + "\n" From a43d14df699255734f9e852fc7cf8f0e2be4cff3 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 04:06:23 -0400 Subject: [PATCH 03/12] Store SPEED-Bench thinking cells as thinking_on/thinking_off srtctl expands zip variants with ruamel (YAML 1.2) and reloads them with a YAML 1.1 loader, where bare on/off become booleans; every cell was rejected with benchmark.env.THINKING 'Not a valid string' (run 36227842725). The client strips the thinking_ prefix. Add a regression test that runs srtctl's own variant expansion round trip for every SPEED-Bench cell. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../vllm/b300-fp4-speedbench/speedbench.yaml | 16 +++++----- .../vllm/b300-fp4-speedbench/speedbench.yaml | 32 +++++++++---------- .../vllm/b300-fp4-speedbench/speedbench.yaml | 32 +++++++++---------- .../vllm/b300-fp4-speedbench/speedbench.yaml | 32 +++++++++---------- .../vllm/b300-fp4-speedbench/speedbench.yaml | 32 +++++++++---------- .../vllm/b300-fp4-speedbench/speedbench.yaml | 32 +++++++++---------- benchmarks/single_node/srt_speedbench.sh | 8 +++++ utils/test_srt_single_node.py | 29 +++++++++++++++-- 8 files changed, 123 insertions(+), 90 deletions(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml index fd53d4e7e2..11d0be82f2 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml @@ -75,14 +75,14 @@ zip_override_thinking_on: benchmark: env: THINKING: - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' MTP: - '1' - '2' diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml index 4e3fb594f5..1adb25d685 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml @@ -78,14 +78,14 @@ zip_override_thinking_off: benchmark: env: THINKING: - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' MTP: - '1' - '2' @@ -112,14 +112,14 @@ zip_override_thinking_on: benchmark: env: THINKING: - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' MTP: - '1' - '2' diff --git a/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml index ac51991360..dc3c7c7290 100644 --- a/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml @@ -79,14 +79,14 @@ zip_override_thinking_off: benchmark: env: THINKING: - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' MTP: - '1' - '2' @@ -113,14 +113,14 @@ zip_override_thinking_on: benchmark: env: THINKING: - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' MTP: - '1' - '2' diff --git a/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml index 68c687ed09..de2b670f0f 100644 --- a/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml @@ -76,14 +76,14 @@ zip_override_thinking_off: benchmark: env: THINKING: - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' MTP: - '1' - '2' @@ -110,14 +110,14 @@ zip_override_thinking_on: benchmark: env: THINKING: - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' MTP: - '1' - '2' diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml index 28625c8769..3c3479e8cd 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml @@ -86,14 +86,14 @@ zip_override_thinking_off: benchmark: env: THINKING: - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' MTP: - '1' - '2' @@ -120,14 +120,14 @@ zip_override_thinking_on: benchmark: env: THINKING: - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' MTP: - '1' - '2' diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml index b7abcfa03d..11d795c3df 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml @@ -97,14 +97,14 @@ zip_override_thinking_off: benchmark: env: THINKING: - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' - - 'off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' + - 'thinking_off' MTP: - '1' - '2' @@ -131,14 +131,14 @@ zip_override_thinking_on: benchmark: env: THINKING: - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' - - 'on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' + - 'thinking_on' MTP: - '1' - '2' diff --git a/benchmarks/single_node/srt_speedbench.sh b/benchmarks/single_node/srt_speedbench.sh index 32b0191d38..e4af545b91 100644 --- a/benchmarks/single_node/srt_speedbench.sh +++ b/benchmarks/single_node/srt_speedbench.sh @@ -27,6 +27,14 @@ check_env_vars \ MODEL CATEGORY SPEEDBENCH_OUTPUT_LEN THINKING MTP \ RESULT_DIR RESULT_FILENAME SRT_FRONTEND_HOST SRT_FRONTEND_PORT +# Recipes store thinking_on/thinking_off: srtctl round-trips zip variants +# through a YAML 1.1 loader, where bare on/off become booleans. +THINKING="${THINKING#thinking_}" +if [[ "$THINKING" != on && "$THINKING" != off ]]; then + echo "ERROR: THINKING must be thinking_on or thinking_off" >&2 + exit 1 +fi + SPEEDBENCH_DIR="/tmp/speed_bench_data" MODEL_KEY="$(basename "$MODEL" | tr '[:upper:]' '[:lower:]')" CONCURRENCY="${SPEEDBENCH_CONCURRENCY:-1}" diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index d25a4caf8c..c8ae36f301 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -487,7 +487,7 @@ def speedbench_point(tmp_path): "MODEL": "deepseek-ai/DeepSeek-V4-Pro", "SPEC_DECODING": "mtp", "TEMPERATURE": "1.0", - "THINKING": "off", + "THINKING": "thinking_off", "MTP": "3", }, }, @@ -581,7 +581,7 @@ def test_speedbench_zip_override_selects_cell(tmp_path): '{"method":"mtp","num_speculative_tokens":2}', ]}}}, "benchmark": {"env": { - "THINKING": ["off", "off"], + "THINKING": ["thinking_off", "thinking_off"], "MTP": ["1", "2"], }}, }} @@ -604,3 +604,28 @@ def test_speedbench_zip_override_selects_cell(tmp_path): # Verify runtime_arguments works on the selected cell. argv = runtime_arguments(f"{path}:zip_override_thinking_off[1]", env) assert any("CATEGORY" in arg for arg in argv) + + +@pytest.mark.parametrize( + "recipe", sorted(Path("benchmarks/single_node/srt-slurm-recipes").glob("*/vllm/b300-fp4-speedbench/speedbench.yaml")) +) +def test_speedbench_cells_keep_string_env_through_srtctl_variant_round_trip(recipe): + # srtctl expands zip variants with ruamel (YAML 1.2) and reloads them with a + # YAML 1.1 loader, where bare on/off become booleans and fail its schema. + from srtctl.core.config import resolve_override_yaml + from srtctl.core.yaml_utils import dump_yaml_with_comments + + raw = yaml.safe_load(recipe.read_text()) + selectors = [ + f"{group}[{index}]" + for group in raw + if group.startswith("zip_override_thinking_") + for index in range(len(raw[group]["benchmark"]["env"]["MTP"])) + ] + assert selectors + for selector in selectors: + [(_, variant)] = resolve_override_yaml(recipe, selector=selector) + reloaded = yaml.safe_load(dump_yaml_with_comments(variant)) + env = reloaded["benchmark"]["env"] + assert all(isinstance(value, str) for value in env.values()), (selector, env) + assert env["THINKING"] in {"thinking_on", "thinking_off"} From 2bad6db7570d5c9f1f515af50e2820488759f144 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 13:01:27 -0400 Subject: [PATCH 04/12] Bind SpeedBench GPUs via CUDA_VISIBLE_DEVICES vLLM v0.21.0 predates --device-ids, which srtctl passes by default, so every cell died with 'unrecognized arguments'. Set engine.set_visible_devices. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../dsr1/vllm/b300-fp4-speedbench/speedbench.yaml | 2 ++ .../dsv4/vllm/b300-fp4-speedbench/speedbench.yaml | 2 ++ .../glm5/vllm/b300-fp4-speedbench/speedbench.yaml | 2 ++ .../glm52/vllm/b300-fp4-speedbench/speedbench.yaml | 2 ++ .../qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml | 2 ++ .../qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml | 2 ++ utils/test_srt_single_node.py | 9 +++++++++ 7 files changed, 21 insertions(+) diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml index 11d0be82f2..40bce2495a 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml @@ -28,6 +28,8 @@ base: engine: type: vllm connector: null + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true health_check: interval_seconds: 10 max_attempts: 360 diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml index 1adb25d685..1a6f607e7c 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml @@ -24,6 +24,8 @@ base: engine: type: vllm connector: null + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true health_check: interval_seconds: 10 max_attempts: 360 diff --git a/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml index dc3c7c7290..b71e1e3917 100644 --- a/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml @@ -27,6 +27,8 @@ base: engine: type: vllm connector: null + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true health_check: interval_seconds: 10 max_attempts: 360 diff --git a/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml index de2b670f0f..f0302662ce 100644 --- a/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml @@ -25,6 +25,8 @@ base: engine: type: vllm connector: null + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true health_check: interval_seconds: 10 max_attempts: 360 diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml index 3c3479e8cd..9b107a2dc5 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml @@ -28,6 +28,8 @@ base: engine: type: vllm connector: null + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true health_check: interval_seconds: 10 max_attempts: 360 diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml index 11d795c3df..cdd234f68b 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml @@ -36,6 +36,8 @@ base: engine: type: vllm connector: null + # The qwen38-flash-next build may predate --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true health_check: interval_seconds: 10 max_attempts: 360 diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index c8ae36f301..e7168dda01 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -629,3 +629,12 @@ def test_speedbench_cells_keep_string_env_through_srtctl_variant_round_trip(reci env = reloaded["benchmark"]["env"] assert all(isinstance(value, str) for value in env.values()), (selector, env) assert env["THINKING"] in {"thinking_on", "thinking_off"} + + +@pytest.mark.parametrize( + "recipe", sorted(Path("benchmarks/single_node/srt-slurm-recipes").glob("*/vllm/b300-fp4-speedbench/speedbench.yaml")) +) +def test_speedbench_recipes_bind_gpus_without_device_ids(recipe): + # vLLM v0.21.0 rejects srtctl's default --device-ids binding at startup. + raw = yaml.safe_load(recipe.read_text()) + assert raw["base"]["engine"]["set_visible_devices"] is True From 13afd0d296c1977810ef7ff5f6ee9d61fdf2c112 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 13:20:39 -0400 Subject: [PATCH 05/12] Fix srtctl validation: move THINKING/MTP from zip env lists to runtime overrides srtctl's schema rejects list values under benchmark.env (only roles.*.args supports zip expansion lists). Consolidate the two zip_override_thinking_off/on groups into a single zip_override_mtp that only varies speculative-config under roles.agg.args. THINKING and MTP are now bound as runtime env overrides via runtime_arguments() in single_node.py, matching the workflow's per-cell loop variables. Co-Authored-By: Claude Opus 4.6 --- .github/workflows/speedbench-al.yml | 4 +- .../vllm/b300-fp4-speedbench/speedbench.yaml | 27 ++------- .../vllm/b300-fp4-speedbench/speedbench.yaml | 56 +------------------ .../vllm/b300-fp4-speedbench/speedbench.yaml | 56 +------------------ .../vllm/b300-fp4-speedbench/speedbench.yaml | 56 +------------------ .../vllm/b300-fp4-speedbench/speedbench.yaml | 56 +------------------ .../vllm/b300-fp4-speedbench/speedbench.yaml | 56 +------------------ infx/srt_slurm/single_node.py | 2 +- utils/test_srt_single_node.py | 36 ++++++------ 9 files changed, 29 insertions(+), 320 deletions(-) diff --git a/.github/workflows/speedbench-al.yml b/.github/workflows/speedbench-al.yml index 563ac7dc27..468f66c2f9 100644 --- a/.github/workflows/speedbench-al.yml +++ b/.github/workflows/speedbench-al.yml @@ -275,7 +275,9 @@ jobs: for mode in $THINKING_MODES; do for mtp in $MTP_LIST; do IDX=$((mtp - 1)) - export SRT_RECIPE="${RECIPE_PATH}:zip_override_thinking_${mode}[${IDX}]" + export THINKING="$mode" + export MTP="$mtp" + export SRT_RECIPE="${RECIPE_PATH}:zip_override_mtp[${IDX}]" export RESULT_FILENAME="speedbench_${mode}_mtp${mtp}" echo "" echo "==========================================" diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml index 40bce2495a..502762d18b 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/vllm/b300-fp4-speedbench/speedbench.yaml @@ -1,7 +1,8 @@ # DeepSeek-R1 B300 vLLM SPEED-Bench AL matrix (native MTP). # -# R1 always reasons (no thinking-off mode), so only zip_override_thinking_on is -# provided. Each entry varies num_speculative_tokens 1..8. +# R1 always reasons (no thinking-off mode); the workflow dispatches only +# thinking-on cells. Each zip_override_mtp entry varies +# num_speculative_tokens 1..8. # # FP4 MoE on Blackwell needs FlashInfer (VLLM_USE_FLASHINFER_MOE_FP4). # R1 is DeepSeek-V3 architecture (plain MLA): no --attention_config.use_fp4_indexer_cache, @@ -61,7 +62,7 @@ base: TOP_P: '0.95' SPEEDBENCH_TRUST_REMOTE_CODE: '1' -zip_override_thinking_on: +zip_override_mtp: roles: agg: args: @@ -74,23 +75,3 @@ zip_override_thinking_on: - '{"method":"mtp","num_speculative_tokens":6}' - '{"method":"mtp","num_speculative_tokens":7}' - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml index 1a6f607e7c..bc9138c429 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml @@ -64,7 +64,7 @@ base: SPEEDBENCH_TRUST_REMOTE_CODE: '1' APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' -zip_override_thinking_off: +zip_override_mtp: roles: agg: args: @@ -77,57 +77,3 @@ zip_override_thinking_off: - '{"method":"mtp","num_speculative_tokens":6}' - '{"method":"mtp","num_speculative_tokens":7}' - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' - -zip_override_thinking_on: - roles: - agg: - args: - speculative-config: - - '{"method":"mtp","num_speculative_tokens":1}' - - '{"method":"mtp","num_speculative_tokens":2}' - - '{"method":"mtp","num_speculative_tokens":3}' - - '{"method":"mtp","num_speculative_tokens":4}' - - '{"method":"mtp","num_speculative_tokens":5}' - - '{"method":"mtp","num_speculative_tokens":6}' - - '{"method":"mtp","num_speculative_tokens":7}' - - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml index b71e1e3917..3bf4eb36cb 100644 --- a/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/glm5/vllm/b300-fp4-speedbench/speedbench.yaml @@ -65,7 +65,7 @@ base: SPEEDBENCH_TRUST_REMOTE_CODE: '1' APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' -zip_override_thinking_off: +zip_override_mtp: roles: agg: args: @@ -78,57 +78,3 @@ zip_override_thinking_off: - '{"method":"mtp","num_speculative_tokens":6}' - '{"method":"mtp","num_speculative_tokens":7}' - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' - -zip_override_thinking_on: - roles: - agg: - args: - speculative-config: - - '{"method":"mtp","num_speculative_tokens":1}' - - '{"method":"mtp","num_speculative_tokens":2}' - - '{"method":"mtp","num_speculative_tokens":3}' - - '{"method":"mtp","num_speculative_tokens":4}' - - '{"method":"mtp","num_speculative_tokens":5}' - - '{"method":"mtp","num_speculative_tokens":6}' - - '{"method":"mtp","num_speculative_tokens":7}' - - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml index f0302662ce..6e5ea8793a 100644 --- a/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml @@ -62,7 +62,7 @@ base: CHAT_TEMPLATE_KWARGS_OFF: '{"enable_thinking": false}' SPEEDBENCH_TRUST_REMOTE_CODE: '1' -zip_override_thinking_off: +zip_override_mtp: roles: agg: args: @@ -75,57 +75,3 @@ zip_override_thinking_off: - '{"method":"mtp","num_speculative_tokens":6}' - '{"method":"mtp","num_speculative_tokens":7}' - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' - -zip_override_thinking_on: - roles: - agg: - args: - speculative-config: - - '{"method":"mtp","num_speculative_tokens":1}' - - '{"method":"mtp","num_speculative_tokens":2}' - - '{"method":"mtp","num_speculative_tokens":3}' - - '{"method":"mtp","num_speculative_tokens":4}' - - '{"method":"mtp","num_speculative_tokens":5}' - - '{"method":"mtp","num_speculative_tokens":6}' - - '{"method":"mtp","num_speculative_tokens":7}' - - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml index 9b107a2dc5..dc4faecfe2 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/vllm/b300-fp4-speedbench/speedbench.yaml @@ -72,7 +72,7 @@ base: SPEEDBENCH_TRUST_REMOTE_CODE: '1' APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' -zip_override_thinking_off: +zip_override_mtp: roles: agg: args: @@ -85,57 +85,3 @@ zip_override_thinking_off: - '{"method":"mtp","num_speculative_tokens":6}' - '{"method":"mtp","num_speculative_tokens":7}' - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' - -zip_override_thinking_on: - roles: - agg: - args: - speculative-config: - - '{"method":"mtp","num_speculative_tokens":1}' - - '{"method":"mtp","num_speculative_tokens":2}' - - '{"method":"mtp","num_speculative_tokens":3}' - - '{"method":"mtp","num_speculative_tokens":4}' - - '{"method":"mtp","num_speculative_tokens":5}' - - '{"method":"mtp","num_speculative_tokens":6}' - - '{"method":"mtp","num_speculative_tokens":7}' - - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml index cdd234f68b..ace5dc2ef2 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.8next/vllm/b300-fp4-speedbench/speedbench.yaml @@ -83,7 +83,7 @@ base: APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' SPEEDBENCH_CONCURRENCY: '64' -zip_override_thinking_off: +zip_override_mtp: roles: agg: args: @@ -96,57 +96,3 @@ zip_override_thinking_off: - '{"method":"mtp","num_speculative_tokens":6}' - '{"method":"mtp","num_speculative_tokens":7}' - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - - 'thinking_off' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' - -zip_override_thinking_on: - roles: - agg: - args: - speculative-config: - - '{"method":"mtp","num_speculative_tokens":1}' - - '{"method":"mtp","num_speculative_tokens":2}' - - '{"method":"mtp","num_speculative_tokens":3}' - - '{"method":"mtp","num_speculative_tokens":4}' - - '{"method":"mtp","num_speculative_tokens":5}' - - '{"method":"mtp","num_speculative_tokens":6}' - - '{"method":"mtp","num_speculative_tokens":7}' - - '{"method":"mtp","num_speculative_tokens":8}' - benchmark: - env: - THINKING: - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - - 'thinking_on' - MTP: - - '1' - - '2' - - '3' - - '4' - - '5' - - '6' - - '7' - - '8' diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index 6f6060885f..58ea09e196 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -195,7 +195,7 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: if collector: # SPEED-Bench collector tunables: bind workflow-level settings into # the per-cell benchmark environment so the client reads them. - for name in ("CATEGORY", "SPEEDBENCH_OUTPUT_LEN"): + for name in ("CATEGORY", "SPEEDBENCH_OUTPUT_LEN", "THINKING", "MTP"): value = environment.get(name, "") if not value: raise ValueError(f"Missing SPEED-Bench input: {name}") diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index e7168dda01..d621a787ea 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -487,8 +487,6 @@ def speedbench_point(tmp_path): "MODEL": "deepseek-ai/DeepSeek-V4-Pro", "SPEC_DECODING": "mtp", "TEMPERATURE": "1.0", - "THINKING": "thinking_off", - "MTP": "3", }, }, } @@ -503,6 +501,7 @@ def speedbench_point(tmp_path): "CONC": "1", "RESULT_FILENAME": "speedbench_off_mtp3", "GPU_MONITOR_INTERVAL": "3", "MODEL_PREFIX": "dsv4", "CATEGORY": "qualitative", "SPEEDBENCH_OUTPUT_LEN": "512", + "THINKING": "off", "MTP": "3", # ISL/OSL/RANDOM_RANGE_RATIO intentionally absent -- collector skips them } return path, recipe, env @@ -554,10 +553,11 @@ def test_speedbench_collector_rejects_missing_output_len(speedbench_point): def test_speedbench_zip_override_selects_cell(tmp_path): - """Collector recipe with zip_override + selector resolves to one cell. + """Collector recipe with zip_override_mtp + selector resolves to one cell. - The workflow constructs the exact selector (e.g. zip_override_thinking_off[1]) + The workflow constructs the exact selector (e.g. zip_override_mtp[1]) for each cell, so select_recipe sees exactly one matching variant. + THINKING and MTP are bound as runtime env overrides, not in the recipe. """ recipe = { "engine": "vllm", @@ -575,15 +575,11 @@ def test_speedbench_zip_override_selects_cell(tmp_path): "env": {"MODEL": "deepseek-ai/DeepSeek-V4-Pro", "SPEC_DECODING": "mtp", "TEMPERATURE": "1.0"}, }, } - raw = {"base": recipe, "zip_override_thinking_off": { + raw = {"base": recipe, "zip_override_mtp": { "roles": {"agg": {"args": {"speculative-config": [ '{"method":"mtp","num_speculative_tokens":1}', '{"method":"mtp","num_speculative_tokens":2}', ]}}}, - "benchmark": {"env": { - "THINKING": ["thinking_off", "thinking_off"], - "MTP": ["1", "2"], - }}, }} path = tmp_path / "speedbench.yaml" path.write_text(yaml.safe_dump(raw)) @@ -596,14 +592,16 @@ def test_speedbench_zip_override_selects_cell(tmp_path): "CONC": "1", "RESULT_FILENAME": "speedbench_off_mtp2", "GPU_MONITOR_INTERVAL": "3", "MODEL_PREFIX": "dsv4", "CATEGORY": "qualitative", "SPEEDBENCH_OUTPUT_LEN": "512", + "THINKING": "off", "MTP": "2", } # The workflow provides an explicit selector: select cell [1] (MTP=2). - config, selected = select_recipe(f"{path}:zip_override_thinking_off[1]", env) - assert "zip_override_thinking_off[1]" in config - assert selected["benchmark"]["env"]["MTP"] == "2" + config, selected = select_recipe(f"{path}:zip_override_mtp[1]", env) + assert "zip_override_mtp[1]" in config # Verify runtime_arguments works on the selected cell. - argv = runtime_arguments(f"{path}:zip_override_thinking_off[1]", env) + argv = runtime_arguments(f"{path}:zip_override_mtp[1]", env) assert any("CATEGORY" in arg for arg in argv) + assert any("THINKING" in arg for arg in argv) + assert any("MTP" in arg for arg in argv) @pytest.mark.parametrize( @@ -612,23 +610,21 @@ def test_speedbench_zip_override_selects_cell(tmp_path): def test_speedbench_cells_keep_string_env_through_srtctl_variant_round_trip(recipe): # srtctl expands zip variants with ruamel (YAML 1.2) and reloads them with a # YAML 1.1 loader, where bare on/off become booleans and fail its schema. + # THINKING and MTP are now runtime env overrides (not in the recipe), so only + # the speculative-config list under roles.agg.args is expanded here. from srtctl.core.config import resolve_override_yaml from srtctl.core.yaml_utils import dump_yaml_with_comments raw = yaml.safe_load(recipe.read_text()) - selectors = [ - f"{group}[{index}]" - for group in raw - if group.startswith("zip_override_thinking_") - for index in range(len(raw[group]["benchmark"]["env"]["MTP"])) - ] + assert "zip_override_mtp" in raw, f"Missing zip_override_mtp in {recipe}" + spec_list = raw["zip_override_mtp"]["roles"]["agg"]["args"]["speculative-config"] + selectors = [f"zip_override_mtp[{i}]" for i in range(len(spec_list))] assert selectors for selector in selectors: [(_, variant)] = resolve_override_yaml(recipe, selector=selector) reloaded = yaml.safe_load(dump_yaml_with_comments(variant)) env = reloaded["benchmark"]["env"] assert all(isinstance(value, str) for value in env.values()), (selector, env) - assert env["THINKING"] in {"thinking_on", "thinking_off"} @pytest.mark.parametrize( From 50f9a78012055cd6eded378b68156865138b7c6e Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 13:27:57 -0400 Subject: [PATCH 06/12] Keep runtime THINKING a string through srtctl --set 13afd0d2 moved THINKING to a runtime --set override, which reintroduced the YAML 1.1 on/off->bool round trip ('Not a valid string'). Bind it as thinking_on/thinking_off; srt_speedbench.sh strips the prefix. Co-Authored-By: Claude Opus 5.5 (1M context) --- infx/srt_slurm/single_node.py | 4 ++++ utils/test_srt_single_node.py | 27 +++++++++++++++++++++++++++ 2 files changed, 31 insertions(+) diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index 58ea09e196..3a8c1202fb 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -199,6 +199,10 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: value = environment.get(name, "") if not value: raise ValueError(f"Missing SPEED-Bench input: {name}") + # srtctl dumps the overridden recipe and reloads it as YAML 1.1, + # where a bare on/off becomes a boolean; the client strips this prefix. + if name == "THINKING" and value in ("on", "off"): + value = f"thinking_{value}" overrides += ["--set", f"benchmark.env.{name}={json.dumps(value)}"] # Chat-template kwargs are optional (dsr1 has no off mode). for name in ("CHAT_TEMPLATE_KWARGS_ON",): diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index d621a787ea..a8de1396c2 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -627,6 +627,33 @@ def test_speedbench_cells_keep_string_env_through_srtctl_variant_round_trip(reci assert all(isinstance(value, str) for value in env.values()), (selector, env) +@pytest.mark.parametrize("mode", ["on", "off"]) +def test_speedbench_runtime_thinking_stays_string_through_srtctl_set(mode): + # The workflow binds THINKING per cell via --set; srtctl dumps the result + # and reloads it as YAML 1.1, where a bare on/off becomes a boolean. + from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides + from srtctl.core.yaml_utils import dump_yaml_with_comments, load_yaml_text_with_comments + + recipe = Path("benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml") + env = { + "FRAMEWORK": "vllm", "MODEL": "deepseek-ai/DeepSeek-V4-Pro", + "IMAGE": "vllm/vllm-openai:v0.21.0", "PRECISION": "fp4", + "TP": "8", "GPU_COUNT": "8", "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", + "EP_SIZE": "1", "DP_ATTENTION": "false", "SPEC_DECODING": "mtp", "IS_AGENTIC": "0", + "RUN_EVAL": "false", "EVAL_ONLY": "false", + "CONC": "1", "RESULT_FILENAME": f"speedbench_{mode}_mtp1", + "GPU_MONITOR_INTERVAL": "3", "MODEL_PREFIX": "dsv4", + "CATEGORY": "coding", "SPEEDBENCH_OUTPUT_LEN": "4096", + "THINKING": mode, "MTP": "1", + } + argv = runtime_arguments(f"{recipe}:zip_override_mtp[0]", env) + sets = [argv[i + 1] for i, arg in enumerate(argv) if arg == "--set"] + document = load_yaml_text_with_comments(recipe.read_text()) + apply_overrides_to_recipe(document, parse_overrides(sets, None)) + reloaded = yaml.safe_load(dump_yaml_with_comments(document)) + assert reloaded["base"]["benchmark"]["env"]["THINKING"] == f"thinking_{mode}" + + @pytest.mark.parametrize( "recipe", sorted(Path("benchmarks/single_node/srt-slurm-recipes").glob("*/vllm/b300-fp4-speedbench/speedbench.yaml")) ) From f1e8d81e6fc05d4c462faca3d793007cd00dbe50 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 13:28:28 -0400 Subject: [PATCH 07/12] Tidy SpeedBench test imports and unused variable Co-Authored-By: Claude Opus 5.5 (1M context) --- utils/test_srt_single_node.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index a8de1396c2..8661ea9f7e 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -595,7 +595,7 @@ def test_speedbench_zip_override_selects_cell(tmp_path): "THINKING": "off", "MTP": "2", } # The workflow provides an explicit selector: select cell [1] (MTP=2). - config, selected = select_recipe(f"{path}:zip_override_mtp[1]", env) + config, _ = select_recipe(f"{path}:zip_override_mtp[1]", env) assert "zip_override_mtp[1]" in config # Verify runtime_arguments works on the selected cell. argv = runtime_arguments(f"{path}:zip_override_mtp[1]", env) @@ -632,7 +632,10 @@ def test_speedbench_runtime_thinking_stays_string_through_srtctl_set(mode): # The workflow binds THINKING per cell via --set; srtctl dumps the result # and reloads it as YAML 1.1, where a bare on/off becomes a boolean. from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides - from srtctl.core.yaml_utils import dump_yaml_with_comments, load_yaml_text_with_comments + from srtctl.core.yaml_utils import ( + dump_yaml_with_comments, + load_yaml_text_with_comments, + ) recipe = Path("benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-speedbench/speedbench.yaml") env = { From df1dff971460f936232645dd2310b657ec591b57 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 14:08:58 -0400 Subject: [PATCH 08/12] Fix SpeedBench client workspace fallback depth srt_speedbench.sh lives in benchmarks/single_node/, two levels below the repo root, but its fallback went up one level (copied from benchmarks/srt_agentic.sh), so both cells sourced /infmax-workspace/benchmarks/benchmarks/benchmark_lib.sh and exited 1 after the server was ready. Adds a test that runs the client with the legacy /workspace and asserts it reaches check_env_vars. Co-Authored-By: Claude Opus 5.5 (1M context) --- benchmarks/single_node/srt_speedbench.sh | 4 ++-- utils/test_srt_single_node.py | 15 +++++++++++++++ 2 files changed, 17 insertions(+), 2 deletions(-) diff --git a/benchmarks/single_node/srt_speedbench.sh b/benchmarks/single_node/srt_speedbench.sh index e4af545b91..cf99765793 100644 --- a/benchmarks/single_node/srt_speedbench.sh +++ b/benchmarks/single_node/srt_speedbench.sh @@ -16,9 +16,9 @@ set -eo pipefail # Jobs inherit the legacy scripts' /workspace, which srt-slurm does not mount; -# fall back to the repo mount this client runs from. +# fall back to the repo mount this client runs from (two levels up). if [[ ! -f "${INFMAX_CONTAINER_WORKSPACE:-}/benchmarks/benchmark_lib.sh" ]]; then - INFMAX_CONTAINER_WORKSPACE="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" + INFMAX_CONTAINER_WORKSPACE="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" fi export INFMAX_CONTAINER_WORKSPACE diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 8661ea9f7e..f0f03eda0d 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -664,3 +664,18 @@ def test_speedbench_recipes_bind_gpus_without_device_ids(recipe): # vLLM v0.21.0 rejects srtctl's default --device-ids binding at startup. raw = yaml.safe_load(recipe.read_text()) assert raw["base"]["engine"]["set_visible_devices"] is True + + +def test_speedbench_client_finds_benchmark_lib_without_legacy_workspace(): + # srt-slurm mounts the repo at /infmax-workspace, not the legacy /workspace; + # the client must fall back to its own checkout and reach check_env_vars. + result = subprocess.run( + ["bash", "benchmarks/single_node/srt_speedbench.sh"], + env={"PATH": os.environ["PATH"], "INFMAX_CONTAINER_WORKSPACE": "/workspace"}, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode != 0 + assert "No such file" not in result.stderr, result.stderr + assert "required environment variables are not set" in result.stdout From f481e3c2226876ecbe4da93f51f8432085269d46 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 13:39:41 -0400 Subject: [PATCH 09/12] feat(speedbench): migrate remaining AL collectors to srt-slurm recipes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port dsv4dspark, dsv4dsparkprob, kimik3, kimik3prob, and minimaxm3 SPEED-Bench acceptance-length collectors from ad-hoc bash scripts to srt-slurm YAML recipes. Remove the legacy BENCH_SCRIPT_OVERRIDE path from the speedbench-al workflow so all nine prefixes now flow through the unified srt-slurm recipe resolver. Move test_speedbench_matrix.py to infx/tests/workflows/. 将 dsv4dspark、dsv4dsparkprob、kimik3、kimik3prob 和 minimaxm3 的 SPEED-Bench 接受长度收集器从临时 bash 脚本迁移到 srt-slurm YAML 配方。 从 speedbench-al 工作流中移除旧版 BENCH_SCRIPT_OVERRIDE 路径,使全部 九个前缀统一通过 srt-slurm 配方解析器运行。将 test_speedbench_matrix.py 迁移至 infx/tests/workflows/。 Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/speedbench-al.yml | 139 ++++--- .../speedbench/dsv4dspark_fp4_b300_vllm.sh | 348 ------------------ .../dsv4dsparkprob_fp4_b300_vllm.sh | 10 - .../speedbench/kimik3_fp4_b300_vllm.sh | 338 ----------------- ...le_method_block_rejection_sample_method.sh | 342 ----------------- .../speedbench/minimaxm3_fp4_b300_vllm.sh | 245 ------------ .../vllm/b300-fp4-speedbench/speedbench.yaml | 87 +++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 82 +++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 91 +++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 87 +++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 86 +++++ infx/tests/srt_slurm/test_srt_single_node.py | 80 ++++ .../workflows}/test_speedbench_matrix.py | 0 runners/launch_b300-dsxe.sh | 5 +- 14 files changed, 577 insertions(+), 1363 deletions(-) delete mode 100755 benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/dsv4dsparkprob_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh delete mode 100755 benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsv4dspark/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsv4dsparkprob/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/kimik3prob/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml rename {utils => infx/tests/workflows}/test_speedbench_matrix.py (100%) diff --git a/.github/workflows/speedbench-al.yml b/.github/workflows/speedbench-al.yml index 468f66c2f9..819efce04c 100644 --- a/.github/workflows/speedbench-al.yml +++ b/.github/workflows/speedbench-al.yml @@ -21,7 +21,7 @@ on: # zizmor: ignore[concurrency-limits] type: string default: 'deepseek-ai/DeepSeek-V4-Pro' model-prefix: - description: "Model prefix; drives launcher MODEL_PATH resolution, exp name, collector script, and artifact names" + description: "Model prefix; drives srt-slurm recipe lookup, exp name, and artifact names" required: false type: string default: 'dsv4' @@ -238,84 +238,69 @@ jobs: source runners/runtime_settings.sh fi - # Migrated native-MTP prefixes use the srt-slurm single-node path. - # Draft-model collectors (dsv4dspark*, kimik3, minimaxm3) keep the - # legacy BENCH_SCRIPT_OVERRIDE path. - _is_srt_slurm_prefix() { - case "$1" in - dsr1|dsv4|glm5|glm52|qwen3.5|qwen3.8next) return 0 ;; - *) return 1 ;; - esac - } - - if _is_srt_slurm_prefix "$MODEL_PREFIX"; then - # --- srt-slurm per-cell collection --- - RECIPE_PATH="benchmarks/single_node/srt-slurm-recipes/${MODEL_PREFIX}/vllm/b300-fp4-speedbench/speedbench.yaml" - if [[ ! -f "$RECIPE_PATH" ]]; then - echo "ERROR: SPEED-Bench recipe not found: $RECIPE_PATH" >&2 - exit 1 - fi - # Plain TP8 is incompatible with the Qwen3.8-Flash-Next FP8 checkpoint's - # 128-wide quantization blocks; its recipe runs TP4 (as the legacy - # collector did), so bind the matrix TP and GPU count to match. - if [[ "$MODEL_PREFIX" == qwen3.8next ]]; then - export TP=4 GPU_COUNT=4 - fi - mkdir -p speedbench_results - ALL_FAILED=true - # Harmless values for launch_srt_single_node env checks; the collector - # recipe and client do not read them. - export ISL=256 - export OSL=256 - export RANDOM_RANGE_RATIO=0.0 - export CONC=1 - export PP_SIZE=1 - export DCP_SIZE=1 - export PCP_SIZE=1 - for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - IDX=$((mtp - 1)) - export THINKING="$mode" - export MTP="$mtp" - export SRT_RECIPE="${RECIPE_PATH}:zip_override_mtp[${IDX}]" - export RESULT_FILENAME="speedbench_${mode}_mtp${mtp}" - echo "" - echo "==========================================" - echo " srt-slurm cell: thinking=${mode} MTP=${mtp}" - echo " SRT_RECIPE=${SRT_RECIPE}" - echo "==========================================" - CELL_RC=0 - bash ./runners/launch_"${RUNNER_NAME%%_*}".sh || CELL_RC=$? - # Collect per-cell artifacts into the results directory. - if [[ -f "${RESULT_FILENAME}.json" ]]; then - mv "${RESULT_FILENAME}.json" "speedbench_results/${RESULT_FILENAME}.json" - ALL_FAILED=false - fi - if [[ -f srt-single-node-logs.tar.gz ]]; then - mv srt-single-node-logs.tar.gz "speedbench_results/srt-logs_${mode}_mtp${mtp}.tar.gz" - fi - if [[ "$CELL_RC" -ne 0 ]]; then - echo " -> cell failed (rc=$CELL_RC), recording N/A" - fi - done + # All collectors use the srt-slurm single-node path with per-model + # recipes under srt-slurm-recipes//vllm/b300-fp4-speedbench/. + RECIPE_PATH="benchmarks/single_node/srt-slurm-recipes/${MODEL_PREFIX}/vllm/b300-fp4-speedbench/speedbench.yaml" + if [[ ! -f "$RECIPE_PATH" ]]; then + echo "ERROR: SPEED-Bench recipe not found: $RECIPE_PATH" >&2 + exit 1 + fi + # Plain TP8 is incompatible with the Qwen3.8-Flash-Next FP8 checkpoint's + # 128-wide quantization blocks; its recipe runs TP4 (as the legacy + # collector did), so bind the matrix TP and GPU count to match. + if [[ "$MODEL_PREFIX" == qwen3.8next ]]; then + export TP=4 GPU_COUNT=4 + fi + mkdir -p speedbench_results + ALL_FAILED=true + # Harmless values for launch_srt_single_node env checks; the collector + # recipe and client do not read them. + export ISL=256 + export OSL=256 + export RANDOM_RANGE_RATIO=0.0 + export CONC=1 + export PP_SIZE=1 + export DCP_SIZE=1 + export PCP_SIZE=1 + for mode in $THINKING_MODES; do + for mtp in $MTP_LIST; do + IDX=$((mtp - 1)) + export THINKING="$mode" + export MTP="$mtp" + export SRT_RECIPE="${RECIPE_PATH}:zip_override_mtp[${IDX}]" + export RESULT_FILENAME="speedbench_${mode}_mtp${mtp}" + echo "" + echo "==========================================" + echo " srt-slurm cell: thinking=${mode} MTP=${mtp}" + echo " SRT_RECIPE=${SRT_RECIPE}" + echo "==========================================" + CELL_RC=0 + bash ./runners/launch_"${RUNNER_NAME%%_*}".sh || CELL_RC=$? + # Collect per-cell artifacts into the results directory. + if [[ -f "${RESULT_FILENAME}.json" ]]; then + mv "${RESULT_FILENAME}.json" "speedbench_results/${RESULT_FILENAME}.json" + ALL_FAILED=false + fi + if [[ -f srt-single-node-logs.tar.gz ]]; then + mv srt-single-node-logs.tar.gz "speedbench_results/srt-logs_${mode}_mtp${mtp}.tar.gz" + fi + if [[ "$CELL_RC" -ne 0 ]]; then + echo " -> cell failed (rc=$CELL_RC), recording N/A" + fi done - if [[ "$ALL_FAILED" == true ]]; then - echo "ERROR: every SPEED-Bench cell failed" >&2 - exit 1 - fi - # Aggregate per-cell result JSONs into the reference YAML. - PYTHONPATH="${GITHUB_WORKSPACE}${PYTHONPATH:+:$PYTHONPATH}" \ - python3 -m infx.workflows.speedbench_matrix \ - --result-dir speedbench_results \ - --model-key "$(basename "$(echo "$MODEL" | tr '[:upper:]' '[:lower:]')")" \ - --thinking-modes "$THINKING_MODES" \ - --mtp-list "$MTP_LIST" \ - > speedbench-reference-al.yaml - else - # --- Legacy collector script --- - export BENCH_SCRIPT_OVERRIDE="benchmarks/single_node/speedbench/${MODEL_PREFIX}_fp4_b300_vllm.sh" - bash ./runners/launch_"${RUNNER_NAME%%_*}".sh + done + if [[ "$ALL_FAILED" == true ]]; then + echo "ERROR: every SPEED-Bench cell failed" >&2 + exit 1 fi + # Aggregate per-cell result JSONs into the reference YAML. + PYTHONPATH="${GITHUB_WORKSPACE}${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.workflows.speedbench_matrix \ + --result-dir speedbench_results \ + --model-key "$(basename "$(echo "$MODEL" | tr '[:upper:]' '[:lower:]')")" \ + --thinking-modes "$THINKING_MODES" \ + --mtp-list "$MTP_LIST" \ + > speedbench-reference-al.yaml if [ ! -f "speedbench-reference-al.yaml" ]; then echo "AL collection failed: speedbench-reference-al.yaml not produced." >&2 diff --git a/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh deleted file mode 100755 index fabc3b421f..0000000000 --- a/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh +++ /dev/null @@ -1,348 +0,0 @@ -#!/usr/bin/env bash - -# DSV4-Pro B300 vLLM SPEED-Bench AL matrix collector for DSpark speculative decoding. -# -# For each thinking mode (on/off) and DSpark speculative-token count, measure the REAL -# acceptance length (AL) on one SPEED-Bench category and emit a YAML matrix in the -# golden_al_distribution shape. DSpark ships as a separate checkpoint -# (deepseek-ai/DeepSeek-V4-Pro-DSpark, 960 GB) with the draft baked in, so there is no -# external draft head and no "model" key in the speculative-config. Every flag that -# affects drafting is byte-identical to the DSV4 MTP collector so the two AL curves -# stay comparable. -# -# Dispatch this collector through speedbench-al.yml. -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DRAFT_SAMPLE_METHOD MODEL MODEL_PATH MTP_LIST \ - OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -# MODEL_PATH is the launcher-resolved weights dir (writable models dir until the -# checkpoint is staged; see the download block below). -SERVE_MODEL="${MODEL_PATH}" - -# Top-level key in the emitted YAML matrix comes from the model basename. -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -# AL is a per-draft accept/reject property independent of batch size, so batch the -# SPEED-Bench pass to cut wall-clock. Nothing sets speculative_disable_by_batch_size, -# so drafting stays on at this batch size. -CONCURRENCY="32" -# Must stay >= CONCURRENCY or requests just queue. Held far below vLLM's default of -# 1024 because that sizes two allocations the memory profiler never sees: the -# rejection sampler's fp32 logits scratch (max_num_seqs * (1 + spec_tokens) * vocab * -# 4B, 4.4 GB at 8 tokens) and the spec-decode CUDA graphs. DSV4-Pro has no room: -# weights + a 100 GiB KV cache already fill 266 of 268 GiB per B300. -MAX_NUM_SEQS="64" -# Reserve device memory for KV cache and speculative verification. -GPU_MEM_UTIL="0.90" -TEMPERATURE="1.0" -# MUST match the golden config: infx/golden_al_distribution/dsv4_mtp.yaml was measured -# with reasoning_effort=high. -# The published recipe uses greedy; probabilistic won at every level on Kimi-K3 -# (infx/golden_al_distribution/kimik3_dspark*.yaml). vLLM accepts exactly these two values -# (vllm/config/speculative.py: DraftSampleMethod). -case "$DRAFT_SAMPLE_METHOD" in - greedy|probabilistic) ;; - *) - echo "CRITICAL: DRAFT_SAMPLE_METHOD must be 'greedy' or 'probabilistic' (got '$DRAFT_SAMPLE_METHOD')" - exit 1 - ;; -esac -# Opt-in rather than tied to draft_sample_method: flipping it to "block" would bundle -# two variables into one measurement, and the forced-AL config has to stay on a -# sampling method TRT-LLM supports too. -REJECTION_SAMPLE_METHOD="${REJECTION_SAMPLE_METHOD:-}" - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi - -# The DSpark checkpoint is not in the launcher's STAGED_MODELS, so MODEL_PATH resolves -# to the writable models dir and the ~960 GB download runs once. Add the basename to -# STAGED_MODELS once the weights are staged on the read-only mount. -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - if [[ ! -w "$(dirname "$MODEL_PATH")" ]]; then - echo "CRITICAL: $MODEL_PATH is empty and $(dirname "$MODEL_PATH") is not writable." - echo "This means the basename is listed in the launcher's STAGED_MODELS but the" - echo "weights were never staged. Either get them staged, or remove it from" - echo "STAGED_MODELS so MODEL_PATH resolves to the writable models dir instead." - exit 1 - fi - echo "=== $MODEL_PATH is empty; downloading $MODEL (~960 GB, first run only) ===" - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -# speed_bench/CustomDataset renders the chat template client-side and posts to -# /v1/completions, so thinking mode must reach apply_chat_template via -# --chat-template-kwargs (native since vllm-project/vllm#44244). Assert rather than -# assume: if the CLI option exists but speed_bench does not forward it, the flag is -# silently ignored and every thinking_on cell reports a non-thinking AL. -assert_chat_template_kwargs_support() { - echo "=== Checking vLLM benchmark --chat-template-kwargs support ===" - python3 - <<'PYEOF' -import sys -import vllm.benchmarks.serve as S -import vllm.benchmarks.datasets.datasets as D - -def read(mod): - with open(mod.__file__) as fh: - return fh.read() - -s_src, d_src = read(S), read(D) - -missing = [] -if '"--chat-template-kwargs"' not in s_src: - missing.append(f"CLI option in {S.__file__}") -if ('chat_template_kwargs=getattr(args' not in d_src - and 'chat_template_kwargs=args.chat_template_kwargs' not in d_src): - missing.append(f"speed_bench forward in {D.__file__}") -if '**(chat_template_kwargs or {})' not in d_src: - missing.append(f"apply_chat_template unpack in {D.__file__}") - -if missing: - print("CRITICAL: this image lacks native --chat-template-kwargs support:") - for item in missing: - print(" missing:", item) - print("thinking_on cells would silently measure a non-thinking AL. Use an") - print("image that contains vllm-project/vllm#44244.") - sys.exit(1) - -print("native --chat-template-kwargs support confirmed") -PYEOF -} - -if [[ " $THINKING_MODES " == *" on "* ]]; then - if ! assert_chat_template_kwargs_support; then - echo "CRITICAL: --chat-template-kwargs preflight failed — aborting" - exit 1 - fi -fi - -# TEP8 as in the published B300 DSpark recipe (vllm-project/recipes). Hard-coded -# rather than driven by EP_SIZE / DP_ATTENTION because speedbench-al.yml exports -# EP_SIZE=1 and DP_ATTENTION=false for every model, which silently turned the recipe -# into plain TP; TP-sharding the FP4 experts costs ~37 GiB per GPU over EP and made -# the num_speculative_tokens=4 cell OOM. AL is unaffected by expert placement. -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -EP_ARGS=(--enable-expert-parallel) -MOE_ARGS=(--moe-backend deep_gemm_mega_moe) - -SPEC_EXTRA="" -if [[ -n "$REJECTION_SAMPLE_METHOD" ]]; then - SPEC_EXTRA=", \"rejection_sample_method\": \"$REJECTION_SAMPLE_METHOD\"" -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -# Descendant PIDs of $1 by PARENT pid. This can never include this script (an -# ancestor of the server), unlike a name-based `pkill -f vllm`, which self-killed -# because the script filename contains "vllm". -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - # Snapshot the worker/EngineCore subprocesses BEFORE killing the parent: once it - # dies the children reparent to init and the tree link is lost. An orphaned - # worker holds GPU memory and OOMs the next server start. - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - if [[ "$mode" == "on" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode DSPARK=$mtp category=$CATEGORY" - echo " draft_sample_method=$DRAFT_SAMPLE_METHOD" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --kv-cache-dtype fp8 - --trust-remote-code - --block-size 256 - --no-enable-prefix-caching - "${EP_ARGS[@]}" - "${MOE_ARGS[@]}" - --compilation-config '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' - --attention_config.use_fp4_indexer_cache True - --tokenizer-mode deepseek_v4 - --tool-call-parser deepseek_v4 - --enable-auto-tool-choice - --reasoning-parser deepseek_v4 - --max-cudagraph-capture-size 2048 - --max-model-len 16384 - --max-num-seqs "$MAX_NUM_SEQS" - --gpu-memory-utilization "$GPU_MEM_UTIL" - --speculative-config "{\"method\": \"dspark\", \"num_speculative_tokens\": $mtp, \"draft_sample_method\": \"$DRAFT_SAMPLE_METHOD\"$SPEC_EXTRA}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - # wait_for_server_ready exits the shell (rather than returning) when the server - # dies; the subshell keeps that exit local so one bad cell does not abort the matrix. - if ! (wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"); then - echo " -> server failed to start (thinking=$mode dspark=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --tokenizer-mode deepseek_v4 \ - --temperature "$TEMPERATURE" \ - "${think_args[@]}" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode DSPARK=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -SPEC_SUMMARY="method=dspark | draft_sample_method=$DRAFT_SAMPLE_METHOD" -if [[ -n "$REJECTION_SAMPLE_METHOD" ]]; then - SPEC_SUMMARY="$SPEC_SUMMARY | rejection_sample_method=$REJECTION_SAMPLE_METHOD" -fi - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | temperature: $TEMPERATURE | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - echo "# speculative-config: $SPEC_SUMMARY" - echo "# Measured on $MODEL_KEY (B300, vLLM DSpark), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (DSpark level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" - -# A matrix where every cell is N/A is a failed collection, not a result: fail the -# job so it is not mistaken for a curve worth reviewing. -MEASURED=0 -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - [[ "${AL_RESULT[${mode}_${mtp}]:-N/A}" != "N/A" ]] && MEASURED=$((MEASURED + 1)) - done -done -if [[ "$MEASURED" -eq 0 ]]; then - echo "CRITICAL: no cell produced an AL value — see the server logs and the" - echo "benchmark client output above." - exit 1 -fi diff --git a/benchmarks/single_node/speedbench/dsv4dsparkprob_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/dsv4dsparkprob_fp4_b300_vllm.sh deleted file mode 100755 index efc6b69a1a..0000000000 --- a/benchmarks/single_node/speedbench/dsv4dsparkprob_fp4_b300_vllm.sh +++ /dev/null @@ -1,10 +0,0 @@ -#!/usr/bin/env bash - -# Probabilistic-drafting arm of the DSV4-Pro DSpark AL collection: identical to -# dsv4dspark_fp4_b300_vllm.sh except draft_sample_method=probabilistic (the recipe -# uses greedy). A separate file only because speedbench-al.yml resolves the collector -# as ${model-prefix}_fp4_b300_vllm.sh; it delegates so the two arms cannot drift. -# Dispatch with model-prefix=dsv4dsparkprob. - -exec env DRAFT_SAMPLE_METHOD=probabilistic \ - bash "$(dirname "$0")/dsv4dspark_fp4_b300_vllm.sh" "$@" diff --git a/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh deleted file mode 100755 index f1ea2cf347..0000000000 --- a/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh +++ /dev/null @@ -1,338 +0,0 @@ -#!/usr/bin/env bash - -# Kimi-K3 B300 vLLM SPEED-Bench AL matrix collector for DSpark speculative decoding. -# -# For each DSpark speculative-token count, measure the REAL acceptance length (AL) on -# one SPEED-Bench category and emit a YAML matrix in the golden_al_distribution shape. -# The Inferact/Kimi-K3-DSpark draft head is downloaded to a writable workspace dir; -# the target (moonshotai/Kimi-K3, FP4) is pre-staged at /scratch/models/Kimi-K3. -# K3 is a thinking model (kimi_k3 reasoning parser defaults enable_thinking=True), -# so the golden curve is collected for thinking_on only. -# -# Usage (inside the Kimi-K3 vLLM container, on a B300 node): -# export MODEL=moonshotai/Kimi-K3 -# bash benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" -GPU_MEM_UTIL="0.90" -MAX_MODEL_LEN="16384" -MAX_NUM_SEQS="512" - -DRAFT_MODEL="Inferact/Kimi-K3-DSpark" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -# AL is concurrency-independent (per-token accept/reject; no spec-disable-by-batch is -# set), so batch the SPEED-Bench pass to stay under the CI wall-time limit; conc=1 -# blew the 8h budget on an earlier Kimi model. -CONCURRENCY="64" -TOP_P="0.95" -# K3 defaults to thinking ON, so the on-cell kwargs are explicit and the off-cell -# kwargs disable it. speedbench-al.yml's thinking-kwargs input defaults to the DSV4 -# value and is exported as CHAT_TEMPLATE_KWARGS_ON; dispatch K3 with -# -f 'thinking-kwargs={"thinking": true}'. -CHAT_TEMPLATE_KWARGS_OFF='{"thinking": false}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export NCCL_DMABUF_ENABLE="0" -export VLLM_ALLREDUCE_USE_FLASHINFER="1" -export VLLM_USE_RUST_FRONTEND="1" -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -# `vllm bench serve` os.execv's the CLIENT into the Rust vllm-rs binary whenever the -# dataset (speed_bench qualifies) and backend are supported, with no opt-out. The -# Rust flag surface lacks --speed-bench-output-len, --save-detailed and -# --chat-template-kwargs, so every cell would die at argument parsing. Call the -# Python entrypoint directly (vllm.benchmarks.serve.main), as the CLI's own Python -# fallback does. The SERVER keeps the Rust frontend; AL is read from /metrics. -BENCH_DRIVER="$RESULTS_DIR/bench_serve_python.py" -mkdir -p "$RESULTS_DIR" -cat > "$BENCH_DRIVER" <<'PYEOF' -# Python-only `vllm bench serve`: bypasses the Rust os.execv delegation in -# vllm/entrypoints/cli/benchmark/serve.py by importing the benchmark directly. -from vllm.benchmarks.serve import add_cli_args, main - -try: # import path moved in newer vLLM - from vllm.utils.argparse_utils import FlexibleArgumentParser -except ImportError: - from vllm.utils import FlexibleArgumentParser - -parser = FlexibleArgumentParser( - description="vllm bench serve (Python entrypoint, no Rust delegation)" -) -add_cli_args(parser) -main(parser.parse_args()) -PYEOF -# No `set -e` here, so a failed redirect would otherwise only surface later as a -# confusing "No such file" from the preflight. -if [[ ! -s "$BENCH_DRIVER" ]]; then - echo "CRITICAL: could not write the benchmark driver to $BENCH_DRIVER — aborting." - exit 1 -fi - -nvidia-smi - -# Kimi-K3 is in the launcher's STAGED_MODELS (read-only /scratch/models/Kimi-K3), -# so this is a no-op in CI; it covers a standalone run with unstaged weights. -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi -fi - -# dirname(MODEL_PATH) can be the read-only staged mount (/scratch/models), so the -# draft must go to a writable workspace dir, not next to the target. -DRAFT_DIR="/workspace/draft_models" -mkdir -p "$DRAFT_DIR" -DRAFT_MODEL_PATH="$DRAFT_DIR/${DRAFT_MODEL##*/}" -if [[ ! -d "$DRAFT_MODEL_PATH" || -z "$(ls -A "$DRAFT_MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$DRAFT_MODEL" --local-dir "$DRAFT_MODEL_PATH" -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -NEED_SHIM=0 -if [[ " $THINKING_MODES " == *" on "* && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then NEED_SHIM=1; fi -if [[ " $THINKING_MODES " == *" off "* && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then NEED_SHIM=1; fi -if [[ "$NEED_SHIM" == "1" ]]; then - if ! apply_chat_template_kwargs_shim; then - echo "CRITICAL: --chat-template-kwargs shim failed — aborting" - exit 1 - fi -fi - -# Preflight the client flags every cell uses; otherwise a CLI mismatch only shows up -# as an all-N/A matrix after eight full server starts (~1h). Probe the driver, not -# `vllm bench serve` (its help exits before the Rust execv), and ask for --help=all -# since plain --help prints only a group summary. -BENCH_HELP="$(python3 "$BENCH_DRIVER" --help=all 2>&1)" -for flag in --speed-bench-category --speed-bench-output-len --chat-template-kwargs --save-detailed; do - if [[ "$BENCH_HELP" != *"$flag"* ]]; then - echo "CRITICAL: the Python benchmark entrypoint does not support $flag — aborting." - echo "--- python3 $BENCH_DRIVER --help=all ---" - echo "$BENCH_HELP" - exit 1 - fi -done -echo "=== Benchmark client flag preflight OK ===" - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi -EP_ARGS=() -if [ "${EP_SIZE}" -gt 1 ]; then - EP_ARGS=(--enable-expert-parallel) -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - local temperature - if [[ "$mode" == "on" ]]; then - temperature=1.0 - if [[ -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - fi - else - temperature=0.6 - if [[ -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - fi - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode DSPARK=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --trust-remote-code - --load-format fastsafetensors - --moe-backend auto - --enable-prefix-caching - --kv-cache-dtype fp8 - "${EP_ARGS[@]}" - --reasoning-parser kimi_k3 - --tool-call-parser kimi_k3 - --enable-auto-tool-choice - --gpu-memory-utilization "$GPU_MEM_UTIL" - --max-num-seqs "$MAX_NUM_SEQS" - --max-model-len "$MAX_MODEL_LEN" - --max-cudagraph-capture-size 256 - --attention-config '{"mla_prefill_backend":"FLASHINFER","use_prefill_query_quantization":true}' - --speculative-config "{\"method\": \"dspark\", \"model\": \"$DRAFT_MODEL_PATH\", \"num_speculative_tokens\": $mtp, \"attention_backend\": \"FLASHINFER_MLA\"}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode dspark=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - python3 "$BENCH_DRIVER" \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$temperature" \ - --top-p "$TOP_P" \ - "${think_args[@]}" - local bench_rc=$? - if [[ $bench_rc -ne 0 ]]; then - echo " -> benchmark client exited rc=$bench_rc (thinking=$mode dspark=$mtp); cell will be N/A" - fi - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode DSPARK=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | top_p: $TOP_P | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on: temperature=1.0, chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo "# thinking_off: temperature=0.6, chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - fi - echo "# Measured on $MODEL_KEY (B300, vLLM DSpark, draft: $DRAFT_MODEL), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (DSpark level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" - -# A matrix where every cell is N/A is a failed collection, not a result: fail the -# job so it is not mistaken for a curve worth reviewing. -MEASURED=0 -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - [[ "${AL_RESULT[${mode}_${mtp}]:-N/A}" != "N/A" ]] && MEASURED=$((MEASURED + 1)) - done -done -if [[ "$MEASURED" -eq 0 ]]; then - echo "CRITICAL: no cell produced an AL value — see the server logs and the" - echo "benchmark client output above." - exit 1 -fi diff --git a/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh b/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh deleted file mode 100755 index 1f1d88a2a8..0000000000 --- a/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh +++ /dev/null @@ -1,342 +0,0 @@ -#!/usr/bin/env bash - -# Kimi-K3 B300 vLLM SPEED-Bench AL matrix collector for DSpark speculative decoding. -# -# For each DSpark speculative-token count, measure the REAL acceptance length (AL) on -# one SPEED-Bench category and emit a YAML matrix in the golden_al_distribution shape. -# The Inferact/Kimi-K3-DSpark draft head is downloaded to a writable workspace dir; -# the target (moonshotai/Kimi-K3, FP4) is pre-staged at /scratch/models/Kimi-K3. -# K3 is a thinking model (kimi_k3 reasoning parser defaults enable_thinking=True), -# so the golden curve is collected for thinking_on only. -# -# VARIANT of kimik3_fp4_b300_vllm.sh (the baseline): the speculative-config also sets -# draft_sample_method=probabilistic and rejection_sample_method=block so baseline and -# variant AL curves can be measured side by side. -# -# Usage (inside the Kimi-K3 vLLM container, on a B300 node): -# export MODEL=moonshotai/Kimi-K3 -# bash benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" -GPU_MEM_UTIL="0.90" -MAX_MODEL_LEN="16384" -MAX_NUM_SEQS="512" - -DRAFT_MODEL="Inferact/Kimi-K3-DSpark" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -# AL is concurrency-independent (per-token accept/reject; no spec-disable-by-batch is -# set), so batch the SPEED-Bench pass to stay under the CI wall-time limit; conc=1 -# blew the 8h budget on an earlier Kimi model. -CONCURRENCY="64" -TOP_P="0.95" -# K3 defaults to thinking ON, so the on-cell kwargs are explicit and the off-cell -# kwargs disable it. speedbench-al.yml's thinking-kwargs input defaults to the DSV4 -# value and is exported as CHAT_TEMPLATE_KWARGS_ON; dispatch K3 with -# -f 'thinking-kwargs={"thinking": true}'. -CHAT_TEMPLATE_KWARGS_OFF='{"thinking": false}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export NCCL_DMABUF_ENABLE="0" -export VLLM_ALLREDUCE_USE_FLASHINFER="1" -export VLLM_USE_RUST_FRONTEND="1" -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -# `vllm bench serve` os.execv's the CLIENT into the Rust vllm-rs binary whenever the -# dataset (speed_bench qualifies) and backend are supported, with no opt-out. The -# Rust flag surface lacks --speed-bench-output-len, --save-detailed and -# --chat-template-kwargs, so every cell would die at argument parsing. Call the -# Python entrypoint directly (vllm.benchmarks.serve.main), as the CLI's own Python -# fallback does. The SERVER keeps the Rust frontend; AL is read from /metrics. -BENCH_DRIVER="$RESULTS_DIR/bench_serve_python.py" -mkdir -p "$RESULTS_DIR" -cat > "$BENCH_DRIVER" <<'PYEOF' -# Python-only `vllm bench serve`: bypasses the Rust os.execv delegation in -# vllm/entrypoints/cli/benchmark/serve.py by importing the benchmark directly. -from vllm.benchmarks.serve import add_cli_args, main - -try: # import path moved in newer vLLM - from vllm.utils.argparse_utils import FlexibleArgumentParser -except ImportError: - from vllm.utils import FlexibleArgumentParser - -parser = FlexibleArgumentParser( - description="vllm bench serve (Python entrypoint, no Rust delegation)" -) -add_cli_args(parser) -main(parser.parse_args()) -PYEOF -# No `set -e` here, so a failed redirect would otherwise only surface later as a -# confusing "No such file" from the preflight. -if [[ ! -s "$BENCH_DRIVER" ]]; then - echo "CRITICAL: could not write the benchmark driver to $BENCH_DRIVER — aborting." - exit 1 -fi - -nvidia-smi - -# Kimi-K3 is in the launcher's STAGED_MODELS (read-only /scratch/models/Kimi-K3), -# so this is a no-op in CI; it covers a standalone run with unstaged weights. -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi -fi - -# dirname(MODEL_PATH) can be the read-only staged mount (/scratch/models), so the -# draft must go to a writable workspace dir, not next to the target. -DRAFT_DIR="/workspace/draft_models" -mkdir -p "$DRAFT_DIR" -DRAFT_MODEL_PATH="$DRAFT_DIR/${DRAFT_MODEL##*/}" -if [[ ! -d "$DRAFT_MODEL_PATH" || -z "$(ls -A "$DRAFT_MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$DRAFT_MODEL" --local-dir "$DRAFT_MODEL_PATH" -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -NEED_SHIM=0 -if [[ " $THINKING_MODES " == *" on "* && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then NEED_SHIM=1; fi -if [[ " $THINKING_MODES " == *" off "* && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then NEED_SHIM=1; fi -if [[ "$NEED_SHIM" == "1" ]]; then - if ! apply_chat_template_kwargs_shim; then - echo "CRITICAL: --chat-template-kwargs shim failed — aborting" - exit 1 - fi -fi - -# Preflight the client flags every cell uses; otherwise a CLI mismatch only shows up -# as an all-N/A matrix after eight full server starts (~1h). Probe the driver, not -# `vllm bench serve` (its help exits before the Rust execv), and ask for --help=all -# since plain --help prints only a group summary. -BENCH_HELP="$(python3 "$BENCH_DRIVER" --help=all 2>&1)" -for flag in --speed-bench-category --speed-bench-output-len --chat-template-kwargs --save-detailed; do - if [[ "$BENCH_HELP" != *"$flag"* ]]; then - echo "CRITICAL: the Python benchmark entrypoint does not support $flag — aborting." - echo "--- python3 $BENCH_DRIVER --help=all ---" - echo "$BENCH_HELP" - exit 1 - fi -done -echo "=== Benchmark client flag preflight OK ===" - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi -EP_ARGS=() -if [ "${EP_SIZE}" -gt 1 ]; then - EP_ARGS=(--enable-expert-parallel) -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - local temperature - if [[ "$mode" == "on" ]]; then - temperature=1.0 - if [[ -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - fi - else - temperature=0.6 - if [[ -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - fi - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode DSPARK=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --trust-remote-code - --load-format fastsafetensors - --moe-backend auto - --enable-prefix-caching - --kv-cache-dtype fp8 - "${EP_ARGS[@]}" - --reasoning-parser kimi_k3 - --tool-call-parser kimi_k3 - --enable-auto-tool-choice - --gpu-memory-utilization "$GPU_MEM_UTIL" - --max-num-seqs "$MAX_NUM_SEQS" - --max-model-len "$MAX_MODEL_LEN" - --max-cudagraph-capture-size 256 - --attention-config '{"mla_prefill_backend":"FLASHINFER","use_prefill_query_quantization":true}' - --speculative-config "{\"method\": \"dspark\", \"model\": \"$DRAFT_MODEL_PATH\", \"num_speculative_tokens\": $mtp, \"attention_backend\": \"FLASHINFER_MLA\", \"draft_sample_method\": \"probabilistic\", \"rejection_sample_method\": \"block\"}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode dspark=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - python3 "$BENCH_DRIVER" \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$temperature" \ - --top-p "$TOP_P" \ - "${think_args[@]}" - local bench_rc=$? - if [[ $bench_rc -ne 0 ]]; then - echo " -> benchmark client exited rc=$bench_rc (thinking=$mode dspark=$mtp); cell will be N/A" - fi - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode DSPARK=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | top_p: $TOP_P | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on: temperature=1.0, chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo "# thinking_off: temperature=0.6, chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - fi - echo "# Measured on $MODEL_KEY (B300, vLLM DSpark, draft: $DRAFT_MODEL), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (DSpark level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" - -# A matrix where every cell is N/A is a failed collection, not a result: fail the -# job so it is not mistaken for a curve worth reviewing. -MEASURED=0 -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - [[ "${AL_RESULT[${mode}_${mtp}]:-N/A}" != "N/A" ]] && MEASURED=$((MEASURED + 1)) - done -done -if [[ "$MEASURED" -eq 0 ]]; then - echo "CRITICAL: no cell produced an AL value — see the server logs and the" - echo "benchmark client output above." - exit 1 -fi diff --git a/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh deleted file mode 100755 index fc5ab0c968..0000000000 --- a/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh +++ /dev/null @@ -1,245 +0,0 @@ -#!/usr/bin/env bash - -# MiniMax-M3 B300 vLLM SPEED-Bench AL matrix collector for EAGLE3 speculative decoding. -# -# For each thinking mode (on/off) and EAGLE3 level (num_speculative_tokens), measure -# the REAL acceptance length (AL) on one SPEED-Bench category and emit a YAML matrix -# in the golden_al_distribution shape. The synthetic value is injected downstream by -# the throughput recipe, not here. -# -# Draft: Inferact/MiniMax-M3-EAGLE3. The EAGLE3 head is MHA and must use FLASH_ATTN -# (FlashInfer only supports page size 128 through its trtllm-gen kernel, which needs -# GQA/MQA); the target keeps its default FlashInfer backend. --block-size 128 is -# mandatory for the MSA sparse/index cache. --language-model-only frees the vision -# encoder's VRAM (text-only benchmark). -# -# The *_fp4_* filename is only the naming convention speedbench-al.yml requires -# (${model-prefix}_fp4_b300_vllm.sh); the staged MiniMax-M3 weights are unquantized -# BF16, so no quantization-specific flags (--moe-backend marlin, --kv-cache-dtype -# fp8) apply. -# -# Dispatch this collector through speedbench-al.yml. -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" -GPU_MEM_UTIL="0.90" - -DRAFT_MODEL="Inferact/MiniMax-M3-EAGLE3" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -CONCURRENCY="1" -# Official MiniMax-M3 sampling: temperature 1.0, top_p 0.95, top_k 40. -TEMPERATURE="1.0" -TOP_P="0.95" -TOP_K="40" -# M3 thinking toggles via the thinking_mode chat_template key. -CHAT_TEMPLATE_KWARGS_OFF='{"thinking_mode": "disabled"}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export VLLM_FLOAT32_MATMUL_PRECISION="high" -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi -if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi - -# dirname(MODEL_PATH) is the read-only staged mount (/scratch/models), so the draft -# must go to a writable workspace dir, not next to the target. -echo "=== Downloading EAGLE3 draft model ($DRAFT_MODEL) ===" -DRAFT_DIR="/workspace/draft_models" -mkdir -p "$DRAFT_DIR" -DRAFT_MODEL_PATH="$DRAFT_DIR/${DRAFT_MODEL##*/}" -if [[ ! -d "$DRAFT_MODEL_PATH" || -z "$(ls -A "$DRAFT_MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$DRAFT_MODEL" --local-dir "$DRAFT_MODEL_PATH" -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP" --enable-expert-parallel) -elif [ "${EP_SIZE}" -gt 1 ]; then - PARALLEL_ARGS=(--tensor-parallel-size "$TP" --enable-expert-parallel) -else - # Plain TP per the official MiniMax-M3 recipe. Do NOT force a MoE backend: the - # staged checkpoint is unquantized BF16, for which marlin is rejected. - PARALLEL_ARGS=(--tensor-parallel-size "$TP") -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - if [[ "$mode" == "on" && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - elif [[ "$mode" == "off" && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode EAGLE3=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --block-size 128 - --language-model-only - --max-cudagraph-capture-size 2048 - --trust-remote-code - --no-enable-prefix-caching - --gpu-memory-utilization "$GPU_MEM_UTIL" - --max-model-len 16384 - --max-num-batched-tokens 16384 - --stream-interval 30 - --speculative-config "{\"method\": \"eagle3\", \"model\": \"$DRAFT_MODEL_PATH\", \"num_speculative_tokens\": $mtp, \"attention_backend\": \"FLASH_ATTN\"}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode eagle3=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$TEMPERATURE" \ - --top-p "$TOP_P" \ - --top-k "$TOP_K" \ - "${think_args[@]}" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode EAGLE3=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | temperature: $TEMPERATURE | top_p: $TOP_P | top_k: $TOP_K | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - echo "# thinking_off chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - echo "# Measured on $MODEL_KEY (B300, vLLM EAGLE3), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (EAGLE3 level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4dspark/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4dspark/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..c81e169e59 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4dspark/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,87 @@ +# DeepSeek-V4-Pro-DSpark B300 vLLM SPEED-Bench AL matrix (DSpark, greedy draft). +# +# DSpark is baked into the DeepSeek-V4-Pro-DSpark checkpoint (no external draft +# model). The legacy collector hard-coded TEP8 (--enable-expert-parallel), but +# the srt-slurm recipe omits it because the workflow exports EP_SIZE=1 and +# validate_recipe enforces the match. AL is expert-placement-independent. +# GPU memory utilization 0.90 and max-num-seqs 64 leave headroom for the DSpark +# rejection sampler's fp32 logits scratch and CUDA graphs. +# SPEEDBENCH_CONCURRENCY=32 batches the SPEED-Bench pass to stay under CI +# wall-time. The client-side --chat-template-kwargs shim is applied when +# APPLY_CHAT_TEMPLATE_KWARGS_SHIM=1. +base: + schema: 2 + name: dsv4dspark-fp4-b300-vllm-speedbench + model: + path: hf:deepseek-ai/DeepSeek-V4-Pro-DSpark + container: vllm/vllm-openai:v0.21.0 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro-DSpark + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + kv-cache-dtype: fp8 + block-size: 256 + no-enable-prefix-caching: true + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' + attention-config: '{"use_fp4_indexer_cache":true}' + tokenizer-mode: deepseek_v4 + tool-call-parser: deepseek_v4 + enable-auto-tool-choice: true + reasoning-parser: deepseek_v4 + max-cudagraph-capture-size: 2048 + max-model-len: 16384 + max-num-seqs: 64 + gpu-memory-utilization: 0.90 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro-DSpark + SPEC_DECODING: mtp + TEMPERATURE: '1.0' + SPEEDBENCH_TOKENIZER_MODE: deepseek_v4 + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + SPEEDBENCH_CONCURRENCY: '32' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"dspark","num_speculative_tokens":1,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":2,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":3,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":6,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":7,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":8,"draft_sample_method":"greedy"}' diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4dsparkprob/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4dsparkprob/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..1181322096 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4dsparkprob/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,82 @@ +# DeepSeek-V4-Pro-DSpark B300 vLLM SPEED-Bench AL matrix (DSpark, probabilistic draft). +# +# Identical to dsv4dspark except draft_sample_method=probabilistic. The greedy +# vs probabilistic comparison isolates the effect of draft-token sampling on +# acceptance length. No rejection_sample_method override (the bash wrapper +# dsv4dsparkprob_fp4_b300_vllm.sh only sets DRAFT_SAMPLE_METHOD=probabilistic). +base: + schema: 2 + name: dsv4dsparkprob-fp4-b300-vllm-speedbench + model: + path: hf:deepseek-ai/DeepSeek-V4-Pro-DSpark + container: vllm/vllm-openai:v0.21.0 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro-DSpark + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + kv-cache-dtype: fp8 + block-size: 256 + no-enable-prefix-caching: true + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' + attention-config: '{"use_fp4_indexer_cache":true}' + tokenizer-mode: deepseek_v4 + tool-call-parser: deepseek_v4 + enable-auto-tool-choice: true + reasoning-parser: deepseek_v4 + max-cudagraph-capture-size: 2048 + max-model-len: 16384 + max-num-seqs: 64 + gpu-memory-utilization: 0.90 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro-DSpark + SPEC_DECODING: mtp + TEMPERATURE: '1.0' + SPEEDBENCH_TOKENIZER_MODE: deepseek_v4 + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + SPEEDBENCH_CONCURRENCY: '32' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"dspark","num_speculative_tokens":1,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":2,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":3,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":4,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":6,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":7,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":8,"draft_sample_method":"probabilistic"}' diff --git a/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..16c253d479 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,91 @@ +# Kimi-K3 B300 vLLM SPEED-Bench AL matrix (DSpark, greedy draft). +# +# External draft model: Inferact/Kimi-K3-DSpark. The target (moonshotai/Kimi-K3) +# is pre-staged; vLLM auto-downloads the draft by its HF id. +# K3 defaults to thinking ON (kimi_k3 reasoning parser), so the off-cell +# CHAT_TEMPLATE_KWARGS_OFF disables it. Per-mode temperatures: 1.0 thinking-on, +# 0.6 thinking-off. SPEEDBENCH_CONCURRENCY=64 keeps collection under the CI +# wall-time limit. +# +# enable-prefix-caching is ON (not no-enable-prefix-caching) to match the +# production K3 recipe; FlashInfer MLA attention backend for both target and draft. +base: + schema: 2 + name: kimik3-fp4-b300-vllm-speedbench + model: + path: hf:moonshotai/Kimi-K3 + container: vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # Bind GPUs via CUDA_VISIBLE_DEVICES for image compatibility. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: moonshotai/Kimi-K3 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + load-format: fastsafetensors + moe-backend: auto + enable-prefix-caching: true + kv-cache-dtype: fp8 + reasoning-parser: kimi_k3 + tool-call-parser: kimi_k3 + enable-auto-tool-choice: true + gpu-memory-utilization: 0.90 + max-num-seqs: 512 + max-model-len: 16384 + max-cudagraph-capture-size: 256 + attention-config: '{"mla_prefill_backend":"FLASHINFER","use_prefill_query_quantization":true}' + env: + NCCL_DMABUF_ENABLE: '0' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: moonshotai/Kimi-K3 + SPEC_DECODING: mtp + TEMPERATURE_ON: '1.0' + TEMPERATURE_OFF: '0.6' + TOP_P: '0.95' + CHAT_TEMPLATE_KWARGS_OFF: '{"thinking": false}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + SPEEDBENCH_CONCURRENCY: '64' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":1,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":3,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":5,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":6,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":8,"attention_backend":"FLASHINFER_MLA"}' diff --git a/benchmarks/single_node/srt-slurm-recipes/kimik3prob/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/kimik3prob/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..acb59e9b04 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/kimik3prob/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,87 @@ +# Kimi-K3 B300 vLLM SPEED-Bench AL matrix (DSpark, probabilistic draft + block verify). +# +# Identical to kimik3 except draft_sample_method=probabilistic and +# rejection_sample_method=block in speculative-config. This matches the +# kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh +# collector and the kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method +# golden curve. +base: + schema: 2 + name: kimik3prob-fp4-b300-vllm-speedbench + model: + path: hf:moonshotai/Kimi-K3 + container: vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # Bind GPUs via CUDA_VISIBLE_DEVICES for image compatibility. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: moonshotai/Kimi-K3 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + load-format: fastsafetensors + moe-backend: auto + enable-prefix-caching: true + kv-cache-dtype: fp8 + reasoning-parser: kimi_k3 + tool-call-parser: kimi_k3 + enable-auto-tool-choice: true + gpu-memory-utilization: 0.90 + max-num-seqs: 512 + max-model-len: 16384 + max-cudagraph-capture-size: 256 + attention-config: '{"mla_prefill_backend":"FLASHINFER","use_prefill_query_quantization":true}' + env: + NCCL_DMABUF_ENABLE: '0' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: moonshotai/Kimi-K3 + SPEC_DECODING: mtp + TEMPERATURE_ON: '1.0' + TEMPERATURE_OFF: '0.6' + TOP_P: '0.95' + CHAT_TEMPLATE_KWARGS_OFF: '{"thinking": false}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + SPEEDBENCH_CONCURRENCY: '64' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":1,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":3,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":5,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":6,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":8,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' diff --git a/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..4f05aac3ec --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,86 @@ +# MiniMax-M3 B300 vLLM SPEED-Bench AL matrix (EAGLE3, MHA draft head). +# +# External draft model: Inferact/MiniMax-M3-EAGLE3. The EAGLE3 head is MHA +# and must use FLASH_ATTN (FlashInfer only supports page size 128 through its +# trtllm-gen kernel, which needs GQA/MQA); the target keeps its default +# FlashInfer backend. --block-size 128 is mandatory for the MSA sparse/index +# cache. --language-model-only frees the vision encoder VRAM (text-only +# benchmark). The staged MiniMax-M3 weights are unquantized BF16, so no +# quantization-specific flags (--moe-backend marlin, --kv-cache-dtype fp8) +# apply. +# +# Official MiniMax-M3 sampling: temperature 1.0, top_p 0.95, top_k 40. +# Thinking toggles via the thinking_mode chat_template key. +base: + schema: 2 + name: minimaxm3-fp4-b300-vllm-speedbench + model: + path: hf:MiniMax/MiniMax-M3 + container: vllm/vllm-openai:nightly-dev-x86_64-cu13-af03963 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # Bind GPUs via CUDA_VISIBLE_DEVICES for image compatibility. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: MiniMax/MiniMax-M3 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + block-size: 128 + language-model-only: true + max-cudagraph-capture-size: 2048 + trust-remote-code: true + no-enable-prefix-caching: true + gpu-memory-utilization: 0.90 + max-model-len: 16384 + max-num-batched-tokens: 16384 + stream-interval: 30 + env: + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: MiniMax/MiniMax-M3 + SPEC_DECODING: mtp + TEMPERATURE: '1.0' + TOP_P: '0.95' + TOP_K: '40' + CHAT_TEMPLATE_KWARGS_OFF: '{"thinking_mode": "disabled"}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":1,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":2,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":4,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":5,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":6,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":7,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":8,"attention_backend":"FLASH_ATTN"}' diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index ab5dbdc615..c559ba0c7b 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -679,3 +679,83 @@ def test_speedbench_client_finds_benchmark_lib_without_legacy_workspace(): assert result.returncode != 0 assert "No such file" not in result.stderr, result.stderr assert "required environment variables are not set" in result.stdout + + +# --- Recipe-render validation for every speedbench prefix --- + +# Per-prefix dispatch environment that mirrors the speedbench-al.yml inputs. +_SPEEDBENCH_PREFIXES = { + "dsv4": { + "MODEL": "deepseek-ai/DeepSeek-V4-Pro", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "dsv4dspark": { + "MODEL": "deepseek-ai/DeepSeek-V4-Pro-DSpark", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "dsv4dsparkprob": { + "MODEL": "deepseek-ai/DeepSeek-V4-Pro-DSpark", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "glm52": { + "MODEL": "nvidia/GLM-5.2-NVFP4", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "kimik3": { + "MODEL": "moonshotai/Kimi-K3", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "kimik3prob": { + "MODEL": "moonshotai/Kimi-K3", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "minimaxm3": { + "MODEL": "MiniMax/MiniMax-M3", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "qwen3.5": { + "MODEL": "nvidia/Qwen3.5-397B-A17B-NVFP4-V2", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "qwen3.8next": { + "MODEL": "Qwen/Qwen3.8-Flash-Next-FP8", + "IMAGE": "vllm/vllm-openai:qwen38-flash-next", + "TP": "4", "GPU_COUNT": "4", + }, +} + + +@pytest.mark.parametrize("prefix", sorted(_SPEEDBENCH_PREFIXES)) +def test_speedbench_recipe_validates_and_renders(prefix): + """Each speedbench recipe selects and validates against the workflow env.""" + recipe = Path( + f"benchmarks/single_node/srt-slurm-recipes/{prefix}/vllm/b300-fp4-speedbench/speedbench.yaml" + ) + assert recipe.exists(), f"Recipe not found: {recipe}" + overrides = _SPEEDBENCH_PREFIXES[prefix] + env = { + "FRAMEWORK": "vllm", + "PRECISION": "fp4", + "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", + "EP_SIZE": "1", "DP_ATTENTION": "false", + "SPEC_DECODING": "mtp", "IS_AGENTIC": "0", + "RUN_EVAL": "false", "EVAL_ONLY": "false", + "CONC": "1", "RESULT_FILENAME": "speedbench_off_mtp1", + "GPU_MONITOR_INTERVAL": "3", "MODEL_PREFIX": prefix, + "CATEGORY": "coding", "SPEEDBENCH_OUTPUT_LEN": "4096", + "THINKING": "off", "MTP": "1", + **overrides, + } + argv = runtime_arguments(f"{recipe}:zip_override_mtp[0]", env) + sets = [argv[i + 1] for i, arg in enumerate(argv) if arg == "--set"] + # Verify THINKING survived as a string through the set binding. + thinking_set = [s for s in sets if "THINKING" in s] + assert any("thinking_off" in s for s in thinking_set), thinking_set diff --git a/utils/test_speedbench_matrix.py b/infx/tests/workflows/test_speedbench_matrix.py similarity index 100% rename from utils/test_speedbench_matrix.py rename to infx/tests/workflows/test_speedbench_matrix.py diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index 7964506032..742b0b619a 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -124,9 +124,8 @@ EXECUTION_PATH=agentic if [[ "$IS_MULTINODE" == true ]]; then EXECUTION_PATH=multinode elif [[ -n "${BENCH_SCRIPT_OVERRIDE:-}" ]]; then - # Legacy SPEED-Bench collectors (draft-model variants: dsv4dspark*, kimik3, - # minimaxm3) explicitly supply their script. Native-MTP collectors migrated - # to the srt-slurm path set SRT_RECIPE instead. + # Legacy path: caller supplies a script via BENCH_SCRIPT_OVERRIDE. + # All SPEED-Bench collectors migrated to srt-slurm recipes (SRT_RECIPE). EXECUTION_PATH=script elif [[ "$IS_AGENTIC" == 0 || -n "${SRT_RECIPE:-}" ]]; then check_env_vars SRT_RECIPE From 079b08326a58fdd8604db5a21e1089652c6a7bbd Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 13:40:50 -0400 Subject: [PATCH 10/12] fix(tests): match speedbench recipe-render test images to actual recipes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The kimik3, kimik3prob, and minimaxm3 speedbench recipes use nightly vLLM images, not v0.21.0. Update the test environment to match so validate_recipe does not reject the image mismatch. kimik3/kimik3prob 和 minimaxm3 speedbench 配方使用夜间构建 vLLM 镜像而非 v0.21.0。更新测试环境以匹配,避免 validate_recipe 因镜像不匹配而拒绝。 Co-Authored-By: Claude Opus 4.6 --- infx/tests/srt_slurm/test_srt_single_node.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index c559ba0c7b..8deffd1a8b 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -707,17 +707,17 @@ def test_speedbench_client_finds_benchmark_lib_without_legacy_workspace(): }, "kimik3": { "MODEL": "moonshotai/Kimi-K3", - "IMAGE": "vllm/vllm-openai:v0.21.0", + "IMAGE": "vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77", "TP": "8", "GPU_COUNT": "8", }, "kimik3prob": { "MODEL": "moonshotai/Kimi-K3", - "IMAGE": "vllm/vllm-openai:v0.21.0", + "IMAGE": "vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77", "TP": "8", "GPU_COUNT": "8", }, "minimaxm3": { "MODEL": "MiniMax/MiniMax-M3", - "IMAGE": "vllm/vllm-openai:v0.21.0", + "IMAGE": "vllm/vllm-openai:nightly-dev-x86_64-cu13-af03963", "TP": "8", "GPU_COUNT": "8", }, "qwen3.5": { From b3e85372d957d8f233500adc9d2ddafbf3a8f08f Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 13:42:53 -0400 Subject: [PATCH 11/12] =?UTF-8?q?feat(speedbench):=20migrate=20remaining?= =?UTF-8?q?=20AL=20collectors=20to=20srt-slurm=20recipes=20/=20=E5=B0=86?= =?UTF-8?q?=E5=89=A9=E4=BD=99=20AL=20=E6=94=B6=E9=9B=86=E5=99=A8=E8=BF=81?= =?UTF-8?q?=E7=A7=BB=E5=88=B0=20srt-slurm=20=E9=85=8D=E6=96=B9?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port dsv4dspark, dsv4dsparkprob, kimik3, kimik3prob and minimaxm3 SPEED-Bench collectors from standalone bash scripts to declarative srt-slurm YAML recipes. Delete the five legacy collector scripts. Update test expectations to match recipe container images (nightly tags for K3 and M3). 将 dsv4dspark、dsv4dsparkprob、kimik3、kimik3prob 和 minimaxm3 SPEED-Bench 收集器从独立 bash 脚本移植到声明式 srt-slurm YAML 配方。 删除五个旧版收集器脚本。更新测试预期值以匹配配方容器镜像 (K3 和 M3 使用 nightly 标签)。 Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/speedbench-al.yml | 139 ++++--- .../speedbench/dsv4dspark_fp4_b300_vllm.sh | 348 ------------------ .../dsv4dsparkprob_fp4_b300_vllm.sh | 10 - .../speedbench/kimik3_fp4_b300_vllm.sh | 338 ----------------- ...le_method_block_rejection_sample_method.sh | 342 ----------------- .../speedbench/minimaxm3_fp4_b300_vllm.sh | 245 ------------ .../vllm/b300-fp4-speedbench/speedbench.yaml | 87 +++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 82 +++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 91 +++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 87 +++++ .../vllm/b300-fp4-speedbench/speedbench.yaml | 86 +++++ infx/tests/srt_slurm/test_srt_single_node.py | 80 ++++ .../workflows}/test_speedbench_matrix.py | 0 runners/launch_b300-dsxe.sh | 5 +- 14 files changed, 577 insertions(+), 1363 deletions(-) delete mode 100755 benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/dsv4dsparkprob_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh delete mode 100755 benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh delete mode 100755 benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsv4dspark/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsv4dsparkprob/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/kimik3prob/vllm/b300-fp4-speedbench/speedbench.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml rename {utils => infx/tests/workflows}/test_speedbench_matrix.py (100%) diff --git a/.github/workflows/speedbench-al.yml b/.github/workflows/speedbench-al.yml index 468f66c2f9..819efce04c 100644 --- a/.github/workflows/speedbench-al.yml +++ b/.github/workflows/speedbench-al.yml @@ -21,7 +21,7 @@ on: # zizmor: ignore[concurrency-limits] type: string default: 'deepseek-ai/DeepSeek-V4-Pro' model-prefix: - description: "Model prefix; drives launcher MODEL_PATH resolution, exp name, collector script, and artifact names" + description: "Model prefix; drives srt-slurm recipe lookup, exp name, and artifact names" required: false type: string default: 'dsv4' @@ -238,84 +238,69 @@ jobs: source runners/runtime_settings.sh fi - # Migrated native-MTP prefixes use the srt-slurm single-node path. - # Draft-model collectors (dsv4dspark*, kimik3, minimaxm3) keep the - # legacy BENCH_SCRIPT_OVERRIDE path. - _is_srt_slurm_prefix() { - case "$1" in - dsr1|dsv4|glm5|glm52|qwen3.5|qwen3.8next) return 0 ;; - *) return 1 ;; - esac - } - - if _is_srt_slurm_prefix "$MODEL_PREFIX"; then - # --- srt-slurm per-cell collection --- - RECIPE_PATH="benchmarks/single_node/srt-slurm-recipes/${MODEL_PREFIX}/vllm/b300-fp4-speedbench/speedbench.yaml" - if [[ ! -f "$RECIPE_PATH" ]]; then - echo "ERROR: SPEED-Bench recipe not found: $RECIPE_PATH" >&2 - exit 1 - fi - # Plain TP8 is incompatible with the Qwen3.8-Flash-Next FP8 checkpoint's - # 128-wide quantization blocks; its recipe runs TP4 (as the legacy - # collector did), so bind the matrix TP and GPU count to match. - if [[ "$MODEL_PREFIX" == qwen3.8next ]]; then - export TP=4 GPU_COUNT=4 - fi - mkdir -p speedbench_results - ALL_FAILED=true - # Harmless values for launch_srt_single_node env checks; the collector - # recipe and client do not read them. - export ISL=256 - export OSL=256 - export RANDOM_RANGE_RATIO=0.0 - export CONC=1 - export PP_SIZE=1 - export DCP_SIZE=1 - export PCP_SIZE=1 - for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - IDX=$((mtp - 1)) - export THINKING="$mode" - export MTP="$mtp" - export SRT_RECIPE="${RECIPE_PATH}:zip_override_mtp[${IDX}]" - export RESULT_FILENAME="speedbench_${mode}_mtp${mtp}" - echo "" - echo "==========================================" - echo " srt-slurm cell: thinking=${mode} MTP=${mtp}" - echo " SRT_RECIPE=${SRT_RECIPE}" - echo "==========================================" - CELL_RC=0 - bash ./runners/launch_"${RUNNER_NAME%%_*}".sh || CELL_RC=$? - # Collect per-cell artifacts into the results directory. - if [[ -f "${RESULT_FILENAME}.json" ]]; then - mv "${RESULT_FILENAME}.json" "speedbench_results/${RESULT_FILENAME}.json" - ALL_FAILED=false - fi - if [[ -f srt-single-node-logs.tar.gz ]]; then - mv srt-single-node-logs.tar.gz "speedbench_results/srt-logs_${mode}_mtp${mtp}.tar.gz" - fi - if [[ "$CELL_RC" -ne 0 ]]; then - echo " -> cell failed (rc=$CELL_RC), recording N/A" - fi - done + # All collectors use the srt-slurm single-node path with per-model + # recipes under srt-slurm-recipes//vllm/b300-fp4-speedbench/. + RECIPE_PATH="benchmarks/single_node/srt-slurm-recipes/${MODEL_PREFIX}/vllm/b300-fp4-speedbench/speedbench.yaml" + if [[ ! -f "$RECIPE_PATH" ]]; then + echo "ERROR: SPEED-Bench recipe not found: $RECIPE_PATH" >&2 + exit 1 + fi + # Plain TP8 is incompatible with the Qwen3.8-Flash-Next FP8 checkpoint's + # 128-wide quantization blocks; its recipe runs TP4 (as the legacy + # collector did), so bind the matrix TP and GPU count to match. + if [[ "$MODEL_PREFIX" == qwen3.8next ]]; then + export TP=4 GPU_COUNT=4 + fi + mkdir -p speedbench_results + ALL_FAILED=true + # Harmless values for launch_srt_single_node env checks; the collector + # recipe and client do not read them. + export ISL=256 + export OSL=256 + export RANDOM_RANGE_RATIO=0.0 + export CONC=1 + export PP_SIZE=1 + export DCP_SIZE=1 + export PCP_SIZE=1 + for mode in $THINKING_MODES; do + for mtp in $MTP_LIST; do + IDX=$((mtp - 1)) + export THINKING="$mode" + export MTP="$mtp" + export SRT_RECIPE="${RECIPE_PATH}:zip_override_mtp[${IDX}]" + export RESULT_FILENAME="speedbench_${mode}_mtp${mtp}" + echo "" + echo "==========================================" + echo " srt-slurm cell: thinking=${mode} MTP=${mtp}" + echo " SRT_RECIPE=${SRT_RECIPE}" + echo "==========================================" + CELL_RC=0 + bash ./runners/launch_"${RUNNER_NAME%%_*}".sh || CELL_RC=$? + # Collect per-cell artifacts into the results directory. + if [[ -f "${RESULT_FILENAME}.json" ]]; then + mv "${RESULT_FILENAME}.json" "speedbench_results/${RESULT_FILENAME}.json" + ALL_FAILED=false + fi + if [[ -f srt-single-node-logs.tar.gz ]]; then + mv srt-single-node-logs.tar.gz "speedbench_results/srt-logs_${mode}_mtp${mtp}.tar.gz" + fi + if [[ "$CELL_RC" -ne 0 ]]; then + echo " -> cell failed (rc=$CELL_RC), recording N/A" + fi done - if [[ "$ALL_FAILED" == true ]]; then - echo "ERROR: every SPEED-Bench cell failed" >&2 - exit 1 - fi - # Aggregate per-cell result JSONs into the reference YAML. - PYTHONPATH="${GITHUB_WORKSPACE}${PYTHONPATH:+:$PYTHONPATH}" \ - python3 -m infx.workflows.speedbench_matrix \ - --result-dir speedbench_results \ - --model-key "$(basename "$(echo "$MODEL" | tr '[:upper:]' '[:lower:]')")" \ - --thinking-modes "$THINKING_MODES" \ - --mtp-list "$MTP_LIST" \ - > speedbench-reference-al.yaml - else - # --- Legacy collector script --- - export BENCH_SCRIPT_OVERRIDE="benchmarks/single_node/speedbench/${MODEL_PREFIX}_fp4_b300_vllm.sh" - bash ./runners/launch_"${RUNNER_NAME%%_*}".sh + done + if [[ "$ALL_FAILED" == true ]]; then + echo "ERROR: every SPEED-Bench cell failed" >&2 + exit 1 fi + # Aggregate per-cell result JSONs into the reference YAML. + PYTHONPATH="${GITHUB_WORKSPACE}${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.workflows.speedbench_matrix \ + --result-dir speedbench_results \ + --model-key "$(basename "$(echo "$MODEL" | tr '[:upper:]' '[:lower:]')")" \ + --thinking-modes "$THINKING_MODES" \ + --mtp-list "$MTP_LIST" \ + > speedbench-reference-al.yaml if [ ! -f "speedbench-reference-al.yaml" ]; then echo "AL collection failed: speedbench-reference-al.yaml not produced." >&2 diff --git a/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh deleted file mode 100755 index fabc3b421f..0000000000 --- a/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh +++ /dev/null @@ -1,348 +0,0 @@ -#!/usr/bin/env bash - -# DSV4-Pro B300 vLLM SPEED-Bench AL matrix collector for DSpark speculative decoding. -# -# For each thinking mode (on/off) and DSpark speculative-token count, measure the REAL -# acceptance length (AL) on one SPEED-Bench category and emit a YAML matrix in the -# golden_al_distribution shape. DSpark ships as a separate checkpoint -# (deepseek-ai/DeepSeek-V4-Pro-DSpark, 960 GB) with the draft baked in, so there is no -# external draft head and no "model" key in the speculative-config. Every flag that -# affects drafting is byte-identical to the DSV4 MTP collector so the two AL curves -# stay comparable. -# -# Dispatch this collector through speedbench-al.yml. -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DRAFT_SAMPLE_METHOD MODEL MODEL_PATH MTP_LIST \ - OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -# MODEL_PATH is the launcher-resolved weights dir (writable models dir until the -# checkpoint is staged; see the download block below). -SERVE_MODEL="${MODEL_PATH}" - -# Top-level key in the emitted YAML matrix comes from the model basename. -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -# AL is a per-draft accept/reject property independent of batch size, so batch the -# SPEED-Bench pass to cut wall-clock. Nothing sets speculative_disable_by_batch_size, -# so drafting stays on at this batch size. -CONCURRENCY="32" -# Must stay >= CONCURRENCY or requests just queue. Held far below vLLM's default of -# 1024 because that sizes two allocations the memory profiler never sees: the -# rejection sampler's fp32 logits scratch (max_num_seqs * (1 + spec_tokens) * vocab * -# 4B, 4.4 GB at 8 tokens) and the spec-decode CUDA graphs. DSV4-Pro has no room: -# weights + a 100 GiB KV cache already fill 266 of 268 GiB per B300. -MAX_NUM_SEQS="64" -# Reserve device memory for KV cache and speculative verification. -GPU_MEM_UTIL="0.90" -TEMPERATURE="1.0" -# MUST match the golden config: infx/golden_al_distribution/dsv4_mtp.yaml was measured -# with reasoning_effort=high. -# The published recipe uses greedy; probabilistic won at every level on Kimi-K3 -# (infx/golden_al_distribution/kimik3_dspark*.yaml). vLLM accepts exactly these two values -# (vllm/config/speculative.py: DraftSampleMethod). -case "$DRAFT_SAMPLE_METHOD" in - greedy|probabilistic) ;; - *) - echo "CRITICAL: DRAFT_SAMPLE_METHOD must be 'greedy' or 'probabilistic' (got '$DRAFT_SAMPLE_METHOD')" - exit 1 - ;; -esac -# Opt-in rather than tied to draft_sample_method: flipping it to "block" would bundle -# two variables into one measurement, and the forced-AL config has to stay on a -# sampling method TRT-LLM supports too. -REJECTION_SAMPLE_METHOD="${REJECTION_SAMPLE_METHOD:-}" - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi - -# The DSpark checkpoint is not in the launcher's STAGED_MODELS, so MODEL_PATH resolves -# to the writable models dir and the ~960 GB download runs once. Add the basename to -# STAGED_MODELS once the weights are staged on the read-only mount. -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - if [[ ! -w "$(dirname "$MODEL_PATH")" ]]; then - echo "CRITICAL: $MODEL_PATH is empty and $(dirname "$MODEL_PATH") is not writable." - echo "This means the basename is listed in the launcher's STAGED_MODELS but the" - echo "weights were never staged. Either get them staged, or remove it from" - echo "STAGED_MODELS so MODEL_PATH resolves to the writable models dir instead." - exit 1 - fi - echo "=== $MODEL_PATH is empty; downloading $MODEL (~960 GB, first run only) ===" - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -# speed_bench/CustomDataset renders the chat template client-side and posts to -# /v1/completions, so thinking mode must reach apply_chat_template via -# --chat-template-kwargs (native since vllm-project/vllm#44244). Assert rather than -# assume: if the CLI option exists but speed_bench does not forward it, the flag is -# silently ignored and every thinking_on cell reports a non-thinking AL. -assert_chat_template_kwargs_support() { - echo "=== Checking vLLM benchmark --chat-template-kwargs support ===" - python3 - <<'PYEOF' -import sys -import vllm.benchmarks.serve as S -import vllm.benchmarks.datasets.datasets as D - -def read(mod): - with open(mod.__file__) as fh: - return fh.read() - -s_src, d_src = read(S), read(D) - -missing = [] -if '"--chat-template-kwargs"' not in s_src: - missing.append(f"CLI option in {S.__file__}") -if ('chat_template_kwargs=getattr(args' not in d_src - and 'chat_template_kwargs=args.chat_template_kwargs' not in d_src): - missing.append(f"speed_bench forward in {D.__file__}") -if '**(chat_template_kwargs or {})' not in d_src: - missing.append(f"apply_chat_template unpack in {D.__file__}") - -if missing: - print("CRITICAL: this image lacks native --chat-template-kwargs support:") - for item in missing: - print(" missing:", item) - print("thinking_on cells would silently measure a non-thinking AL. Use an") - print("image that contains vllm-project/vllm#44244.") - sys.exit(1) - -print("native --chat-template-kwargs support confirmed") -PYEOF -} - -if [[ " $THINKING_MODES " == *" on "* ]]; then - if ! assert_chat_template_kwargs_support; then - echo "CRITICAL: --chat-template-kwargs preflight failed — aborting" - exit 1 - fi -fi - -# TEP8 as in the published B300 DSpark recipe (vllm-project/recipes). Hard-coded -# rather than driven by EP_SIZE / DP_ATTENTION because speedbench-al.yml exports -# EP_SIZE=1 and DP_ATTENTION=false for every model, which silently turned the recipe -# into plain TP; TP-sharding the FP4 experts costs ~37 GiB per GPU over EP and made -# the num_speculative_tokens=4 cell OOM. AL is unaffected by expert placement. -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -EP_ARGS=(--enable-expert-parallel) -MOE_ARGS=(--moe-backend deep_gemm_mega_moe) - -SPEC_EXTRA="" -if [[ -n "$REJECTION_SAMPLE_METHOD" ]]; then - SPEC_EXTRA=", \"rejection_sample_method\": \"$REJECTION_SAMPLE_METHOD\"" -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -# Descendant PIDs of $1 by PARENT pid. This can never include this script (an -# ancestor of the server), unlike a name-based `pkill -f vllm`, which self-killed -# because the script filename contains "vllm". -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - # Snapshot the worker/EngineCore subprocesses BEFORE killing the parent: once it - # dies the children reparent to init and the tree link is lost. An orphaned - # worker holds GPU memory and OOMs the next server start. - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - if [[ "$mode" == "on" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode DSPARK=$mtp category=$CATEGORY" - echo " draft_sample_method=$DRAFT_SAMPLE_METHOD" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --kv-cache-dtype fp8 - --trust-remote-code - --block-size 256 - --no-enable-prefix-caching - "${EP_ARGS[@]}" - "${MOE_ARGS[@]}" - --compilation-config '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' - --attention_config.use_fp4_indexer_cache True - --tokenizer-mode deepseek_v4 - --tool-call-parser deepseek_v4 - --enable-auto-tool-choice - --reasoning-parser deepseek_v4 - --max-cudagraph-capture-size 2048 - --max-model-len 16384 - --max-num-seqs "$MAX_NUM_SEQS" - --gpu-memory-utilization "$GPU_MEM_UTIL" - --speculative-config "{\"method\": \"dspark\", \"num_speculative_tokens\": $mtp, \"draft_sample_method\": \"$DRAFT_SAMPLE_METHOD\"$SPEC_EXTRA}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - # wait_for_server_ready exits the shell (rather than returning) when the server - # dies; the subshell keeps that exit local so one bad cell does not abort the matrix. - if ! (wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"); then - echo " -> server failed to start (thinking=$mode dspark=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --tokenizer-mode deepseek_v4 \ - --temperature "$TEMPERATURE" \ - "${think_args[@]}" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode DSPARK=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -SPEC_SUMMARY="method=dspark | draft_sample_method=$DRAFT_SAMPLE_METHOD" -if [[ -n "$REJECTION_SAMPLE_METHOD" ]]; then - SPEC_SUMMARY="$SPEC_SUMMARY | rejection_sample_method=$REJECTION_SAMPLE_METHOD" -fi - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | temperature: $TEMPERATURE | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - echo "# speculative-config: $SPEC_SUMMARY" - echo "# Measured on $MODEL_KEY (B300, vLLM DSpark), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (DSpark level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" - -# A matrix where every cell is N/A is a failed collection, not a result: fail the -# job so it is not mistaken for a curve worth reviewing. -MEASURED=0 -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - [[ "${AL_RESULT[${mode}_${mtp}]:-N/A}" != "N/A" ]] && MEASURED=$((MEASURED + 1)) - done -done -if [[ "$MEASURED" -eq 0 ]]; then - echo "CRITICAL: no cell produced an AL value — see the server logs and the" - echo "benchmark client output above." - exit 1 -fi diff --git a/benchmarks/single_node/speedbench/dsv4dsparkprob_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/dsv4dsparkprob_fp4_b300_vllm.sh deleted file mode 100755 index efc6b69a1a..0000000000 --- a/benchmarks/single_node/speedbench/dsv4dsparkprob_fp4_b300_vllm.sh +++ /dev/null @@ -1,10 +0,0 @@ -#!/usr/bin/env bash - -# Probabilistic-drafting arm of the DSV4-Pro DSpark AL collection: identical to -# dsv4dspark_fp4_b300_vllm.sh except draft_sample_method=probabilistic (the recipe -# uses greedy). A separate file only because speedbench-al.yml resolves the collector -# as ${model-prefix}_fp4_b300_vllm.sh; it delegates so the two arms cannot drift. -# Dispatch with model-prefix=dsv4dsparkprob. - -exec env DRAFT_SAMPLE_METHOD=probabilistic \ - bash "$(dirname "$0")/dsv4dspark_fp4_b300_vllm.sh" "$@" diff --git a/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh deleted file mode 100755 index f1ea2cf347..0000000000 --- a/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh +++ /dev/null @@ -1,338 +0,0 @@ -#!/usr/bin/env bash - -# Kimi-K3 B300 vLLM SPEED-Bench AL matrix collector for DSpark speculative decoding. -# -# For each DSpark speculative-token count, measure the REAL acceptance length (AL) on -# one SPEED-Bench category and emit a YAML matrix in the golden_al_distribution shape. -# The Inferact/Kimi-K3-DSpark draft head is downloaded to a writable workspace dir; -# the target (moonshotai/Kimi-K3, FP4) is pre-staged at /scratch/models/Kimi-K3. -# K3 is a thinking model (kimi_k3 reasoning parser defaults enable_thinking=True), -# so the golden curve is collected for thinking_on only. -# -# Usage (inside the Kimi-K3 vLLM container, on a B300 node): -# export MODEL=moonshotai/Kimi-K3 -# bash benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" -GPU_MEM_UTIL="0.90" -MAX_MODEL_LEN="16384" -MAX_NUM_SEQS="512" - -DRAFT_MODEL="Inferact/Kimi-K3-DSpark" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -# AL is concurrency-independent (per-token accept/reject; no spec-disable-by-batch is -# set), so batch the SPEED-Bench pass to stay under the CI wall-time limit; conc=1 -# blew the 8h budget on an earlier Kimi model. -CONCURRENCY="64" -TOP_P="0.95" -# K3 defaults to thinking ON, so the on-cell kwargs are explicit and the off-cell -# kwargs disable it. speedbench-al.yml's thinking-kwargs input defaults to the DSV4 -# value and is exported as CHAT_TEMPLATE_KWARGS_ON; dispatch K3 with -# -f 'thinking-kwargs={"thinking": true}'. -CHAT_TEMPLATE_KWARGS_OFF='{"thinking": false}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export NCCL_DMABUF_ENABLE="0" -export VLLM_ALLREDUCE_USE_FLASHINFER="1" -export VLLM_USE_RUST_FRONTEND="1" -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -# `vllm bench serve` os.execv's the CLIENT into the Rust vllm-rs binary whenever the -# dataset (speed_bench qualifies) and backend are supported, with no opt-out. The -# Rust flag surface lacks --speed-bench-output-len, --save-detailed and -# --chat-template-kwargs, so every cell would die at argument parsing. Call the -# Python entrypoint directly (vllm.benchmarks.serve.main), as the CLI's own Python -# fallback does. The SERVER keeps the Rust frontend; AL is read from /metrics. -BENCH_DRIVER="$RESULTS_DIR/bench_serve_python.py" -mkdir -p "$RESULTS_DIR" -cat > "$BENCH_DRIVER" <<'PYEOF' -# Python-only `vllm bench serve`: bypasses the Rust os.execv delegation in -# vllm/entrypoints/cli/benchmark/serve.py by importing the benchmark directly. -from vllm.benchmarks.serve import add_cli_args, main - -try: # import path moved in newer vLLM - from vllm.utils.argparse_utils import FlexibleArgumentParser -except ImportError: - from vllm.utils import FlexibleArgumentParser - -parser = FlexibleArgumentParser( - description="vllm bench serve (Python entrypoint, no Rust delegation)" -) -add_cli_args(parser) -main(parser.parse_args()) -PYEOF -# No `set -e` here, so a failed redirect would otherwise only surface later as a -# confusing "No such file" from the preflight. -if [[ ! -s "$BENCH_DRIVER" ]]; then - echo "CRITICAL: could not write the benchmark driver to $BENCH_DRIVER — aborting." - exit 1 -fi - -nvidia-smi - -# Kimi-K3 is in the launcher's STAGED_MODELS (read-only /scratch/models/Kimi-K3), -# so this is a no-op in CI; it covers a standalone run with unstaged weights. -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi -fi - -# dirname(MODEL_PATH) can be the read-only staged mount (/scratch/models), so the -# draft must go to a writable workspace dir, not next to the target. -DRAFT_DIR="/workspace/draft_models" -mkdir -p "$DRAFT_DIR" -DRAFT_MODEL_PATH="$DRAFT_DIR/${DRAFT_MODEL##*/}" -if [[ ! -d "$DRAFT_MODEL_PATH" || -z "$(ls -A "$DRAFT_MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$DRAFT_MODEL" --local-dir "$DRAFT_MODEL_PATH" -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -NEED_SHIM=0 -if [[ " $THINKING_MODES " == *" on "* && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then NEED_SHIM=1; fi -if [[ " $THINKING_MODES " == *" off "* && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then NEED_SHIM=1; fi -if [[ "$NEED_SHIM" == "1" ]]; then - if ! apply_chat_template_kwargs_shim; then - echo "CRITICAL: --chat-template-kwargs shim failed — aborting" - exit 1 - fi -fi - -# Preflight the client flags every cell uses; otherwise a CLI mismatch only shows up -# as an all-N/A matrix after eight full server starts (~1h). Probe the driver, not -# `vllm bench serve` (its help exits before the Rust execv), and ask for --help=all -# since plain --help prints only a group summary. -BENCH_HELP="$(python3 "$BENCH_DRIVER" --help=all 2>&1)" -for flag in --speed-bench-category --speed-bench-output-len --chat-template-kwargs --save-detailed; do - if [[ "$BENCH_HELP" != *"$flag"* ]]; then - echo "CRITICAL: the Python benchmark entrypoint does not support $flag — aborting." - echo "--- python3 $BENCH_DRIVER --help=all ---" - echo "$BENCH_HELP" - exit 1 - fi -done -echo "=== Benchmark client flag preflight OK ===" - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi -EP_ARGS=() -if [ "${EP_SIZE}" -gt 1 ]; then - EP_ARGS=(--enable-expert-parallel) -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - local temperature - if [[ "$mode" == "on" ]]; then - temperature=1.0 - if [[ -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - fi - else - temperature=0.6 - if [[ -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - fi - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode DSPARK=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --trust-remote-code - --load-format fastsafetensors - --moe-backend auto - --enable-prefix-caching - --kv-cache-dtype fp8 - "${EP_ARGS[@]}" - --reasoning-parser kimi_k3 - --tool-call-parser kimi_k3 - --enable-auto-tool-choice - --gpu-memory-utilization "$GPU_MEM_UTIL" - --max-num-seqs "$MAX_NUM_SEQS" - --max-model-len "$MAX_MODEL_LEN" - --max-cudagraph-capture-size 256 - --attention-config '{"mla_prefill_backend":"FLASHINFER","use_prefill_query_quantization":true}' - --speculative-config "{\"method\": \"dspark\", \"model\": \"$DRAFT_MODEL_PATH\", \"num_speculative_tokens\": $mtp, \"attention_backend\": \"FLASHINFER_MLA\"}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode dspark=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - python3 "$BENCH_DRIVER" \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$temperature" \ - --top-p "$TOP_P" \ - "${think_args[@]}" - local bench_rc=$? - if [[ $bench_rc -ne 0 ]]; then - echo " -> benchmark client exited rc=$bench_rc (thinking=$mode dspark=$mtp); cell will be N/A" - fi - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode DSPARK=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | top_p: $TOP_P | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on: temperature=1.0, chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo "# thinking_off: temperature=0.6, chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - fi - echo "# Measured on $MODEL_KEY (B300, vLLM DSpark, draft: $DRAFT_MODEL), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (DSpark level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" - -# A matrix where every cell is N/A is a failed collection, not a result: fail the -# job so it is not mistaken for a curve worth reviewing. -MEASURED=0 -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - [[ "${AL_RESULT[${mode}_${mtp}]:-N/A}" != "N/A" ]] && MEASURED=$((MEASURED + 1)) - done -done -if [[ "$MEASURED" -eq 0 ]]; then - echo "CRITICAL: no cell produced an AL value — see the server logs and the" - echo "benchmark client output above." - exit 1 -fi diff --git a/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh b/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh deleted file mode 100755 index 1f1d88a2a8..0000000000 --- a/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh +++ /dev/null @@ -1,342 +0,0 @@ -#!/usr/bin/env bash - -# Kimi-K3 B300 vLLM SPEED-Bench AL matrix collector for DSpark speculative decoding. -# -# For each DSpark speculative-token count, measure the REAL acceptance length (AL) on -# one SPEED-Bench category and emit a YAML matrix in the golden_al_distribution shape. -# The Inferact/Kimi-K3-DSpark draft head is downloaded to a writable workspace dir; -# the target (moonshotai/Kimi-K3, FP4) is pre-staged at /scratch/models/Kimi-K3. -# K3 is a thinking model (kimi_k3 reasoning parser defaults enable_thinking=True), -# so the golden curve is collected for thinking_on only. -# -# VARIANT of kimik3_fp4_b300_vllm.sh (the baseline): the speculative-config also sets -# draft_sample_method=probabilistic and rejection_sample_method=block so baseline and -# variant AL curves can be measured side by side. -# -# Usage (inside the Kimi-K3 vLLM container, on a B300 node): -# export MODEL=moonshotai/Kimi-K3 -# bash benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" -GPU_MEM_UTIL="0.90" -MAX_MODEL_LEN="16384" -MAX_NUM_SEQS="512" - -DRAFT_MODEL="Inferact/Kimi-K3-DSpark" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -# AL is concurrency-independent (per-token accept/reject; no spec-disable-by-batch is -# set), so batch the SPEED-Bench pass to stay under the CI wall-time limit; conc=1 -# blew the 8h budget on an earlier Kimi model. -CONCURRENCY="64" -TOP_P="0.95" -# K3 defaults to thinking ON, so the on-cell kwargs are explicit and the off-cell -# kwargs disable it. speedbench-al.yml's thinking-kwargs input defaults to the DSV4 -# value and is exported as CHAT_TEMPLATE_KWARGS_ON; dispatch K3 with -# -f 'thinking-kwargs={"thinking": true}'. -CHAT_TEMPLATE_KWARGS_OFF='{"thinking": false}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export NCCL_DMABUF_ENABLE="0" -export VLLM_ALLREDUCE_USE_FLASHINFER="1" -export VLLM_USE_RUST_FRONTEND="1" -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -# `vllm bench serve` os.execv's the CLIENT into the Rust vllm-rs binary whenever the -# dataset (speed_bench qualifies) and backend are supported, with no opt-out. The -# Rust flag surface lacks --speed-bench-output-len, --save-detailed and -# --chat-template-kwargs, so every cell would die at argument parsing. Call the -# Python entrypoint directly (vllm.benchmarks.serve.main), as the CLI's own Python -# fallback does. The SERVER keeps the Rust frontend; AL is read from /metrics. -BENCH_DRIVER="$RESULTS_DIR/bench_serve_python.py" -mkdir -p "$RESULTS_DIR" -cat > "$BENCH_DRIVER" <<'PYEOF' -# Python-only `vllm bench serve`: bypasses the Rust os.execv delegation in -# vllm/entrypoints/cli/benchmark/serve.py by importing the benchmark directly. -from vllm.benchmarks.serve import add_cli_args, main - -try: # import path moved in newer vLLM - from vllm.utils.argparse_utils import FlexibleArgumentParser -except ImportError: - from vllm.utils import FlexibleArgumentParser - -parser = FlexibleArgumentParser( - description="vllm bench serve (Python entrypoint, no Rust delegation)" -) -add_cli_args(parser) -main(parser.parse_args()) -PYEOF -# No `set -e` here, so a failed redirect would otherwise only surface later as a -# confusing "No such file" from the preflight. -if [[ ! -s "$BENCH_DRIVER" ]]; then - echo "CRITICAL: could not write the benchmark driver to $BENCH_DRIVER — aborting." - exit 1 -fi - -nvidia-smi - -# Kimi-K3 is in the launcher's STAGED_MODELS (read-only /scratch/models/Kimi-K3), -# so this is a no-op in CI; it covers a standalone run with unstaged weights. -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi -fi - -# dirname(MODEL_PATH) can be the read-only staged mount (/scratch/models), so the -# draft must go to a writable workspace dir, not next to the target. -DRAFT_DIR="/workspace/draft_models" -mkdir -p "$DRAFT_DIR" -DRAFT_MODEL_PATH="$DRAFT_DIR/${DRAFT_MODEL##*/}" -if [[ ! -d "$DRAFT_MODEL_PATH" || -z "$(ls -A "$DRAFT_MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$DRAFT_MODEL" --local-dir "$DRAFT_MODEL_PATH" -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -NEED_SHIM=0 -if [[ " $THINKING_MODES " == *" on "* && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then NEED_SHIM=1; fi -if [[ " $THINKING_MODES " == *" off "* && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then NEED_SHIM=1; fi -if [[ "$NEED_SHIM" == "1" ]]; then - if ! apply_chat_template_kwargs_shim; then - echo "CRITICAL: --chat-template-kwargs shim failed — aborting" - exit 1 - fi -fi - -# Preflight the client flags every cell uses; otherwise a CLI mismatch only shows up -# as an all-N/A matrix after eight full server starts (~1h). Probe the driver, not -# `vllm bench serve` (its help exits before the Rust execv), and ask for --help=all -# since plain --help prints only a group summary. -BENCH_HELP="$(python3 "$BENCH_DRIVER" --help=all 2>&1)" -for flag in --speed-bench-category --speed-bench-output-len --chat-template-kwargs --save-detailed; do - if [[ "$BENCH_HELP" != *"$flag"* ]]; then - echo "CRITICAL: the Python benchmark entrypoint does not support $flag — aborting." - echo "--- python3 $BENCH_DRIVER --help=all ---" - echo "$BENCH_HELP" - exit 1 - fi -done -echo "=== Benchmark client flag preflight OK ===" - -PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1) -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP") -fi -EP_ARGS=() -if [ "${EP_SIZE}" -gt 1 ]; then - EP_ARGS=(--enable-expert-parallel) -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - local temperature - if [[ "$mode" == "on" ]]; then - temperature=1.0 - if [[ -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - fi - else - temperature=0.6 - if [[ -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - fi - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode DSPARK=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --trust-remote-code - --load-format fastsafetensors - --moe-backend auto - --enable-prefix-caching - --kv-cache-dtype fp8 - "${EP_ARGS[@]}" - --reasoning-parser kimi_k3 - --tool-call-parser kimi_k3 - --enable-auto-tool-choice - --gpu-memory-utilization "$GPU_MEM_UTIL" - --max-num-seqs "$MAX_NUM_SEQS" - --max-model-len "$MAX_MODEL_LEN" - --max-cudagraph-capture-size 256 - --attention-config '{"mla_prefill_backend":"FLASHINFER","use_prefill_query_quantization":true}' - --speculative-config "{\"method\": \"dspark\", \"model\": \"$DRAFT_MODEL_PATH\", \"num_speculative_tokens\": $mtp, \"attention_backend\": \"FLASHINFER_MLA\", \"draft_sample_method\": \"probabilistic\", \"rejection_sample_method\": \"block\"}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode dspark=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - python3 "$BENCH_DRIVER" \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$temperature" \ - --top-p "$TOP_P" \ - "${think_args[@]}" - local bench_rc=$? - if [[ $bench_rc -ne 0 ]]; then - echo " -> benchmark client exited rc=$bench_rc (thinking=$mode dspark=$mtp); cell will be N/A" - fi - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode DSPARK=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | top_p: $TOP_P | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on: temperature=1.0, chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo "# thinking_off: temperature=0.6, chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - fi - echo "# Measured on $MODEL_KEY (B300, vLLM DSpark, draft: $DRAFT_MODEL), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (DSpark level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" - -# A matrix where every cell is N/A is a failed collection, not a result: fail the -# job so it is not mistaken for a curve worth reviewing. -MEASURED=0 -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - [[ "${AL_RESULT[${mode}_${mtp}]:-N/A}" != "N/A" ]] && MEASURED=$((MEASURED + 1)) - done -done -if [[ "$MEASURED" -eq 0 ]]; then - echo "CRITICAL: no cell produced an AL value — see the server logs and the" - echo "benchmark client output above." - exit 1 -fi diff --git a/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh deleted file mode 100755 index fc5ab0c968..0000000000 --- a/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh +++ /dev/null @@ -1,245 +0,0 @@ -#!/usr/bin/env bash - -# MiniMax-M3 B300 vLLM SPEED-Bench AL matrix collector for EAGLE3 speculative decoding. -# -# For each thinking mode (on/off) and EAGLE3 level (num_speculative_tokens), measure -# the REAL acceptance length (AL) on one SPEED-Bench category and emit a YAML matrix -# in the golden_al_distribution shape. The synthetic value is injected downstream by -# the throughput recipe, not here. -# -# Draft: Inferact/MiniMax-M3-EAGLE3. The EAGLE3 head is MHA and must use FLASH_ATTN -# (FlashInfer only supports page size 128 through its trtllm-gen kernel, which needs -# GQA/MQA); the target keeps its default FlashInfer backend. --block-size 128 is -# mandatory for the MSA sparse/index cache. --language-model-only frees the vision -# encoder's VRAM (text-only benchmark). -# -# The *_fp4_* filename is only the naming convention speedbench-al.yml requires -# (${model-prefix}_fp4_b300_vllm.sh); the staged MiniMax-M3 weights are unquantized -# BF16, so no quantization-specific flags (--moe-backend marlin, --kv-cache-dtype -# fp8) apply. -# -# Dispatch this collector through speedbench-al.yml. -# -# Required collection settings come from speedbench-al.yml. - -set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars \ - CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ - MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP - -SERVE_MODEL="${MODEL_PATH}" -GPU_MEM_UTIL="0.90" - -DRAFT_MODEL="Inferact/MiniMax-M3-EAGLE3" - -MODEL_KEY="$(basename "$SERVE_MODEL" | tr '[:upper:]' '[:lower:]')" -CONCURRENCY="1" -# Official MiniMax-M3 sampling: temperature 1.0, top_p 0.95, top_k 40. -TEMPERATURE="1.0" -TOP_P="0.95" -TOP_K="40" -# M3 thinking toggles via the thinking_mode chat_template key. -CHAT_TEMPLATE_KWARGS_OFF='{"thinking_mode": "disabled"}' - -SPEEDBENCH_DIR="/workspace/speed_bench_data" -RESULTS_DIR="/workspace/speedbench_results" - -export VLLM_FLOAT32_MATMUL_PRECISION="high" -export VLLM_ENGINE_READY_TIMEOUT_S=3600 - -mkdir -p "$RESULTS_DIR" -nvidia-smi -if [[ "$SERVE_MODEL" != /* ]]; then hf download "$SERVE_MODEL"; fi - -# dirname(MODEL_PATH) is the read-only staged mount (/scratch/models), so the draft -# must go to a writable workspace dir, not next to the target. -echo "=== Downloading EAGLE3 draft model ($DRAFT_MODEL) ===" -DRAFT_DIR="/workspace/draft_models" -mkdir -p "$DRAFT_DIR" -DRAFT_MODEL_PATH="$DRAFT_DIR/${DRAFT_MODEL##*/}" -if [[ ! -d "$DRAFT_MODEL_PATH" || -z "$(ls -A "$DRAFT_MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$DRAFT_MODEL" --local-dir "$DRAFT_MODEL_PATH" -fi - -echo "=== Downloading SPEED-Bench dataset ===" -pip install -q datasets tiktoken -curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \ - | python3 - --config qualitative --output_dir "$SPEEDBENCH_DIR" - -if [[ ! -f "$SPEEDBENCH_DIR/qualitative.jsonl" ]]; then - echo "CRITICAL: SPEED-Bench download failed — $SPEEDBENCH_DIR/qualitative.jsonl not found" - exit 1 -fi - -if [ "${DP_ATTENTION}" = "true" ]; then - PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP" --enable-expert-parallel) -elif [ "${EP_SIZE}" -gt 1 ]; then - PARALLEL_ARGS=(--tensor-parallel-size "$TP" --enable-expert-parallel) -else - # Plain TP per the official MiniMax-M3 recipe. Do NOT force a MoE backend: the - # staged checkpoint is unquantized BF16, for which marlin is rejected. - PARALLEL_ARGS=(--tensor-parallel-size "$TP") -fi - -fetch_metric() { - local port="$1" name="$2" - curl -s "http://localhost:${port}/metrics" \ - | grep -oP "${name}\\{[^}]*\\} \\K[0-9.]+" || echo "0" -} - -SERVER_PID="" -_descendants() { - local pid="$1" child - for child in $(pgrep -P "$pid" 2>/dev/null || true); do - echo "$child" - _descendants "$child" - done -} -cleanup_server() { - if [[ -n "$SERVER_PID" ]]; then - local descendants - descendants=$(_descendants "$SERVER_PID") - kill "$SERVER_PID" 2>/dev/null || true - wait "$SERVER_PID" 2>/dev/null || true - local pid - for pid in $descendants; do - kill -9 "$pid" 2>/dev/null || true - done - local waited=0 - while [[ $waited -lt 120 ]]; do - local used - used=$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sort -rn | head -1) - if [[ -z "$used" || "$used" -lt 2000 ]]; then break; fi - sleep 3; waited=$((waited + 3)) - done - SERVER_PID="" - fi -} -trap 'cleanup_server' EXIT - -start_gpu_monitor - -declare -A AL_RESULT - -run_cell() { - local mode="$1" mtp="$2" - local think_args=() - if [[ "$mode" == "on" && -n "$CHAT_TEMPLATE_KWARGS_ON" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_ON") - elif [[ "$mode" == "off" && -n "$CHAT_TEMPLATE_KWARGS_OFF" ]]; then - think_args=(--chat-template-kwargs "$CHAT_TEMPLATE_KWARGS_OFF") - fi - - echo "" - echo "==========================================" - echo " Cell: thinking=$mode EAGLE3=$mtp category=$CATEGORY" - echo "==========================================" - - local serve_args=( - --host 0.0.0.0 --port "$PORT" - "${PARALLEL_ARGS[@]}" - --pipeline-parallel-size 1 - --block-size 128 - --language-model-only - --max-cudagraph-capture-size 2048 - --trust-remote-code - --no-enable-prefix-caching - --gpu-memory-utilization "$GPU_MEM_UTIL" - --max-model-len 16384 - --max-num-batched-tokens 16384 - --stream-interval 30 - --speculative-config "{\"method\": \"eagle3\", \"model\": \"$DRAFT_MODEL_PATH\", \"num_speculative_tokens\": $mtp, \"attention_backend\": \"FLASH_ATTN\"}" - ) - - local server_log="$RESULTS_DIR/server_${mode}_mtp${mtp}.log" - vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & - SERVER_PID=$! - - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then - echo " -> server failed to start (thinking=$mode eagle3=$mtp), recording N/A" - AL_RESULT["${mode}_${mtp}"]="N/A" - cleanup_server - return - fi - - local acc_before drf_before acc_after drf_after - acc_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_before=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - vllm bench serve \ - --model "$SERVE_MODEL" \ - --port "$PORT" \ - --dataset-name speed_bench \ - --dataset-path "$SPEEDBENCH_DIR" \ - --speed-bench-category "$CATEGORY" \ - --speed-bench-output-len "$SPEEDBENCH_OUTPUT_LEN" \ - --num-prompts -1 \ - --max-concurrency "$CONCURRENCY" \ - --save-result \ - --save-detailed \ - --result-dir "$RESULTS_DIR" \ - --result-filename "speedbench_${mode}_mtp${mtp}" \ - --trust-remote-code \ - --temperature "$TEMPERATURE" \ - --top-p "$TOP_P" \ - --top-k "$TOP_K" \ - "${think_args[@]}" - - acc_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_accepted_tokens_total") - drf_after=$(fetch_metric "$PORT" "vllm:spec_decode_num_drafts_total") - - local delta_acc delta_drf al - delta_acc=$(awk "BEGIN {printf \"%d\", $acc_after - $acc_before}") - delta_drf=$(awk "BEGIN {printf \"%d\", $drf_after - $drf_before}") - if [[ "$delta_drf" -gt 0 ]]; then - al=$(awk "BEGIN {printf \"%.2f\", 1 + ($delta_acc / $delta_drf)}") - else - al="N/A" - fi - echo " -> thinking=$mode EAGLE3=$mtp AL=$al (accepted=$delta_acc drafts=$delta_drf)" - AL_RESULT["${mode}_${mtp}"]="$al" - - cleanup_server -} - -for mode in $THINKING_MODES; do - for mtp in $MTP_LIST; do - run_cell "$mode" "$mtp" - done -done - -stop_gpu_monitor - -emit_mode_block() { - local mode="$1" - for mtp in $MTP_LIST; do - echo " $mtp: ${AL_RESULT[${mode}_${mtp}]:-N/A}" - done -} - -{ - echo "# Acceptance Length (AL) reference values measured with SPEED-Bench." - echo "# dataset: $CATEGORY | temperature: $TEMPERATURE | top_p: $TOP_P | top_k: $TOP_K | output_len: $SPEEDBENCH_OUTPUT_LEN" - echo "# thinking_on chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_ON" - echo "# thinking_off chat_template_kwargs: $CHAT_TEMPLATE_KWARGS_OFF" - echo "# Measured on $MODEL_KEY (B300, vLLM EAGLE3), per num_speculative_tokens." - echo "# Auto-generated by benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh (speedbench-al.yml)." - echo "#" - echo "# key = num_speculative_tokens (EAGLE3 level); value = golden AL" - echo "${MODEL_KEY}:" - if [[ " $THINKING_MODES " == *" on "* ]]; then - echo " thinking_on:" - emit_mode_block on - fi - if [[ " $THINKING_MODES " == *" off "* ]]; then - echo " thinking_off:" - emit_mode_block off - fi -} > "$OUT_YAML" - -echo "" -echo "==========================================" -echo " SPEED-Bench AL matrix written to: $OUT_YAML" -echo "==========================================" -cat "$OUT_YAML" diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4dspark/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4dspark/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..c81e169e59 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4dspark/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,87 @@ +# DeepSeek-V4-Pro-DSpark B300 vLLM SPEED-Bench AL matrix (DSpark, greedy draft). +# +# DSpark is baked into the DeepSeek-V4-Pro-DSpark checkpoint (no external draft +# model). The legacy collector hard-coded TEP8 (--enable-expert-parallel), but +# the srt-slurm recipe omits it because the workflow exports EP_SIZE=1 and +# validate_recipe enforces the match. AL is expert-placement-independent. +# GPU memory utilization 0.90 and max-num-seqs 64 leave headroom for the DSpark +# rejection sampler's fp32 logits scratch and CUDA graphs. +# SPEEDBENCH_CONCURRENCY=32 batches the SPEED-Bench pass to stay under CI +# wall-time. The client-side --chat-template-kwargs shim is applied when +# APPLY_CHAT_TEMPLATE_KWARGS_SHIM=1. +base: + schema: 2 + name: dsv4dspark-fp4-b300-vllm-speedbench + model: + path: hf:deepseek-ai/DeepSeek-V4-Pro-DSpark + container: vllm/vllm-openai:v0.21.0 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro-DSpark + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + kv-cache-dtype: fp8 + block-size: 256 + no-enable-prefix-caching: true + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' + attention-config: '{"use_fp4_indexer_cache":true}' + tokenizer-mode: deepseek_v4 + tool-call-parser: deepseek_v4 + enable-auto-tool-choice: true + reasoning-parser: deepseek_v4 + max-cudagraph-capture-size: 2048 + max-model-len: 16384 + max-num-seqs: 64 + gpu-memory-utilization: 0.90 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro-DSpark + SPEC_DECODING: mtp + TEMPERATURE: '1.0' + SPEEDBENCH_TOKENIZER_MODE: deepseek_v4 + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + SPEEDBENCH_CONCURRENCY: '32' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"dspark","num_speculative_tokens":1,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":2,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":3,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":4,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":6,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":7,"draft_sample_method":"greedy"}' + - '{"method":"dspark","num_speculative_tokens":8,"draft_sample_method":"greedy"}' diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4dsparkprob/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4dsparkprob/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..1181322096 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4dsparkprob/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,82 @@ +# DeepSeek-V4-Pro-DSpark B300 vLLM SPEED-Bench AL matrix (DSpark, probabilistic draft). +# +# Identical to dsv4dspark except draft_sample_method=probabilistic. The greedy +# vs probabilistic comparison isolates the effect of draft-token sampling on +# acceptance length. No rejection_sample_method override (the bash wrapper +# dsv4dsparkprob_fp4_b300_vllm.sh only sets DRAFT_SAMPLE_METHOD=probabilistic). +base: + schema: 2 + name: dsv4dsparkprob-fp4-b300-vllm-speedbench + model: + path: hf:deepseek-ai/DeepSeek-V4-Pro-DSpark + container: vllm/vllm-openai:v0.21.0 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro-DSpark + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + kv-cache-dtype: fp8 + block-size: 256 + no-enable-prefix-caching: true + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","custom_ops":["all"]}' + attention-config: '{"use_fp4_indexer_cache":true}' + tokenizer-mode: deepseek_v4 + tool-call-parser: deepseek_v4 + enable-auto-tool-choice: true + reasoning-parser: deepseek_v4 + max-cudagraph-capture-size: 2048 + max-model-len: 16384 + max-num-seqs: 64 + gpu-memory-utilization: 0.90 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro-DSpark + SPEC_DECODING: mtp + TEMPERATURE: '1.0' + SPEEDBENCH_TOKENIZER_MODE: deepseek_v4 + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + SPEEDBENCH_CONCURRENCY: '32' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"dspark","num_speculative_tokens":1,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":2,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":3,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":4,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":6,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":7,"draft_sample_method":"probabilistic"}' + - '{"method":"dspark","num_speculative_tokens":8,"draft_sample_method":"probabilistic"}' diff --git a/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..16c253d479 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,91 @@ +# Kimi-K3 B300 vLLM SPEED-Bench AL matrix (DSpark, greedy draft). +# +# External draft model: Inferact/Kimi-K3-DSpark. The target (moonshotai/Kimi-K3) +# is pre-staged; vLLM auto-downloads the draft by its HF id. +# K3 defaults to thinking ON (kimi_k3 reasoning parser), so the off-cell +# CHAT_TEMPLATE_KWARGS_OFF disables it. Per-mode temperatures: 1.0 thinking-on, +# 0.6 thinking-off. SPEEDBENCH_CONCURRENCY=64 keeps collection under the CI +# wall-time limit. +# +# enable-prefix-caching is ON (not no-enable-prefix-caching) to match the +# production K3 recipe; FlashInfer MLA attention backend for both target and draft. +base: + schema: 2 + name: kimik3-fp4-b300-vllm-speedbench + model: + path: hf:moonshotai/Kimi-K3 + container: vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # Bind GPUs via CUDA_VISIBLE_DEVICES for image compatibility. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: moonshotai/Kimi-K3 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + load-format: fastsafetensors + moe-backend: auto + enable-prefix-caching: true + kv-cache-dtype: fp8 + reasoning-parser: kimi_k3 + tool-call-parser: kimi_k3 + enable-auto-tool-choice: true + gpu-memory-utilization: 0.90 + max-num-seqs: 512 + max-model-len: 16384 + max-cudagraph-capture-size: 256 + attention-config: '{"mla_prefill_backend":"FLASHINFER","use_prefill_query_quantization":true}' + env: + NCCL_DMABUF_ENABLE: '0' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: moonshotai/Kimi-K3 + SPEC_DECODING: mtp + TEMPERATURE_ON: '1.0' + TEMPERATURE_OFF: '0.6' + TOP_P: '0.95' + CHAT_TEMPLATE_KWARGS_OFF: '{"thinking": false}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + SPEEDBENCH_CONCURRENCY: '64' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":1,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":3,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":5,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":6,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"attention_backend":"FLASHINFER_MLA"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":8,"attention_backend":"FLASHINFER_MLA"}' diff --git a/benchmarks/single_node/srt-slurm-recipes/kimik3prob/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/kimik3prob/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..acb59e9b04 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/kimik3prob/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,87 @@ +# Kimi-K3 B300 vLLM SPEED-Bench AL matrix (DSpark, probabilistic draft + block verify). +# +# Identical to kimik3 except draft_sample_method=probabilistic and +# rejection_sample_method=block in speculative-config. This matches the +# kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh +# collector and the kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method +# golden curve. +base: + schema: 2 + name: kimik3prob-fp4-b300-vllm-speedbench + model: + path: hf:moonshotai/Kimi-K3 + container: vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # Bind GPUs via CUDA_VISIBLE_DEVICES for image compatibility. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: moonshotai/Kimi-K3 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + trust-remote-code: true + load-format: fastsafetensors + moe-backend: auto + enable-prefix-caching: true + kv-cache-dtype: fp8 + reasoning-parser: kimi_k3 + tool-call-parser: kimi_k3 + enable-auto-tool-choice: true + gpu-memory-utilization: 0.90 + max-num-seqs: 512 + max-model-len: 16384 + max-cudagraph-capture-size: 256 + attention-config: '{"mla_prefill_backend":"FLASHINFER","use_prefill_query_quantization":true}' + env: + NCCL_DMABUF_ENABLE: '0' + VLLM_ALLREDUCE_USE_FLASHINFER: '1' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: moonshotai/Kimi-K3 + SPEC_DECODING: mtp + TEMPERATURE_ON: '1.0' + TEMPERATURE_OFF: '0.6' + TOP_P: '0.95' + CHAT_TEMPLATE_KWARGS_OFF: '{"thinking": false}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + SPEEDBENCH_CONCURRENCY: '64' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":1,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":3,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":4,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":5,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":6,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + - '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":8,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' diff --git a/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml new file mode 100644 index 0000000000..4f05aac3ec --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml @@ -0,0 +1,86 @@ +# MiniMax-M3 B300 vLLM SPEED-Bench AL matrix (EAGLE3, MHA draft head). +# +# External draft model: Inferact/MiniMax-M3-EAGLE3. The EAGLE3 head is MHA +# and must use FLASH_ATTN (FlashInfer only supports page size 128 through its +# trtllm-gen kernel, which needs GQA/MQA); the target keeps its default +# FlashInfer backend. --block-size 128 is mandatory for the MSA sparse/index +# cache. --language-model-only frees the vision encoder VRAM (text-only +# benchmark). The staged MiniMax-M3 weights are unquantized BF16, so no +# quantization-specific flags (--moe-backend marlin, --kv-cache-dtype fp8) +# apply. +# +# Official MiniMax-M3 sampling: temperature 1.0, top_p 0.95, top_k 40. +# Thinking toggles via the thinking_mode chat_template key. +base: + schema: 2 + name: minimaxm3-fp4-b300-vllm-speedbench + model: + path: hf:MiniMax/MiniMax-M3 + container: vllm/vllm-openai:nightly-dev-x86_64-cu13-af03963 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + # Bind GPUs via CUDA_VISIBLE_DEVICES for image compatibility. + set_visible_devices: true + health_check: + interval_seconds: 10 + max_attempts: 360 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: MiniMax/MiniMax-M3 + tensor-parallel-size: 8 + data-parallel-size: 1 + pipeline-parallel-size: 1 + block-size: 128 + language-model-only: true + max-cudagraph-capture-size: 2048 + trust-remote-code: true + no-enable-prefix-caching: true + gpu-memory-utilization: 0.90 + max-model-len: 16384 + max-num-batched-tokens: 16384 + stream-interval: 30 + env: + VLLM_FLOAT32_MATMUL_PRECISION: high + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_speedbench.sh + env: + MODEL: MiniMax/MiniMax-M3 + SPEC_DECODING: mtp + TEMPERATURE: '1.0' + TOP_P: '0.95' + TOP_K: '40' + CHAT_TEMPLATE_KWARGS_OFF: '{"thinking_mode": "disabled"}' + SPEEDBENCH_TRUST_REMOTE_CODE: '1' + APPLY_CHAT_TEMPLATE_KWARGS_SHIM: '1' + +zip_override_mtp: + roles: + agg: + args: + speculative-config: + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":1,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":2,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":3,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":4,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":5,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":6,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":7,"attention_backend":"FLASH_ATTN"}' + - '{"method":"eagle3","model":"Inferact/MiniMax-M3-EAGLE3","num_speculative_tokens":8,"attention_backend":"FLASH_ATTN"}' diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index ab5dbdc615..8deffd1a8b 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -679,3 +679,83 @@ def test_speedbench_client_finds_benchmark_lib_without_legacy_workspace(): assert result.returncode != 0 assert "No such file" not in result.stderr, result.stderr assert "required environment variables are not set" in result.stdout + + +# --- Recipe-render validation for every speedbench prefix --- + +# Per-prefix dispatch environment that mirrors the speedbench-al.yml inputs. +_SPEEDBENCH_PREFIXES = { + "dsv4": { + "MODEL": "deepseek-ai/DeepSeek-V4-Pro", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "dsv4dspark": { + "MODEL": "deepseek-ai/DeepSeek-V4-Pro-DSpark", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "dsv4dsparkprob": { + "MODEL": "deepseek-ai/DeepSeek-V4-Pro-DSpark", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "glm52": { + "MODEL": "nvidia/GLM-5.2-NVFP4", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "kimik3": { + "MODEL": "moonshotai/Kimi-K3", + "IMAGE": "vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77", + "TP": "8", "GPU_COUNT": "8", + }, + "kimik3prob": { + "MODEL": "moonshotai/Kimi-K3", + "IMAGE": "vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77", + "TP": "8", "GPU_COUNT": "8", + }, + "minimaxm3": { + "MODEL": "MiniMax/MiniMax-M3", + "IMAGE": "vllm/vllm-openai:nightly-dev-x86_64-cu13-af03963", + "TP": "8", "GPU_COUNT": "8", + }, + "qwen3.5": { + "MODEL": "nvidia/Qwen3.5-397B-A17B-NVFP4-V2", + "IMAGE": "vllm/vllm-openai:v0.21.0", + "TP": "8", "GPU_COUNT": "8", + }, + "qwen3.8next": { + "MODEL": "Qwen/Qwen3.8-Flash-Next-FP8", + "IMAGE": "vllm/vllm-openai:qwen38-flash-next", + "TP": "4", "GPU_COUNT": "4", + }, +} + + +@pytest.mark.parametrize("prefix", sorted(_SPEEDBENCH_PREFIXES)) +def test_speedbench_recipe_validates_and_renders(prefix): + """Each speedbench recipe selects and validates against the workflow env.""" + recipe = Path( + f"benchmarks/single_node/srt-slurm-recipes/{prefix}/vllm/b300-fp4-speedbench/speedbench.yaml" + ) + assert recipe.exists(), f"Recipe not found: {recipe}" + overrides = _SPEEDBENCH_PREFIXES[prefix] + env = { + "FRAMEWORK": "vllm", + "PRECISION": "fp4", + "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", + "EP_SIZE": "1", "DP_ATTENTION": "false", + "SPEC_DECODING": "mtp", "IS_AGENTIC": "0", + "RUN_EVAL": "false", "EVAL_ONLY": "false", + "CONC": "1", "RESULT_FILENAME": "speedbench_off_mtp1", + "GPU_MONITOR_INTERVAL": "3", "MODEL_PREFIX": prefix, + "CATEGORY": "coding", "SPEEDBENCH_OUTPUT_LEN": "4096", + "THINKING": "off", "MTP": "1", + **overrides, + } + argv = runtime_arguments(f"{recipe}:zip_override_mtp[0]", env) + sets = [argv[i + 1] for i, arg in enumerate(argv) if arg == "--set"] + # Verify THINKING survived as a string through the set binding. + thinking_set = [s for s in sets if "THINKING" in s] + assert any("thinking_off" in s for s in thinking_set), thinking_set diff --git a/utils/test_speedbench_matrix.py b/infx/tests/workflows/test_speedbench_matrix.py similarity index 100% rename from utils/test_speedbench_matrix.py rename to infx/tests/workflows/test_speedbench_matrix.py diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index 7964506032..742b0b619a 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -124,9 +124,8 @@ EXECUTION_PATH=agentic if [[ "$IS_MULTINODE" == true ]]; then EXECUTION_PATH=multinode elif [[ -n "${BENCH_SCRIPT_OVERRIDE:-}" ]]; then - # Legacy SPEED-Bench collectors (draft-model variants: dsv4dspark*, kimik3, - # minimaxm3) explicitly supply their script. Native-MTP collectors migrated - # to the srt-slurm path set SRT_RECIPE instead. + # Legacy path: caller supplies a script via BENCH_SCRIPT_OVERRIDE. + # All SPEED-Bench collectors migrated to srt-slurm recipes (SRT_RECIPE). EXECUTION_PATH=script elif [[ "$IS_AGENTIC" == 0 || -n "${SRT_RECIPE:-}" ]]; then check_env_vars SRT_RECIPE From 9fc4be503c99b95aab406dc8854493aee1e94681 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 14:15:52 -0400 Subject: [PATCH 12/12] =?UTF-8?q?fix(speedbench):=20fix=20porting=20regres?= =?UTF-8?q?sions=20in=205=20AL=20collector=20recipes=20/=20=E4=BF=AE?= =?UTF-8?q?=E5=A4=8D=205=20=E4=B8=AA=20AL=20=E6=94=B6=E9=9B=86=E5=99=A8?= =?UTF-8?q?=E9=85=8D=E6=96=B9=E7=9A=84=E8=BF=81=E7=A7=BB=E5=9B=9E=E5=BD=92?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - glm52: use nightly-dev-x86_64-cu13-3696c77 (GLM-5.2 needs transformers newer than v0.21.0 for the layer_types config field) - minimaxm3: replace expired nightly af03963 with v0.21.0 (matches old bash collector; EAGLE3 supported in v0.21.0) - launcher: replace hardcoded RadixArk/Qwen3.8-Flash-Next-NVFP4 model check with generic fallback: any model not staged on NVMe and not present (or empty) on shared storage falls through to hf: download, fixing dsv4dspark, dsv4dsparkprob, and qwen3.8next - tests: update IMAGE expectations to match recipe changes Co-Authored-By: Claude Opus 4.6 --- .../glm52/vllm/b300-fp4-speedbench/speedbench.yaml | 3 +-- .../minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml | 4 ++-- infx/tests/srt_slurm/test_srt_single_node.py | 4 ++-- runners/launch_b300-dsxe.sh | 6 ++++-- 4 files changed, 9 insertions(+), 8 deletions(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml index 6e5ea8793a..af2b8701fa 100644 --- a/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/glm52/vllm/b300-fp4-speedbench/speedbench.yaml @@ -10,7 +10,7 @@ base: name: glm52-fp4-b300-vllm-speedbench model: path: hf:nvidia/GLM-5.2-NVFP4 - container: vllm/vllm-openai:v0.21.0 + container: vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77 precision: fp4 resources: gpu_type: b300 @@ -25,7 +25,6 @@ base: engine: type: vllm connector: null - # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. set_visible_devices: true health_check: interval_seconds: 10 diff --git a/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml b/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml index 4f05aac3ec..9fd8405937 100644 --- a/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/b300-fp4-speedbench/speedbench.yaml @@ -16,7 +16,7 @@ base: name: minimaxm3-fp4-b300-vllm-speedbench model: path: hf:MiniMax/MiniMax-M3 - container: vllm/vllm-openai:nightly-dev-x86_64-cu13-af03963 + container: vllm/vllm-openai:v0.21.0 precision: fp4 resources: gpu_type: b300 @@ -31,7 +31,7 @@ base: engine: type: vllm connector: null - # Bind GPUs via CUDA_VISIBLE_DEVICES for image compatibility. + # vLLM v0.21.0 predates --device-ids; bind GPUs via CUDA_VISIBLE_DEVICES. set_visible_devices: true health_check: interval_seconds: 10 diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index 8deffd1a8b..857fbf3e4e 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -702,7 +702,7 @@ def test_speedbench_client_finds_benchmark_lib_without_legacy_workspace(): }, "glm52": { "MODEL": "nvidia/GLM-5.2-NVFP4", - "IMAGE": "vllm/vllm-openai:v0.21.0", + "IMAGE": "vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77", "TP": "8", "GPU_COUNT": "8", }, "kimik3": { @@ -717,7 +717,7 @@ def test_speedbench_client_finds_benchmark_lib_without_legacy_workspace(): }, "minimaxm3": { "MODEL": "MiniMax/MiniMax-M3", - "IMAGE": "vllm/vllm-openai:nightly-dev-x86_64-cu13-af03963", + "IMAGE": "vllm/vllm-openai:v0.21.0", "TP": "8", "GPU_COUNT": "8", }, "qwen3.5": { diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index 742b0b619a..cf5573cf46 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -142,8 +142,10 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then # Not staged on every node's NVMe; read the shared copy. SRT_MODEL_PATH="$SHARED_MODEL_ROOT/${MODEL##*/}" fi - # Not staged on node-local NVMe: the engine downloads it into the shared HF cache. - if [[ "$MODEL" == RadixArk/Qwen3.8-Flash-Next-NVFP4 ]]; then + # Not staged on node-local NVMe and not pre-copied to shared storage + # (or present as an empty directory from a failed download): let the + # engine download via the mounted HF cache. + if [[ "$SRT_MODEL_PATH" != hf:* && ( ! -d "$SRT_MODEL_PATH" || -z "$(ls -A "$SRT_MODEL_PATH" 2>/dev/null)" ) ]]; then SRT_MODEL_PATH="hf:$MODEL" fi SRT_SQUASH_FILE="$SQUASH_DIR/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh"