diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index b9a66f22ba..9656bf30f1 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -110,6 +110,7 @@ env: HF_HUB_CACHE: '/mnt/hf_hub_cache/' EXP_NAME: ${{ fromJSON(inputs.config).exp-name }} RECIPE_FINGERPRINT: ${{ fromJSON(inputs.config).recipe-fingerprint || '' }} + SRT_RECIPE: ${{ fromJSON(inputs.config).srt-recipe || '' }} MODEL: ${{ fromJSON(inputs.config).model }} THINKING_MODE: thinking_on MODEL_PREFIX: ${{ fromJSON(inputs.config).model-prefix }} @@ -399,6 +400,9 @@ jobs: server.log results/*.log results/*_config.json + srt-single-node-logs.tar.gz + srt-single-node-submission.json + srt-slurm-sha.txt if-no-files-found: ignore - name: Upload GPU metrics diff --git a/MODELS.md b/MODELS.md index 619e3fefc9..6d7019f19a 100644 --- a/MODELS.md +++ b/MODELS.md @@ -58,6 +58,8 @@ Rationale: `dsv4` carries the largest single-turn footprint in the repository. 4 **Deprecation parity audit (2026-09-21):** Active master configs and benchmark-script locations match the enacted retirements above and in the support matrix below. GLM-5.1 B200 TileRT remains the documented exception to the earlier GLM-5/5.1 and 1k1k retirements. Conditional A/B baseline retirement remains pending; non-speculative Pareto contributors remain supported. The broader routing audit also removed stale retired-model branches from launchers/runtime settings and a GLM-5-only environment override, and corrected workflow/agent guidance that still recommended retired coverage. SPEED-Bench collectors, historical result readers, and the explicitly retained recipe YAMLs remain available. Deprecated configs are consolidated in [`configs/deprecated/amd-master.yaml`](configs/deprecated/amd-master.yaml) and [`configs/deprecated/nvidia-master.yaml`](configs/deprecated/nvidia-master.yaml). +**Single-node SRT-only cutover (2026-09-22):** Active single-node fixed-sequence recipes now use SRT-Slurm. The two Docker-only Qwen3.5 RTX PRO 6000 FP4 configs (with and without MTP) are retired, with their original settings preserved in `configs/deprecated/nvidia-master.yaml` and their scripts in `benchmarks/single_node/fixed_seq_len/deprecated/`. The unused `rtx6000pro-lat` runner mappings, launcher, and runtime settings are removed. Qwen3.5 remains active on the other supported Slurm pools; AgentX and multi-node coverage are unchanged. + ## Scenarios | Scenario | ISL/OSL | Status | diff --git a/MODELS_zh.md b/MODELS_zh.md index da8c72b08f..a21efc727e 100644 --- a/MODELS_zh.md +++ b/MODELS_zh.md @@ -58,6 +58,8 @@ InferenceX-e2e 运行在数量固定且有限的 GPU 资源池上,并由一支 **弃用状态一致性核查(2026-09-21):** 启用的主配置及基准测试脚本位置与上述已执行的退役事项和下方支持矩阵一致。GLM-5.1 B200 TileRT 仍是文档明确保留的例外,不受此前 GLM-5/5.1 和 1k1k 退役范围限制。有条件的 A/B 基线退役仍待执行;对 Pareto 前沿有贡献的非投机解码配置继续受支持。进一步的路由核查还移除了启动器和运行时设置中遗留的退役模型分支及 GLM-5 专用环境覆盖,并修正了仍推荐退役配置的工作流和智能体指南。SPEED-Bench 采集器、历史结果读取逻辑及明确保留的配方 YAML 继续保留。弃用配置现统一归档至 [`configs/deprecated/amd-master.yaml`](configs/deprecated/amd-master.yaml) 和 [`configs/deprecated/nvidia-master.yaml`](configs/deprecated/nvidia-master.yaml)。 +**单节点切换为仅使用 SRT(2026-09-22):** 活跃的单节点定长配方现统一使用 SRT-Slurm。两个仅支持 Docker 的 Qwen3.5 RTX PRO 6000 FP4 配置(启用和关闭 MTP)已退役,原始设置保留在 `configs/deprecated/nvidia-master.yaml`,脚本保留在 `benchmarks/single_node/fixed_seq_len/deprecated/`。已移除不再使用的 `rtx6000pro-lat` runner 映射、启动器及运行时设置。Qwen3.5 在其他受支持的 Slurm 池上继续启用;AgentX 和多节点覆盖保持不变。 + ## 场景 | 场景 | ISL/OSL | 状态 | diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index e5765862ed..ed708ede8e 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -425,6 +425,24 @@ GPU_METRICS_CSV="${GPU_METRICS_CSV:-gpu_metrics.csv}" NVIDIA_GPU_MONITOR_QUERY="timestamp,index,power.draw,temperature.gpu,clocks.current.sm,clocks.current.memory,utilization.gpu,utilization.memory" export GPU_METRICS_CSV +# Keep one AMD CSV header and forward each complete row immediately. Some awk +# implementations buffer pipe input even with fflush(), losing the final ticks +# when the monitor stops. +_filter_amd_smi_metrics() { + local line header_seen=false + while IFS= read -r line; do + if [[ "$line" == timestamp,* ]]; then + if [[ "$header_seen" == true ]]; then + continue + fi + header_seen=true + fi + if [[ "$header_seen" == true ]]; then + printf '%s\n' "$line" + fi + done +} + # Background nvidia-smi/amd-smi sampler writing CSV. # Usage: start_gpu_monitor [--output /path/to/output.csv] [--interval 1] start_gpu_monitor() { @@ -457,10 +475,9 @@ start_gpu_monitor() { elif command -v amd-smi &>/dev/null; then GPU_MONITOR_VENDOR="amd" # amd-smi is Python and block-buffers stdout; without PYTHONUNBUFFERED the - # trailing ticks were lost at kill (measured on MI355X). awk keeps the first - # CSV header, drops repeated ones, and flushes every row for the same reason. + # trailing ticks were lost at kill (measured on MI355X). PYTHONUNBUFFERED=1 amd-smi metric -p -c -t -u -w "$interval" --csv 2>/dev/null \ - | awk '/^timestamp,/{if(!h){print;h=1};next} h{print;fflush()}' > "$output" & + | _filter_amd_smi_metrics > "$output" & GPU_MONITOR_PID=$! # Hardware energy-accumulator + identity snapshots; the end-side twin in # stop_gpu_monitor lets auditors cross-check the integrated energy @@ -800,6 +817,7 @@ run_benchmark_serving() { local model="" local port="" local backend="" + local base_url="" local endpoint="" local input_len="" local output_len="" @@ -834,6 +852,10 @@ run_benchmark_serving() { endpoint="$2" shift 2 ;; + --base-url) + base_url="$2" + shift 2 + ;; --input-len) input_len="$2" shift 2 @@ -954,12 +976,16 @@ run_benchmark_serving() { num_prompts="$max_concurrency" fi + if [[ -z "$base_url" ]]; then + base_url="http://0.0.0.0:$port" + fi + local benchmark_cmd=( env PYTHONPATH="$workspace_dir${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench_serving.benchmark_serving --model "$model" --backend "$backend" - --base-url "http://0.0.0.0:$port" + --base-url "$base_url" --dataset-name random --random-input-len "$input_len" --random-output-len "$output_len" diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_rtx6000pro_sglang.sh b/benchmarks/single_node/fixed_seq_len/deprecated/qwen3.5_fp4_rtx6000pro_sglang.sh similarity index 100% rename from benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_rtx6000pro_sglang.sh rename to benchmarks/single_node/fixed_seq_len/deprecated/qwen3.5_fp4_rtx6000pro_sglang.sh diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_rtx6000pro_sglang_mtp.sh b/benchmarks/single_node/fixed_seq_len/deprecated/qwen3.5_fp4_rtx6000pro_sglang_mtp.sh similarity index 100% rename from benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_rtx6000pro_sglang_mtp.sh rename to benchmarks/single_node/fixed_seq_len/deprecated/qwen3.5_fp4_rtx6000pro_sglang_mtp.sh diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200.sh deleted file mode 100644 index 429d6872e4..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200.sh +++ /dev/null @@ -1,102 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ "$DP_ATTENTION" != "true" && "$DP_ATTENTION" != "false" ]]; then - echo "DP_ATTENTION must be true or false; got '$DP_ATTENTION'" >&2 - exit 1 -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -CHUNKED_PREFILL_SIZE=16384 -SGLANG_PARALLEL_ARGS=( - --tensor-parallel-size="$TP" - --data-parallel-size=1 -) -SGLANG_DPA_ARGS=() - -if [[ "$DP_ATTENTION" == "true" ]]; then - SCHEDULER_RECV_INTERVAL=1 - CHUNKED_PREFILL_SIZE=32768 - SGLANG_PARALLEL_ARGS=( - --tensor-parallel-size="$TP" - --data-parallel-size="$TP" - --enable-dp-attention - --enable-dp-attention-local-control-broadcast - --enable-dp-lm-head - ) - SGLANG_DPA_ARGS=( - --schedule-conservativeness 3.33 - --enable-prefill-delayer - ) -fi - -echo "TP: $TP, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION, CONC: $CONC, ISL: $ISL, OSL: $OSL" -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CHUNKED_PREFILL_SIZE: $CHUNKED_PREFILL_SIZE" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -SGLANG_RADIX_FORCE_MISS=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL --host 0.0.0.0 --port $PORT --trust-remote-code \ -"${SGLANG_PARALLEL_ARGS[@]}" \ ---cuda-graph-max-bs 256 --max-running-requests 256 --mem-fraction-static 0.85 --kv-cache-dtype fp8_e4m3 \ ---chunked-prefill-size "$CHUNKED_PREFILL_SIZE" \ ---ep-size $EP_SIZE --quantization modelopt_fp4 --enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---enable-symm-mem --disable-piecewise-cuda-graph --attention-backend trtllm_mla --moe-runner-backend flashinfer_trtllm --stream-interval 10 "${SGLANG_DPA_ARGS[@]}" $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $((CONC * 10)) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_mtp.sh deleted file mode 100755 index 2a68ebbce2..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_mtp.sh +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ "$DP_ATTENTION" != "true" && "$DP_ATTENTION" != "false" ]]; then - echo "DP_ATTENTION must be true or false; got '$DP_ATTENTION'" >&2 - exit 1 -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -CHUNKED_PREFILL_SIZE=16384 -SGLANG_PARALLEL_ARGS=( - --tensor-parallel-size="$TP" - --data-parallel-size=1 -) -SGLANG_DPA_ARGS=() - -if [[ "$DP_ATTENTION" == "true" ]]; then - SCHEDULER_RECV_INTERVAL=1 - CHUNKED_PREFILL_SIZE=32768 - SGLANG_PARALLEL_ARGS=( - --tensor-parallel-size="$TP" - --data-parallel-size="$TP" - --enable-dp-attention - --enable-dp-attention-local-control-broadcast - --enable-dp-lm-head - ) - SGLANG_DPA_ARGS=( - --schedule-conservativeness 3.33 - --enable-prefill-delayer - ) -fi - -echo "TP: $TP, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION, CONC: $CONC, ISL: $ISL, OSL: $OSL" -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CHUNKED_PREFILL_SIZE: $CHUNKED_PREFILL_SIZE" - -SPECULATIVE_NUM_STEPS=2 -SPECULATIVE_DRAFT_TOKENS=3 -SPECULATIVE_EAGLE_TOPK=1 - -export SGLANG_ENABLE_SPEC_V2=1 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -SGLANG_RADIX_FORCE_MISS=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL --host 0.0.0.0 --port $PORT --trust-remote-code \ -"${SGLANG_PARALLEL_ARGS[@]}" \ ---cuda-graph-max-bs 256 --max-running-requests 256 --mem-fraction-static 0.85 --kv-cache-dtype fp8_e4m3 \ ---chunked-prefill-size "$CHUNKED_PREFILL_SIZE" \ ---ep-size $EP_SIZE --quantization modelopt_fp4 --enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---disable-piecewise-cuda-graph --attention-backend trtllm_mla --moe-runner-backend flashinfer_trtllm --stream-interval 10 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps $SPECULATIVE_NUM_STEPS \ ---speculative-num-draft-tokens $SPECULATIVE_DRAFT_TOKENS \ ---speculative-eagle-topk $SPECULATIVE_EAGLE_TOPK \ -"${SGLANG_DPA_ARGS[@]}" $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $((CONC * 10)) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt.sh deleted file mode 100644 index 346b21b56d..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt.sh +++ /dev/null @@ -1,127 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -CUDA_GRAPH_MAX_BATCH_SIZE=$CONC -MOE_BACKEND="TRTLLM" -PIECEWISE_CUDA_GRAPHS="false" - -if [[ "$ISL" == "1024" && "$OSL" == "1024" ]]; then - if [[ "$TP" == "8" && "$EP_SIZE" == "8" ]]; then - PIECEWISE_CUDA_GRAPHS="true" - fi -fi - -if [[ "$DP_ATTENTION" == "true" ]]; then - MOE_BACKEND="CUTLASS" - CUDA_GRAPH_MAX_BATCH_SIZE=$(( CONC < 4 ? CONC : CONC / 4 )) -fi - -echo "MOE_BACKEND set to '$MOE_BACKEND'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp4.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: $CUDA_GRAPH_MAX_BATCH_SIZE -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: 0.8 - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -start_gpu_monitor - -set -x - -MAX_NUM_TOKENS=$(( ($CONC+$ISL+64+63)/64*64 )) -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) -MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -if [[ "$PIECEWISE_CUDA_GRAPHS" == "true" ]]; then - capture_tokens=(1 2 4 8 16 32 64 128) - capture_tokens+=( $(seq 256 256 $MAX_NUM_TOKENS)) - CAPTURE_TOKENS_LIST=$(printf "%s, " "${capture_tokens[@]}") - - cat << EOF >> $EXTRA_CONFIG_FILE -torch_compile_config: - capture_num_tokens: [${CAPTURE_TOKENS_LIST%, }] - enable_piecewise_cuda_graph: true -EOF -fi - -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt_mtp.sh deleted file mode 100644 index 5a34591884..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt_mtp.sh +++ /dev/null @@ -1,136 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -MOE_BACKEND="TRTLLM" -PIECEWISE_CUDA_GRAPHS="false" -MAX_BATCH_SIZE=$CONC -MTP=3 - -if [[ "$DP_ATTENTION" == "true" ]]; then - MAX_BATCH_SIZE=$(( CONC < 4 ? CONC : CONC / 4 )) - MOE_BACKEND="CUTLASS" - MTP=1 -fi - -echo "MOE_BACKEND='$MOE_BACKEND', MTP='$MTP'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp4-mtp.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: $MAX_BATCH_SIZE -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: 0.8 - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -speculative_config: - decoding_type: MTP - num_nextn_predict_layers: ${MTP} -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -MAX_NUM_TOKENS=$(( ((MTP+1)*MAX_BATCH_SIZE+ISL+64+63)/64*64 )) - -if [[ "$ISL" == "1024" && "$OSL" == "1024" ]]; then - if [[ $CONC == 32 || $CONC == 64 ]]; then - PIECEWISE_CUDA_GRAPHS="true" - elif [[ $CONC == 128 && $DP_ATTENTION == "false" ]]; then - PIECEWISE_CUDA_GRAPHS="true" - fi -fi - -if [[ "$PIECEWISE_CUDA_GRAPHS" == "true" ]]; then - capture_tokens=(1 2 4 8 16 32 64 128) - capture_tokens+=( $(seq 256 256 $MAX_NUM_TOKENS)) - if [ $((MAX_NUM_TOKENS%256)) -ne 0 ]; then - capture_tokens+=($MAX_NUM_TOKENS) - fi - CAPTURE_TOKENS_LIST=$(printf "%s, " "${capture_tokens[@]}") - - cat << EOF >> $EXTRA_CONFIG_FILE -torch_compile_config: - capture_num_tokens: [${CAPTURE_TOKENS_LIST%, }] - enable_piecewise_cuda_graph: true -EOF -fi # end of set of configs using piecewise_cuda_graphs - -start_gpu_monitor - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -set -x -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size=$MAX_BATCH_SIZE \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b300.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b300.sh deleted file mode 100644 index 6e5b5a6f0b..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b300.sh +++ /dev/null @@ -1,81 +0,0 @@ -#!/usr/bin/env bash - -# https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 has no B300-specific recipe; this reuses the B200 SGLang tuning. - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT --trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 \ ---cuda-graph-max-bs 256 --max-running-requests 256 --mem-fraction-static 0.85 --kv-cache-dtype fp8_e4m3 \ ---chunked-prefill-size 16384 \ ---ep-size $EP_SIZE --quantization modelopt_fp4 --enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---enable-symm-mem --disable-radix-cache --attention-backend trtllm_mla --moe-runner-backend flashinfer_trtllm --stream-interval 10 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $((CONC * 10)) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x.sh deleted file mode 100644 index 0e77d3069f..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x.sh +++ /dev/null @@ -1,74 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER=1 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT4 - -PREFILL_SIZE=196608 -if [[ "$ISL" == "8192" && "$OSL" == "1024" ]]; then - if [[ "$CONC" -gt "32" ]]; then - PREFILL_SIZE=32768 - fi -fi - -SERVER_LOG=/workspace/server.log - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -python3 -m sglang.launch_server --model-path=$MODEL --trust-remote-code \ ---host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP \ ---chunked-prefill-size=$PREFILL_SIZE \ ---mem-fraction-static=0.8 \ ---disable-radix-cache \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=$PREFILL_SIZE \ ---cuda-graph-max-bs=128 \ ---attention-backend aiter \ ---kv-cache-dtype fp8_e4m3 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom.sh deleted file mode 100644 index 50427bd858..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom.sh +++ /dev/null @@ -1,77 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CALCULATED_MAX_MODEL_LEN=" --max-model-len $EVAL_MAX_MODEL_LEN " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor - -set -x - -BLOCK_SIZE=16 -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --block-size $BLOCK_SIZE > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom_mtp.sh deleted file mode 100644 index 83a95def55..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom_mtp.sh +++ /dev/null @@ -1,80 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CALCULATED_MAX_MODEL_LEN=" --max-model-len $EVAL_MAX_MODEL_LEN " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor - -set -x - -export AMDGCN_USE_BUFFER_OPS=1 - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --method mtp \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_mtp.sh deleted file mode 100755 index a72351c59e..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_mtp.sh +++ /dev/null @@ -1,84 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 -export SGLANG_ENABLE_SPEC_V2=1 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT4 - -PREFILL_SIZE=196608 -if [[ "$ISL" == "8192" && "$OSL" == "1024" ]]; then - if [[ "$CONC" -gt "32" ]]; then - PREFILL_SIZE=32768 - fi -fi - -SERVER_LOG=/workspace/server.log - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -python3 -m sglang.launch_server --model-path=$MODEL --trust-remote-code \ ---host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP \ ---ep-size $EP_SIZE \ ---chunked-prefill-size=$PREFILL_SIZE \ ---mem-fraction-static=0.8 \ ---disable-radix-cache \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=$PREFILL_SIZE \ ---cuda-graph-max-bs=128 \ ---attention-backend aiter \ ---kv-cache-dtype fp8_e4m3 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ -$EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200.sh deleted file mode 100644 index d9a2dc7f61..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200.sh +++ /dev/null @@ -1,101 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGL_ENABLE_JIT_DEEPGEMM=false -export SGLANG_ENABLE_FLASHINFER_GEMM=true -SERVER_LOG=/workspace/server.log - -if [[ $TP -eq 8 ]]; then - if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 - else - SCHEDULER_RECV_INTERVAL=10 - fi - - # Capped so KV memory is not reserved for requests that never run. - MAX_RUNNING_REQUESTS=128 - CUDA_GRAPH_MAX_BATCH_SIZE=128 - - MEM_FRAC_STATIC=0.82 - CHUNKED_PREFILL_SIZE=32768 - MAX_PREFILL_TOKENS=32768 -elif [[ $TP -eq 4 ]]; then - if [[ $ISL -ne 8192 ]] || [[ $OSL -ne 1024 ]]; then - echo "TP=4 not yet supported for ISL=$ISL OSL=$OSL!" - exit 1 - fi - - # Capped so KV memory is not reserved for requests that never run. - MAX_RUNNING_REQUESTS=32 - CUDA_GRAPH_MAX_BATCH_SIZE=32 - - MEM_FRAC_STATIC=0.95 - CHUNKED_PREFILL_SIZE=8192 - MAX_PREFILL_TOKENS=8192 - - SCHEDULER_RECV_INTERVAL=10 -else - echo "Unrecognized TP size $TP!" - exit 1 -fi -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP --data-parallel-size=1 \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE --max-running-requests $MAX_RUNNING_REQUESTS \ ---mem-fraction-static $MEM_FRAC_STATIC --kv-cache-dtype fp8_e4m3 --chunked-prefill-size $CHUNKED_PREFILL_SIZE --max-prefill-tokens $MAX_PREFILL_TOKENS \ ---enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL --disable-radix-cache \ ---attention-backend trtllm_mla --stream-interval 30 --ep-size $EP_SIZE --moe-runner-backend flashinfer_trtllm --quantization fp8 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_mtp.sh deleted file mode 100755 index 39a3e61659..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_mtp.sh +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_ENABLE_JIT_DEEPGEMM=false - -SERVER_LOG=/workspace/server.log - -if [[ $TP -ne 8 ]]; then - echo "MTP only supports TP=8, got TP=$TP!" - exit 1 -fi - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -# Capped so KV memory is not reserved for requests that never run. -MAX_RUNNING_REQUESTS=512 -CUDA_GRAPH_MAX_BATCH_SIZE=512 - -MEM_FRAC_STATIC=0.82 -CHUNKED_PREFILL_SIZE=16384 -MAX_PREFILL_TOKENS=16384 - -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -SPECULATIVE_NUM_STEPS=2 -SPECULATIVE_DRAFT_TOKENS=3 -SPECULATIVE_EAGLE_TOPK=1 - -SGLANG_ENABLE_SPEC_V2=1 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server \ - --model-path=$MODEL \ - --host=0.0.0.0 \ - --port=$PORT \ - --tensor-parallel-size=$TP \ - --data-parallel-size=1 \ - --cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE \ - --max-running-requests $MAX_RUNNING_REQUESTS \ - --mem-fraction-static $MEM_FRAC_STATIC \ - --kv-cache-dtype fp8_e4m3 \ - --chunked-prefill-size $CHUNKED_PREFILL_SIZE \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --enable-flashinfer-allreduce-fusion \ - --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ - --disable-radix-cache \ - --fp8-gemm-backend=flashinfer_trtllm \ - --attention-backend trtllm_mla \ - --stream-interval 30 \ - --ep-size $EP_SIZE \ - --moe-runner-backend flashinfer_trtllm \ - --quantization fp8 \ - --speculative-algorithm EAGLE \ - --speculative-num-steps $SPECULATIVE_NUM_STEPS \ - --speculative-num-draft-tokens $SPECULATIVE_DRAFT_TOKENS \ - --speculative-eagle-topk $SPECULATIVE_EAGLE_TOPK \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt.sh deleted file mode 100644 index d4b66e8cc3..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt.sh +++ /dev/null @@ -1,145 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Guards against OOM. -export TLLM_OVERRIDE_LAYER_NUM=61 - -CUDA_GRAPH_MAX_BATCH_SIZE=$CONC -MOE_BACKEND="TRTLLM" -PIECEWISE_CUDA_GRAPHS="false" -DELAY_BATCHING="false" -KV_CACHE_FREE_MEM_FRACTION=0.8 - -if [[ "$ISL" == "1024" && "$OSL" == "1024" ]]; then - if [[ $CONC -ge 64 ]]; then - PIECEWISE_CUDA_GRAPHS="true" - DELAY_BATCHING="true" - fi -elif [[ "$ISL" == "8192" && "$OSL" == "1024" ]]; then - if [[ $CONC -ge 64 ]]; then - PIECEWISE_CUDA_GRAPHS="true" - fi - if [[ "$TP" == "4" ]]; then - KV_CACHE_FREE_MEM_FRACTION=0.75 - fi -fi - -echo "MOE_BACKEND set to '$MOE_BACKEND'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp8.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: $CUDA_GRAPH_MAX_BATCH_SIZE -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: $KV_CACHE_FREE_MEM_FRACTION - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -if [[ "$DELAY_BATCHING" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -batch_wait_timeout_iters: 40 -batch_wait_max_tokens_ratio: 0.8 -EOF -fi - -start_gpu_monitor - -set -x - -MAX_NUM_TOKENS=$(( ($CONC+$ISL+64+63)/64*64 )) -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) -MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -if [[ "$PIECEWISE_CUDA_GRAPHS" == "true" ]]; then - capture_tokens=(1 2 4 8 16 32 64 128) - capture_tokens+=( $(seq 256 256 $MAX_NUM_TOKENS)) - if [ $((MAX_NUM_TOKENS%256)) -ne 0 ]; then - capture_tokens+=($MAX_NUM_TOKENS) - fi - CAPTURE_TOKENS_LIST=$(printf "%s, " "${capture_tokens[@]}") - - cat << EOF >> $EXTRA_CONFIG_FILE -torch_compile_config: - capture_num_tokens: [${CAPTURE_TOKENS_LIST%, }] - enable_piecewise_cuda_graph: true -EOF -fi - -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt_mtp.sh deleted file mode 100644 index b945a0a6a5..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt_mtp.sh +++ /dev/null @@ -1,145 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -MOE_BACKEND="TRTLLM" -PIECEWISE_CUDA_GRAPHS="true" -MAX_BATCH_SIZE=$CONC -KV_CACHE_FREE_MEM_FRACTION=0.8 -MTP=3 - -if [[ "$DP_ATTENTION" == "true" ]]; then - MOE_BACKEND="DEEPGEMM" - PIECEWISE_CUDA_GRAPHS="false" - MAX_BATCH_SIZE=$(( CONC < 8 ? CONC : CONC / 8 )) - KV_CACHE_FREE_MEM_FRACTION=0.7 - # Configurable MoE backend has better comms under attention DP. - export ENABLE_CONFIGURABLE_MOE=1 - MTP=1 -fi - -# Low-CONC cases do not benefit from piecewise CUDA graphs. -if [[ "$ISL" == "1024" && "$OSL" == "1024" ]]; then - if [[ $CONC -le 4 ]]; then - PIECEWISE_CUDA_GRAPHS="false" - fi -elif [[ "$ISL" == "8192" && "$OSL" == "1024" ]]; then - if [[ $CONC -le 16 ]]; then - PIECEWISE_CUDA_GRAPHS="false" - fi -fi - - -echo "MOE_BACKEND='$MOE_BACKEND', MTP='$MTP'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp8-mtp.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: $MAX_BATCH_SIZE -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: $KV_CACHE_FREE_MEM_FRACTION - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -speculative_config: - decoding_type: MTP - num_nextn_predict_layers: ${MTP} -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" -fi - -MAX_NUM_TOKENS=$(( ((MTP+1)*MAX_BATCH_SIZE+ISL+64+63)/64*64 )) -if [ "${EVAL_ONLY}" = "true" ]; then - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -if [[ "$PIECEWISE_CUDA_GRAPHS" == "true" ]]; then - capture_tokens=(1 2 4 8 16 32 64 128) - capture_tokens+=( $(seq 256 256 $MAX_NUM_TOKENS)) - if [ $((MAX_NUM_TOKENS%256)) -ne 0 ]; then - capture_tokens+=($MAX_NUM_TOKENS) - fi - CAPTURE_TOKENS_LIST=$(printf "%s, " "${capture_tokens[@]}") - - cat << EOF >> $EXTRA_CONFIG_FILE -torch_compile_config: - capture_num_tokens: [${CAPTURE_TOKENS_LIST%, }] - enable_piecewise_cuda_graph: true -EOF -fi -start_gpu_monitor - -set -x -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size=$MAX_BATCH_SIZE \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300.sh deleted file mode 100644 index b29aa07310..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300.sh +++ /dev/null @@ -1,111 +0,0 @@ -#!/usr/bin/env bash - -# https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 has no B300-specific recipe; this reuses the B200 SGLang tuning. - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - - -export SGL_ENABLE_JIT_DEEPGEMM=false -export SGLANG_ENABLE_FLASHINFER_GEMM=true -SERVER_LOG=/workspace/server.log - -if [[ $TP -eq 8 ]]; then - if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 - else - SCHEDULER_RECV_INTERVAL=10 - fi - - # Capped so KV memory is not reserved for requests that never run. - MAX_RUNNING_REQUESTS=128 - CUDA_GRAPH_MAX_BATCH_SIZE=128 - - MEM_FRAC_STATIC=0.82 - CHUNKED_PREFILL_SIZE=32768 - MAX_PREFILL_TOKENS=32768 -elif [[ $TP -eq 4 ]]; then - if [[ $ISL -ne 8192 ]] || [[ $OSL -ne 1024 ]]; then - echo "TP=4 not yet supported for ISL=$ISL OSL=$OSL!" - exit 1 - fi - - # Capped so KV memory is not reserved for requests that never run. - MAX_RUNNING_REQUESTS=32 - CUDA_GRAPH_MAX_BATCH_SIZE=32 - - MEM_FRAC_STATIC=0.95 - CHUNKED_PREFILL_SIZE=8192 - MAX_PREFILL_TOKENS=8192 - - SCHEDULER_RECV_INTERVAL=10 -else - echo "Unrecognized TP size $TP!" - exit 1 -fi -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---tensor-parallel-size $TP --data-parallel-size 1 \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE --max-running-requests $MAX_RUNNING_REQUESTS \ ---mem-fraction-static $MEM_FRAC_STATIC --kv-cache-dtype fp8_e4m3 --chunked-prefill-size $CHUNKED_PREFILL_SIZE --max-prefill-tokens $MAX_PREFILL_TOKENS \ ---enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL --disable-radix-cache \ ---attention-backend trtllm_mla --stream-interval 30 --ep-size $EP_SIZE --moe-runner-backend flashinfer_trtllm --quantization fp8 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300_mtp.sh deleted file mode 100755 index 4915d17224..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300_mtp.sh +++ /dev/null @@ -1,124 +0,0 @@ -#!/usr/bin/env bash - -# https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 has no B300-specific recipe; this reuses the B200 SGLang tuning. - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - - -export SGLANG_ENABLE_JIT_DEEPGEMM=false - -SERVER_LOG=/workspace/server.log - -if [[ $TP -ne 8 ]]; then - echo "MTP only supports TP=8, got TP=$TP!" - exit 1 -fi - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -# Capped so KV memory is not reserved for requests that never run. -MAX_RUNNING_REQUESTS=512 -CUDA_GRAPH_MAX_BATCH_SIZE=512 - -MEM_FRAC_STATIC=0.82 -CHUNKED_PREFILL_SIZE=16384 -MAX_PREFILL_TOKENS=16384 - -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -SPECULATIVE_NUM_STEPS=2 -SPECULATIVE_DRAFT_TOKENS=3 -SPECULATIVE_EAGLE_TOPK=1 - -SGLANG_ENABLE_SPEC_V2=1 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server \ - --model-path $MODEL_PATH --served-model-name $MODEL \ - --host 0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --data-parallel-size 1 \ - --cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE \ - --max-running-requests $MAX_RUNNING_REQUESTS \ - --mem-fraction-static $MEM_FRAC_STATIC \ - --kv-cache-dtype fp8_e4m3 \ - --chunked-prefill-size $CHUNKED_PREFILL_SIZE \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --enable-flashinfer-allreduce-fusion \ - --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ - --disable-radix-cache \ - --fp8-gemm-backend flashinfer_trtllm \ - --attention-backend trtllm_mla \ - --stream-interval 30 \ - --ep-size $EP_SIZE \ - --moe-runner-backend flashinfer_trtllm \ - --quantization fp8 \ - --speculative-algorithm EAGLE \ - --speculative-num-steps $SPECULATIVE_NUM_STEPS \ - --speculative-num-draft-tokens $SPECULATIVE_DRAFT_TOKENS \ - --speculative-eagle-topk $SPECULATIVE_EAGLE_TOPK \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh deleted file mode 100644 index 0bec9b79e6..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -pip3 install --user --break-system-packages sentencepiece - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi -SERVER_LOG=/workspace/server.log - -start_gpu_monitor - -export TORCH_CUDA_ARCH_LIST="9.0" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi - -set -x -if [[ $ISL -eq 1024 && $OSL -eq 1024 ]]; then - PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL \ - --host 0.0.0.0 --port $PORT --trust-remote-code \ - --tensor-parallel-size=$TP --data-parallel-size=1 \ - --disable-radix-cache --max-running-requests 512 --cuda-graph-max-bs 512 \ - --chunked-prefill-size 32768 --max-prefill-tokens 32768 --mem-fraction-static 0.82 \ - --attention-backend flashinfer --stream-interval 10 \ - --decode-log-interval 1 \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & -else - PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL \ - --host 0.0.0.0 --port $PORT --trust-remote-code \ - --tensor-parallel-size=$TP --data-parallel-size=1 \ - --disable-radix-cache --max-running-requests 256 --cuda-graph-max-bs 256 \ - --chunked-prefill-size 32768 --max-prefill-tokens 32768 --mem-fraction-static 0.82 \ - --attention-backend flashinfer --stream-interval 10 \ - --decode-log-interval 1 \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & -fi - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_mtp.sh deleted file mode 100755 index f073983d5e..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_mtp.sh +++ /dev/null @@ -1,95 +0,0 @@ -#!/usr/bin/env bash - -# No trtllm_mla attention path on H200 in this image, so attention stays on flashinfer. - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -pip3 install --user --break-system-packages sentencepiece - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -if [[ $TP -ne 8 ]]; then - echo "MTP only supports TP=8, got TP=$TP!" - exit 1 -fi - -SERVER_LOG=/workspace/server.log - -SPECULATIVE_NUM_STEPS=2 -SPECULATIVE_DRAFT_TOKENS=3 -SPECULATIVE_EAGLE_TOPK=1 - -export SGLANG_ENABLE_SPEC_V2=1 -export TORCH_CUDA_ARCH_LIST="9.0" - -start_gpu_monitor - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi - -set -x -if [[ $ISL -eq 1024 && $OSL -eq 1024 ]]; then - MAX_RUNNING_REQUESTS=512 - CUDA_GRAPH_MAX_BS=512 -else - MAX_RUNNING_REQUESTS=256 - CUDA_GRAPH_MAX_BS=256 -fi - -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL \ ---host 0.0.0.0 --port $PORT --trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 \ ---ep-size $EP_SIZE \ ---disable-radix-cache \ ---max-running-requests $MAX_RUNNING_REQUESTS \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BS \ ---chunked-prefill-size 32768 --max-prefill-tokens 32768 --mem-fraction-static 0.82 \ ---attention-backend flashinfer --stream-interval 10 \ ---decode-log-interval 1 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps $SPECULATIVE_NUM_STEPS \ ---speculative-num-draft-tokens $SPECULATIVE_DRAFT_TOKENS \ ---speculative-eagle-topk $SPECULATIVE_EAGLE_TOPK \ -$EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt.sh deleted file mode 100644 index 4e1095bb79..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt.sh +++ /dev/null @@ -1,102 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -MOE_BACKEND="CUTLASS" - -echo "MOE_BACKEND set to '$MOE_BACKEND'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp8.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: 128 -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: 0.75 - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -start_gpu_monitor - -set -x - -MAX_NUM_TOKENS=$(( (CONC + ISL + 64 + 63) / 64 * 64 )) -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) -MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -PYTHONNOUSERSITE=1 mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt_mtp.sh deleted file mode 100644 index 339468da0f..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt_mtp.sh +++ /dev/null @@ -1,121 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -MOE_BACKEND="CUTLASS" - -if [[ "$DP_ATTENTION" == "true" ]]; then - MTP=1 -else - MTP=3 -fi - -echo "MOE_BACKEND='$MOE_BACKEND', MTP='$MTP'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp8-mtp.yml" - -if [[ "$ISL" == "8192" && "$DP_ATTENTION" == "true" ]]; then - export PYTORCH_CUDA_ALLOC_CONF="max_split_size_mb:8192" -fi - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: 128 -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: 0.75 - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -speculative_config: - decoding_type: MTP - num_nextn_predict_layers: ${MTP} -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -if [[ "$DP_ATTENTION" == "true" ]]; then - MAX_BATCH_SIZE=$((CONC/TP)) - if [[ $MAX_BATCH_SIZE -lt 1 ]]; then - MAX_BATCH_SIZE=1 - fi -else - MAX_BATCH_SIZE=$CONC -fi - -MAX_NUM_TOKENS=$(( ((MTP+1)*MAX_BATCH_SIZE+ISL+64+63)/64*64 )) - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size=$MAX_BATCH_SIZE \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi300x.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi300x.sh deleted file mode 100644 index 4b2950ade1..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi300x.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Reference: https://rocm.docs.amd.com/en/docs-7.0-rc1/preview/benchmark-docker/inference-sglang-deepseek-r1-fp8.html#run-the-inference-benchmark - -# On MEC firmware older than 177 RCCL cannot reclaim scratch memory and crashes; disable it. -# See https://rocm.docs.amd.com/en/docs-6.4.3/about/release-notes.html#amdgpu-driver-updates -version=`rocm-smi --showfw | grep MEC | head -n 1 | awk '{print $NF}'` -if [[ "$version" == "" || $version -lt 177 ]]; then - export HSA_NO_SCRATCH_RECLAIM=1 -fi - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 - -SERVER_LOG=/workspace/server.log - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -python3 -m sglang.launch_server \ ---model-path=$MODEL --host=0.0.0.0 --port=$PORT --trust-remote-code \ ---tensor-parallel-size=$TP \ ---mem-fraction-static=0.8 \ ---cuda-graph-max-bs=128 \ ---chunked-prefill-size=131072 \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=131072 \ ---kv-cache-dtype fp8_e4m3 \ ---attention-backend aiter \ ---disable-radix-cache $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh deleted file mode 100644 index 0f6a801136..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -SERVER_LOG=/workspace/server.log -PORT=8888 -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Reference: https://rocm.docs.amd.com/en/docs-7.0-rc1/preview/benchmark-docker/inference-sglang-deepseek-r1-fp8.html#run-the-inference-benchmark - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 - -start_gpu_monitor - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi - -set -x -python3 -m sglang.launch_server \ ---model-path=$MODEL --host=0.0.0.0 --port=$PORT --trust-remote-code \ ---tensor-parallel-size=$TP \ ---mem-fraction-static=0.8 \ ---cuda-graph-max-bs=128 \ ---chunked-prefill-size=131072 \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=131072 \ ---kv-cache-dtype fp8_e4m3 \ ---attention-backend aiter \ ---disable-radix-cache \ -$EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x_mtp.sh deleted file mode 100755 index df72ca869e..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x_mtp.sh +++ /dev/null @@ -1,77 +0,0 @@ -#!/usr/bin/bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -SERVER_LOG=/workspace/server.log -PORT=8888 -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 -export SGLANG_ENABLE_SPEC_V2=1 - -start_gpu_monitor - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi - -set -x -python3 -m sglang.launch_server \ ---model-path=$MODEL --host=0.0.0.0 --port=$PORT --trust-remote-code \ ---tensor-parallel-size=$TP \ ---ep-size $EP_SIZE \ ---mem-fraction-static=0.8 \ ---cuda-graph-max-bs=128 \ ---chunked-prefill-size=131072 \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=131072 \ ---kv-cache-dtype fp8_e4m3 \ ---attention-backend aiter \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---disable-radix-cache \ -$EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x.sh deleted file mode 100644 index 979ab0d0dd..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Reference: https://rocm.docs.amd.com/en/docs-7.0-docker/benchmark-docker/inference-sglang-deepseek-r1-fp8.html - -export SGLANG_USE_AITER=1 -export RCCL_MSCCL_ENABLE=0 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT4 - -SERVER_LOG=/workspace/server.log - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --trust-remote-code \ - --chunked-prefill-size 196608 \ - --mem-fraction-static 0.8 --disable-radix-cache \ - --num-continuous-decode-steps 8 \ - --max-prefill-tokens 196608 \ - --kv-cache-dtype fp8_e4m3 \ - --cuda-graph-max-bs "$CONC" $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom.sh deleted file mode 100644 index 50427bd858..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom.sh +++ /dev/null @@ -1,77 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CALCULATED_MAX_MODEL_LEN=" --max-model-len $EVAL_MAX_MODEL_LEN " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor - -set -x - -BLOCK_SIZE=16 -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --block-size $BLOCK_SIZE > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom_mtp.sh deleted file mode 100644 index f58b7bd534..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom_mtp.sh +++ /dev/null @@ -1,80 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -CALCULATED_MAX_MODEL_LEN="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CALCULATED_MAX_MODEL_LEN=" --max-model-len $EVAL_MAX_MODEL_LEN " -fi - -PARALLEL_ARGS=(-tp "$TP") #TP -if [ "$DP_ATTENTION" = "true" ]; then - if [ "$EP_SIZE" -gt 1 ]; then #DP+EP - PARALLEL_ARGS=(-tp "$TP" --enable-expert-parallel --enable-dp-attention ) - else #DP+TP - PARALLEL_ARGS=(-tp "$TP" --enable-dp-attention ) - fi -fi - -SPEC_ARGS=(--method mtp --num-speculative-tokens 3 ) - -start_gpu_monitor - -set -x - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - "${PARALLEL_ARGS[@]}" \ - "${SPEC_ARGS[@]}" \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN \ - --no-enable_prefix_caching \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -export PYTHONDONTWRITEBYTECODE=1 -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_mtp.sh deleted file mode 100755 index feee07896b..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_mtp.sh +++ /dev/null @@ -1,86 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Reference: https://rocm.docs.amd.com/en/docs-7.0-docker/benchmark-docker/inference-sglang-deepseek-r1-fp8.html - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 -export SGLANG_ENABLE_SPEC_V2=1 -export RCCL_MSCCL_ENABLE=0 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT4 - -SERVER_LOG=/workspace/server.log - -# Keep server-side speculative decoding capacity aligned with the matrix row. -MAX_RUNNING_REQUESTS="$CONC" -CUDA_GRAPH_MAX_BATCH_SIZE="$CONC" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --ep-size $EP_SIZE \ - --trust-remote-code \ - --chunked-prefill-size 196608 \ - --mem-fraction-static 0.8 --disable-radix-cache \ - --num-continuous-decode-steps 8 \ - --max-prefill-tokens 196608 \ - --kv-cache-dtype fp8_e4m3 \ - --cuda-graph-max-bs "$CUDA_GRAPH_MAX_BATCH_SIZE" \ - --max-running-requests "$MAX_RUNNING_REQUESTS" \ - --speculative-algorithm EAGLE \ - --speculative-num-steps 3 \ - --speculative-eagle-topk 1 \ - --speculative-num-draft-tokens 4 \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200.sh deleted file mode 100755 index 904e1eca8e..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200.sh +++ /dev/null @@ -1,78 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 --expert-parallel-size=$EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization modelopt_fp4 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---attention-backend trtllm_mha \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs $CONC \ ---max-prefill-tokens 16384 \ ---chunked-prefill-size 16384 \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval $( [[ $CONC -gt 4 ]] && echo 30 || echo 10 ) \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_mtp.sh deleted file mode 100755 index 53a33b5473..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_mtp.sh +++ /dev/null @@ -1,98 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -LINEAR_ATTN_ARGS=() -if [[ "$TP" == "2" && "$EP_SIZE" == "2" ]]; then - case "$CONC" in - 16|32|64) - LINEAR_ATTN_ARGS=( - --linear-attn-backend triton - --linear-attn-decode-backend flashinfer - --linear-attn-prefill-backend flashinfer - ) - ;; - esac -fi - -set -x -SGLANG_ENABLE_SPEC_V2=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 --expert-parallel-size=$EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization modelopt_fp4 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ -"${LINEAR_ATTN_ARGS[@]}" \ ---attention-backend trtllm_mha \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs $CONC \ ---max-running-requests $CONC \ ---max-prefill-tokens 16384 \ ---chunked-prefill-size 16384 \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval $( [[ $CONC -gt 4 ]] && echo 30 || echo 10 ) \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt.sh deleted file mode 100644 index 7fc3599434..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt.sh +++ /dev/null @@ -1,142 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="qwen3.5-fp4-trt.yml" - -if [[ "$DP_ATTENTION" == "true" ]]; then - case "$TP" in - 4) MAX_BATCH_SIZE=256 ;; # tp4 / ep4 with attention DP - 8) MAX_BATCH_SIZE=128 ;; # tp8 / ep8 with attention DP - *) MAX_BATCH_SIZE=$(( CONC > 16 ? CONC : 16 )) ;; - esac -elif [[ "$TP" == "2" ]]; then - if [[ "$EP_SIZE" == "2" && "$CONC" -ge 32 ]]; then - MAX_BATCH_SIZE=32 # tp2 / ep2 at high concurrency - else - MAX_BATCH_SIZE=256 # tp2 / ep1, or tp2 / ep2 at low concurrency - fi -elif [[ "$TP" -ge 4 ]]; then - MAX_BATCH_SIZE=512 # tp>=4 without attention DP -else - MAX_BATCH_SIZE=$(( CONC > 16 ? CONC : 16 )) -fi - -if [[ "$DP_ATTENTION" == "true" ]]; then - MOE_BACKEND="CUTEDSL" - MODE_CONFIG="attention_dp_config: - enable_balance: true - batching_wait_iters: 10 - timeout_iters: 500" -else - MOE_BACKEND="TRTLLM" - MODE_CONFIG="batch_wait_timeout_iters: 50 -batch_wait_max_tokens_ratio: 0.45" -fi - -cat > "$EXTRA_CONFIG_FILE" << EOF -backend: pytorch -print_iter_log: true -enable_layerwise_nvtx_marker: false -disable_overlap_scheduler: false -enable_iter_perf_stats: true -enable_chunked_prefill: false -stream_interval: 20 -num_postprocess_workers: 4 -enable_attention_dp: $DP_ATTENTION -scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - context_chunking_policy: FIRST_COME_FIRST_SERVED -kv_cache_config: - free_gpu_memory_fraction: 0.9 - enable_block_reuse: false - dtype: fp8 -cuda_graph_config: - enable_padding: true - max_batch_size: $MAX_BATCH_SIZE -moe_config: - backend: $MOE_BACKEND - use_low_precision_moe_combine: true -$MODE_CONFIG -EOF - -echo "Generated config file contents:" -cat "$EXTRA_CONFIG_FILE" - -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) - -case "${ISL}_${OSL}" in - 8192_1024) MAX_NUM_TOKENS=32768 ;; - 1024_1024) MAX_NUM_TOKENS=16384 ;; - *) - MAX_NUM_TOKENS=$(( ISL + OSL + 256 )) - MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - ;; -esac - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve "$MODEL" --port="$PORT" \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size "$MAX_BATCH_SIZE" \ - --max_seq_len="$MAX_MODEL_LEN" \ - --max_num_tokens="$MAX_NUM_TOKENS" \ - --tp_size="$TP" --ep_size="$EP_SIZE" \ - --extra_llm_api_options="$EXTRA_CONFIG_FILE" \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$(( CONC * 10 ))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt_mtp.sh deleted file mode 100644 index 2300dd1b35..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt_mtp.sh +++ /dev/null @@ -1,159 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -# MTP speculative decode requires the FlashInfer GDN prefill path to be disabled. -export TLLM_USE_FLASHINFER_GDN_PREFILL="0" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="qwen3.5-fp4-trt-mtp.yml" -NUM_NEXTN_PREDICT_LAYERS=3 - -# KV-cache memory fractions below are tuned empirically per layout. -if [[ "$DP_ATTENTION" == "true" ]]; then - MAX_BATCH_SIZE=$(( CONC / 8 )) - MOE_BACKEND="CUTEDSL" - if (( CONC >= 1024 )); then KV_MEMORY_FRACTION=0.8; else KV_MEMORY_FRACTION=0.9; fi - MODE_CONFIG="enable_attention_dp: true -attention_dp_config: - enable_balance: true - batching_wait_iters: 10 - timeout_iters: 500" -else - MAX_BATCH_SIZE="$CONC" - MOE_BACKEND="TRTLLM" - case "${ISL}_tp${TP}_ep${EP_SIZE}" in - 1024_tp2_ep1) KV_MEMORY_FRACTION=0.6 ;; - 1024_tp2_ep2) KV_MEMORY_FRACTION=0.75 ;; - 1024_tp8_ep8) KV_MEMORY_FRACTION=0.8 ;; - 8192_tp2_ep1) KV_MEMORY_FRACTION=0.7 ;; - 8192_tp2_ep2) KV_MEMORY_FRACTION=0.6 ;; - 8192_tp4_ep4) KV_MEMORY_FRACTION=0.75 ;; - 8192_tp8_ep8) KV_MEMORY_FRACTION=0.8 ;; - *) KV_MEMORY_FRACTION=0.8 ;; - esac - # Short-context runs hold less in flight, so they use a tighter token ratio before flushing a batch. - case "$ISL" in - 1024) BATCH_WAIT_MAX_TOKENS_RATIO=0.0625 ;; - *) BATCH_WAIT_MAX_TOKENS_RATIO=0.45 ;; - esac - MODE_CONFIG="batch_wait_timeout_iters: 50 -batch_wait_max_tokens_ratio: $BATCH_WAIT_MAX_TOKENS_RATIO" -fi - -cat > "$EXTRA_CONFIG_FILE" << EOF -backend: pytorch -print_iter_log: true -enable_layerwise_nvtx_marker: false -disable_overlap_scheduler: false -enable_iter_perf_stats: true -enable_chunked_prefill: false -stream_interval: 20 -num_postprocess_workers: 4 -scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - context_chunking_policy: FIRST_COME_FIRST_SERVED -kv_cache_config: - free_gpu_memory_fraction: $KV_MEMORY_FRACTION - enable_block_reuse: false - dtype: fp8 -cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 -moe_config: - backend: $MOE_BACKEND - use_low_precision_moe_combine: true -speculative_config: - decoding_type: MTP - num_nextn_predict_layers: $NUM_NEXTN_PREDICT_LAYERS -$MODE_CONFIG -EOF - -echo "Generated config file contents:" -cat "$EXTRA_CONFIG_FILE" - -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) - -case "${ISL}_${OSL}" in - 8192_1024) MAX_NUM_TOKENS=32768 ;; - 1024_1024) MAX_NUM_TOKENS=16384 ;; - *) - MAX_NUM_TOKENS=$(( ISL + OSL + 256 )) - MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - ;; -esac - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve "$MODEL" --port="$PORT" \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size "$MAX_BATCH_SIZE" \ - --max_seq_len="$MAX_MODEL_LEN" \ - --max_num_tokens="$MAX_NUM_TOKENS" \ - --tp_size="$TP" --ep_size="$EP_SIZE" \ - --extra_llm_api_options="$EXTRA_CONFIG_FILE" \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$(( CONC * 10 ))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300.sh deleted file mode 100755 index 0f04f7f6e2..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300.sh +++ /dev/null @@ -1,108 +0,0 @@ -#!/usr/bin/env bash - -# Follows https://cookbook.sglang.io/autoregressive/Qwen/Qwen3.5 - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - - -export NCCL_NVLS_ENABLE=1 -export SGL_ENABLE_JIT_DEEPGEMM=false -export SGLANG_ENABLE_FLASHINFER_GEMM=true -export PYTHONUNBUFFERED=1 - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -MEM_FRAC_STATIC=0.8 -CHUNKED_PREFILL_SIZE=32768 -MAX_PREFILL_TOKENS=32768 -CUDA_GRAPH_MAX_BATCH_SIZE=$CONC -MAX_RUNNING_REQUESTS=128 -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -if [[ $TP -eq 8 ]]; then - EXTRA_ARGS="--enable-flashinfer-allreduce-fusion" -else - EXTRA_ARGS="" -fi - -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 --ep-size $EP_SIZE \ ---reasoning-parser qwen3 \ ---tool-call-parser qwen3_coder \ ---mamba-scheduler-strategy no_buffer \ ---quantization modelopt_fp4 --fp4-gemm-backend flashinfer_cutlass \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE --max-running-requests $MAX_RUNNING_REQUESTS \ ---mem-fraction-static $MEM_FRAC_STATIC --chunked-prefill-size $CHUNKED_PREFILL_SIZE --max-prefill-tokens $MAX_PREFILL_TOKENS \ ---context-length $CONTEXT_LENGTH --disable-radix-cache \ ---attention-backend trtllm_mha --mm-attention-backend triton_attn --moe-runner-backend flashinfer_trtllm \ -$EXTRA_ARGS --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---tokenizer-worker-num 6 --stream-interval 30 > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300_mtp.sh deleted file mode 100755 index 0c0befe0dc..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300_mtp.sh +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env bash - -# Follows https://cookbook.sglang.io/autoregressive/Qwen/Qwen3.5 - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - - -export NCCL_NVLS_ENABLE=1 -export SGL_ENABLE_JIT_DEEPGEMM=false -export SGLANG_ENABLE_FLASHINFER_GEMM=true -export PYTHONUNBUFFERED=1 - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -MEM_FRAC_STATIC=0.8 -CHUNKED_PREFILL_SIZE=32768 -MAX_PREFILL_TOKENS=32768 -CUDA_GRAPH_MAX_BATCH_SIZE=$CONC -MAX_RUNNING_REQUESTS=128 -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -if [[ $TP -eq 8 ]]; then - EXTRA_ARGS="--enable-flashinfer-allreduce-fusion" -else - EXTRA_ARGS="" -fi - -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 --ep-size $EP_SIZE \ ---reasoning-parser qwen3 \ ---tool-call-parser qwen3_coder \ ---mamba-scheduler-strategy no_buffer \ ---quantization modelopt_fp4 --fp4-gemm-backend flashinfer_cutlass \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE --max-running-requests $MAX_RUNNING_REQUESTS \ ---mem-fraction-static $MEM_FRAC_STATIC --chunked-prefill-size $CHUNKED_PREFILL_SIZE --max-prefill-tokens $MAX_PREFILL_TOKENS \ ---context-length $CONTEXT_LENGTH --disable-radix-cache \ ---attention-backend trtllm_mha --mm-attention-backend triton_attn --moe-runner-backend flashinfer_trtllm \ -$EXTRA_ARGS --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---tokenizer-worker-num 6 --stream-interval 30 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ -> $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x.sh deleted file mode 100644 index 9351dd3d7a..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER=1 -export SGLANG_USE_AITER_UNIFIED_ATTN=1 -export AITER_FLYDSL_FORCE=1 -export SGLANG_MAMBA_SSM_DTYPE=bfloat16 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT8 - -SERVER_LOG=/workspace/server.log -MEM_FRAC_STATIC=0.8 - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context -fi - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server --model-path=$MODEL --trust-remote-code \ ---host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP \ ---attention-backend aiter \ ---mem-fraction-static $MEM_FRAC_STATIC \ ---model-loader-extra-config '{"enable_multithread_load": true}' \ ---watchdog-timeout 1200 \ ---disable-radix-cache \ ---max-running-requests $CONC \ ---page-size 16 \ ---kv-cache-dtype fp8_e4m3 \ -> $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" --sleep-interval 60 - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_atom.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_atom.sh deleted file mode 100644 index 98f2416dcc..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_atom.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor -MEM_FRAC_STATIC=0.9 - -set -x - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --gpu-memory-utilization $MEM_FRAC_STATIC \ - --trust-remote-code \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -export PYTHONDONTWRITEBYTECODE=1 -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --trust-remote-code - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_mtp.sh deleted file mode 100755 index 649eb121ab..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_mtp.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -hf download "$MODEL" - -export SGLANG_USE_AITER=1 -export SGLANG_USE_AITER_UNIFIED_ATTN=1 -export AITER_FLYDSL_FORCE=1 -export SGLANG_MAMBA_SSM_DTYPE=bfloat16 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT8 - -SERVER_LOG=/workspace/server.log -MEM_FRAC_STATIC=0.8 - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context -fi - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server --model-path=$MODEL --trust-remote-code \ ---host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP \ ---attention-backend aiter \ ---mem-fraction-static $MEM_FRAC_STATIC \ ---model-loader-extra-config '{"enable_multithread_load": true}' \ ---watchdog-timeout 1200 \ ---disable-radix-cache \ ---max-running-requests $CONC \ ---page-size 16 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---kv-cache-dtype fp8_e4m3 \ -> $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" --sleep-interval 60 - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200.sh deleted file mode 100755 index 8e8771369b..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200.sh +++ /dev/null @@ -1,79 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 --expert-parallel-size=$EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization fp8 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---mamba-full-memory-ratio 0.37 \ ---linear-attn-prefill-backend flashinfer \ ---attention-backend trtllm_mha \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs-decode $CONC \ ---max-prefill-tokens 32768 \ ---chunked-prefill-size 32768 \ ---mem-fraction-static 0.86 \ ---stream-interval 50 \ ---scheduler-recv-interval $( [[ $CONC -gt 4 ]] && echo 30 || echo 10 ) \ ---tokenizer-worker-num 6 \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200_mtp.sh deleted file mode 100755 index 5f30b28545..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200_mtp.sh +++ /dev/null @@ -1,86 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -SGLANG_ENABLE_SPEC_V2=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 --expert-parallel-size=$EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization fp8 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---attention-backend trtllm_mha \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs-decode $CONC \ ---max-running-requests $CONC \ ---max-prefill-tokens 32768 \ ---chunked-prefill-size 32768 \ ---mamba-full-memory-ratio 0.37 \ ---linear-attn-prefill-backend flashinfer \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval $( [[ $CONC -gt 4 ]] && echo 30 || echo 10 ) \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300.sh deleted file mode 100644 index ce0ef07dcd..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300.sh +++ /dev/null @@ -1,88 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 --expert-parallel-size $EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization fp8 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---attention-backend trtllm_mha \ ---mm-attention-backend triton_attn \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs $CONC \ ---max-running-requests $CONC \ ---max-prefill-tokens 16384 \ ---chunked-prefill-size 16384 \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval 10 \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL_PATH \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300_mtp.sh deleted file mode 100644 index ffe9ea2c0c..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300_mtp.sh +++ /dev/null @@ -1,92 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -SGLANG_ENABLE_SPEC_V2=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 --expert-parallel-size $EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization fp8 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---attention-backend trtllm_mha \ ---mm-attention-backend triton_attn \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs $CONC \ ---max-running-requests $CONC \ ---max-prefill-tokens 16384 \ ---chunked-prefill-size 16384 \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval 10 \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL_PATH \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100.sh deleted file mode 100755 index c94fbd5793..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100.sh +++ /dev/null @@ -1,123 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -MAX_SEQ_LEN=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_SEQ_LEN="$EVAL_MAX_MODEL_LEN" -fi - -PARALLEL_ARGS=(--tp "$TP") -if [ "${EP_SIZE}" -gt 1 ]; then - PARALLEL_ARGS+=(--expert-parallel-size "$EP_SIZE") -fi - -SCHEDULER_RECV_INTERVAL= -case "$CONC" in - 1|2|4) - SCHEDULER_RECV_INTERVAL=2 - ;; - 8) - SCHEDULER_RECV_INTERVAL=60 - ;; - 16) - SCHEDULER_RECV_INTERVAL=30 - ;; - 32) - SCHEDULER_RECV_INTERVAL=1200 - ;; - 64) - SCHEDULER_RECV_INTERVAL=600 - ;; - 128|256) - SCHEDULER_RECV_INTERVAL=1920 - ;; - *) - echo "Unsupported CONC=$CONC for qwen3.5 FP8 H100 SGLang recipe" >&2 - exit 1 - ;; -esac - -SCHEDULER_ARGS=() -if [ -n "$SCHEDULER_RECV_INTERVAL" ]; then - SCHEDULER_ARGS=(--scheduler-recv-interval "$SCHEDULER_RECV_INTERVAL") -fi - -echo "TP: $TP, EP_SIZE: $EP_SIZE, CONC: $CONC, ISL: $ISL, OSL: $OSL, MAX_SEQ_LEN: $MAX_SEQ_LEN" -echo "SCHEDULER_RECV_INTERVAL: ${SCHEDULER_RECV_INTERVAL:-none}" -echo "SCHEDULER_ARGS: ${SCHEDULER_ARGS[*]}" - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server \ - --model "$MODEL" \ - --host 0.0.0.0 \ - --port "$PORT" \ - "${PARALLEL_ARGS[@]}" \ - --reasoning-parser qwen3 \ - --tool-call-parser qwen3_coder \ - --enable-flashinfer-allreduce-fusion \ - --max-running-requests 256 \ - --chunked-prefill-size 16384 \ - --decode-log-interval 1 \ - --mem-fraction-static 0.8 \ - --cuda-graph-max-bs "$CONC" \ - --context-length "$MAX_SEQ_LEN" \ - --kv-cache-dtype fp8_e4m3 \ - --quantization fp8 \ - --attention-backend flashinfer \ - --stream-interval 50 \ - --tokenizer-worker-num 6 \ - --mamba-ssm-dtype bfloat16 \ - --disable-radix-cache \ - --enable-symm-mem \ - --trust-remote-code \ - "${SCHEDULER_ARGS[@]}" \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100_mtp.sh deleted file mode 100755 index 5860d601cb..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100_mtp.sh +++ /dev/null @@ -1,91 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_ENABLE_SPEC_V2=1 - -SERVER_LOG=/workspace/server.log -MAX_SEQ_LEN=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_SEQ_LEN="$EVAL_MAX_MODEL_LEN" -fi - -echo "CONC: $CONC, ISL: $ISL, OSL: $OSL, MAX_SEQ_LEN: $MAX_SEQ_LEN" - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server \ - --model "$MODEL" \ - --host 0.0.0.0 \ - --port "$PORT" \ - --tp "$TP" \ - --expert-parallel-size "$EP_SIZE" \ - --reasoning-parser qwen3 \ - --tool-call-parser qwen3_coder \ - --enable-flashinfer-allreduce-fusion \ - --max-running-requests 64 \ - --chunked-prefill-size 8192 \ - --decode-log-interval 1 \ - --mem-fraction-static 0.75 \ - --cuda-graph-max-bs "$CONC" \ - --context-length "$MAX_SEQ_LEN" \ - --kv-cache-dtype fp8_e4m3 \ - --quantization fp8 \ - --attention-backend flashinfer \ - --stream-interval 50 \ - --tokenizer-worker-num 6 \ - --mamba-ssm-dtype bfloat16 \ - --disable-radix-cache \ - --trust-remote-code \ - --speculative-algorithm EAGLE \ - --speculative-num-steps 3 \ - --speculative-eagle-topk 1 \ - --speculative-num-draft-tokens 4 \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200.sh deleted file mode 100644 index 0e1a4b581d..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200.sh +++ /dev/null @@ -1,84 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -MAX_SEQ_LEN=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_SEQ_LEN="$EVAL_MAX_MODEL_LEN" -fi - -echo "CONC: $CONC, ISL: $ISL, OSL: $OSL, MAX_SEQ_LEN: $MAX_SEQ_LEN" - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server \ - --model "$MODEL" \ - --host 0.0.0.0 \ - --port "$PORT" \ - --tp "$TP" \ - --expert-parallel-size "$EP_SIZE" \ - --reasoning-parser qwen3 \ - --tool-call-parser qwen3_coder \ - --enable-flashinfer-allreduce-fusion \ - --max-running-requests 128 \ - --chunked-prefill-size 16384 \ - --decode-log-interval 1 \ - --mem-fraction-static 0.8 \ - --cuda-graph-max-bs "$CONC" \ - --context-length "$MAX_SEQ_LEN" \ - --kv-cache-dtype fp8_e4m3 \ - --quantization fp8 \ - --attention-backend flashinfer \ - --stream-interval 50 \ - --tokenizer-worker-num 6 \ - --mamba-ssm-dtype bfloat16 \ - --disable-radix-cache \ - --trust-remote-code \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200_mtp.sh deleted file mode 100644 index 46e04ebe84..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200_mtp.sh +++ /dev/null @@ -1,89 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - MAX_MODEL_LEN - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -SPECULATIVE_NUM_STEPS=3 -SPECULATIVE_DRAFT_TOKENS=4 -SPECULATIVE_EAGLE_TOPK=1 - -echo "CONC: $CONC, ISL: $ISL, OSL: $OSL, MAX_MODEL_LEN: $MAX_MODEL_LEN" - -start_gpu_monitor - -set -x -SGLANG_ENABLE_SPEC_V2=1 python3 -m sglang.launch_server \ - --model "$MODEL" \ - --host 0.0.0.0 \ - --port "$PORT" \ - --tp "$TP" \ - --expert-parallel-size "$EP_SIZE" \ - --reasoning-parser qwen3 \ - --tool-call-parser qwen3_coder \ - --enable-flashinfer-allreduce-fusion \ - --max-running-requests 128 \ - --chunked-prefill-size 16384 \ - --mem-fraction-static 0.8 \ - --cuda-graph-max-bs "$CONC" \ - --context-length "$MAX_MODEL_LEN" \ - --kv-cache-dtype fp8_e4m3 \ - --quantization fp8 \ - --attention-backend flashinfer \ - --stream-interval 50 \ - --tokenizer-worker-num 6 \ - --mamba-ssm-dtype bfloat16 \ - --disable-radix-cache \ - --trust-remote-code \ - --speculative-algorithm EAGLE \ - --speculative-num-steps "$SPECULATIVE_NUM_STEPS" \ - --speculative-num-draft-tokens "$SPECULATIVE_DRAFT_TOKENS" \ - --speculative-eagle-topk "$SPECULATIVE_EAGLE_TOPK" \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --use-chat-template \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - export EVAL_CONCURRENT_REQUESTS="$CONC" - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi300x.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi300x.sh deleted file mode 100755 index 1ca367d51c..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi300x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) -MAX_PREFILL_TOKENS=32768 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -# Recipe source: https://www.linkedin.com/feed/update/urn:li:activity:7429203734389280768/ -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --data-parallel-size 1 \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --scheduler-recv-interval 30 \ - --mem-fraction-static 0.75 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x.sh deleted file mode 100755 index 1ca367d51c..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) -MAX_PREFILL_TOKENS=32768 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -# Recipe source: https://www.linkedin.com/feed/update/urn:li:activity:7429203734389280768/ -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --data-parallel-size 1 \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --scheduler-recv-interval 30 \ - --mem-fraction-static 0.75 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x_mtp.sh deleted file mode 100755 index edd44a705d..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x_mtp.sh +++ /dev/null @@ -1,78 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - EP_SIZE \ - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) -MAX_PREFILL_TOKENS=32768 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -# Recipe source: https://www.linkedin.com/feed/update/urn:li:activity:7429203734389280768/ -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --ep-size $EP_SIZE \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --scheduler-recv-interval 30 \ - --speculative-algorithm EAGLE \ - --speculative-num-steps 3 \ - --speculative-eagle-topk 1 \ - --speculative-num-draft-tokens 4 \ - --mem-fraction-static 0.75 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - EP_SIZE \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x.sh deleted file mode 100644 index 1c4e1dceb8..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER_UNIFIED_ATTN=1 -export SGLANG_USE_AITER=1 - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --ep-size $EP_SIZE \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --max-running-requests $CONC \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --chunked-prefill-size 32768 \ - --scheduler-recv-interval 30 \ - --mem-fraction-static 0.8 \ - --model-loader-extra-config '{"enable_multithread_load": true}' \ - --page-size 16 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom.sh deleted file mode 100644 index 98f2416dcc..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor -MEM_FRAC_STATIC=0.9 - -set -x - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --gpu-memory-utilization $MEM_FRAC_STATIC \ - --trust-remote-code \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -export PYTHONDONTWRITEBYTECODE=1 -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --trust-remote-code - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom_mtp.sh deleted file mode 100644 index 7948322313..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom_mtp.sh +++ /dev/null @@ -1,79 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor -MEM_FRAC_STATIC=0.9 - -set -x - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --gpu-memory-utilization $MEM_FRAC_STATIC \ - --method mtp \ - --num-speculative-tokens 3 \ - --trust-remote-code \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -export PYTHONDONTWRITEBYTECODE=1 -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --trust-remote-code \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_mtp.sh deleted file mode 100755 index 2a20db1868..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_mtp.sh +++ /dev/null @@ -1,82 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER_UNIFIED_ATTN=1 -export SGLANG_USE_AITER=1 - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --ep-size $EP_SIZE \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --max-running-requests $CONC \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --chunked-prefill-size 32768 \ - --scheduler-recv-interval 30 \ - --mem-fraction-static 0.8 \ - --model-loader-extra-config '{"enable_multithread_load": true}' \ - --page-size 16 \ - --speculative-algorithm EAGLE \ - --speculative-num-steps 3 \ - --speculative-eagle-topk 1 \ - --speculative-num-draft-tokens 4 \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..2d11e33ab1 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml @@ -0,0 +1,44 @@ +base: + schema: 2 + name: dsr1-fp4-mi355x-atom-mtp-8k1k + model: + path: hf:amd/DeepSeek-R1-0528-MXFP4 + container: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + method: mtp + env: + OMP_NUM_THREADS: '1' + AMDGCN_USE_BUFFER_OPS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/DeepSeek-R1-0528-MXFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml new file mode 100644 index 0000000000..9769e663d7 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml @@ -0,0 +1,50 @@ +base: + schema: 2 + name: dsr1-fp4-mi355x-atom-8k1k + model: + path: hf:amd/DeepSeek-R1-0528-MXFP4-Preview + container: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + block-size: 16 + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/DeepSeek-R1-0528-MXFP4-Preview + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + benchmark: + env: + CONC: ['4'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..40bc67ac50 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml @@ -0,0 +1,44 @@ +base: + schema: 2 + name: dsr1-fp8-mi355x-atom-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: rocm/atom:rocm7.2.4_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.3 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + method: mtp + num-speculative-tokens: 3 + kv_cache_dtype: fp8 + no-enable_prefix_caching: true + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml new file mode 100644 index 0000000000..cdcc9646dc --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml @@ -0,0 +1,43 @@ +base: + schema: 2 + name: dsr1-fp8-mi355x-atom-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + block-size: 16 + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..88ae02628c --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml @@ -0,0 +1,83 @@ +base: + schema: 2 + name: dsr1-fp4-b200-sglang-mtp-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: lmsysorg/sglang:v0.5.16-cu130 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + cuda-graph-max-bs: 256 + max-running-requests: 256 + mem-fraction-static: 0.85 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + expert-parallel-size: 1 + quantization: modelopt_fp4 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-piecewise-cuda-graph: true + attention-backend: trtllm_mla + moe-runner-backend: flashinfer_trtllm + stream-interval: 10 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-num-draft-tokens: 3 + speculative-eagle-topk: 1 + served-model-name: nvidia/DeepSeek-R1-0528-FP4-V2 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_RADIX_FORCE_MISS: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + chunked-prefill-size: 32768 + data-parallel-size: 4 + enable-dp-attention: true + enable-dp-attention-local-control-broadcast: true + enable-dp-lm-head: true + enable-prefill-delayer: true + expert-parallel-size: 4 + schedule-conservativeness: 3.33 + scheduler-recv-interval: 1 + benchmark: + env: + CONC: ['64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml new file mode 100644 index 0000000000..98888e3f91 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml @@ -0,0 +1,79 @@ +base: + schema: 2 + name: dsr1-fp4-b200-sglang-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + cuda-graph-max-bs: 256 + max-running-requests: 256 + mem-fraction-static: 0.85 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + expert-parallel-size: 1 + quantization: modelopt_fp4 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + enable-symm-mem: true + disable-piecewise-cuda-graph: true + attention-backend: trtllm_mla + moe-runner-backend: flashinfer_trtllm + stream-interval: 10 + served-model-name: nvidia/DeepSeek-R1-0528-FP4-V2 + enable-metrics: false + env: + SGLANG_RADIX_FORCE_MISS: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + chunked-prefill-size: 32768 + data-parallel-size: 4 + enable-dp-attention: true + enable-dp-attention-local-control-broadcast: true + enable-dp-lm-head: true + enable-prefill-delayer: true + expert-parallel-size: 4 + schedule-conservativeness: 3.33 + scheduler-recv-interval: 1 + benchmark: + env: + CONC: ['64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..54d62161ee --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml @@ -0,0 +1,68 @@ +base: + schema: 2 + name: dsr1-fp8-b200-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + data-parallel-size: 1 + cuda-graph-max-bs: 512 + max-running-requests: 512 + mem-fraction-static: 0.82 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-radix-cache: true + fp8-gemm-backend: flashinfer_trtllm + attention-backend: trtllm_mla + stream-interval: 30 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_trtllm + quantization: fp8 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-num-draft-tokens: 3 + speculative-eagle-topk: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30, 30, 30, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128', '256', '512'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml new file mode 100644 index 0000000000..301b150f54 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml @@ -0,0 +1,73 @@ +base: + schema: 2 + name: dsr1-fp8-b200-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + data-parallel-size: 1 + cuda-graph-max-bs: 128 + max-running-requests: 128 + mem-fraction-static: 0.82 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-radix-cache: true + attention-backend: trtllm_mla + stream-interval: 30 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_trtllm + quantization: fp8 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1: + benchmark: + env: + CONC: ['1', '2', '4'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + chunked-prefill-size: 8192 + cuda-graph-max-bs: 32 + max-prefill-tokens: 8192 + max-running-requests: 32 + mem-fraction-static: 0.95 + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml new file mode 100644 index 0000000000..4399686cde --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml @@ -0,0 +1,73 @@ +base: + schema: 2 + name: dsr1-fp4-b300-sglang-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: nvidia/DeepSeek-R1-0528-FP4-V2 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + cuda-graph-max-bs: 256 + max-running-requests: 256 + mem-fraction-static: 0.85 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + expert-parallel-size: 4 + quantization: modelopt_fp4 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + enable-symm-mem: true + disable-radix-cache: true + attention-backend: trtllm_mla + moe-runner-backend: flashinfer_trtllm + stream-interval: 10 + enable-metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep4: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128'] +zip_override_tp8_ep8: + roles: + agg: + gpus: 8 + args: + expert-parallel-size: 8 + scheduler-recv-interval: [10, 10, 10, 10, 30] + tensor-parallel-size: 8 + benchmark: + env: + CONC: ['1', '2', '4', '8', '16'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..952ec213c9 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml @@ -0,0 +1,68 @@ +base: + schema: 2 + name: dsr1-fp8-b300-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.15.post1-cu130 + precision: fp8 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-R1-0528 + tensor-parallel-size: 8 + data-parallel-size: 1 + cuda-graph-max-bs: 512 + max-running-requests: 512 + mem-fraction-static: 0.82 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-radix-cache: true + fp8-gemm-backend: flashinfer_trtllm + attention-backend: trtllm_mla + stream-interval: 30 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_trtllm + quantization: fp8 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-num-draft-tokens: 3 + speculative-eagle-topk: 1 + enable-metrics: false + env: + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30, 30, 30, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128', '256', '512'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml new file mode 100644 index 0000000000..0fbaf5bb8a --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml @@ -0,0 +1,73 @@ +base: + schema: 2 + name: dsr1-fp8-b300-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-R1-0528 + tensor-parallel-size: 8 + data-parallel-size: 1 + cuda-graph-max-bs: 128 + max-running-requests: 128 + mem-fraction-static: 0.82 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-radix-cache: true + attention-backend: trtllm_mla + stream-interval: 30 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_trtllm + quantization: fp8 + enable-metrics: false + env: + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1: + benchmark: + env: + CONC: ['1', '2', '4'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + chunked-prefill-size: 8192 + cuda-graph-max-bs: 32 + max-prefill-tokens: 8192 + max-running-requests: 32 + mem-fraction-static: 0.95 + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..7818ad85df --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml @@ -0,0 +1,60 @@ +base: + schema: 2 + name: dsr1-fp8-h200-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + disable-radix-cache: true + max-running-requests: 256 + cuda-graph-max-bs: 256 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + mem-fraction-static: 0.82 + attention-backend: flashinfer + stream-interval: 10 + decode-log-interval: 1 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-num-draft-tokens: 3 + speculative-eagle-topk: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + TORCH_CUDA_ARCH_LIST: '9.0' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml new file mode 100644 index 0000000000..23bbfa1d27 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml @@ -0,0 +1,55 @@ +base: + schema: 2 + name: dsr1-fp8-h200-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + data-parallel-size: 1 + disable-radix-cache: true + max-running-requests: 256 + cuda-graph-max-bs: 256 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + mem-fraction-static: 0.82 + attention-backend: flashinfer + stream-interval: 10 + decode-log-interval: 1 + expert-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + TORCH_CUDA_ARCH_LIST: '9.0' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml new file mode 100644 index 0000000000..78e9a937a0 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml @@ -0,0 +1,54 @@ +base: + schema: 2 + name: dsr1-fp8-mi300x-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-rocm700-mi30x + precision: fp8 + resources: + gpu_type: mi300x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 128 + chunked-prefill-size: 131072 + num-continuous-decode-steps: 4 + max-prefill-tokens: 131072 + kv-cache-dtype: fp8_e4m3 + attention-backend: aiter + disable-radix-cache: true + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..449ba41109 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml @@ -0,0 +1,59 @@ +base: + schema: 2 + name: dsr1-fp8-mi325x-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-rocm700-mi30x + precision: fp8 + resources: + gpu_type: mi325x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + expert-parallel-size: 1 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 128 + chunked-prefill-size: 131072 + num-continuous-decode-steps: 4 + max-prefill-tokens: 131072 + kv-cache-dtype: fp8_e4m3 + attention-backend: aiter + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + disable-radix-cache: true + data-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + SGLANG_ENABLE_SPEC_V2: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml new file mode 100644 index 0000000000..06b99ea82f --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml @@ -0,0 +1,54 @@ +base: + schema: 2 + name: dsr1-fp8-mi325x-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.19-rocm700-mi30x + precision: fp8 + resources: + gpu_type: mi325x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 128 + chunked-prefill-size: 131072 + num-continuous-decode-steps: 4 + max-prefill-tokens: 131072 + kv-cache-dtype: fp8_e4m3 + attention-backend: aiter + disable-radix-cache: true + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..18cf9bfd4f --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml @@ -0,0 +1,65 @@ +base: + schema: 2 + name: dsr1-fp4-mi355x-sglang-mtp-8k1k + model: + path: hf:amd/DeepSeek-R1-0528-MXFP4 + container: lmsysorg/sglang:v0.5.12-rocm700-mi35x + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + expert-parallel-size: 1 + chunked-prefill-size: 196608 + mem-fraction-static: 0.8 + disable-radix-cache: true + num-continuous-decode-steps: 4 + max-prefill-tokens: 196608 + cuda-graph-max-bs: 128 + attention-backend: aiter + kv-cache-dtype: fp8_e4m3 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + data-parallel-size: 1 + served-model-name: amd/DeepSeek-R1-0528-MXFP4 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + SGLANG_ENABLE_SPEC_V2: '1' + ROCM_QUICK_REDUCE_QUANTIZATION: INT4 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/DeepSeek-R1-0528-MXFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + chunked-prefill-size: [196608, 196608, 196608, 196608, 32768] + max-prefill-tokens: [196608, 196608, 196608, 196608, 32768] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml new file mode 100644 index 0000000000..75d0a62f57 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml @@ -0,0 +1,70 @@ +base: + schema: 2 + name: dsr1-fp4-mi355x-sglang-8k1k + model: + path: hf:amd/DeepSeek-R1-0528-MXFP4-Preview + container: lmsysorg/sglang:v0.5.12-rocm700-mi35x + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + chunked-prefill-size: 196608 + mem-fraction-static: 0.8 + disable-radix-cache: true + num-continuous-decode-steps: 4 + max-prefill-tokens: 196608 + cuda-graph-max-bs: 128 + attention-backend: aiter + kv-cache-dtype: fp8_e4m3 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: amd/DeepSeek-R1-0528-MXFP4-Preview + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + ROCM_QUICK_REDUCE_QUANTIZATION: INT4 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/DeepSeek-R1-0528-MXFP4-Preview + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + chunked-prefill-size: [196608, 196608, 196608, 196608, 32768] + max-prefill-tokens: [196608, 196608, 196608, 196608, 32768] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + args: + chunked-prefill-size: [196608, 196608, 196608, 196608, 32768] + max-prefill-tokens: [196608, 196608, 196608, 196608, 32768] + tensor-parallel-size: 8 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..ae4ba64142 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml @@ -0,0 +1,67 @@ +base: + schema: 2 + name: dsr1-fp8-mi355x-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-rocm700-mi35x + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + attention-backend: aiter + tensor-parallel-size: 8 + expert-parallel-size: 1 + trust-remote-code: true + chunked-prefill-size: 196608 + mem-fraction-static: 0.8 + disable-radix-cache: true + num-continuous-decode-steps: 8 + max-prefill-tokens: 196608 + kv-cache-dtype: fp8_e4m3 + cuda-graph-max-bs: 4 + max-running-requests: 4 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + data-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + SGLANG_ENABLE_SPEC_V2: '1' + RCCL_MSCCL_ENABLE: '0' + ROCM_QUICK_REDUCE_QUANTIZATION: INT4 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + max-running-requests: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml new file mode 100644 index 0000000000..07cb2a1783 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml @@ -0,0 +1,69 @@ +base: + schema: 2 + name: dsr1-fp8-mi355x-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-rocm700-mi35x + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + attention-backend: aiter + tensor-parallel-size: 4 + trust-remote-code: true + chunked-prefill-size: 196608 + mem-fraction-static: 0.8 + disable-radix-cache: true + num-continuous-decode-steps: 8 + max-prefill-tokens: 196608 + kv-cache-dtype: fp8_e4m3 + cuda-graph-max-bs: 32 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + RCCL_MSCCL_ENABLE: '0' + ROCM_QUICK_REDUCE_QUANTIZATION: INT4 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda-graph-max-bs: [32, 64] + benchmark: + env: + CONC: ['32', '64'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + tensor-parallel-size: 8 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..d2a484832f --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml @@ -0,0 +1,151 @@ +base: + schema: 2 + name: dsr1-fp4-b200-trt-mtp-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: nvidia/DeepSeek-R1-0528-FP4-V2 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + cuda_graph_config: + enable_padding: true + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.8 + enable_block_reuse: false + stream_interval: 10 + speculative_config: + decoding_type: MTP + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [4, 8, 16] + enable_attention_dp: false + moe_config: + backend: TRTLLM + speculative_config: + num_nextn_predict_layers: 3 + max_batch_size: [4, 8, 16] + max_num_tokens: 8320 + tensor_parallel_size: 4 + moe_expert_parallel_size: 1 + benchmark: + env: + CONC: ['4', '8', '16'] +zip_override_tp4_ep4: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 32 + enable_attention_dp: false + moe_config: + backend: TRTLLM + speculative_config: + num_nextn_predict_layers: 3 + max_batch_size: 32 + max_num_tokens: 8384 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + benchmark: + env: + CONC: ['32'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 64 + enable_attention_dp: true + moe_config: + backend: CUTLASS + speculative_config: + num_nextn_predict_layers: 1 + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_batch_size: 64 + max_num_tokens: 8384 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + benchmark: + env: + CONC: ['256'] +zip_override_tp8_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 4 + enable_attention_dp: false + moe_config: + backend: TRTLLM + speculative_config: + num_nextn_predict_layers: 3 + max_batch_size: 4 + max_num_tokens: 8320 + tensor_parallel_size: 8 + moe_expert_parallel_size: 1 + gpus: 8 + benchmark: + env: + CONC: ['4'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [16, 32, 64] + enable_attention_dp: true + moe_config: + backend: CUTLASS + speculative_config: + num_nextn_predict_layers: 1 + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_batch_size: [16, 32, 64] + max_num_tokens: [8320, 8320, 8384] + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml new file mode 100644 index 0000000000..bfee802b20 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml @@ -0,0 +1,134 @@ +base: + schema: 2 + name: dsr1-fp4-b200-trt-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: nvidia/DeepSeek-R1-0528-FP4-V2 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + cuda_graph_config: + enable_padding: true + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.8 + enable_block_reuse: false + stream_interval: 10 + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [4, 8, 16, 32] + enable_attention_dp: false + moe_config: + backend: TRTLLM + max_num_tokens: 8320 + tensor_parallel_size: 4 + moe_expert_parallel_size: 1 + benchmark: + env: + CONC: ['4', '8', '16', '32'] +zip_override_tp4_ep4: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 32 + enable_attention_dp: false + moe_config: + backend: TRTLLM + max_num_tokens: 8320 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + benchmark: + env: + CONC: ['32'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 64 + enable_attention_dp: true + moe_config: + backend: CUTLASS + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_num_tokens: 8512 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + benchmark: + env: + CONC: ['256'] +zip_override_tp8_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 4 + enable_attention_dp: false + moe_config: + backend: TRTLLM + max_num_tokens: 8320 + tensor_parallel_size: 8 + moe_expert_parallel_size: 1 + gpus: 8 + benchmark: + env: + CONC: ['4'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [32, 64] + enable_attention_dp: true + moe_config: + backend: CUTLASS + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_num_tokens: [8384, 8512] + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..70d909c580 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml @@ -0,0 +1,92 @@ +base: + schema: 2 + name: dsr1-fp8-b200-trt-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: deepseek-ai/DeepSeek-R1-0528 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + enable_padding: true + enable_attention_dp: false + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.8 + enable_block_reuse: false + stream_interval: 10 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + tensor_parallel_size: 8 + moe_expert_parallel_size: 1 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp8_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [4, 8, 16] + max_batch_size: [4, 8, 16] + max_num_tokens: 8320 + benchmark: + env: + CONC: ['4', '8', '16'] +zip_override_tp8_ep1_graphs: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [32, 64, 128, 256] + torch_compile_config: + capture_num_tokens: [[1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, 2048, + 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, 5888, + 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, 8192, 8384], [1, 2, 4, 8, 16, 32, 64, 128, + 256, 512, 768, 1024, 1280, 1536, 1792, 2048, 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, + 4352, 4608, 4864, 5120, 5376, 5632, 5888, 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, + 8192, 8448, 8512], [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, + 2048, 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, + 5888, 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, 8192, 8448, 8704, 8768], [1, 2, 4, + 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, 2048, 2304, 2560, 2816, 3072, + 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, 5888, 6144, 6400, 6656, 6912, + 7168, 7424, 7680, 7936, 8192, 8448, 8704, 8960, 9216, 9280]] + enable_piecewise_cuda_graph: true + max_batch_size: [32, 64, 128, 256] + max_num_tokens: [8384, 8512, 8768, 9280] + benchmark: + env: + CONC: ['32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml new file mode 100644 index 0000000000..9223814440 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml @@ -0,0 +1,104 @@ +base: + schema: 2 + name: dsr1-fp8-b200-trt-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: deepseek-ai/DeepSeek-R1-0528 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + enable_padding: true + enable_attention_dp: false + print_iter_log: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + stream_interval: 10 + moe_config: + backend: TRTLLM + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + moe_expert_parallel_size: 1 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + env: + TLLM_OVERRIDE_LAYER_NUM: '61' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1_graphs: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [64, 128, 256] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + torch_compile_config: + capture_num_tokens: [[1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, 2048, + 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, 5888, + 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, 8192, 8320], [1, 2, 4, 8, 16, 32, 64, 128, + 256, 512, 768, 1024, 1280, 1536, 1792, 2048, 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, + 4352, 4608, 4864, 5120, 5376, 5632, 5888, 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, + 8192, 8384], [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, 2048, + 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, 5888, + 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, 8192, 8448, 8512]] + enable_piecewise_cuda_graph: true + max_num_tokens: [8320, 8384, 8512] + tensor_parallel_size: 8 + benchmark: + env: + CONC: ['64', '128', '256'] +zip_override_tp4_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [8, 16, 32] + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_num_tokens: 8320 + tensor_parallel_size: 4 + gpus: 4 + benchmark: + env: + CONC: ['8', '16', '32'] +zip_override_tp8_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [4, 8] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_num_tokens: 8320 + tensor_parallel_size: 8 + benchmark: + env: + CONC: ['4', '8'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..6e6685878d --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml @@ -0,0 +1,88 @@ +base: + schema: 2 + name: dsr1-fp8-h200-trt-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: deepseek-ai/DeepSeek-R1-0528 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.75 + enable_block_reuse: false + stream_interval: 10 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp8_ep8: + roles: + agg: + args: + enable_attention_dp: false + speculative_config: + num_nextn_predict_layers: 3 + max_batch_size: [4, 8, 16, 32] + max_num_tokens: [8320, 8320, 8320, 8384] + benchmark: + env: + CONC: ['4', '8', '16', '32'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + enable_attention_dp: true + speculative_config: + num_nextn_predict_layers: 1 + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_batch_size: [8, 16, 32] + max_num_tokens: 8320 + env: + PYTORCH_CUDA_ALLOC_CONF: max_split_size_mb:8192 + benchmark: + env: + CONC: ['64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml new file mode 100644 index 0000000000..f544e6bbbc --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml @@ -0,0 +1,77 @@ +base: + schema: 2 + name: dsr1-fp8-h200-trt-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: deepseek-ai/DeepSeek-R1-0528 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.75 + enable_block_reuse: false + stream_interval: 10 + moe_config: + backend: CUTLASS + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + max_num_tokens: 8320 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep8: + roles: + agg: + args: + enable_attention_dp: false + benchmark: + env: + CONC: ['4', '8', '16', '32'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + enable_attention_dp: true + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + benchmark: + env: + CONC: ['64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml new file mode 100644 index 0000000000..7677e9a68a --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml @@ -0,0 +1,51 @@ +base: + schema: 2 + name: qwen3.5-fp4-mi355x-atom-8k1k + model: + path: hf:amd/Qwen3.5-397B-A17B-MXFP4 + container: rocm/atom:rocm7.2.2_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.2.post + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + gpu-memory-utilization: 0.9 + trust-remote-code: true + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code + env: + MODEL: amd/Qwen3.5-397B-A17B-MXFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp2_ep1: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + benchmark: + env: + CONC: ['4', '8', '16'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..6e5b336ee5 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml @@ -0,0 +1,53 @@ +base: + schema: 2 + name: qwen3.5-fp8-mi355x-atom-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: rocm/atom:rocm7.2.2_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.2.post + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + gpu-memory-utilization: 0.9 + method: mtp + num-speculative-tokens: 3 + trust-remote-code: true + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml new file mode 100644 index 0000000000..6c6fb60a3e --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml @@ -0,0 +1,58 @@ +base: + schema: 2 + name: qwen3.5-fp8-mi355x-atom-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + gpu-memory-utilization: 0.9 + trust-remote-code: true + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp2_ep1: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..8604043522 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml @@ -0,0 +1,95 @@ +base: + schema: 2 + name: qwen3.5-fp4-b200-sglang-mtp-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs: 4 + max-running-requests: 4 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + context-length: 9236 + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + benchmark: + env: + CONC: ['4'] +zip_override_tp2_ep1: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + max-running-requests: [4, 8, 16, 32, 64] + scheduler-recv-interval: [10, 30, 30, 30, 30] + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] +zip_override_tp2_ep2: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [16, 32, 64] + expert-parallel-size: 2 + linear-attn-backend: triton + linear-attn-decode-backend: flashinfer + linear-attn-prefill-backend: flashinfer + max-running-requests: [16, 32, 64] + scheduler-recv-interval: 30 + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml new file mode 100644 index 0000000000..39801bfa8e --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml @@ -0,0 +1,72 @@ +base: + schema: 2 + name: qwen3.5-fp4-b200-sglang-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs: 4 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + context-length: 9236 + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + enable-metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + benchmark: + env: + CONC: ['4'] +zip_override_tp2_ep1: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + scheduler-recv-interval: [10, 30, 30, 30, 30, 30] + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..8ea8fbae21 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml @@ -0,0 +1,81 @@ +base: + schema: 2 + name: qwen3.5-fp8-b200-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs-decode: 4 + max-running-requests: 4 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + mamba-full-memory-ratio: 0.37 + linear-attn-prefill-backend: flashinfer + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: Qwen/Qwen3.5-397B-A17B-FP8 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + context-length: 9236 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp8_ep1: + benchmark: + env: + CONC: ['4'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + cuda-graph-max-bs-decode: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + scheduler-recv-interval: [10, 30, 30, 30, 30, 30, 30] + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml new file mode 100644 index 0000000000..68df2a0c0a --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml @@ -0,0 +1,77 @@ +base: + schema: 2 + name: qwen3.5-fp8-b200-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-full-memory-ratio: 0.37 + linear-attn-prefill-backend: flashinfer + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs-decode: 1 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + mem-fraction-static: 0.86 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + context-length: 9236 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1: + roles: + agg: + args: + cuda-graph-max-bs-decode: [1, 2, 4] + benchmark: + env: + CONC: ['1', '2', '4'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + cuda-graph-max-bs-decode: [2, 4, 8, 16, 32, 64, 128, 256, 512, 320, 384, 448, 640] + scheduler-recv-interval: [10, 10, 30, 30, 30, 30, 30, 30, 30, 30, 30, 30, 30] + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['2', '4', '8', '16', '32', '64', '128', '256', '512', '320', '384', '448', '640'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..3297e6b7df --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml @@ -0,0 +1,90 @@ +base: + schema: 2 + name: qwen3.5-fp4-b300-sglang-mtp-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + mamba-scheduler-strategy: no_buffer + quantization: modelopt_fp4 + fp4-gemm-backend: flashinfer_cutlass + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + cuda-graph-max-bs: 4 + max-running-requests: 128 + mem-fraction-static: 0.8 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + context-length: 9236 + disable-radix-cache: true + attention-backend: trtllm_mha + mm-attention-backend: triton_attn + moe-runner-backend: flashinfer_trtllm + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + stream-interval: 30 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + enable-metrics: false + env: + NCCL_NVLS_ENABLE: '1' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONUNBUFFERED: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + scheduler-recv-interval: [10, 10, 30, 30, 30, 30] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] +zip_override_tp2_ep2: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + expert-parallel-size: 2 + scheduler-recv-interval: [10, 10, 30, 30, 30, 30] + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml new file mode 100644 index 0000000000..00af900818 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml @@ -0,0 +1,86 @@ +base: + schema: 2 + name: qwen3.5-fp4-b300-sglang-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + mamba-scheduler-strategy: no_buffer + quantization: modelopt_fp4 + fp4-gemm-backend: flashinfer_cutlass + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + cuda-graph-max-bs: 4 + max-running-requests: 128 + mem-fraction-static: 0.8 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + context-length: 9236 + disable-radix-cache: true + attention-backend: trtllm_mha + mm-attention-backend: triton_attn + moe-runner-backend: flashinfer_trtllm + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + stream-interval: 30 + enable-metrics: false + env: + NCCL_NVLS_ENABLE: '1' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONUNBUFFERED: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + scheduler-recv-interval: [10, 10, 30, 30, 30, 30] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] +zip_override_tp2_ep2: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + expert-parallel-size: 2 + scheduler-recv-interval: [10, 10, 30, 30, 30, 30] + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..aa6af8a4b7 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml @@ -0,0 +1,73 @@ +base: + schema: 2 + name: qwen3.5-fp8-b300-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + mm-attention-backend: triton_attn + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs: 4 + max-running-requests: 4 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: Qwen/Qwen3.5-397B-A17B-FP8 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + context-length: 9236 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml new file mode 100644 index 0000000000..98f877110c --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml @@ -0,0 +1,68 @@ +base: + schema: 2 + name: qwen3.5-fp8-b300-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + mm-attention-backend: triton_attn + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs: 4 + max-running-requests: 4 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: Qwen/Qwen3.5-397B-A17B-FP8 + context-length: 9236 + enable-metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..ff59b74291 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml @@ -0,0 +1,69 @@ +base: + schema: 2 + name: qwen3.5-fp8-h100-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp8 + resources: + gpu_type: h100 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + expert-parallel-size: 8 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-flashinfer-allreduce-fusion: true + max-running-requests: 64 + chunked-prefill-size: 8192 + decode-log-interval: 1 + mem-fraction-static: 0.75 + cuda-graph-max-bs: 4 + context-length: 9236 + kv-cache-dtype: fp8_e4m3 + quantization: fp8 + attention-backend: flashinfer + stream-interval: 50 + tokenizer-worker-num: 6 + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + trust-remote-code: true + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32] + benchmark: + env: + CONC: ['4', '8', '16', '32'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml new file mode 100644 index 0000000000..9f5dac04cc --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml @@ -0,0 +1,76 @@ +base: + schema: 2 + name: qwen3.5-fp8-h100-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp8 + resources: + gpu_type: h100 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-flashinfer-allreduce-fusion: true + max-running-requests: 256 + chunked-prefill-size: 16384 + decode-log-interval: 1 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 1 + context-length: 9236 + kv-cache-dtype: fp8_e4m3 + quantization: fp8 + attention-backend: flashinfer + stream-interval: 50 + tokenizer-worker-num: 6 + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + enable-symm-mem: true + trust-remote-code: true + scheduler-recv-interval: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1: + roles: + agg: + args: + cuda-graph-max-bs: [1, 2, 4, 8] + scheduler-recv-interval: [2, 2, 2, 60] + benchmark: + env: + CONC: ['1', '2', '4', '8'] +zip_override_tp8_ep8: + roles: + agg: + args: + cuda-graph-max-bs: [16, 32, 64, 128, 256] + expert-parallel-size: 8 + scheduler-recv-interval: [30, 1200, 600, 1920, 1920] + benchmark: + env: + CONC: ['16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..e5bf9adca9 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml @@ -0,0 +1,68 @@ +base: + schema: 2 + name: qwen3.5-fp8-h200-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + expert-parallel-size: 8 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-flashinfer-allreduce-fusion: true + max-running-requests: 128 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 4 + context-length: 9472 + kv-cache-dtype: fp8_e4m3 + quantization: fp8 + attention-backend: flashinfer + stream-interval: 50 + tokenizer-worker-num: 6 + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + trust-remote-code: true + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-num-draft-tokens: 4 + speculative-eagle-topk: 1 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml new file mode 100644 index 0000000000..a7c1df61c2 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml @@ -0,0 +1,63 @@ +base: + schema: 2 + name: qwen3.5-fp8-h200-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + expert-parallel-size: 8 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-flashinfer-allreduce-fusion: true + max-running-requests: 128 + chunked-prefill-size: 16384 + decode-log-interval: 1 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 4 + context-length: 9236 + kv-cache-dtype: fp8_e4m3 + quantization: fp8 + attention-backend: flashinfer + stream-interval: 50 + tokenizer-worker-num: 6 + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + trust-remote-code: true + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml new file mode 100644 index 0000000000..6691e5ee92 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml @@ -0,0 +1,56 @@ +base: + schema: 2 + name: qwen3.5-fp8-mi300x-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-rocm720-mi30x + precision: fp8 + resources: + gpu_type: mi300x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + attention-backend: aiter + tensor-parallel-size: 8 + data-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + cuda-graph-max-bs: 4 + disable-radix-cache: true + max-prefill-tokens: 32768 + scheduler-recv-interval: 30 + mem-fraction-static: 0.75 + context-length: 9236 + expert-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..bd0dcd6623 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml @@ -0,0 +1,60 @@ +base: + schema: 2 + name: qwen3.5-fp8-mi325x-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-rocm720-mi30x + precision: fp8 + resources: + gpu_type: mi325x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + attention-backend: aiter + tensor-parallel-size: 8 + expert-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + cuda-graph-max-bs: 4 + disable-radix-cache: true + max-prefill-tokens: 32768 + scheduler-recv-interval: 30 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + mem-fraction-static: 0.75 + context-length: 9236 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml new file mode 100644 index 0000000000..7d4ef623dc --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml @@ -0,0 +1,56 @@ +base: + schema: 2 + name: qwen3.5-fp8-mi325x-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-rocm720-mi30x + precision: fp8 + resources: + gpu_type: mi325x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + attention-backend: aiter + tensor-parallel-size: 8 + data-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + cuda-graph-max-bs: 4 + disable-radix-cache: true + max-prefill-tokens: 32768 + scheduler-recv-interval: 30 + mem-fraction-static: 0.75 + context-length: 9236 + expert-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..244b8d64dc --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml @@ -0,0 +1,75 @@ +base: + schema: 2 + name: qwen3.5-fp4-mi355x-sglang-mtp-8k1k + model: + path: hf:amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + container: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913 + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + trust-remote-code: true + tensor-parallel-size: 2 + attention-backend: aiter + mem-fraction-static: 0.8 + model-loader-extra-config: '{"enable_multithread_load": true}' + watchdog-timeout: 1200 + disable-radix-cache: true + max-running-requests: 4 + page-size: 16 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + kv-cache-dtype: fp8_e4m3 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_USE_AITER_UNIFIED_ATTN: '1' + AITER_FLYDSL_FORCE: '1' + SGLANG_MAMBA_SSM_DTYPE: bfloat16 + ROCM_QUICK_REDUCE_QUANTIZATION: INT8 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp2_ep1: + roles: + agg: + args: + max-running-requests: [4, 8, 16, 32, 64, 128] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + max-running-requests: [4, 8, 16] + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['4', '8', '16'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml new file mode 100644 index 0000000000..7ea8244e0d --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml @@ -0,0 +1,71 @@ +base: + schema: 2 + name: qwen3.5-fp4-mi355x-sglang-8k1k + model: + path: hf:amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + container: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913 + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + trust-remote-code: true + tensor-parallel-size: 2 + attention-backend: aiter + mem-fraction-static: 0.8 + model-loader-extra-config: '{"enable_multithread_load": true}' + watchdog-timeout: 1200 + disable-radix-cache: true + max-running-requests: 4 + page-size: 16 + kv-cache-dtype: fp8_e4m3 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_USE_AITER_UNIFIED_ATTN: '1' + AITER_FLYDSL_FORCE: '1' + SGLANG_MAMBA_SSM_DTYPE: bfloat16 + ROCM_QUICK_REDUCE_QUANTIZATION: INT8 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp2_ep1: + roles: + agg: + args: + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + max-running-requests: [4, 8, 16] + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['4', '8', '16'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..488994e3bf --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml @@ -0,0 +1,67 @@ +base: + schema: 2 + name: qwen3.5-fp8-mi355x-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + attention-backend: aiter + tensor-parallel-size: 4 + expert-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + max-running-requests: 4 + cuda-graph-max-bs: 4 + disable-radix-cache: true + chunked-prefill-size: 32768 + scheduler-recv-interval: 30 + mem-fraction-static: 0.8 + model-loader-extra-config: '{"enable_multithread_load": true}' + page-size: 16 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + context-length: 9236 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_USE_AITER_UNIFIED_ATTN: '1' + SGLANG_USE_AITER: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml new file mode 100644 index 0000000000..c753c2b465 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml @@ -0,0 +1,63 @@ +base: + schema: 2 + name: qwen3.5-fp8-mi355x-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + attention-backend: aiter + tensor-parallel-size: 4 + expert-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + max-running-requests: 4 + cuda-graph-max-bs: 4 + disable-radix-cache: true + chunked-prefill-size: 32768 + scheduler-recv-interval: 30 + mem-fraction-static: 0.8 + model-loader-extra-config: '{"enable_multithread_load": true}' + page-size: 16 + context-length: 9236 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_USE_AITER_UNIFIED_ATTN: '1' + SGLANG_USE_AITER: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..9d0c9375cd --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml @@ -0,0 +1,154 @@ +base: + schema: 2 + name: qwen3.5-fp4-b200-trt-mtp-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc18 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: nvidia/Qwen3.5-397B-A17B-NVFP4 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + backend: pytorch + print_iter_log: true + enable_layerwise_nvtx_marker: false + disable_overlap_scheduler: false + enable_iter_perf_stats: true + enable_chunked_prefill: false + stream_interval: 20 + num_postprocess_workers: 4 + scheduler_config: + capacity_scheduler_policy: MAX_UTILIZATION + context_chunking_policy: FIRST_COME_FIRST_SERVED + kv_cache_config: + enable_block_reuse: false + dtype: fp8 + cuda_graph_config: + enable_padding: true + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128] + moe_config: + use_low_precision_moe_combine: true + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + trust_remote_code: true + max_seq_len: 9472 + max_num_tokens: 32768 + pipeline_parallel_size: 1 + return_perf_metrics: false + env: + TLLM_USE_FLASHINFER_GDN_PREFILL: '0' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp2_ep1: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: 0.7 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 4 + tensor_parallel_size: 2 + moe_expert_parallel_size: 1 + enable_attention_dp: false + benchmark: + env: + CONC: ['4'] +zip_override_tp2_ep2: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: 0.6 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: [8, 16] + tensor_parallel_size: 2 + moe_expert_parallel_size: 2 + enable_attention_dp: false + benchmark: + env: + CONC: ['8', '16'] +zip_override_tp4_ep4: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: 0.75 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 4 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + gpus: 4 + benchmark: + env: + CONC: ['4'] +zip_override_tp8_ep8: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 4 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + gpus: 8 + benchmark: + env: + CONC: ['4'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: [0.9, 0.9, 0.8] + moe_config: + backend: CUTEDSL + enable_attention_dp: true + attention_dp_config: + enable_balance: true + batching_wait_iters: 10 + timeout_iters: 500 + max_batch_size: [16, 32, 128] + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['128', '256', '1024'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml new file mode 100644 index 0000000000..6b705c00cb --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml @@ -0,0 +1,169 @@ +base: + schema: 2 + name: qwen3.5-fp4-b200-trt-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc18 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: nvidia/Qwen3.5-397B-A17B-NVFP4 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + backend: pytorch + print_iter_log: true + enable_layerwise_nvtx_marker: false + disable_overlap_scheduler: false + enable_iter_perf_stats: true + enable_chunked_prefill: false + stream_interval: 20 + num_postprocess_workers: 4 + scheduler_config: + capacity_scheduler_policy: MAX_UTILIZATION + context_chunking_policy: FIRST_COME_FIRST_SERVED + kv_cache_config: + free_gpu_memory_fraction: 0.9 + enable_block_reuse: false + dtype: fp8 + cuda_graph_config: + enable_padding: true + moe_config: + use_low_precision_moe_combine: true + trust_remote_code: true + max_seq_len: 9472 + max_num_tokens: 32768 + pipeline_parallel_size: 1 + return_perf_metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp2_ep1: + roles: + agg: + args: + enable_attention_dp: false + cuda_graph_config: + max_batch_size: 256 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 256 + tensor_parallel_size: 2 + moe_expert_parallel_size: 1 + benchmark: + env: + CONC: ['4', '16'] +zip_override_tp4_ep1: + roles: + agg: + args: + enable_attention_dp: false + cuda_graph_config: + max_batch_size: 512 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 512 + tensor_parallel_size: 4 + moe_expert_parallel_size: 1 + gpus: 4 + benchmark: + env: + CONC: ['4'] +zip_override_tp2_ep2: + roles: + agg: + args: + enable_attention_dp: false + cuda_graph_config: + max_batch_size: [256, 32] + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: [256, 32] + tensor_parallel_size: 2 + moe_expert_parallel_size: 2 + benchmark: + env: + CONC: ['8', '32'] +zip_override_tp8_ep8: + roles: + agg: + args: + enable_attention_dp: false + cuda_graph_config: + max_batch_size: 512 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 512 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['4'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + enable_attention_dp: true + cuda_graph_config: + max_batch_size: 256 + moe_config: + backend: CUTEDSL + attention_dp_config: + enable_balance: true + batching_wait_iters: 10 + timeout_iters: 500 + max_batch_size: 256 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + gpus: 4 + benchmark: + env: + CONC: ['1024'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + enable_attention_dp: true + cuda_graph_config: + max_batch_size: 128 + moe_config: + backend: CUTEDSL + attention_dp_config: + enable_balance: true + batching_wait_iters: 10 + timeout_iters: 500 + max_batch_size: 128 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['256', '512', '1024'] diff --git a/benchmarks/single_node/srt_eval.sh b/benchmarks/single_node/srt_eval.sh new file mode 100644 index 0000000000..e6c8af8af4 --- /dev/null +++ b/benchmarks/single_node/srt_eval.sh @@ -0,0 +1,27 @@ +#!/usr/bin/env bash + +# SRT owns readiness and lifecycle; InferenceX owns evaluation and its artifacts. +set -eo pipefail +if [[ $# != 2 || -z "$1" || -z "$2" ]]; then + echo "Usage: $0 endpoint status-file" >&2 + exit 1 +fi +SRT_EVAL_STATUS_FILE="$2" +trap 'rc=$?; printf "%s\n" "$rc" > "$SRT_EVAL_STATUS_FILE"' EXIT + +source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" +check_env_vars MODEL MODEL_NAME CONC TP EP_SIZE DP_ATTENTION IS_MULTINODE MAX_MODEL_LEN +export PORT="${1##*:}" +if [[ ! "$PORT" =~ ^[1-9][0-9]*$ || "$IS_MULTINODE" != false ]]; then + echo "ERROR: single-node eval requires a local endpoint and single-node metadata" >&2 + exit 1 +fi +cd "$INFERENCEX_REPO_ROOT" +if [[ -d /model ]]; then + export MODEL_PATH=/model +fi + +eval_rc=0 +run_eval --framework lm-eval --port "$PORT" || eval_rc=$? +append_lm_eval_summary || eval_rc=1 +exit "$eval_rc" diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh new file mode 100644 index 0000000000..0bc79b9ab4 --- /dev/null +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -0,0 +1,64 @@ +#!/usr/bin/env bash + +# SRT owns the server lifecycle; retain the existing InferenceX client and sampler. +set -eo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only +check_env_vars MODEL CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME RESULT_DIR \ + SRT_FRONTEND_HOST SRT_FRONTEND_PORT RUN_EVAL EVAL_ONLY GPU_MONITOR_INTERVAL USE_CHAT_TEMPLATE FRAMEWORK +for name in RUN_EVAL EVAL_ONLY; do + if [[ "${!name}" != true && "${!name}" != false ]]; then + echo "ERROR: $name must be true or false" >&2 + exit 1 + fi +done +case "$FRAMEWORK" in + sglang|atom) CLIENT_BACKEND=vllm ;; + trt) CLIENT_BACKEND=openai ;; + *) echo "ERROR: unsupported fixed-sequence FRAMEWORK: $FRAMEWORK" >&2; exit 1 ;; +esac +SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" +CLIENT_ARGS=() +for argument in "$@"; do + case "$argument" in + --trust-remote-code) CLIENT_ARGS+=("$argument") ;; + *) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;; + esac +done +case "$USE_CHAT_TEMPLATE" in + true) CLIENT_ARGS+=(--use-chat-template) ;; + false) ;; + *) echo "ERROR: USE_CHAT_TEMPLATE must be true or false" >&2; exit 1 ;; +esac + +for name in CONC ISL OSL SRT_FRONTEND_PORT GPU_MONITOR_INTERVAL; do + if [[ ! "${!name}" =~ ^[1-9][0-9]*$ ]]; then + echo "ERROR: $name must be a positive integer" >&2 + exit 1 + fi +done + +if [[ ! -d "$RESULT_DIR" ]]; then + echo "ERROR: RESULT_DIR must be an existing runtime-provided directory" >&2 + exit 1 +fi + +source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" +cd "$INFERENCEX_REPO_ROOT" +pip3 install --break-system-packages sentencepiece datasets pandas + +start_gpu_monitor --output "$RESULT_DIR/gpu_metrics.csv" --interval "$SRT_MONITOR_INTERVAL" +trap 'rc=$?; stop_gpu_monitor; exit "$rc"' EXIT + +run_benchmark_serving \ + --model "$MODEL" \ + --port "$SRT_FRONTEND_PORT" \ + --base-url "http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" \ + --backend "$CLIENT_BACKEND" \ + --input-len "$ISL" \ + --output-len "$OSL" \ + --random-range-ratio "$RANDOM_RANGE_RATIO" \ + --num-prompts "$((CONC * 10))" \ + --max-concurrency "$CONC" \ + --result-filename "$RESULT_FILENAME" \ + --result-dir "$RESULT_DIR" \ + "${CLIENT_ARGS[@]}" diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index 421d55c483..a786bed1e5 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -11,8 +11,14 @@ dsr1-fp4-mi355x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, conc-start: 4, conc-end: 64 } - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 4 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml dsr1-fp4-mi355x-sglang-mtp: image: lmsysorg/sglang:v0.5.12-rocm700-mi35x @@ -27,7 +33,12 @@ dsr1-fp4-mi355x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml dsr1-fp4-mi355x-atom: image: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 @@ -42,8 +53,16 @@ dsr1-fp4-mi355x-atom: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256 } - - { tp: 8, ep: 1, conc-start: 4, conc-end: 4 } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml dsr1-fp4-mi355x-atom-mtp: image: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 @@ -60,7 +79,11 @@ dsr1-fp4-mi355x-atom-mtp: osl: 1024 search-space: #- { tp: 4, conc-start: 32, conc-end: 256, spec-decoding: mtp } - - { tp: 8, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 8 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml dsr1-fp8-mi300x-sglang: image: lmsysorg/sglang:v0.5.12-rocm700-mi30x @@ -75,7 +98,10 @@ dsr1-fp8-mi300x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml dsr1-fp8-mi325x-sglang: image: lmsysorg/sglang:v0.5.19-rocm700-mi30x @@ -90,7 +116,10 @@ dsr1-fp8-mi325x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml dsr1-fp8-mi355x-sglang: image: lmsysorg/sglang:v0.5.12-rocm700-mi35x @@ -105,8 +134,14 @@ dsr1-fp8-mi355x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, conc-start: 32, conc-end: 64 } - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 4 + conc-start: 32 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml dsr1-fp8-mi355x-sglang-mtp: image: lmsysorg/sglang:v0.5.12-rocm700-mi35x @@ -121,7 +156,12 @@ dsr1-fp8-mi355x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml qwen3.5-fp8-mi325x-sglang: image: lmsysorg/sglang:v0.5.12-rocm720-mi30x @@ -136,7 +176,10 @@ qwen3.5-fp8-mi325x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml qwen3.5-fp8-mi355x-sglang: image: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 @@ -151,7 +194,11 @@ qwen3.5-fp8-mi355x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256 } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml qwen3.5-fp8-mi355x-sglang-mtp: image: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 @@ -166,7 +213,12 @@ qwen3.5-fp8-mi355x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml # MI325X official matrix selected from the complete 62-point fast sweep. TP2 # peaks at c4, TP4/TEP4 at c40, and TP8/TEP8 at c64; the next point beyond each @@ -202,9 +254,21 @@ qwen3.5-fp8-mi355x-atom: - isl: 8192 osl: 1024 search-space: - - { tp: 2, ep: 1, conc-start: 4, conc-end: 256 } - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256 } - - { tp: 8, ep: 1, conc-start: 4, conc-end: 256 } + - tp: 2 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml qwen3.5-fp8-mi355x-atom-mtp: image: rocm/atom:rocm7.2.2_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.2.post @@ -219,8 +283,18 @@ qwen3.5-fp8-mi355x-atom-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } - - { tp: 8, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml qwen3.5-fp8-mi355x-sglang-disagg: image: lmsysorg/sglang:v0.5.16-rocm720-mi35x @@ -278,8 +352,14 @@ qwen3.5-fp4-mi355x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 2, conc-start: 4, conc-end: 256 } - - { tp: 4, conc-start: 4, conc-end: 16 } + - tp: 2 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml + - tp: 4 + conc-start: 4 + conc-end: 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml qwen3.5-fp4-mi355x-atom: image: rocm/atom:rocm7.2.2_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.2.post @@ -294,8 +374,14 @@ qwen3.5-fp4-mi355x-atom: - isl: 8192 osl: 1024 search-space: - - { tp: 2, conc-start: 4, conc-end: 256 } - - { tp: 4, conc-start: 4, conc-end: 16 } + - tp: 2 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml + - tp: 4 + conc-start: 4 + conc-end: 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml qwen3.5-fp4-mi355x-sglang-mtp: image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913 @@ -310,8 +396,16 @@ qwen3.5-fp4-mi355x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 2, conc-start: 4, conc-end: 128, spec-decoding: mtp } - - { tp: 4, conc-start: 4, conc-end: 16, spec-decoding: mtp } + - tp: 2 + conc-start: 4 + conc-end: 128 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml + - tp: 4 + conc-start: 4 + conc-end: 16 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml qwen3.5-fp4-mi355x-sglang-agentic-mtp: image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260915 @@ -376,7 +470,10 @@ qwen3.5-fp8-mi300x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml # AgentX Pareto sweep for Qwen3.5 FP8 on MI300X. The fast discovery run peaks # at c24; c32 records the post-knee throughput and latency cliff. @@ -409,7 +506,10 @@ dsr1-fp8-mi355x-atom: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 128 } + - tp: 8 + conc-start: 4 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml dsr1-fp8-mi355x-atom-mtp: image: rocm/atom:rocm7.2.4_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.3 @@ -424,7 +524,11 @@ dsr1-fp8-mi355x-atom-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 8 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml dsr1-fp8-mi355x-sglang-disagg: image: rocm/sgl-dev:sglang-0.5.9-rocm720-mi35x-mori-0227-2 @@ -963,7 +1067,12 @@ dsr1-fp8-mi325x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml qwen3.5-fp8-mi325x-sglang-mtp: image: lmsysorg/sglang:v0.5.12-rocm720-mi30x @@ -978,7 +1087,12 @@ qwen3.5-fp8-mi325x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml dsv4-fp4-mi355x-vllm-agentic-mtp: image: vllm/vllm-openai-rocm:nightly-eed1f3d0c6043bd494424a22443ee198dd56f657 diff --git a/configs/deprecated/nvidia-master.yaml b/configs/deprecated/nvidia-master.yaml index a72715dd2f..9029f6235a 100644 --- a/configs/deprecated/nvidia-master.yaml +++ b/configs/deprecated/nvidia-master.yaml @@ -16549,3 +16549,40 @@ qwen3.5-bf16-b300-sglang-mtp: search-space: - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } - { tp: 4, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + +# Retired on 2026-09-22: Docker fixed-sequence coverage is outside the SRT-only cutover. +# Qwen3.5-397B-A17B NVFP4 single-node SGLang sweep using 4 of 8 RTX PRO +# 6000 Blackwell GPUs. Both arms use ordinary NCCL collectives on PCIe. +qwen3.5-fp4-rtx6000pro-sglang: + image: lmsysorg/sglang:v0.5.16-cu130 + model: nvidia/Qwen3.5-397B-A17B-NVFP4 + model-prefix: qwen3.5 + runner: rtx6000pro-lat + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 4, conc-list: [1, 4, 16, 64] } + - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64] } + +# Same sweep with the built-in MTP draft head driven through SGLang's EAGLE +# speculative path. +qwen3.5-fp4-rtx6000pro-sglang-mtp: + image: lmsysorg/sglang:v0.5.16-cu130 + model: nvidia/Qwen3.5-397B-A17B-NVFP4 + model-prefix: qwen3.5 + runner: rtx6000pro-lat + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } + - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 02cd60d23e..7ee2ca0bee 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -872,8 +872,17 @@ dsr1-fp4-b200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 1, conc-end: 32 } - - { tp: 4, ep: 4, dp-attn: true, conc-start: 64, conc-end: 256 } + - tp: 4 + ep: 1 + conc-start: 1 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-start: 64 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml # agentic-coding: temporarily disabled — blocked by e2e-tests.yml artifact # name mismatch (downloads `agentic_*` but benchmark-tmpl.yml uploads as # `bmk_agentic_*`). Re-enable once that workflow is aligned. @@ -896,8 +905,19 @@ dsr1-fp4-b200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 1, conc-end: 32, spec-decoding: mtp } - - { tp: 4, ep: 4, dp-attn: true, conc-start: 64, conc-end: 256, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 1 + conc-end: 32 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-start: 64 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml dsv4-fp4-b200-sglang-agentic-hicache-mtp: image: lmsysorg/sglang:v0.5.19-cu130 @@ -948,8 +968,16 @@ dsr1-fp4-b300-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 4, conc-start: 1, conc-end: 128 } - - { tp: 8, ep: 8, conc-start: 1, conc-end: 16 } + - tp: 4 + ep: 4 + conc-start: 1 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml + - tp: 8 + ep: 8 + conc-start: 1 + conc-end: 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml dsr1-fp4-b200-trt: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -967,11 +995,31 @@ dsr1-fp4-b200-trt: # low concurrency cases use TP only # concurrency 32 uses TP & EP # high concurrency cases use TP & EP & DP-ATTN - - { tp: 4, conc-start: 4, conc-end: 32 } - - { tp: 4, ep: 4, conc-start: 32, conc-end: 32 } - - { tp: 4, ep: 4, dp-attn: true, conc-start: 256, conc-end: 256 } - - { tp: 8, conc-start: 4, conc-end: 4 } - - { tp: 8, ep: 8, dp-attn: true, conc-start: 128, conc-end: 256 } + - tp: 4 + conc-start: 4 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml + - tp: 4 + ep: 4 + conc-start: 32 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-start: 256 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml + - tp: 8 + conc-start: 4 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-start: 128 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml dsr1-fp4-b200-trt-mtp: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -987,12 +1035,37 @@ dsr1-fp4-b200-trt-mtp: osl: 1024 search-space: # TP=4 configurations - - { tp: 4, conc-start: 4, conc-end: 16, spec-decoding: mtp } - - { tp: 4, ep: 4, conc-start: 32, conc-end: 32, spec-decoding: mtp } - - { tp: 4, ep: 4, dp-attn: true, conc-start: 256, conc-end: 256, spec-decoding: mtp } + - tp: 4 + conc-start: 4 + conc-end: 16 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 4 + ep: 4 + conc-start: 32 + conc-end: 32 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-start: 256 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml # TP=8 configurations - - { tp: 8, conc-start: 4, conc-end: 4, spec-decoding: mtp } - - { tp: 8, ep: 8, dp-attn: true, conc-start: 64, conc-end: 256, spec-decoding: mtp } + - tp: 8 + conc-start: 4 + conc-end: 4 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-start: 64 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml dsr1-fp8-b200-sglang: image: lmsysorg/sglang:v0.5.12-cu130 @@ -1007,8 +1080,16 @@ dsr1-fp8-b200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 4 } - - { tp: 4, ep: 1, conc-start: 1, conc-end: 32 } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 1 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml # NOTE: At the time of submission, https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 # does not have a B300-specific recipe, so this config reuses the existing DSR1 FP8 @@ -1026,8 +1107,16 @@ dsr1-fp8-b300-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 4 } - - { tp: 4, ep: 1, conc-start: 1, conc-end: 32 } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 1 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml dsv4-fp4-b300-sglang-agentic-hicache-mtp: image: lmsysorg/sglang:nightly-dev-20260901-07c8f729 @@ -1069,9 +1158,23 @@ qwen3.5-fp8-b200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 1, conc-end: 4 } - - { tp: 4, ep: 1, conc-start: 2, conc-end: 512 } - - { tp: 4, ep: 1, conc-list: [320, 384, 448, 640] } + - tp: 8 + conc-start: 1 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 2 + conc-end: 512 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-list: + - 320 + - 384 + - 448 + - 640 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml qwen3.5-fp8-b200-sglang-agentic-mtp: image: lmsysorg/sglang:nightly-dev-cu13-20260914-4358a161 @@ -1103,8 +1206,16 @@ qwen3.5-fp4-b200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 4 } - - { tp: 2, ep: 1, conc-start: 4, conc-end: 128 } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml + - tp: 2 + ep: 1 + conc-start: 4 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml qwen3.5-fp4-b200-sglang-mtp: image: lmsysorg/sglang:v0.5.14-cu130 @@ -1119,9 +1230,26 @@ qwen3.5-fp4-b200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 4, spec-decoding: mtp } - - { tp: 2, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } - - { tp: 2, ep: 2, conc-list: [16, 32, 64], spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 4 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml + - tp: 2 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml + - tp: 2 + ep: 2 + conc-list: + - 16 + - 32 + - 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml qwen3.5-fp8-b200-sglang-mtp: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 @@ -1136,8 +1264,18 @@ qwen3.5-fp8-b200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 4, spec-decoding: mtp } - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 4 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml qwen3.5-fp8-b300-sglang-mtp: image: lmsysorg/sglang:v0.5.12-cu130 @@ -1152,7 +1290,12 @@ qwen3.5-fp8-b300-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml qwen3.5-fp8-b300-sglang: image: lmsysorg/sglang:v0.5.12-cu130 @@ -1167,7 +1310,11 @@ qwen3.5-fp8-b300-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256 } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml qwen3.5-fp4-b300-sglang: image: lmsysorg/sglang:v0.5.14-cu130 @@ -1182,44 +1329,16 @@ qwen3.5-fp4-b300-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 128 } - - { tp: 2, ep: 2, conc-start: 4, conc-end: 128 } - -# Qwen3.5-397B-A17B NVFP4 single-node SGLang sweep using 4 of 8 RTX PRO -# 6000 Blackwell GPUs. Both arms use ordinary NCCL collectives on PCIe. -qwen3.5-fp4-rtx6000pro-sglang: - image: lmsysorg/sglang:v0.5.16-cu130 - model: nvidia/Qwen3.5-397B-A17B-NVFP4 - model-prefix: qwen3.5 - runner: rtx6000pro-lat - precision: fp4 - framework: sglang - multinode: false - scenarios: - fixed-seq-len: - - isl: 8192 - osl: 1024 - search-space: - - { tp: 4, conc-list: [1, 4, 16, 64] } - - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64] } - -# Same sweep with the built-in MTP draft head driven through SGLang's EAGLE -# speculative path. -qwen3.5-fp4-rtx6000pro-sglang-mtp: - image: lmsysorg/sglang:v0.5.16-cu130 - model: nvidia/Qwen3.5-397B-A17B-NVFP4 - model-prefix: qwen3.5 - runner: rtx6000pro-lat - precision: fp4 - framework: sglang - multinode: false - scenarios: - fixed-seq-len: - - isl: 8192 - osl: 1024 - search-space: - - { tp: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } - - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml + - tp: 2 + ep: 2 + conc-start: 4 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml qwen3.5-fp4-b300-sglang-mtp: image: lmsysorg/sglang:v0.5.19-cu130 @@ -1234,8 +1353,18 @@ qwen3.5-fp4-b300-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 128, spec-decoding: mtp } - - { tp: 2, ep: 2, conc-start: 4, conc-end: 128, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 128 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml + - tp: 2 + ep: 2 + conc-start: 4 + conc-end: 128 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml # Kimi K3 is a 2.8T MXFP4 MoE served on H200 nodes. # These are the three aggregated strategy classes from the official H200 @@ -1343,7 +1472,12 @@ dsr1-fp8-b200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 512, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 512 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml # NOTE: At the time of submission, https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 # does not have a B300-specific recipe, so this config reuses the existing DSR1 FP8 @@ -1361,7 +1495,12 @@ dsr1-fp8-b300-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 512, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 512 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml kimik3-fp4-b300-vllm-agentic-dspark: # TP8 x DCP8 with Mooncake as the external KV tier. The recipe drafts with @@ -1405,9 +1544,21 @@ dsr1-fp8-b200-trt: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 64, conc-end: 256 } - - { tp: 4, ep: 1, conc-start: 8, conc-end: 32 } - - { tp: 8, ep: 1, conc-start: 4, conc-end: 8 } + - tp: 8 + ep: 1 + conc-start: 64 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 8 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 8 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml dsr1-fp8-b200-trt-mtp: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -1423,7 +1574,12 @@ dsr1-fp8-b200-trt-mtp: osl: 1024 search-space: # TP8 for all points - - { tp: 8, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml dsr1-fp8-h200-sglang: image: lmsysorg/sglang:v0.5.12-cu130 @@ -1438,7 +1594,10 @@ dsr1-fp8-h200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml dsr1-fp8-h200-sglang-mtp: image: lmsysorg/sglang:v0.5.19-cu130 @@ -1453,7 +1612,12 @@ dsr1-fp8-h200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml # DeepSeek-V4-Pro AgentX on one aggregated TP8 H200 worker. Keep the serving # topology fixed and sweep only concurrency to produce the Pareto curve. @@ -1519,7 +1683,11 @@ qwen3.5-fp8-h200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml qwen3.5-fp8-h200-sglang-mtp: image: lmsysorg/sglang:v0.5.14-cu130 @@ -1534,7 +1702,12 @@ qwen3.5-fp8-h200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 8, conc-start: 4, conc-end: 128, spec-decoding: mtp } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 128 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml dsr1-fp8-h200-trt: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -1551,8 +1724,17 @@ dsr1-fp8-h200-trt: osl: 1024 # If CONC > 32, then DP_ATTN=true search-space: - - { tp: 8, ep: 8, conc-start: 4, conc-end: 32 } - - { tp: 8, ep: 8, dp-attn: true, conc-start: 64, conc-end: 64 } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-start: 64 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml dsr1-fp8-h200-trt-mtp: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -1569,8 +1751,19 @@ dsr1-fp8-h200-trt-mtp: osl: 1024 search-space: # If CONC >= 64, then DP_ATTN=true, MTP=1 - - { tp: 8, ep: 8, conc-start: 4, conc-end: 32, spec-decoding: mtp } - - { tp: 8, ep: 8, dp-attn: true, conc-start: 64, conc-end: 256, spec-decoding: mtp } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 32 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-start: 64 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml dsr1-fp8-h200-dynamo-trt: image: nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1 @@ -3950,8 +4143,16 @@ qwen3.5-fp8-h100-sglang: osl: 1024 require-power: true search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 8 } - - { tp: 8, ep: 8, conc-start: 16, conc-end: 256 } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 8 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml + - tp: 8 + ep: 8 + conc-start: 16 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml qwen3.5-fp8-h100-sglang-mtp: image: lmsysorg/sglang:v0.5.19-cu130 @@ -3967,7 +4168,12 @@ qwen3.5-fp8-h100-sglang-mtp: osl: 1024 require-power: true search-space: - - { tp: 8, ep: 8, conc-start: 4, conc-end: 32, spec-decoding: mtp } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 32 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml qwen3.5-fp4-gb300-dynamo-sglang: image: lmsysorg/sglang:v0.5.14-cu130 @@ -5117,12 +5323,42 @@ qwen3.5-fp4-b200-trt: - isl: 8192 osl: 1024 search-space: - - { tp: 2, ep: 1, conc-list: [4, 16] } - - { tp: 4, ep: 1, conc-list: [4] } - - { tp: 2, ep: 2, conc-list: [8, 32] } - - { tp: 8, ep: 8, conc-list: [4] } - - { tp: 4, ep: 4, dp-attn: true, conc-list: [1024] } - - { tp: 8, ep: 8, dp-attn: true, conc-list: [256, 512, 1024] } + - tp: 2 + ep: 1 + conc-list: + - 4 + - 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 4 + ep: 1 + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 2 + ep: 2 + conc-list: + - 8 + - 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 8 + ep: 8 + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-list: + - 1024 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-list: + - 256 + - 512 + - 1024 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml qwen3.5-fp4-b200-trt-mtp: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc18 @@ -5137,11 +5373,40 @@ qwen3.5-fp4-b200-trt-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 2, ep: 1, spec-decoding: "mtp", conc-list: [4] } - - { tp: 2, ep: 2, spec-decoding: "mtp", conc-list: [8, 16] } - - { tp: 4, ep: 4, spec-decoding: "mtp", conc-list: [4] } - - { tp: 8, ep: 8, spec-decoding: "mtp", conc-list: [4] } - - { tp: 8, ep: 8, dp-attn: true, spec-decoding: "mtp", conc-list: [128, 256, 1024] } + - tp: 2 + ep: 1 + spec-decoding: mtp + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 2 + ep: 2 + spec-decoding: mtp + conc-list: + - 8 + - 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 4 + ep: 4 + spec-decoding: mtp + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 8 + ep: 8 + spec-decoding: mtp + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + spec-decoding: mtp + conc-list: + - 128 + - 256 + - 1024 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml minimaxm3-fp8-h100-vllm-agentic-mtp: image: vllm/vllm-openai:v0.27.1 diff --git a/configs/runners.yaml b/configs/runners.yaml index 2b85b54431..b9ede289c5 100644 --- a/configs/runners.yaml +++ b/configs/runners.yaml @@ -25,16 +25,16 @@ labels: h200: - h200-cw_00 - h200-cw_01 - - h200-dgxc-slurm_0 - - h200-dgxc-slurm_1 - - h200-dgxc-slurm_2 - - h200-dgxc-slurm_3 - - h200-dgxc-slurm_4 - - h200-dgxc-slurm_5 - - h200-dgxc-slurm_6 - - h200-dgxc-slurm_7 - - h200-dgxc-slurm_8 - - h200-dgxc-slurm_9 + - h200-dgxc-slurm_00 + - h200-dgxc-slurm_01 + - h200-dgxc-slurm_02 + - h200-dgxc-slurm_03 + - h200-dgxc-slurm_04 + - h200-dgxc-slurm_05 + - h200-dgxc-slurm_06 + - h200-dgxc-slurm_07 + - h200-dgxc-slurm_08 + - h200-dgxc-slurm_09 - h200-dgxc-slurm_10 - h200-dgxc-slurm_11 - h200-dgxc-slurm_12 @@ -138,10 +138,6 @@ labels: - gb300-nv_15 - gb300-nv_16 - gb300-nv_17 - rtx6000pro: - - rtx6000pro-lat_00 - rtx6000pro-lat: - - rtx6000pro-lat_00 cluster:h100-cw: - h100-cw_00 - h100-cw_01 @@ -170,16 +166,16 @@ labels: - h200-cw_00 - h200-cw_01 cluster:h200-dgxc: - - h200-dgxc-slurm_0 - - h200-dgxc-slurm_1 - - h200-dgxc-slurm_2 - - h200-dgxc-slurm_3 - - h200-dgxc-slurm_4 - - h200-dgxc-slurm_5 - - h200-dgxc-slurm_6 - - h200-dgxc-slurm_7 - - h200-dgxc-slurm_8 - - h200-dgxc-slurm_9 + - h200-dgxc-slurm_00 + - h200-dgxc-slurm_01 + - h200-dgxc-slurm_02 + - h200-dgxc-slurm_03 + - h200-dgxc-slurm_04 + - h200-dgxc-slurm_05 + - h200-dgxc-slurm_06 + - h200-dgxc-slurm_07 + - h200-dgxc-slurm_08 + - h200-dgxc-slurm_09 - h200-dgxc-slurm_10 - h200-dgxc-slurm_11 - h200-dgxc-slurm_12 @@ -239,8 +235,6 @@ labels: - gb300-nv_15 - gb300-nv_16 - gb300-nv_17 - cluster:rtx6000pro-lat: - - rtx6000pro-lat_00 cluster:mi300x-amd: - mi300x-amd_00 - mi300x-amd_01 @@ -299,9 +293,6 @@ hardware: cluster:gb300-nv: available-cpu-dram-mib: 860_160 gpus-per-node: 4 - cluster:rtx6000pro-lat: - available-cpu-dram-mib: 1_500_000 - gpus-per-node: 8 cluster:mi300x-amd: available-cpu-dram-mib: 1_547_820 gpus-per-node: 8 diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 3d129b289b..9f083eae9e 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -35,6 +35,15 @@ git submodule update --init To upgrade, fetch and check out the desired commit inside the relevant submodule, then commit the updated submodule pointer in InferenceX. Benchmark workflows already initialize submodules. Slurm launchers make a local Git clone for each job so recipe staging and runtime writes do not modify the submodule, and record the actual commit for result provenance. NVIDIA setup clones locally; TileRT setup fetches its pinned fork commit over the network. +Single-node fixed-sequence recipes use NVIDIA upstream srt-slurm. ATOM recipes use +the native `atomesh` frontend with one aggregate worker and +`enable_multiple_frontends: false`. The router's pinned official image belongs in +`frontend.container_image`: older benchmark worker images do not include AToMesh. +Keep `model.container` aligned with the master config's worker `image`; changing the +router image does not require changing the worker image. TRT-LLM recipes use native +`engine.served_model_name`, without duplicating that flag in `roles.agg.extra_args`. +The former fork's direct ATOM frontend is not required. + ### Cluster profiles Launchers that use srt-slurm keep their cluster configuration in diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index a530d17153..83fd0bee97 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -35,6 +35,13 @@ git submodule update --init 升级时,在对应子模块中获取并检出目标提交,再将更新后的子模块指针提交到 InferenceX。基准测试工作流已配置为自动初始化子模块。Slurm 启动器为每个作业创建本地 Git 克隆,避免配方准备和运行时写入修改子模块,并记录实际提交以供结果溯源。NVIDIA 启动器使用本地克隆;TileRT 启动器通过网络获取固定的分支提交。 +单节点固定序列长度配方使用 NVIDIA 上游 srt-slurm。ATOM 配方使用原生 `atomesh` +frontend、一个聚合 worker,并设置 `enable_multiple_frontends: false`。旧版基准 worker +镜像不包含 AToMesh,因此通过 `frontend.container_image` 单独固定路由器的官方镜像。 +`model.container` 必须与主配置中的 worker `image` 一致;更换路由器镜像无需更换 worker +镜像。TRT-LLM 配方使用原生 `engine.served_model_name`,不再通过 `roles.agg.extra_args` +重复传入该参数。不再依赖此前分叉中的 ATOM 直连 frontend。 + ## 规程索引 1. [准备 worktree](#准备-worktree) diff --git a/infx/matrix/generate.py b/infx/matrix/generate.py index 5bbad46d3c..af7ec486fd 100644 --- a/infx/matrix/generate.py +++ b/infx/matrix/generate.py @@ -850,6 +850,8 @@ def _fixed_sequence_entries( Fields.SPEC_DECODING.value: spec_decoding, } ) + if benchmark.get(Fields.SRT_RECIPE.value) is not None: + entry[Fields.SRT_RECIPE.value] = benchmark[Fields.SRT_RECIPE.value] entry.update( { Fields.EXP_NAME.value: f"{model_code}_{seq_len_to_str(isl, osl)}", diff --git a/infx/matrix/validation.py b/infx/matrix/validation.py index d3503c7f2f..7f774570b9 100644 --- a/infx/matrix/validation.py +++ b/infx/matrix/validation.py @@ -45,6 +45,7 @@ class Fields(Enum): # Search-space/benchmark fields TP = "tp" + SRT_RECIPE = "srt-recipe" PP = "pp" DCP_SIZE = "dcp-size" PCP_SIZE = "pcp-size" @@ -162,6 +163,7 @@ class SingleNodeMatrixEntry(BaseModel): model_config = ConfigDict(extra="forbid", populate_by_name=True) image: str + srt_recipe: str | None = Field(default=None, alias=Fields.SRT_RECIPE.value, min_length=1) model: str model_prefix: str = Field(alias=Fields.MODEL_PREFIX.value) precision: str @@ -534,6 +536,7 @@ class SingleNodeSearchSpaceEntry(BaseModel): model_config = ConfigDict(extra="forbid", populate_by_name=True) tp: int + srt_recipe: str | None = Field(default=None, alias=Fields.SRT_RECIPE.value, min_length=1) pp: int = Field(default=1, gt=0, strict=True) dcp_size: int = Field(default=1, alias=Fields.DCP_SIZE.value, gt=0, strict=True) pcp_size: int = Field(default=1, alias=Fields.PCP_SIZE.value, gt=0, strict=True) diff --git a/infx/srt_slurm/cluster_config.py b/infx/srt_slurm/cluster_config.py index 4ce1154400..4990692541 100644 --- a/infx/srt_slurm/cluster_config.py +++ b/infx/srt_slurm/cluster_config.py @@ -37,6 +37,7 @@ def main() -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("profile", type=Path) parser.add_argument("output", type=Path) + parser.add_argument("--exclusive", action="store_true", help="Request exclusive Slurm nodes") parser.add_argument("--var", nargs=2, action="append", default=[], metavar=("NAME", "VALUE")) parser.add_argument("--model", nargs=2, action="append", default=[], metavar=("ALIAS", "PATH")) parser.add_argument( @@ -55,6 +56,8 @@ def main() -> None: dict(args.var), {"model_paths": args.model, "containers": args.container, "default_mounts": args.mount}, ) + if args.exclusive: + config["use_exclusive_sbatch_directive"] = True args.output.write_text(yaml.safe_dump(config, sort_keys=False)) except (OSError, ValueError, KeyError, yaml.YAMLError) as exc: parser.error(str(exc)) diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py new file mode 100644 index 0000000000..31655f6dc7 --- /dev/null +++ b/infx/srt_slurm/single_node.py @@ -0,0 +1,209 @@ +"""Bind a native single-node SRT recipe to one fixed-sequence matrix point.""" + +from __future__ import annotations + +import argparse +import json +import os +import re +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +import yaml + +from infx.srt_slurm.synthetic_acceptance import ENGINES, selected_recipes, spec_parameters + +SINGLE_NODE_ENGINES = {**ENGINES, "atom": "atom"} + + +def parallelism_constraints( + engine: str, args: Mapping[str, Any], environment: Mapping[str, str] +) -> dict[str, tuple[Any, Any]]: + """Read each engine's native topology fields without translating the recipe.""" + tp, ep = int(environment["TP"]), int(environment["EP_SIZE"]) + dp_attention = environment["DP_ATTENTION"] == "true" + if engine == "sglang": + return { + "tensor-parallel-size": (args["tensor-parallel-size"], tp), + "data-parallel-size": (args.get("data-parallel-size", 1), tp if dp_attention else 1), + "expert-parallel-size": (args.get("expert-parallel-size", args.get("ep-size", 1)), ep), + "DP_ATTENTION": (args.get("enable-dp-attention", False), dp_attention), + } + if engine == "trtllm": + return { + "tensor_parallel_size": (args["tensor_parallel_size"], tp), + "moe_expert_parallel_size": (args["moe_expert_parallel_size"], ep), + "pipeline_parallel_size": (args.get("pipeline_parallel_size", 1), 1), + "DP_ATTENTION": (args.get("enable_attention_dp", False), dp_attention), + } + if engine == "atom": + if ep not in {1, tp}: + raise ValueError("ATOM expert parallelism must be 1 or TP") + return { + "enable-expert-parallel": (args.get("enable-expert-parallel", False), ep > 1), + "DP_ATTENTION": (args.get("enable-dp-attention", False), dp_attention), + } + raise ValueError(f"Unsupported single-node SRT engine: {engine!r}") + + +def select_recipe(config: str, environment: Mapping[str, str]) -> tuple[str, dict[str, Any]]: + """Resolve a matrix point to one native variant, never submit an entire sweep.""" + path, _, selector = config.partition(":") + raw = yaml.safe_load(Path(path).read_text()) + if not isinstance(raw, dict): + raise ValueError("Recipe must be a mapping") + recipes = selected_recipes(raw, selector or None) + matches = [] + errors = [] + for name, recipe in recipes: + try: + validate_recipe(recipe, environment) + except ValueError as exc: + errors.append(f"{name}: {exc}") + else: + matches.append((f"{path}:{name}" if name else path, recipe)) + if len(matches) != 1: + detail = "; ".join(errors) if not matches else ", ".join(name for name, _ in matches) + raise ValueError(f"Expected exactly one matching single-node SRT recipe; {detail}") + return matches[0] + + +def validate_recipe(recipe: dict[str, Any], environment: Mapping[str, str]) -> None: + """Reject metadata mismatches without overwriting recipe-owned server settings.""" + role = recipe["roles"]["agg"] + args = role["args"] + benchmark = recipe["benchmark"] + workload = benchmark["env"] + engine_config = recipe["engine"] + engine = engine_config["type"] if isinstance(engine_config, dict) else engine_config + if environment["FRAMEWORK"] not in {"sglang", "trt", "atom"}: + raise ValueError(f"Unsupported single-node framework: {environment['FRAMEWORK']!r}") + spec = spec_parameters(role, engine) + if spec and spec["method"] not in {"eagle", "nextn", "mtp"}: + raise ValueError("Single-node SRT supports only native MTP or no speculation") + speculation = "mtp" if spec else "none" + expected = { + "engine": (engine, SINGLE_NODE_ENGINES[environment["FRAMEWORK"]]), + "model": (recipe["model"]["path"], f"hf:{environment['MODEL']}"), + "image": (recipe["model"]["container"], environment["IMAGE"]), + "precision": (recipe["model"]["precision"], environment["PRECISION"]), + **parallelism_constraints(engine, args, environment), + "gpus": (role["gpus"], int(environment["GPU_COUNT"])), + "nodes": (role["nodes"], 1), + "workers": (role["workers"], 1), + "roles": (set(recipe["roles"]), {"agg"}), + "benchmark type": (benchmark["type"], "custom"), + "benchmark MODEL": (workload["MODEL"], environment["MODEL"]), + "SPEC_DECODING": (speculation, environment["SPEC_DECODING"]), + "USE_CHAT_TEMPLATE": (workload["USE_CHAT_TEMPLATE"], "true" if spec else "false"), + } + if "CONC" in workload: + expected["CONC"] = (str(workload["CONC"]), environment["CONC"]) + if engine == "atom": + # Native ATOM derives -tp from the aggregate worker's GPU allocation. + expected["ATOM TP"] = (role["gpus"], int(environment["TP"])) + for name in ("ISL", "OSL", "RANDOM_RANGE_RATIO"): + expected[name] = (str(workload[name]), environment[name]) + # Multi-node and AgentX workloads use their existing connector. + for name, value in { + "PP_SIZE": "1", + "DCP_SIZE": "1", + "PCP_SIZE": "1", + "IS_AGENTIC": "0", + }.items(): + expected[name] = (environment[name], value) + for name, (actual, wanted) in expected.items(): + if actual != wanted: + raise ValueError(f"Single-node SRT {name}: recipe/matrix {actual!r} != {wanted!r}") + + +def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: + """Bind only runtime-owned values after validating the selected recipe.""" + _, recipe = select_recipe(config, environment) + for name in ("RUN_EVAL", "EVAL_ONLY", "DP_ATTENTION"): + if environment[name] not in {"true", "false"}: + raise ValueError(f"{name} must be true or false") + # Exclusive nodes include idle GPUs. Restrict each server/client step to + # the serving GPU count so client-side power collection sees the same set. + overrides = ["--set", f"srun_options.gpus-per-node={json.dumps(environment['GPU_COUNT'])}"] + # Match the legacy container working directory using the existing repo mount. + # PyTorch's generated module imports fail from / with PYTHONPYCACHEPREFIX set. + overrides += ["--set", 'srun_options.container-workdir="/infmax-workspace"'] + if environment.get("SRT_SRUN_OPTIONS"): + options = json.loads(environment["SRT_SRUN_OPTIONS"]) + if not isinstance(options, dict) or any( + not re.fullmatch(r"[a-z][a-z0-9-]*", key) or not isinstance(value, str) + for key, value in options.items() + ): + raise ValueError("SRT_SRUN_OPTIONS must map option names to string values") + # Native --set preserves whole mappings as JSON strings for engine + # flags. Runtime option mappings therefore need individual leaf sets. + for key, value in options.items(): + overrides += ["--set", f"srun_options.{key}={json.dumps(value)}"] + for name in ( + "CONC", + "RESULT_FILENAME", + "GPU_MONITOR_INTERVAL", + "RUN_EVAL", + "EVAL_ONLY", + "FRAMEWORK", + ): + value = environment[name] + if not value: + raise ValueError(f"Missing runtime input: {name}") + # Native --set broadcasts into zip groups. CONC already matched above; + # replacing its list could collapse the selected variant's index. + if name == "CONC" and name in recipe["benchmark"]["env"]: + continue + overrides += ["--set", f"benchmark.env.{name}={json.dumps(value)}"] + if environment["EVAL_ONLY"] == "true": + context = int(environment["MAX_MODEL_LEN"]) + if context <= 0: + raise ValueError("MAX_MODEL_LEN must be positive") + context_keys = { + "sglang": ("context-length",), + "trt": ("max_seq_len", "max_num_tokens"), + "atom": ("max-model-len",), + }[environment["FRAMEWORK"]] + for key in context_keys: + overrides += ["--set", f"roles.agg.args.{key}={context}"] + return [*overrides, "--set", 'benchmark.env.RESULT_DIR="/logs"'] + + +def submission_fields(path: Path) -> tuple[str, str]: + """Accept exactly one successful native JSON submission, never scrape prose.""" + record = json.loads(path.read_text()) + if record.get("status") != "submitted": + raise ValueError("SRT did not submit a job") + job_id = str(record["slurm_job_id"]) + output = str(record["output_dir"]) + if not job_id.isascii() or not job_id.isdecimal() or int(job_id) <= 0: + raise ValueError("Invalid SRT Slurm job ID") + if not Path(output).is_absolute() or "\n" in output: + raise ValueError("SRT output directory must be absolute") + return job_id, output + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + commands = parser.add_subparsers(dest="command", required=True) + prepare = commands.add_parser("prepare") + prepare.add_argument("recipe") + prepare.add_argument("output", type=Path) + submitted = commands.add_parser("submission") + submitted.add_argument("manifest", type=Path) + parsed = parser.parse_args() + try: + if parsed.command == "prepare": + config, _ = select_recipe(parsed.recipe, os.environ) + arguments = runtime_arguments(parsed.recipe, os.environ) + parsed.output.write_bytes("\0".join([config, *arguments, ""]).encode()) + else: + print("\n".join(submission_fields(parsed.manifest))) + except (OSError, ValueError, KeyError, TypeError, yaml.YAMLError) as exc: + parser.error(str(exc)) + + +if __name__ == "__main__": + main() diff --git a/infx/srt_slurm/synthetic_acceptance.py b/infx/srt_slurm/synthetic_acceptance.py index 95d06b7763..0a9aa05403 100644 --- a/infx/srt_slurm/synthetic_acceptance.py +++ b/infx/srt_slurm/synthetic_acceptance.py @@ -19,6 +19,7 @@ GOLDEN_DIR = Path(__file__).resolve().parents[2] / "golden_al_distribution" ENGINES = { + "sglang": "sglang", "vllm": "vllm", "dynamo-vllm": "vllm", "dynamo-sglang": "sglang", @@ -35,6 +36,14 @@ def spec_parameters(role: Mapping[str, Any], engine: str) -> dict[str, Any]: args = role.get("args", {}) + if engine == "atom": + method = args.get("method") + if not method: + return {} + return { + "method": str(method).lower(), + "num_speculative_tokens": args.get("num-speculative-tokens"), + } if engine == "vllm": raw = args.get("speculative-config") if raw is None: @@ -226,7 +235,7 @@ def plan_commands( """Build native arguments for every selected variant before submitting jobs.""" command = ["srtctl", "apply", *arguments] if framework not in ENGINES: - return [command] + return [[*command, "--file", config]] from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides path, _, selector = config.partition(":") diff --git a/runners/launch_b200-cw.sh b/runners/launch_b200-cw.sh index 8be2bfd9dd..655599e755 100644 --- a/runners/launch_b200-cw.sh +++ b/runners/launch_b200-cw.sh @@ -1,7 +1,23 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC + +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + HF_HUB_CACHE_MOUNT=/tmp/gharunner/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + SRT_SQUASH_FILE="/tmp/gharunner/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node b200-cw + exit $? +fi export HF_HUB_CACHE_MOUNT="/tmp/gharunner/hf-hub-cache" export PORT=8888 diff --git a/runners/launch_b200-nb.sh b/runners/launch_b200-nb.sh index 5ea7691e39..419af005de 100644 --- a/runners/launch_b200-nb.sh +++ b/runners/launch_b200-nb.sh @@ -1,7 +1,24 @@ #!/usr/bin/bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC + +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + HF_HUB_CACHE_MOUNT=/mnt/data/gharunners/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + unset SRT_SQUASH_FILE + export UCX_NET_DEVICES=eth0 + launch_srt_single_node b200-nb + exit $? +fi HF_HUB_CACHE_MOUNT="/mnt/data/gharunners/hf-hub-cache/" PARTITION="main" diff --git a/runners/launch_b200-nscale-slurm.sh b/runners/launch_b200-nscale-slurm.sh index cd238f252e..f010f76765 100755 --- a/runners/launch_b200-nscale-slurm.sh +++ b/runners/launch_b200-nscale-slurm.sh @@ -4,13 +4,14 @@ # # The reusable workflows run runners/launch_${RUNNER_NAME%%_*}.sh, so every # b200-nscale-slurm_* runner enters here and this is the pool's only launcher. -# Three execution paths share the file and are selected once, below: +# Execution paths share the file and are selected once, below: # native-srt multi-node lanes whose srt-slurm recipes are maintained # against this cluster (DSV4 / Kimi K3 / GLM-5.2 # FP4 and GLM-5.1 FP8 TileRT) # multinode-srt every other multi-node job, through srt-slurm with the # cluster-wide model table -# single-node salloc + srun of the benchmarks/single_node script +# native-single-node fixed-sequence jobs, which require an SRT recipe +# agentic salloc + srun of the existing AgentX script source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 check_env_vars EVAL_ONLY IS_AGENTIC IS_MULTINODE RUN_EVAL # Exported for this pool by runners/runtime_settings.sh. @@ -49,8 +50,11 @@ if uses_native_srt_lane; then LAUNCH_PATH="native-srt" elif [[ "$IS_MULTINODE" == "true" ]]; then LAUNCH_PATH="multinode-srt" +elif [[ "$IS_AGENTIC" == "0" ]]; then + check_env_vars SRT_RECIPE + LAUNCH_PATH="native-single-node" else - LAUNCH_PATH="single-node" + LAUNCH_PATH="agentic" fi echo "B200 Nscale launch path: $LAUNCH_PATH" @@ -148,6 +152,15 @@ else exit 1 fi +if [[ "$LAUNCH_PATH" == native-single-node ]]; then + HF_HUB_CACHE_MOUNT=/data/home/sa-shared/gharunners/hf-hub-cache + SRT_MODEL_PATH="$MODEL_PATH" + SRT_SQUASH_FILE="$B200_SQUASH_DIR/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node b200-nscale-slurm \ + --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" + exit $? +fi + # --------------------------------------------------------------------------- # Container import helpers shared by both srt-slurm paths # --------------------------------------------------------------------------- @@ -754,10 +767,10 @@ run_multinode_srt() { } # --------------------------------------------------------------------------- -# single-node: salloc + srun of the benchmarks/single_node script +# agentic: salloc + srun of the existing AgentX script # --------------------------------------------------------------------------- -run_single_node() { +run_agentic() { # The runner lease reserves the Slurm nodes before this single-node job is # submitted to the Nscale batch_1 partition. check_env_vars SALLOC_TIME_LIMIT GPU_COUNT @@ -834,5 +847,5 @@ run_single_node() { case "$LAUNCH_PATH" in native-srt) run_native_srt_lane ;; multinode-srt) run_multinode_srt ;; - single-node) run_single_node ;; + agentic) run_agentic ;; esac diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index a9c21d65f9..ca270b217e 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -8,7 +8,7 @@ source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 # B300 DSXE Slurm cluster (dsxe-sa-b300-prd0); runners run as sa-gha-runner. # Cluster-specific facts live in this block. Multi-node jobs go through -# srt-slurm/srtctl, single-node jobs through salloc + pyxis. +# srt-slurm/srtctl; AgentX and explicit collector scripts retain salloc + pyxis. SLURM_PARTITION="batch_1" SLURM_ACCOUNT="benchmark" @@ -77,7 +77,6 @@ STAGED_MODELS=( Qwen3.8-2.4T-A95B-FP8 ) -mkdir -p "$SQUASH_DIR" set -x # Keep this definition above the IS_MULTINODE branch: both paths call it, and @@ -94,6 +93,8 @@ import_squash_image() { local sqsh="$2" local lock="${2}.lock" + mkdir -p "$SQUASH_DIR" + if unsquashfs -l "$sqsh" > /dev/null 2>&1; then echo "Squash file already present, skipping import: $sqsh" return 0 @@ -120,7 +121,29 @@ import_squash_image() { test -r "$sqsh" || { echo "Error: squash file not readable: $sqsh" >&2; exit 1; } } -if [[ "$IS_MULTINODE" == "true" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ -n "${BENCH_SCRIPT_OVERRIDE:-}" ]]; then + # SPEED-Bench collectors explicitly supply their script outside this migration. + EXECUTION_PATH=script +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi + +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + check_env_vars B300_HF_CACHE_HOST_DIR + HF_HUB_CACHE_MOUNT="$B300_HF_CACHE_HOST_DIR/hub" + SRT_MODEL_PATH="$MODEL_ROOT/${MODEL##*/}" + if [[ "$MODEL" == nvidia/DeepSeek-R1-0528-FP4-V2 ]]; then + SRT_MODEL_PATH="$MODEL_ROOT/DeepSeek-R1-0528-NVFP4-v2" + fi + SRT_SQUASH_FILE="$SQUASH_DIR/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node b300-dsxe \ + --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" \ + --var MODEL_ROOT "$MODEL_ROOT" +elif [[ "$EXECUTION_PATH" == multinode ]]; then if [[ $FRAMEWORK != "dynamo-sglang" && $FRAMEWORK != "dynamo-trt" && $FRAMEWORK != "dynamo-vllm" ]]; then echo "Unsupported framework: $FRAMEWORK. Supported frameworks are: dynamo-trt, dynamo-sglang, dynamo-vllm" diff --git a/runners/launch_h100-cw.sh b/runners/launch_h100-cw.sh index 26c4c2ff11..693211a505 100644 --- a/runners/launch_h100-cw.sh +++ b/runners/launch_h100-cw.sh @@ -1,7 +1,23 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC + +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + HF_HUB_CACHE_MOUNT=/mnt/vast/gharunner/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + SRT_SQUASH_FILE="/mnt/vast/gharunner/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node h100-cw + exit $? +fi export HF_HUB_CACHE_MOUNT="/mnt/vast/gharunner/hf-hub-cache" PARTITION="h100" diff --git a/runners/launch_h100-dgxc-slurm.sh b/runners/launch_h100-dgxc-slurm.sh index 3eccf3eba5..6571df5224 100644 --- a/runners/launch_h100-dgxc-slurm.sh +++ b/runners/launch_h100-dgxc-slurm.sh @@ -1,7 +1,7 @@ #!/usr/bin/bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars EVAL_ONLY IS_MULTINODE RUN_EVAL SALLOC_TIME_LIMIT +check_env_vars EVAL_ONLY IS_MULTINODE RUN_EVAL SALLOC_TIME_LIMIT IS_AGENTIC set -e # shellcheck source=runners/slurm_utils.sh @@ -14,7 +14,22 @@ SPEC_SUFFIX=$([[ "$SPEC_DECODING" == "mtp" ]] && printf '_mtp' || printf '') set -x -if [[ "$IS_MULTINODE" == "true" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi + +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + HF_HUB_CACHE_MOUNT=/mnt/nfs/sa-shared/gharunners/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + SRT_SQUASH_FILE="/mnt/nfs/lustre/containers/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node h100-dgxc-slurm \ + --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" \ + --var CONTAINER_KEY "$IMAGE" +elif [[ "$EXECUTION_PATH" == multinode ]]; then # Recipes name HF model IDs; resolve them to pre-staged paths so the shared # cluster does not re-download. SRT_SLURM_MODEL_PREFIX must match the diff --git a/runners/launch_h200-cw.sh b/runners/launch_h200-cw.sh index 2ee0b76f6e..d101457309 100644 --- a/runners/launch_h200-cw.sh +++ b/runners/launch_h200-cw.sh @@ -1,7 +1,23 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC + +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + HF_HUB_CACHE_MOUNT=/mnt/vast/gharunner/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + SRT_SQUASH_FILE="/mnt/vast/gharunner/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node h200-cw + exit $? +fi export HF_HUB_CACHE_MOUNT="/mnt/vast/gharunner/hf-hub-cache" export AIPERF_MMAP_CACHE_HOST_PATH="/mnt/vast/gharunner/ai-perf-cache" diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index c71d7ec0e9..7ad5fc35a8 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -1,7 +1,7 @@ #!/usr/bin/bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars EVAL_ONLY IS_MULTINODE REQUIRE_POWER RUN_EVAL SALLOC_TIME_LIMIT +check_env_vars EVAL_ONLY IS_MULTINODE REQUIRE_POWER RUN_EVAL SALLOC_TIME_LIMIT IS_AGENTIC set -eo pipefail SLURM_PARTITION="main" @@ -15,7 +15,22 @@ set -x source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 -if [[ "$IS_MULTINODE" == "true" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi + +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + SRT_SQUASH_FILE="/data/containers/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node h200-dgxc-slurm \ + --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" \ + --var AIPERF_MMAP_CACHE_HOST_PATH "$AIPERF_MMAP_CACHE_HOST_PATH" \ + --var HF_HUB_CACHE_MOUNT "$HF_HUB_CACHE_MOUNT" --var CONTAINER_KEY "$IMAGE" + +elif [[ "$EXECUTION_PATH" == multinode ]]; then if [[ -z "${CONFIG_FILE:-}" ]]; then echo "Error: CONFIG_FILE is not set. The srt-slurm path requires a CONFIG_FILE in additional-settings." >&2 diff --git a/runners/launch_mi300x-amd.sh b/runners/launch_mi300x-amd.sh index 6d88f743cd..e881f41dc2 100644 --- a/runners/launch_mi300x-amd.sh +++ b/runners/launch_mi300x-amd.sh @@ -1,9 +1,29 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC set -eo pipefail +# Select native fixed-sequence execution before the retained AgentX/multi-node paths. +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + check_env_vars GITHUB_WORKSPACE MODEL IMAGE + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + export HF_HUB_CACHE_MOUNT=/raid/inferencex/models/hub + export SRT_MODEL_PATH="hf:$MODEL" + export SALLOC_TIME_LIMIT=180 + export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' + SRT_SQUASH_FILE="/raid/inferencex/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node mi300x-amd --var GITHUB_WORKSPACE "$GITHUB_WORKSPACE" + exit $? +fi + export HF_HUB_CACHE_MOUNT="/raid/inferencex/models/hub" export AIPERF_MMAP_CACHE_MOUNT="/raid/inferencex/aiperf-mmap-cache" export AIPERF_DATASET_MMAP_CACHE_DIR="/aiperf_mmap_cache" diff --git a/runners/launch_mi325x-amds.sh b/runners/launch_mi325x-amds.sh index e2cd6501a3..8063915ea5 100644 --- a/runners/launch_mi325x-amds.sh +++ b/runners/launch_mi325x-amds.sh @@ -1,9 +1,29 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC set -eo pipefail +# Select native fixed-sequence execution before the retained AgentX/multi-node paths. +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + check_env_vars GITHUB_WORKSPACE MODEL IMAGE + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + export HF_HUB_CACHE_MOUNT=/raid/hf-hub-cache/ + export SRT_MODEL_PATH="hf:$MODEL" + export SALLOC_TIME_LIMIT=480 + export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' + SRT_SQUASH_FILE="/raid/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node mi325x-amds + exit $? +fi + export HF_HUB_CACHE_MOUNT="/raid/hf-hub-cache/" PARTITION="compute" diff --git a/runners/launch_mi355x-amds.sh b/runners/launch_mi355x-amds.sh index 9920bb9d90..8971a58f6b 100644 --- a/runners/launch_mi355x-amds.sh +++ b/runners/launch_mi355x-amds.sh @@ -3,6 +3,26 @@ source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 check_env_vars EVAL_ONLY IS_AGENTIC IS_MULTINODE KEEP_LOGS RUN_EVAL +# Select native fixed-sequence execution before the retained AgentX/multi-node paths. +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + check_env_vars GITHUB_WORKSPACE MODEL IMAGE + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + export HF_HUB_CACHE_MOUNT=/var/lib/hf-hub-cache/ + export SRT_MODEL_PATH="hf:$MODEL" + export SALLOC_TIME_LIMIT=500 + export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' + SRT_SQUASH_FILE="/var/lib/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node mi355x-amds + exit $? +fi + scancel_sync() { local jobid=$1 local timeout=${2:-600} diff --git a/runners/launch_rtx6000pro-lat.sh b/runners/launch_rtx6000pro-lat.sh deleted file mode 100755 index 32e244b88c..0000000000 --- a/runners/launch_rtx6000pro-lat.sh +++ /dev/null @@ -1,161 +0,0 @@ -#!/usr/bin/bash - -source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE -set -eo pipefail - -# This runner executes directly on the single RTX PRO 6000 GPU node. Docker -# therefore owns a separate image cache from the node's RKE2/containerd cache. -check_env_vars HF_HUB_CACHE_MOUNT -check_env_vars HF_HUB_CACHE -check_env_vars PORT - -# NCCL 2.28.9 segfaults while probing this node's bnxt_re RDMA devices. -# Disable that RDMA path by default while preserving local CUDA P2P/SHM and -# allowing an explicit caller override. -check_env_vars NCCL_IB_DISABLE - -: "${GITHUB_WORKSPACE:?GITHUB_WORKSPACE must be set}" -: "${IMAGE:?IMAGE must be set}" -: "${EXP_NAME:?EXP_NAME must be set}" -: "${PRECISION:?PRECISION must be set}" - -mkdir -p "$HF_HUB_CACHE_MOUNT" - -check_env_vars GPU_COUNT -if [[ ! "$GPU_COUNT" =~ ^[1-9][0-9]*$ ]]; then - echo "GPU_COUNT must be a positive integer, got: $GPU_COUNT" >&2 - exit 1 -fi - -export CUDA_VISIBLE_DEVICES -CUDA_VISIBLE_DEVICES="$(seq -s, 0 "$((GPU_COUNT - 1))")" - -# Some Slurm/enroot configs spell registry paths as nvcr.io#namespace/image. -# Docker requires the normal slash form. -DOCKER_IMAGE="${IMAGE//#//}" - -SPEC_SUFFIX="" -if [[ "${SPEC_DECODING:-}" == "mtp" ]]; then - SPEC_SUFFIX="_mtp" -fi - -check_env_vars SCENARIO_SUBDIR -SCENARIO_SUBDIR="${SCENARIO_SUBDIR#/}" -SCENARIO_SUBDIR="${SCENARIO_SUBDIR%/}/" -# Prefer a framework-tagged script so engines can coexist; fall back to the -# untagged name for scripts not yet retagged. -BENCH_BASE="benchmarks/single_node/${SCENARIO_SUBDIR}${EXP_NAME%%_*}_${PRECISION}_rtx6000pro" -BENCH_SCRIPT="${BENCH_BASE}_${FRAMEWORK:-}${SPEC_SUFFIX}.sh" -if [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then - BENCH_SCRIPT="${BENCH_BASE}${SPEC_SUFFIX}.sh" -fi - -if [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then - echo "Benchmark script not found: $GITHUB_WORKSPACE/$BENCH_SCRIPT" >&2 - exit 1 -fi - -check_env_vars RUNNER_NAME -server_name="bmk-server-${RUNNER_NAME}" -server_name="${server_name//[^a-zA-Z0-9_.-]/-}" - -cleanup() { - docker rm -f "$server_name" >/dev/null 2>&1 || true -} -trap cleanup EXIT - -# Clear a container left behind by a cancelled or interrupted workflow. -cleanup - -check_env_vars INFERENCEX_RUNTIME_ENV_VARS -RUNTIME_ENV_ARGS=() -for runtime_var in $INFERENCEX_RUNTIME_ENV_VARS; do - check_env_vars "$runtime_var" - RUNTIME_ENV_ARGS+=(--env "$runtime_var") -done - -docker run \ - "${RUNTIME_ENV_ARGS[@]}" \ - --env IS_MULTINODE \ - --env REQUIRE_POWER \ - --env INFMAX_CONTAINER_WORKSPACE \ - --env AIPERF_EXPERIMENTAL_FAST \ - --rm \ - --pull=missing \ - --name="$server_name" \ - --runtime=nvidia \ - --gpus="$GPU_COUNT" \ - --network=host \ - --ipc=host \ - --privileged \ - --shm-size=32g \ - --ulimit memlock=-1 \ - --ulimit stack=67108864 \ - --security-opt seccomp=unconfined \ - --cap-add=SYS_PTRACE \ - --volume "$HF_HUB_CACHE_MOUNT:$HF_HUB_CACHE" \ - --volume "$GITHUB_WORKSPACE:/workspace/" \ - --workdir=/workspace/ \ - --env HF_TOKEN \ - --env HF_HUB_CACHE \ - --env MODEL \ - --env MODEL_PREFIX \ - --env MODEL_PATH \ - --env TP \ - --env PP_SIZE \ - --env DCP_SIZE \ - --env PCP_SIZE \ - --env EP_SIZE \ - --env DP_SIZE \ - --env DP_ATTENTION \ - --env GPU_COUNT \ - --env CONC \ - --env MAX_MODEL_LEN \ - --env ISL \ - --env OSL \ - --env FRAMEWORK \ - --env PRECISION \ - --env DISAGG \ - --env SPEC_DECODING \ - --env NUM_SPEC_TOKENS \ - --env RUN_EVAL \ - --env EVAL_ONLY \ - --env EVAL_FRAMEWORK \ - --env EVAL_LIMIT \ - --env EVAL_SUITE \ - --env EVAL_MAX_MODEL_LEN \ - --env RUNNER_TYPE \ - --env RUNNER_NAME \ - --env RESULT_FILENAME \ - --env RESULT_DIR \ - --env RANDOM_RANGE_RATIO \ - --env GPU_MEM_UTIL \ - --env AIPERF_FAILED_REQUEST_THRESHOLD \ - --env KV_OFFLOADING \ - --env KV_OFFLOAD_BACKEND \ - --env KV_OFFLOAD_BACKEND_METADATA \ - --env ROUTER_METADATA \ - --env KV_P2P_TRANSFER \ - --env TOTAL_CPU_DRAM_GB \ - --env DURATION \ - --env SCENARIO_TYPE \ - --env SCENARIO_SUBDIR \ - --env IS_AGENTIC \ - --env SWEBENCH_GEN_MODE \ - --env SWEBENCH_USE_MODAL \ - --env MODAL_TOKEN_ID \ - --env MODAL_TOKEN_SECRET \ - --env PROFILE \ - --env SGLANG_TORCH_PROFILER_DIR \ - --env VLLM_TORCH_PROFILER_DIR \ - --env VLLM_RPC_TIMEOUT \ - --env PYTHONDONTWRITEBYTECODE \ - --env PYTHONPYCACHEPREFIX=/tmp/pycache/ \ - --env PORT="$PORT" \ - --env CUDA_DEVICE_ORDER=PCI_BUS_ID \ - --env CUDA_VISIBLE_DEVICES \ - --env NCCL_IB_DISABLE \ - --entrypoint=/bin/bash \ - "$DOCKER_IMAGE" \ - "$BENCH_SCRIPT" diff --git a/runners/runtime_settings.sh b/runners/runtime_settings.sh index ab2ccf616f..f39004be9c 100644 --- a/runners/runtime_settings.sh +++ b/runners/runtime_settings.sh @@ -46,6 +46,10 @@ case "${RUNNER_NAME%%_*}" in fi fi export HF_HUB_CACHE_MOUNT=/models/gharunners/hf-hub-cache + case "$MODEL_PREFIX/$PRECISION" in + dsr1/fp8) export SRT_MODEL_PATH=/models/DeepSeek-R1-0528 ;; + qwen3.5/fp8) export SRT_MODEL_PATH="$HF_HUB_CACHE_MOUNT/Qwen3.5-397B-A17B-FP8" ;; + esac export AIPERF_MMAP_CACHE_HOST_PATH=/home/sa-shared/gharunners/ai-perf-cache export DSV4_MODEL_PATH="$HF_HUB_CACHE_MOUNT/DeepSeek-V4-Pro" export GLM52_FP8_MODEL_PATH=/models/GLM-5.2-FP8 @@ -59,8 +63,4 @@ case "${RUNNER_NAME%%_*}" in check_env_vars GITHUB_WORKSPACE export BENCHMARK_LOGS_DIR="$GITHUB_WORKSPACE/benchmark_logs" ;; - rtx6000pro-lat) - export HF_HUB_CACHE_MOUNT=/var/lib/inferencex/hf-hub-cache - export NCCL_IB_DISABLE=1 - ;; esac diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 3ab61b00e2..17b0525f86 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -39,9 +39,8 @@ setup_srt_slurm() { return 1 fi local destination="$1" framework="$2" uses_power="$3" - check_env_vars INFERENCEX_RUNTIME_ENV_VARS AIPERF_DRAIN_TIMEOUT_SECONDS AIPERF_DRAIN_POLL_SECONDS EVAL_ONLY - local eval_passthrough - eval_passthrough=$(python3 - <<'PYENV' + check_env_vars INFERENCEX_RUNTIME_ENV_VARS EVAL_ONLY + SRT_EVAL_PASSTHROUGH=$(python3 - <<'PYENV' import json import os @@ -49,16 +48,17 @@ names = [ "EVAL_FRAMEWORK", "EVAL_CONC", "EVAL_LIMIT", "EVAL_SUITE", "SWEBENCH_GEN_MODE", "SWEBENCH_USE_MODAL", "MODAL_TOKEN_ID", "MODAL_TOKEN_SECRET", "IS_AGENTIC", "SCENARIO_TYPE", + "TP", "EP_SIZE", "DP_ATTENTION", "PP_SIZE", "DCP_SIZE", "PCP_SIZE", "CONC", ] print(json.dumps(names + os.environ["INFERENCEX_RUNTIME_ENV_VARS"].split())) PYENV ) || return 1 - SRTCTL_EVAL_ARGS+=(--set "post_eval.passthrough_env=$eval_passthrough") + SRTCTL_EVAL_ARGS+=(--set "post_eval.passthrough_env=$SRT_EVAL_PASSTHROUGH") # Custom benchmarks inherit exported workflow settings through sbatch/srun; # native recipe environment and benchmark.env retain their override priority. local source="$INFERENCEX_SLURM_UTILS_DIR/../utils/srt-slurm" if [[ "$framework" == "tilert" ]]; then - # Sole fork exception until NVIDIA supports the TileRT backend and router. + # TileRT still needs its legacy runtime until the native backend and router land. SRT_SLURM_COMMIT=6bc3f306bdafa1edfb5dded2fcda8f1ccede1bde git init "$destination" || return 1 git -C "$destination" remote add origin https://github.com/SemiAnalysisAI/srt-slurm.git || return 1 @@ -73,6 +73,12 @@ PYENV SRTCTL_EVAL_ARGS+=(--set benchmark.stream_output=true) # A local clone keeps job writes isolated and preserves upstream Git provenance. git clone --no-hardlinks "$source" "$destination" || return 1 + # Temporary fixes awaiting upstream merge; see runners/srt-slurm/patches/README.md. + local patch + for patch in "$GITHUB_WORKSPACE"/runners/srt-slurm/patches/*.patch; do + [[ -e "$patch" ]] || continue + git -C "$destination" apply "$patch" || return 1 + done fi cd "$destination" || return 1 [[ "$(git rev-parse HEAD)" == "$SRT_SLURM_COMMIT" ]] || return 1 @@ -128,6 +134,105 @@ apply_srt_recipe() { "$config" "$framework" -- "$@" } +# One native submission per fixed-sequence matrix point, shared across Slurm pools. +launch_srt_single_node() { + set -eo pipefail + local profile="$1" + shift + check_env_vars GITHUB_WORKSPACE SRT_RECIPE FRAMEWORK MODEL MODEL_PREFIX IMAGE PRECISION \ + TP PP_SIZE DCP_SIZE PCP_SIZE EP_SIZE DP_ATTENTION GPU_COUNT IS_AGENTIC SPEC_DECODING \ + CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME GPU_MONITOR_INTERVAL SRT_MODEL_PATH \ + HF_HUB_CACHE_MOUNT HF_HUB_CACHE SALLOC_TIME_LIMIT + SRT_SINGLE_NODE_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-single.XXXXXX") + SRTCTL_ROOT="$SRT_SINGLE_NODE_ROOT/checkout" + export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" + setup_srt_slurm "$SRTCTL_ROOT" "$FRAMEWORK" 0 + if ! command -v uv >/dev/null; then + curl -LsSf https://astral.sh/uv/install.sh | sh + source "$HOME/.local/bin/env" + fi + uv venv .venv + source .venv/bin/activate + uv pip install -e . + export PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" + + python3 -m infx.srt_slurm.single_node prepare "$GITHUB_WORKSPACE/$SRT_RECIPE" "$SRT_SINGLE_NODE_ROOT/arguments" + mapfile -d '' -t SRT_RUNTIME_ARGS < "$SRT_SINGLE_NODE_ROOT/arguments" + SRT_SELECTED_RECIPE="${SRT_RUNTIME_ARGS[0]}" + SRT_RUNTIME_ARGS=("${SRT_RUNTIME_ARGS[@]:1}") + SRT_RUNTIME_ARGS+=( + --set 'post_eval.command=["bash", "{infmax_workspace}/benchmarks/single_node/srt_eval.sh", "{endpoint}", "/logs/infx-eval-exit-code"]' + --set "post_eval.passthrough_env=$SRT_EVAL_PASSTHROUGH" + ) + # Reuse only a valid cache for this exact image. Missing caches are imported + # by native Pyxis inside the same benchmark allocation. + SRT_CONTAINER="$IMAGE" + if [[ -n "${SRT_SQUASH_FILE:-}" && -r "$SRT_SQUASH_FILE" ]] && unsquashfs -s "$SRT_SQUASH_FILE" >/dev/null 2>&1; then + SRT_CONTAINER="$SRT_SQUASH_FILE" + fi + python3 -m infx.srt_slurm.cluster_config \ + "$INFERENCEX_SLURM_UTILS_DIR/srt-slurm/${profile}.yaml" srtslurm.yaml \ + --var SRTCTL_ROOT "$SRTCTL_ROOT" --var SQUASH_FILE "$SRT_CONTAINER" \ + --var IMAGE "$IMAGE" --var NGINX_SQUASH_FILE nginx:1.27.4 \ + --var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" \ + --model "hf:$MODEL" "$SRT_MODEL_PATH" --container "$IMAGE" "$SRT_CONTAINER" \ + --mount "$HF_HUB_CACHE_MOUNT" "$HF_HUB_CACHE" --exclusive "$@" + make setup ARCH=x86_64 + + SRT_JOB_ID="" + SRT_JOB_OUTPUT="" + finish_native_single_node() { + local rc=$? artifact + trap - EXIT + # Submission may succeed immediately before cancellation or a client error. + if [[ -z "$SRT_JOB_ID" ]] && python3 -m infx.srt_slurm.single_node submission \ + "$GITHUB_WORKSPACE/srt-single-node-submission.json" > "$SRT_SINGLE_NODE_ROOT/submission-fields" 2>/dev/null; then + mapfile -t SRT_SUBMISSION < "$SRT_SINGLE_NODE_ROOT/submission-fields" + SRT_JOB_ID="${SRT_SUBMISSION[0]}" + SRT_JOB_OUTPUT="${SRT_SUBMISSION[1]}" + fi + if [[ -n "$SRT_JOB_ID" ]] && slurm_job_is_active "$SRT_JOB_ID"; then + scancel "$SRT_JOB_ID" || true + fi + if [[ -n "$SRT_JOB_OUTPUT" && -d "$SRT_JOB_OUTPUT" ]]; then + bundle_server_logs "$SRT_JOB_OUTPUT" "$GITHUB_WORKSPACE/srt-single-node-logs.tar.gz" + for artifact in "$SRT_JOB_OUTPUT/logs/$RESULT_FILENAME.json" "$SRT_JOB_OUTPUT"/logs/gpu_metrics*; do + [[ -f "$artifact" ]] || continue + copy_to_workspace "$artifact" "$GITHUB_WORKSPACE/$(basename "$artifact")" || rc=1 + done + fi + exit "$rc" + } + trap finish_native_single_node EXIT + trap 'exit 130' INT + trap 'exit 143' TERM + local submission_rc=0 + apply_srt_recipe "$SRT_SELECTED_RECIPE" "$FRAMEWORK" \ + --json --yes --output "$SRT_SINGLE_NODE_ROOT/outputs" "${SRT_RUNTIME_ARGS[@]}" \ + > "$GITHUB_WORKSPACE/srt-single-node-submission.json" || submission_rc=$? + if (( submission_rc != 0 )); then + cat "$GITHUB_WORKSPACE/srt-single-node-submission.json" >&2 + return "$submission_rc" + fi + python3 -m infx.srt_slurm.single_node submission "$GITHUB_WORKSPACE/srt-single-node-submission.json" \ + > "$SRT_SINGLE_NODE_ROOT/submission-fields" + mapfile -t SRT_SUBMISSION < "$SRT_SINGLE_NODE_ROOT/submission-fields" + SRT_JOB_ID="${SRT_SUBMISSION[0]}" + SRT_JOB_OUTPUT="${SRT_SUBMISSION[1]}" + stream_slurm_job_log "$SRT_JOB_ID" "$SRT_JOB_OUTPUT/logs/sweep_${SRT_JOB_ID}.log" + verify_slurm_job_status "$SRT_JOB_ID" + # Native SRT treats post-throughput eval failure as non-fatal. InferenceX + # requires every requested eval to finish successfully, including staging. + if [[ "$RUN_EVAL" == true || "$EVAL_ONLY" == true ]]; then + test -f "$SRT_JOB_OUTPUT/logs/infx-eval-exit-code" + test "$(cat "$SRT_JOB_OUTPUT/logs/infx-eval-exit-code")" = 0 + fi + if [[ "$EVAL_ONLY" != true ]]; then + test -s "$SRT_JOB_OUTPUT/logs/$RESULT_FILENAME.json" + fi + +} + slurm_job_is_active() { local job_id="$1" squeue -j "$job_id" --noheader 2>/dev/null | grep -q "$job_id" @@ -162,10 +267,29 @@ verify_slurm_job_status() { local job_id="$1" # Disappearing from squeue means terminal, not successful. Accounting can # lag briefly; inspect only the allocation, never successful service steps. - local attempt accounting state exit_code + local attempt accounting state exit_code controller field controller_job_id + local -a controller_fields for attempt in {1..10}; do accounting=$(sacct -X -n -P -j "$job_id" --format=State,ExitCode 2>/dev/null) || accounting="" IFS='|' read -r state exit_code <<< "$accounting" + if [[ -z "$state" ]]; then + # Some pools do not expose slurmdbd. The controller retains recent + # terminal allocations; require its state and exit code, too. + controller=$(scontrol show job -o "$job_id" 2>/dev/null) || controller="" + controller_job_id="" + read -r -a controller_fields <<< "$controller" + for field in "${controller_fields[@]}"; do + case "$field" in + JobId=*) controller_job_id="${field#JobId=}" ;; + JobState=*) state="${field#JobState=}" ;; + ExitCode=*) exit_code="${field#ExitCode=}" ;; + esac + done + if [[ "$controller_job_id" != "$job_id" ]]; then + state="" + exit_code="" + fi + fi case "$state" in COMPLETED) if [[ "$exit_code" == "0:0" ]]; then diff --git a/runners/srt-slurm/b200-cw.yaml b/runners/srt-slurm/b200-cw.yaml new file mode 100644 index 0000000000..881d227fef --- /dev/null +++ b/runners/srt-slurm/b200-cw.yaml @@ -0,0 +1,6 @@ +default_partition: b200 +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: "" +srtctl_root: ${SRTCTL_ROOT} +use_exclusive_sbatch_directive: true diff --git a/runners/srt-slurm/b200-nb.yaml b/runners/srt-slurm/b200-nb.yaml new file mode 100644 index 0000000000..7ad1a06fdf --- /dev/null +++ b/runners/srt-slurm/b200-nb.yaml @@ -0,0 +1,6 @@ +default_partition: main +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: "" +srtctl_root: ${SRTCTL_ROOT} +use_exclusive_sbatch_directive: true diff --git a/runners/srt-slurm/h100-cw.yaml b/runners/srt-slurm/h100-cw.yaml new file mode 100644 index 0000000000..85f05a44b0 --- /dev/null +++ b/runners/srt-slurm/h100-cw.yaml @@ -0,0 +1,6 @@ +default_partition: h100 +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: "" +srtctl_root: ${SRTCTL_ROOT} +use_exclusive_sbatch_directive: true diff --git a/runners/srt-slurm/h200-cw.yaml b/runners/srt-slurm/h200-cw.yaml new file mode 100644 index 0000000000..ac2631ccc2 --- /dev/null +++ b/runners/srt-slurm/h200-cw.yaml @@ -0,0 +1,6 @@ +default_partition: h200 +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: "" +srtctl_root: ${SRTCTL_ROOT} +use_exclusive_sbatch_directive: true diff --git a/runners/srt-slurm/hooks/mi300x-amd/setup.sh b/runners/srt-slurm/hooks/mi300x-amd/setup.sh new file mode 100755 index 0000000000..1eeb9bfe32 --- /dev/null +++ b/runners/srt-slurm/hooks/mi300x-amd/setup.sh @@ -0,0 +1,12 @@ +#!/usr/bin/env bash +set -eo pipefail + +# RCCL cannot reclaim scratch memory on MEC firmware older than 177 and crashes. +# See https://rocm.docs.amd.com/en/docs-6.4.3/about/release-notes.html#amdgpu-driver-updates +minimum=177 +firmware=$(rocm-smi --showfw | awk '/MEC firmware version/ {print $NF}' | sort -n | head -n 1) +if [[ -z "$firmware" || "$firmware" -lt "$minimum" ]]; then + echo "[$(hostname -s)] MEC firmware ${firmware:-unknown} is older than $minimum" >&2 + exit 1 +fi +echo "[$(hostname -s)] MEC firmware $firmware" diff --git a/runners/srt-slurm/mi300x-amd.yaml b/runners/srt-slurm/mi300x-amd.yaml new file mode 100644 index 0000000000..9bb06997c8 --- /dev/null +++ b/runners/srt-slurm/mi300x-amd.yaml @@ -0,0 +1,17 @@ +default_partition: compute-0 +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: '' +visible_devices_env: ROCR_VISIBLE_DEVICES +srtctl_root: ${SRTCTL_ROOT} +default_mounts: + /dev/kfd: /dev/kfd + /dev/dri: /dev/dri +default_sbatch_directives: + cpus-per-task: '128' +use_gpus_per_node_directive: true +use_segment_sbatch_directive: false +use_exclusive_sbatch_directive: true +default_host_setup: + commands: + - bash "${GITHUB_WORKSPACE}/runners/srt-slurm/hooks/mi300x-amd/setup.sh" diff --git a/runners/srt-slurm/mi325x-amds.yaml b/runners/srt-slurm/mi325x-amds.yaml new file mode 100644 index 0000000000..4c8681a2d7 --- /dev/null +++ b/runners/srt-slurm/mi325x-amds.yaml @@ -0,0 +1,17 @@ +default_partition: compute +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: '' +visible_devices_env: ROCR_VISIBLE_DEVICES +srtctl_root: ${SRTCTL_ROOT} +default_mounts: + /dev/kfd: /dev/kfd + /dev/dri: /dev/dri +default_sbatch_directives: + cpus-per-task: '256' +use_gpus_per_node_directive: true +use_segment_sbatch_directive: false +use_exclusive_sbatch_directive: true +default_bash_preamble: |- + export XDG_CACHE_HOME="/tmp/xdg-cache-$$SLURM_JOB_ID" + export TRITON_CACHE_DIR="/tmp/triton-cache-$$SLURM_JOB_ID" diff --git a/runners/srt-slurm/mi355x-amds.yaml b/runners/srt-slurm/mi355x-amds.yaml new file mode 100644 index 0000000000..f6a0df40e8 --- /dev/null +++ b/runners/srt-slurm/mi355x-amds.yaml @@ -0,0 +1,14 @@ +default_partition: compute +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: '' +visible_devices_env: ROCR_VISIBLE_DEVICES +srtctl_root: ${SRTCTL_ROOT} +default_mounts: + /dev/kfd: /dev/kfd + /dev/dri: /dev/dri +default_sbatch_directives: + cpus-per-task: '128' +use_gpus_per_node_directive: true +use_segment_sbatch_directive: false +use_exclusive_sbatch_directive: true diff --git a/runners/srt-slurm/patches/504-post-eval-srun-options.patch b/runners/srt-slurm/patches/504-post-eval-srun-options.patch new file mode 100644 index 0000000000..196295312d --- /dev/null +++ b/runners/srt-slurm/patches/504-post-eval-srun-options.patch @@ -0,0 +1,25 @@ +diff --git a/docs/config-reference.md b/docs/config-reference.md +index 17cfccba8..d6ddf2874 100644 +--- a/docs/config-reference.md ++++ b/docs/config-reference.md +@@ -2014,6 +2014,8 @@ post_eval: + + `MODEL_NAME` (the served model name) and `EVAL_CONC` are always set by srtctl. `srtctl dry-run` prints the effective dispatch. + ++Eval steps inherit the recipe's `srun_options`, including container flags such as `container-writable`. ++ + --- + + ## services +diff --git a/src/srtctl/cli/do_sweep.py b/src/srtctl/cli/do_sweep.py +index 4ff19ade6..ceba6ad6b 100644 +--- a/src/srtctl/cli/do_sweep.py ++++ b/src/srtctl/cli/do_sweep.py +@@ -603,6 +603,7 @@ def _run_post_eval(self, stop_event: threading.Event) -> int: + container_image=str(self.runtime.container_image), + container_mounts=self.runtime.container_mounts, + env_to_set=env_to_set, ++ srun_options=self.runtime.srun_options, + het_group=self.runtime.nodes.het_group_for(self.runtime.nodes.head), + ) + diff --git a/runners/srt-slurm/patches/README.md b/runners/srt-slurm/patches/README.md new file mode 100644 index 0000000000..cb619d0506 --- /dev/null +++ b/runners/srt-slurm/patches/README.md @@ -0,0 +1,9 @@ +# srt-slurm patches + +`setup_srt_slurm()` in [`runners/slurm_utils.sh`](../../slurm_utils.sh) applies every `*.patch` here to the job's srt-slurm clone after checking out the pinned submodule. TileRT jobs use the fork checkout and skip these patches. + +Each patch is a temporary fix for an open upstream PR. When the PR merges and the submodule pin includes it, delete the patch and its row. + +| Patch | Upstream PR | Fix | +|-------|-------------|-----| +| `504-post-eval-srun-options.patch` | [NVIDIA/srt-slurm#504](https://github.com/NVIDIA/srt-slurm/pull/504) | Forward recipe `srun_options` (e.g. `container-writable`) to post-eval steps | diff --git a/utils/matrix_logic/test_generate_sweep_configs.py b/utils/matrix_logic/test_generate_sweep_configs.py index 4d16b62bfd..a09db34998 100644 --- a/utils/matrix_logic/test_generate_sweep_configs.py +++ b/utils/matrix_logic/test_generate_sweep_configs.py @@ -190,6 +190,20 @@ def test_multinode_node_count_prefers_recipe_roles( # Test Fixtures # ============================================================================= + +@pytest.mark.parametrize("command", ["full-sweep", "test-config"]) +def test_srt_recipe_selection_stays_with_its_scenario( + sample_single_node_config, sample_runner_config, full_sweep_args_both, command, +): + key, config = next(iter(sample_single_node_config.items())) + config["scenarios"]["fixed-seq-len"][0]["search-space"][0]["srt-recipe"] = "pilot.yaml:base" + vars(full_sweep_args_both).update(config_keys=[key], no_evals=True) + generate = generate_full_sweep if command == "full-sweep" else generate_test_config_sweep + rows = generate(full_sweep_args_both, sample_single_node_config, sample_runner_config) + assert rows + assert {row.get("srt-recipe") for row in rows if row["isl"] == 1024} == {"pilot.yaml:base"} + assert {row.get("srt-recipe") for row in rows if row["isl"] == 8192} == {None} + @pytest.fixture def sample_single_node_config(): """Single node config based on dsr1-fp8-mi300x-sglang.""" diff --git a/utils/srt-slurm b/utils/srt-slurm index 2ac4eb1367..8dace5f959 160000 --- a/utils/srt-slurm +++ b/utils/srt-slurm @@ -1 +1 @@ -Subproject commit 2ac4eb1367dd2a78f597a72ca91afe4211d76b38 +Subproject commit 8dace5f9596907a5075bf056251563b2e9563e7d diff --git a/utils/test_process_result.py b/utils/test_process_result.py index 46af25f41a..51d1f128c9 100644 --- a/utils/test_process_result.py +++ b/utils/test_process_result.py @@ -1,9 +1,11 @@ """Exercise the fixed-sequence module CLI with controlled environment and artifacts.""" import json import os +import select import signal import subprocess import sys +import time from pathlib import Path import pytest @@ -1033,6 +1035,42 @@ def test_multinode_internal_error_preserves_validation( "message": "forced import failure" if fail_import else "forced aggregation failure", } + def test_amd_csv_filter_streams_complete_rows_before_eof(self): + """A live producer must not leave telemetry buffered until shutdown.""" + benchmark_lib = REPO_ROOT / "benchmarks/benchmark_lib.sh" + expected = b"timestamp,gpu,socket_power\n123,0,400\n124,0,410\n" + with subprocess.Popen( + ["bash", "-c", f"source {str(benchmark_lib)!r}; _filter_amd_smi_metrics"], + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + env={"PATH": os.environ["PATH"], "PYTHONDONTWRITEBYTECODE": "1"}, + ) as process: + try: + process.stdin.write( + b"diagnostic before header\ntimestamp,gpu,socket_power\n123,0,400\n" + b"timestamp,gpu,socket_power\n124,0,410\n125,0,4" + ) + process.stdin.flush() + received = b"" + deadline = time.monotonic() + 5 + while len(received) < len(expected): + ready, _, _ = select.select( + [process.stdout], [], [], max(0, deadline - time.monotonic()) + ) + assert ready, "CSV rows remained buffered while the producer was open" + chunk = os.read(process.stdout.fileno(), 4096) + assert chunk, "filter exited before consuming the live stream" + received += chunk + assert received == expected + process.stdin.close() + assert process.wait(timeout=5) == 0 + assert process.stdout.read() == b"" # Discard the incomplete final row. + finally: + if process.poll() is None: + process.kill() + process.wait(timeout=5) + def test_stop_gpu_monitor_appends_final_nvidia_sample(self, tmp_path): """Stopping between 1 Hz ticks still records one post-benchmark sample.""" fake_bin = tmp_path / "bin" diff --git a/utils/test_srt_cluster_config.py b/utils/test_srt_cluster_config.py index a6229e28b8..ebfa78f46e 100644 --- a/utils/test_srt_cluster_config.py +++ b/utils/test_srt_cluster_config.py @@ -13,6 +13,20 @@ ROOT = Path(__file__).resolve().parents[1] +def test_exclusive_allocation_override(tmp_path): + profile = tmp_path / "profile.yaml" + output = tmp_path / "cluster.yaml" + profile.write_text("use_exclusive_sbatch_directive: false\ndefault_partition: test\n") + result = subprocess.run( + [sys.executable, "-m", "infx.srt_slurm.cluster_config", str(profile), str(output), "--exclusive"], + cwd=ROOT, capture_output=True, text=True, + ) + assert result.returncode == 0, result.stderr + assert yaml.safe_load(output.read_text()) == { + "use_exclusive_sbatch_directive": True, "default_partition": "test", + } + + @pytest.mark.parametrize("power", ["0", "1", "missing-exporter"]) def test_launcher_writes_job_local_cluster_config(tmp_path: Path, power: str) -> None: runner_dir = tmp_path / "runners" diff --git a/utils/test_srt_fixed_sequence.py b/utils/test_srt_fixed_sequence.py new file mode 100644 index 0000000000..84575b1db2 --- /dev/null +++ b/utils/test_srt_fixed_sequence.py @@ -0,0 +1,208 @@ +"""Run the shared client against stubbed external benchmark/GPU processes.""" + +import json +import os +import shutil +import subprocess +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +CLIENT = ROOT / "benchmarks/single_node/srt_fixed_sequence.sh" + + +@pytest.fixture +def client_environment(tmp_path): + binaries = tmp_path / "bin" + binaries.mkdir() + benchmark = binaries / "python3" + benchmark.write_text( + f"#!{sys.executable}\n" + "import json, os, pathlib, sys\n" + "pathlib.Path(os.environ['CAPTURE']).write_text(json.dumps(sys.argv[1:]))\n" + "sys.exit(int(os.environ['CLIENT_EXIT']))\n" + ) + benchmark.chmod(0o755) + for name, body in { + "pip3": "exit 0\n", + "nvidia-smi": "printf 'timestamp,index,power.draw\\n'\n", + }.items(): + binary = binaries / name + binary.write_text(f"#!/bin/bash\n{body}") + binary.chmod(0o755) + env = { + **os.environ, + "PATH": f"{binaries}:{os.environ['PATH']}", + "MODEL": "test/model", + "CONC": "3", + "ISL": "128", + "OSL": "64", + "RANDOM_RANGE_RATIO": "0.5", + "RESULT_FILENAME": "test-result", + "RESULT_DIR": str(tmp_path), + "SRT_FRONTEND_HOST": "10.2.3.4", + "SRT_FRONTEND_PORT": "9444", + "RUN_EVAL": "false", + "EVAL_ONLY": "false", + "GPU_MONITOR_INTERVAL": "2", + "USE_CHAT_TEMPLATE": "false", + "FRAMEWORK": "sglang", + "IS_AGENTIC": "0", + "SCENARIO_TYPE": "fixed-seq-len", + "CLIENT_EXIT": "0", + "CAPTURE": str(tmp_path / "argv.json"), + } + for key in ("PROFILE", "INFERENCEX_SERVER_PID", "INFERENCEX_SERVER_STATE"): + env.pop(key, None) + return env + + +@pytest.mark.parametrize("exit_code,chat_template,framework,backend,extra", [ + (0, "false", "sglang", "vllm", []), (7, "false", "sglang", "vllm", []), + (0, "true", "trt", "openai", []), + (0, "true", "atom", "vllm", ["--trust-remote-code"]), +]) +def test_native_endpoint_preserves_client_settings_and_failure( + client_environment, exit_code, chat_template, framework, backend, extra +): + env = {**client_environment, "CLIENT_EXIT": str(exit_code), "USE_CHAT_TEMPLATE": chat_template, + "FRAMEWORK": framework} + result = subprocess.run( + ["bash", str(CLIENT), *extra], env=env, capture_output=True, text=True + ) + assert result.returncode == exit_code, result.stderr + argv = json.loads(Path(env["CAPTURE"]).read_text()) + assert argv == [ + "-m", + "infx.bench_serving.benchmark_serving", + "--model", + "test/model", + "--backend", + backend, + "--base-url", + "http://10.2.3.4:9444", + "--dataset-name", + "random", + "--random-input-len", + "128", + "--random-output-len", + "64", + "--random-range-ratio", + "0.5", + "--num-prompts", + "30", + "--max-concurrency", + "3", + "--request-rate", + "inf", + "--ignore-eos", + "--save-result", + "--num-warmups", + "6", + "--percentile-metrics", + "ttft,tpot,itl,e2el", + "--result-dir", + env["RESULT_DIR"], + "--result-filename", + "test-result.json", + ] + (["--use-chat-template"] if chat_template == "true" else []) + extra + assert ( + (Path(env["RESULT_DIR"]) / "gpu_metrics.csv") + .read_text() + .startswith("timestamp") + ) + + +@pytest.mark.parametrize( + ("key", "value", "error"), + [ + ("MODEL", None, "MODEL"), + ("GPU_MONITOR_INTERVAL", None, "GPU_MONITOR_INTERVAL"), + ("USE_CHAT_TEMPLATE", "yes", "USE_CHAT_TEMPLATE must be true or false"), + ("CONC", "0", "CONC must be a positive integer"), + ("RUN_EVAL", "yes", "RUN_EVAL must be true or false"), + ("EVAL_ONLY", "yes", "EVAL_ONLY must be true or false"), + ("FRAMEWORK", "unknown", "unsupported fixed-sequence FRAMEWORK"), + ], +) +def test_invalid_runtime_inputs_fail_before_the_client( + client_environment, key, value, error +): + env = dict(client_environment) + if value is None: + env.pop(key) + else: + env[key] = value + result = subprocess.run( + ["bash", str(CLIENT)], env=env, capture_output=True, text=True + ) + assert result.returncode != 0 + assert error in result.stdout + result.stderr + assert not Path(env["CAPTURE"]).exists() + + +def test_legacy_client_keeps_its_local_endpoint(client_environment): + env = client_environment + result = subprocess.run( + [ + "bash", + "-c", + """source "$1/benchmarks/benchmark_lib.sh" +run_benchmark_serving --model test/model --port 8888 --backend vllm \\ + --input-len 128 --output-len 64 --random-range-ratio 0.5 \\ + --num-prompts 30 --max-concurrency 3 --result-filename old --result-dir "$RESULT_DIR" +""", + "bash", + str(ROOT), + ], + env=env, + capture_output=True, + text=True, + ) + assert result.returncode == 0, result.stderr + argv = json.loads(Path(env["CAPTURE"]).read_text()) + assert argv[argv.index("--base-url") + 1] == "http://0.0.0.0:8888" + + +@pytest.mark.parametrize("eval_exit", [0, 7]) +def test_native_post_eval_preserves_results_topology_and_failure(client_environment, tmp_path, eval_exit): + workspace = tmp_path / "repo" + scripts = workspace / "benchmarks/single_node" + scripts.mkdir(parents=True) + shutil.copyfile(ROOT / "benchmarks/benchmark_lib.sh", scripts.parent / "benchmark_lib.sh") + shutil.copyfile(ROOT / "benchmarks/single_node/srt_eval.sh", scripts / "srt_eval.sh") + python = tmp_path / "bin/python3" + python.write_text( + f"#!{sys.executable}\n" + "import json, os, pathlib, sys\n" + "args = sys.argv[1:]\n" + "assert args[:2] == ['-m', 'lm_eval']\n" + "pathlib.Path(os.environ['CAPTURE']).write_text(json.dumps(args))\n" + "output = pathlib.Path(args[args.index('--output_path') + 1])\n" + "output.mkdir(parents=True, exist_ok=True)\n" + "(output / 'results_fixture.json').write_text('{\"score\":0.75}')\n" + "sys.exit(int(os.environ['CLIENT_EXIT']))\n" + ) + env = { + **client_environment, "CLIENT_EXIT": str(eval_exit), "MODEL_NAME": "served-model", + "TP": "4", "EP_SIZE": "4", "DP_ATTENTION": "true", "IS_MULTINODE": "false", + "MAX_MODEL_LEN": "8192", "EVAL_MAX_MODEL_LEN": "8192", "OPENAI_API_KEY": "EMPTY", + "INFERENCEX_LM_EVAL_RUNTIME_READY": "true", "EVAL_ONLY": "true", "RUN_EVAL": "true", + "EVAL_RESULT_DIR": str(tmp_path / "eval-output"), "FRAMEWORK": "sglang", "PRECISION": "fp8", + } + status = tmp_path / "eval-status" + result = subprocess.run( + ["bash", str(scripts / "srt_eval.sh"), "http://localhost:9444", str(status)], + env=env, capture_output=True, text=True, + ) + assert result.returncode == eval_exit, result.stderr + assert status.read_text() == f"{eval_exit}\n" + assert json.loads((workspace / "results_fixture.json").read_text()) == {"score": 0.75} + metadata = json.loads((workspace / "meta_env.json").read_text()) + assert (metadata["tp"], metadata["ep"], metadata["dp_attention"], metadata["conc"]) == (4, 4, True, 3) + argv = json.loads(Path(env["CAPTURE"]).read_text()) + model_args = argv[argv.index("--model_args") + 1] + assert "model=served-model,base_url=http://0.0.0.0:9444/v1/chat/completions" in model_args + assert "num_concurrent=3" in model_args diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py new file mode 100644 index 0000000000..979f286daf --- /dev/null +++ b/utils/test_srt_single_node.py @@ -0,0 +1,455 @@ +"""Behavioral checks for binding a matrix point to a native SRT recipe.""" + +import copy +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest +import yaml + +from infx.srt_slurm.single_node import runtime_arguments, select_recipe, submission_fields +from infx.srt_slurm.synthetic_acceptance import plan_commands, selected_recipes + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) +from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides + + +@pytest.fixture +def point(tmp_path): + recipe = { + "engine": "sglang", + "resources": {"gpus_per_node": 8}, + "model": {"path": "hf:test/model", "container": "test:tag", "precision": "fp8"}, + "roles": {"agg": { + "nodes": 1, "workers": 1, "gpus": 4, + "args": {"tensor-parallel-size": 4, "data-parallel-size": 1, "max-running-requests": 32}, + }}, + "benchmark": {"type": "custom", "env": { + "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", + "USE_CHAT_TEMPLATE": "false", + }}, + } + path = tmp_path / "recipe.yaml" + path.write_text(yaml.safe_dump({"base": recipe, "zip_override_conc": { + "benchmark": {"env": {"CONC": ["2", "4"]}}, + }})) + env = { + "FRAMEWORK": "sglang", "MODEL": "test/model", "IMAGE": "test:tag", "PRECISION": "fp8", + "TP": "4", "GPU_COUNT": "4", "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", + "EP_SIZE": "1", "DP_ATTENTION": "false", "SPEC_DECODING": "none", "IS_AGENTIC": "0", + "RUN_EVAL": "false", "EVAL_ONLY": "false", "ISL": "256", "OSL": "64", + "RANDOM_RANGE_RATIO": "0.5", "CONC": "2", "RESULT_FILENAME": "point-identity", + "GPU_MONITOR_INTERVAL": "3", "MODEL_PREFIX": "test", + } + return path, recipe, env + + +def test_native_binding_submits_one_point_and_keeps_server_settings(point): + path, recipe, env = point + argv = runtime_arguments(f"{path}:base", env) + overrides = parse_overrides(argv[1::2], []) + actual = copy.deepcopy(recipe) + apply_overrides_to_recipe(actual, overrides) + assert actual["srun_options"] == { + "gpus-per-node": "4", "container-workdir": "/infmax-workspace", + } + assert actual["benchmark"]["env"] == { + "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", + "USE_CHAT_TEMPLATE": "false", + "CONC": "2", "RESULT_FILENAME": "point-identity", "GPU_MONITOR_INTERVAL": "3", + "RUN_EVAL": "false", "EVAL_ONLY": "false", "RESULT_DIR": "/logs", + "FRAMEWORK": "sglang", + } + assert actual["roles"]["agg"]["args"] == { + "tensor-parallel-size": 4, "data-parallel-size": 1, "max-running-requests": 32, + } + commands = plan_commands(f"{path}:base", "sglang", ["--json", "--yes", *argv], env) + assert commands == [["srtctl", "apply", "--json", "--yes", *argv, "--file", f"{path}:base"]] + + +@pytest.mark.parametrize("field,value,message", [ + ("TP", "2", "tensor-parallel-size"), ("IMAGE", "other:tag", "image"), + ("ISL", "128", "ISL"), ("RUN_EVAL", "yes", "RUN_EVAL"), + ("PP_SIZE", "2", "PP_SIZE"), ("RESULT_FILENAME", "", "Missing runtime input"), + ("EP_SIZE", "2", "expert-parallel-size"), ("SPEC_DECODING", "mtp", "SPEC_DECODING"), +]) +def test_mismatched_point_fails_before_submission(point, field, value, message): + path, _, env = point + with pytest.raises(ValueError, match=message): + runtime_arguments(f"{path}:base", {**env, field: value}) + + +def test_native_variants_select_only_the_matching_matrix_point(point): + path, _, env = point + config, recipe = select_recipe(str(path), {**env, "CONC": "4"}) + assert config == f"{path}:zip_override_conc[1]" + assert recipe["benchmark"]["env"]["CONC"] == "4" + argv = runtime_arguments(config, {**env, "CONC": "4"}) + assert plan_commands(config, "sglang", ["--json", *argv], env) == [[ + "srtctl", "apply", "--json", *argv, "--file", f"{path}:zip_override_conc[1]", + ]] + with pytest.raises(ValueError, match="exactly one"): + select_recipe(str(path), {**env, "CONC": "8"}) + + +def test_ambiguous_native_variants_are_rejected(point): + path, recipe, env = point + path.write_text(yaml.safe_dump({"base": recipe, "override_first": {}, "override_second": {}})) + with pytest.raises(ValueError, match="exactly one"): + runtime_arguments(str(path), env) + + +def test_mtp_binding_uses_real_verification_and_preserves_expert_parallelism(point): + path, recipe, env = point + recipe["roles"]["agg"]["args"].update({ + "expert-parallel-size": 4, "speculative-algorithm": "EAGLE", + "speculative-num-steps": 2, "speculative-num-draft-tokens": 3, + }) + recipe["roles"]["agg"]["env"] = {"SGLANG_SIMULATE_ACC_LEN": "2.5"} + recipe["benchmark"]["env"]["USE_CHAT_TEMPLATE"] = "true" + path.write_text(yaml.safe_dump({"base": recipe})) + env = {**env, "EP_SIZE": "4", "SPEC_DECODING": "mtp"} + argv = runtime_arguments(f"{path}:base", env) + commands = plan_commands(f"{path}:base", "sglang", ["--json", *argv], env) + assert commands == [[ + "srtctl", "apply", "--json", *argv, "--file", f"{path}:base", + "--unset", "roles.agg.env.SGLANG_SIMULATE_ACC_LEN", + ]] + recipe["benchmark"]["env"]["USE_CHAT_TEMPLATE"] = "false" + path.write_text(yaml.safe_dump({"base": recipe})) + with pytest.raises(ValueError, match="USE_CHAT_TEMPLATE"): + runtime_arguments(f"{path}:base", env) + + +def test_concurrency_selector_keeps_graph_capture_coupled_to_client(point): + path, recipe, env = point + path.write_text(yaml.safe_dump({"base": recipe, "zip_override_conc": { + "roles": {"agg": {"args": {"cuda-graph-max-bs": [2, 4]}}}, + "benchmark": {"env": {"CONC": ["2", "4"]}}, + }})) + argv = runtime_arguments(f"{path}:zip_override_conc[1]", {**env, "CONC": "4"}) + actual = selected_recipes(yaml.safe_load(path.read_text()), "zip_override_conc[1]")[0][1] + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual["roles"]["agg"]["args"]["cuda-graph-max-bs"] == 4 + assert actual["benchmark"]["env"]["CONC"] == "4" + with pytest.raises(ValueError, match="CONC"): + runtime_arguments(f"{path}:zip_override_conc[1]", env) + + +def test_eval_binding_changes_context_without_changing_selected_concurrency(point): + path, recipe, env = point + recipe["roles"]["agg"]["args"]["context-length"] = 512 + path.write_text(yaml.safe_dump({"base": recipe, "zip_override_conc": { + "benchmark": {"env": {"CONC": ["2", "4"]}}, + }})) + env = {**env, "EVAL_ONLY": "true", "RUN_EVAL": "true", "CONC": "4", "MAX_MODEL_LEN": "1024"} + config, _ = select_recipe(str(path), env) + argv = runtime_arguments(config, env) + raw = yaml.safe_load(path.read_text()) + apply_overrides_to_recipe(raw, parse_overrides(argv[1::2], [])) + actual = selected_recipes(raw, "zip_override_conc[1]")[0][1] + assert actual["roles"]["agg"]["args"]["context-length"] == 1024 + assert actual["benchmark"]["env"]["CONC"] == "4" + assert len(plan_commands(config, "sglang", ["--json", *argv], env)) == 1 + + +def test_dp_attention_is_validated_without_replacing_recipe_topology(point): + path, recipe, env = point + recipe["roles"]["agg"]["args"].update({ + "data-parallel-size": 4, "expert-parallel-size": 4, "enable-dp-attention": True, + }) + path.write_text(yaml.safe_dump({"base": recipe})) + env = {**env, "DP_ATTENTION": "true", "EP_SIZE": "4"} + actual = copy.deepcopy(recipe) + argv = runtime_arguments(f"{path}:base", env) + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual["roles"]["agg"]["args"] == { + "tensor-parallel-size": 4, "data-parallel-size": 4, + "max-running-requests": 32, "expert-parallel-size": 4, "enable-dp-attention": True, + } + with pytest.raises(ValueError, match="data-parallel-size|DP_ATTENTION"): + runtime_arguments(f"{path}:base", {**env, "DP_ATTENTION": "false"}) + + +def test_trt_binding_keeps_engine_options_and_sets_eval_token_budget(point): + path, recipe, env = point + recipe["engine"] = {"type": "trtllm", "served_model_name": "test/model"} + recipe["roles"]["agg"]["args"] = { + "tensor_parallel_size": 4, "moe_expert_parallel_size": 4, + "enable_attention_dp": True, "max_seq_len": 512, "max_num_tokens": 256, + "speculative_config": {"decoding_type": "MTP", "num_nextn_predict_layers": 3}, + "cuda_graph_config": {"batch_sizes": [1, 2, 4]}, + } + recipe["roles"]["agg"]["env"] = {"TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS": "3"} + recipe["benchmark"]["env"]["USE_CHAT_TEMPLATE"] = "true" + path.write_text(yaml.safe_dump({"base": recipe})) + env = {**env, "FRAMEWORK": "trt", "EP_SIZE": "4", "DP_ATTENTION": "true", + "SPEC_DECODING": "mtp", "EVAL_ONLY": "true", "MAX_MODEL_LEN": "1024"} + argv = runtime_arguments(f"{path}:base", env) + actual = copy.deepcopy(recipe) + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual["roles"]["agg"]["args"] == { + "tensor_parallel_size": 4, "moe_expert_parallel_size": 4, + "enable_attention_dp": True, "max_seq_len": 1024, "max_num_tokens": 1024, + "speculative_config": {"decoding_type": "MTP", "num_nextn_predict_layers": 3}, + "cuda_graph_config": {"batch_sizes": [1, 2, 4]}, + } + assert plan_commands(f"{path}:base", "trt", ["--json", *argv], env) == [[ + "srtctl", "apply", "--json", *argv, "--file", f"{path}:base", + "--unset", "roles.agg.env.TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS", + ]] + with pytest.raises(ValueError, match="moe_expert_parallel_size"): + runtime_arguments(f"{path}:base", {**env, "EP_SIZE": "1"}) + + +def test_atom_binding_uses_allocation_tp_and_native_mtp_arguments(point): + path, recipe, env = point + recipe["engine"] = "atom" + recipe["roles"]["agg"]["args"] = { + "method": "mtp", "num-speculative-tokens": 3, "kv_cache_dtype": "fp8", + "enable-expert-parallel": True, "enable-dp-attention": True, + } + recipe["benchmark"]["env"]["USE_CHAT_TEMPLATE"] = "true" + path.write_text(yaml.safe_dump({"base": recipe})) + env = {**env, "FRAMEWORK": "atom", "EP_SIZE": "4", "DP_ATTENTION": "true", + "SPEC_DECODING": "mtp", "EVAL_ONLY": "true", "MAX_MODEL_LEN": "2048"} + argv = runtime_arguments(f"{path}:base", env) + actual = copy.deepcopy(recipe) + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual["roles"]["agg"]["args"] == { + "method": "mtp", "num-speculative-tokens": 3, "kv_cache_dtype": "fp8", + "enable-expert-parallel": True, "enable-dp-attention": True, "max-model-len": 2048, + } + assert plan_commands(f"{path}:base", "atom", ["--json", *argv], env) == [[ + "srtctl", "apply", "--json", *argv, "--file", f"{path}:base", + ]] + for changes, error in [ + ({"EP_SIZE": "2"}, "expert parallelism"), + ({"EP_SIZE": "1"}, "enable-expert-parallel"), + ({"TP": "8", "EP_SIZE": "8"}, "ATOM TP"), + ({"DP_ATTENTION": "false"}, "DP_ATTENTION"), + ]: + with pytest.raises(ValueError, match=error): + runtime_arguments(f"{path}:base", {**env, **changes}) + + +@pytest.mark.parametrize("record,expected", [ + ({"status": "submitted", "slurm_job_id": "42", "output_dir": "/shared/42"}, ("42", "/shared/42")), + ({"status": "error"}, None), + ({"status": "submitted", "slurm_job_id": "42;43", "output_dir": "/shared/42"}, None), + ({"status": "submitted", "slurm_job_id": "42", "output_dir": "relative"}, None), +]) +def test_submission_manifest(tmp_path, record, expected): + path = tmp_path / "submission.json" + path.write_text(json.dumps(record)) + if expected is None: + with pytest.raises(ValueError): + submission_fields(path) + else: + assert submission_fields(path) == expected + + +@pytest.mark.parametrize("pool,failure", [ + ("h200-dgxc-slurm", "none"), ("h200-dgxc-slurm", "allocation"), + ("h200-dgxc-slurm", "submission"), ("h200-dgxc-slurm", "bootstrap"), + ("h200-cw", "none"), ("h100-cw", "none"), ("h100-dgxc-slurm", "none"), + ("b200-cw", "none"), ("b200-nb", "none"), ("b200-nscale-slurm", "none"), + ("b200-nscale-slurm", "agentic"), + ("b300-dsxe", "none"), + ("mi300x-amd", "none"), ("mi325x-amds", "none"), ("mi355x-amds", "none"), +] + [(pool, "missing-recipe") for pool in ( + "b200-cw", "b200-nb", "b200-nscale-slurm", "b300-dsxe", "h100-cw", + "h100-dgxc-slurm", "h200-cw", "h200-dgxc-slurm", "mi300x-amd", + "mi325x-amds", "mi355x-amds", +)]) +def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, pool, failure): + path, _, point_env = point + binaries = tmp_path / "bin" + binaries.mkdir() + model = tmp_path / "model" + model.mkdir() + (model / "config.json").write_text("{}") + (tmp_path / "benchmarks").symlink_to(ROOT / "benchmarks", target_is_directory=True) + capture = tmp_path / "cancelled" + # Only external executables are stubbed; run the real pool launcher, shared + # setup/profile/acceptance helpers, binder, and artifact collection. + scripts = { + "git": 'if [[ "$1" == clone ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', + "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', + "make": '[[ "$TEST_FAILURE" == bootstrap ]] && exit 13; mkdir -p bin; touch bin/uv', + "squeue": '[[ "$TEST_FAILURE" == submission || "$TEST_FAILURE" == agentic ]] && echo "42"; exit 0', + "salloc": 'echo "Granted job allocation 42"', + "sacct": 'if [[ "$TEST_FAILURE" == allocation ]]; then echo "FAILED|1:0"; else echo "COMPLETED|0:0"; fi', + "scancel": 'printf "%s\\n" "$@" >> "$CANCEL_CAPTURE"', + "tail": 'exit 0', + } + for name, script in scripts.items(): + binary = binaries / name + binary.write_text(f"#!/usr/bin/env bash\n{script}\n") + binary.chmod(0o755) + srtctl = binaries / "srtctl" + srtctl.write_text( + f"#!{sys.executable}\n" + "import json, os, pathlib, sys\n" + "assert pathlib.Path('bin/uv').is_file(), 'native bootstrap was skipped'\n" + "output = pathlib.Path(sys.argv[sys.argv.index('--output') + 1]) / '42'\n" + "logs = output / 'logs'\n" + "logs.mkdir(parents=True)\n" + "(logs / 'sweep_42.log').write_text('benchmark complete\\n')\n" + "(logs / (os.environ['RESULT_FILENAME'] + '.json')).write_text('{\"completed\":2}')\n" + "(logs / 'gpu_metrics.csv').write_text('gpu,power\\n0,300\\n')\n" + "(logs / 'gpu_metrics_context.json').write_text('{\"device_count\":4}')\n" + "print(json.dumps({'status':'submitted', 'slurm_job_id':'42', 'output_dir':str(output)}))\n" + "sys.exit(7 if os.environ['TEST_FAILURE'] == 'submission' else 0)\n" + ) + srtctl.chmod(0o755) + srun = binaries / "srun" + srun.write_text(f"#!{sys.executable}\n" + + "import json, os, pathlib, sys\n" + "with pathlib.Path(os.environ['SRUN_CAPTURE']).open('a') as f: f.write(json.dumps(sys.argv[1:])+'\\n')\n") + srun.chmod(0o755) + env = { + **os.environ, **point_env, + "PATH": f"{binaries}:{Path(sys.executable).parent}:{os.environ['PATH']}", + "PYTHONPATH": f"{ROOT}:{ROOT / 'utils/srt-slurm/src'}", + "GITHUB_WORKSPACE": str(tmp_path), "SRT_RECIPE": f"{path.name}:base", + "IS_MULTINODE": "false", "REQUIRE_POWER": "1", "SALLOC_TIME_LIMIT": "10", + "HF_HUB_CACHE_MOUNT": str(tmp_path), "AIPERF_MMAP_CACHE_HOST_PATH": str(tmp_path), + "HF_HUB_CACHE": "/hf", "SRT_MODEL_PATH": str(model), "MODEL_PREFIX": "dsr1", + "SLURM_ACCOUNT": "fixture", "SLURM_PARTITION": "fixture", + "B200_SQUASH_DIR": str(tmp_path), "B300_HF_CACHE_HOST_DIR": str(tmp_path), + "B300_HF_CACHE_CONTAINER_DIR": "/hf", "ENROOT_IMPORT_TIME_LIMIT": "10", + "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", + "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), + "SRUN_CAPTURE": str(tmp_path / "srun.jsonl"), + "KEEP_LOGS": "0", + } + env.pop("AIPERF_DRAIN_TIMEOUT_SECONDS", None) + env.pop("AIPERF_DRAIN_POLL_SECONDS", None) + env.pop("BENCH_SCRIPT_OVERRIDE", None) + if failure == "missing-recipe": + env.pop("SRT_RECIPE") + if failure == "agentic": + env.update(IS_AGENTIC="1", SCENARIO_SUBDIR="agentic/", EXP_NAME="fixture_agentic", + RUNNER_NAME="fixture_00", SRT_RECIPE="unused.yaml") + result = subprocess.run( + ["bash", str(ROOT / f"runners/launch_{pool}.sh")], cwd=tmp_path, + env=env, capture_output=True, text=True, timeout=30, + ) + assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0}[failure], result.stderr + if failure == "agentic": + calls = [json.loads(line) for line in Path(env["SRUN_CAPTURE"]).read_text().splitlines()] + assert calls[-1][-2:] == ["bash", "benchmarks/single_node/agentic/fixture_fp8_b200.sh"] + assert "--jobid=42" in calls[-1] + assert not (tmp_path / "srt-single-node-submission.json").exists() + return + if failure == "missing-recipe": + assert "SRT_RECIPE" in result.stdout + assert not (tmp_path / "srt-single-node-submission.json").exists() + assert not (tmp_path / "point-identity.json").exists() + assert not capture.exists() + return + if failure == "bootstrap": + assert not (tmp_path / "srt-single-node-submission.json").exists() + assert not capture.exists() + return + assert json.loads((tmp_path / "point-identity.json").read_text()) == {"completed": 2} + assert (tmp_path / "gpu_metrics.csv").read_text() == "gpu,power\n0,300\n" + assert json.loads((tmp_path / "gpu_metrics_context.json").read_text()) == {"device_count": 4} + assert (tmp_path / "srt-single-node-logs.tar.gz").stat().st_size > 0 + cluster_config = yaml.safe_load(next(tmp_path.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) + assert cluster_config["containers"]["test:tag"] == "test:tag" + assert cluster_config["use_exclusive_sbatch_directive"] is True + assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") + + +def test_runtime_container_options_remain_native_mapping(point): + path, recipe, env = point + env = {**env, "SRT_SRUN_OPTIONS": json.dumps({ + "container-remap-root": "", "container-writable": "", "container-workdir": "/custom", + })} + argv = runtime_arguments(f"{path}:base", env) + actual = copy.deepcopy(recipe) + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual['srun_options'] == { + 'gpus-per-node': '4', 'container-remap-root': '', 'container-writable': '', + 'container-workdir': '/custom', + } + with pytest.raises(ValueError, match='must map option names to string values'): + runtime_arguments(f"{path}:base", {**env, 'SRT_SRUN_OPTIONS': '{"container-remap-root": true}'}) + + +@pytest.mark.parametrize("controller,expected", [ + ("JobId=42 JobState=COMPLETED ExitCode=0:0", 0), + ("JobId=42 JobState=FAILED ExitCode=1:0", 1), + ("JobId=42 JobState=COMPLETED ExitCode=0:9", 1), + ("JobId=43 JobState=COMPLETED ExitCode=0:0", 1), + ("", 1), +]) +def test_terminal_allocation_without_accounting(tmp_path, controller, expected): + binaries = tmp_path / "bin" + binaries.mkdir() + for name, body in { + "sacct": "exit 1", + "scontrol": 'printf "%s\\n" "$CONTROLLER_RECORD"', + "sleep": "exit 0", + }.items(): + binary = binaries / name + binary.write_text(f"#!/usr/bin/env bash\n{body}\n") + binary.chmod(0o755) + result = subprocess.run( + ["bash", "-c", 'source "$1"; verify_slurm_job_status 42', "bash", str(ROOT / "runners/slurm_utils.sh")], + env={**os.environ, "PATH": f"{binaries}:{os.environ['PATH']}", "CONTROLLER_RECORD": controller}, + capture_output=True, text=True, timeout=5, + ) + assert result.returncode == expected, result.stdout + result.stderr + if expected: + assert "ERROR:" in result.stderr + + +@pytest.mark.parametrize("collector", [False, True]) +def test_b300_keeps_agentic_and_explicit_collector_dispatch(tmp_path, collector): + binaries = tmp_path / "bin" + binaries.mkdir() + for name, body in { + # Host image-cache directories and Slurm operations are external here. + "mkdir": "exit 0", + "unsquashfs": "exit 0", + "salloc": 'echo "Granted job allocation 42"', + "scancel": 'printf "%s\\n" "$@" > "$CANCEL_CAPTURE"', + }.items(): + binary = binaries / name + binary.write_text(f"#!/usr/bin/env bash\n{body}\n") + binary.chmod(0o755) + srun = binaries / "srun" + srun.write_text(f"#!{sys.executable}\n" + + "import json, os, pathlib, sys\n" + "with pathlib.Path(os.environ['SRUN_CAPTURE']).open('a') as f: f.write(json.dumps(sys.argv[1:])+'\\n')\n" + "sys.exit(7 if '--container-image' in ' '.join(sys.argv) else 0)\n") + srun.chmod(0o755) + env = {**os.environ, "PATH": f"{binaries}:{os.environ['PATH']}", + "GITHUB_WORKSPACE": str(tmp_path), "B300_HF_CACHE_HOST_DIR": str(tmp_path / "cache"), + "B300_HF_CACHE_CONTAINER_DIR": "/cache", "RUNNER_NAME": "fixture_00", + "ENROOT_IMPORT_TIME_LIMIT": "10", "SALLOC_TIME_LIMIT": "10", "IS_MULTINODE": "false", + "IS_AGENTIC": "0" if collector else "1", "EVAL_ONLY": "false", "RUN_EVAL": "false", + "MODEL": "test/DeepSeek-V4-Pro", "MODEL_PREFIX": "fixture", "PRECISION": "fp4", + "FRAMEWORK": "vllm", "EXP_NAME": "fixture_workload", "IMAGE": "fixture:tag", + "SPEC_DECODING": "none", "GPU_COUNT": "4", + "SCENARIO_SUBDIR": "fixed_seq_len/" if collector else "agentic/", + "SRUN_CAPTURE": str(tmp_path / "srun.jsonl"), "CANCEL_CAPTURE": str(tmp_path / "cancelled")} + env.pop("SRT_RECIPE", None) + env.pop("BENCH_SCRIPT_OVERRIDE", None) + if collector: + env["BENCH_SCRIPT_OVERRIDE"] = "benchmarks/single_node/speedbench/fixture.py" + result = subprocess.run(["bash", str(ROOT / "runners/launch_b300-dsxe.sh")], cwd=tmp_path, + env=env, capture_output=True, text=True, timeout=10) + assert result.returncode == 7, result.stdout + result.stderr + calls = [json.loads(line) for line in Path(env["SRUN_CAPTURE"]).read_text().splitlines()] + expected = "benchmarks/single_node/speedbench/fixture.py" if collector else "benchmarks/single_node/agentic/fixture_fp4_b300.sh" + assert calls[-1][-2:] == ["bash", expected] + assert "--jobid=42" in calls[-1] + assert (tmp_path / "cancelled").read_text() == "42\n"