diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index 74dce63495..fb219570fc 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -19,6 +19,115 @@ check_env_vars() { fi } +# Report live members of explicitly owned process groups. Zombies cannot hold +# output pipes open. Do not use leader liveness: a router can orphan its workers. +_background_process_groups_alive() { + local groups=" $* " + local listing + listing=$(ps -eo pgid=,stat=) || return 1 + awk -v groups="$groups" ' + index(groups, " " $1 " ") && $2 !~ /^[ZX]/ { alive[$1] = 1 } + END { for (group in alive) print group } + ' <<< "$listing" +} + +# Called only after benchmark/eval work ends. Preserve its exit status while +# bounding teardown of the setsid groups recorded by the launcher. Grace periods +# are explicit arguments, independent of benchmark duration and server readiness. +stop_background_process_groups() { + local work_status="$1" term_grace="$2" kill_grace="$3" + shift 3 + local pgid own_pgid remaining deadline cleanup_status=0 + local groups=("$@") + if [[ ! "$work_status" =~ ^[0-9]+$ || ! "$term_grace" =~ ^[0-9]+$ || ! "$kill_grace" =~ ^[0-9]+$ ]]; then + echo "ERROR: invalid process-group cleanup status or grace period" >&2 + return 1 + fi + if ! own_pgid=$(ps -o pgid= -p "$$"); then + if [[ "$work_status" -ne 0 ]]; then return "$work_status"; fi + return 1 + fi + own_pgid="${own_pgid//[[:space:]]/}" + for pgid in "${groups[@]}"; do + if [[ ! "$pgid" =~ ^[1-9][0-9]*$ || "$pgid" -le 1 || "$pgid" == "$own_pgid" ]]; then + echo "ERROR: refusing unsafe process-group cleanup: '$pgid'" >&2 + if [[ "$work_status" -ne 0 ]]; then return "$work_status"; fi + return 1 + fi + done + if [[ ${#groups[@]} -eq 0 ]]; then return "$work_status"; fi + + echo "Stopping owned process groups: ${groups[*]}" + for pgid in "${groups[@]}"; do + kill -TERM -- "-$pgid" 2>/dev/null || true + done + deadline=$((SECONDS + term_grace)) + while true; do + remaining=$(_background_process_groups_alive "${groups[@]}") || { cleanup_status=1; break; } + [[ -n "$remaining" && $SECONDS -lt $deadline ]] || break + sleep 1 + done + if [[ -n "$remaining" ]]; then + echo "TERM grace expired; force-stopping owned process groups: $remaining" + for pgid in $remaining; do + kill -KILL -- "-$pgid" 2>/dev/null || true + done + deadline=$((SECONDS + kill_grace)) + while true; do + remaining=$(_background_process_groups_alive "${groups[@]}") || { cleanup_status=1; break; } + [[ -n "$remaining" && $SECONDS -lt $deadline ]] || break + sleep 1 + done + if [[ -n "$remaining" ]]; then + echo "ERROR: process groups still alive after KILL grace: $remaining" >&2 + cleanup_status=1 + fi + fi + if [[ "$work_status" -ne 0 ]]; then return "$work_status"; fi + return "$cleanup_status" +} + +# EXIT handler for launchers that own explicit setsid groups and optionally one +# auxiliary daemon PID. Disable this handler before exiting to avoid recursion. +# Run auxiliary cleanup even when group cleanup fails, preserving the work code. +exit_after_background_process_cleanup() { + local work_status="$1" term_grace="$2" kill_grace="$3" auxiliary_pid="$4" + shift 4 + local final_status + trap - EXIT + if stop_background_process_groups "$work_status" "$term_grace" "$kill_grace" "$@"; then + final_status=0 + else + final_status=$? + fi + if [[ -n "$auxiliary_pid" ]]; then + kill "$auxiliary_pid" 2>/dev/null || true + fi + exit "$final_status" +} + +# Finish preflight on every allocated node before any server container starts its +# peer-readiness deadline. A failed node prevents the entire serving step. +run_amd_multinode_after_preflight() { + local nodelist="$1" node_count="$2" preflight_script="$3" + local container_filter="$4" skip_gpu_sanity="$5" + shift 5 + local preflight_rc + if srun --nodelist="$nodelist" \ + --nodes="$node_count" --ntasks="$node_count" --ntasks-per-node=1 \ + --kill-on-bad-exit=1 --unbuffered \ + bash "$preflight_script" "$container_filter" "$skip_gpu_sanity"; then + echo "[preflight] all nodes ready; launching server containers" + else + preflight_rc=$? + echo "[preflight][ERROR] node preflight failed; no server containers launched" >&2 + return "$preflight_rc" + fi + srun --nodelist="$nodelist" \ + --nodes="$node_count" --ntasks="$node_count" --ntasks-per-node=1 \ + --kill-on-bad-exit=1 --signal=TERM@30 --unbuffered "$@" +} + # Launchers may load only input validation, without benchmark initialization. if [[ "${1-}" == "--validation-only" ]]; then return 0 diff --git a/benchmarks/multi_node/amd_utils/job.slurm b/benchmarks/multi_node/amd_utils/job.slurm index c12535776c..d1f5ea11a7 100755 --- a/benchmarks/multi_node/amd_utils/job.slurm +++ b/benchmarks/multi_node/amd_utils/job.slurm @@ -583,11 +583,10 @@ if [[ -n "${CLIENT_IMAGE:-}" ]]; then srun --nodelist="$SELECTED_NODELIST_SRUN" bash -c 'eval "$DOCKER_CMD_DETECT"; $DOCKER_CMD pull '"$CLIENT_IMAGE"' >/dev/null 2>&1 || true' 2>/dev/null || true fi -srun \ - --nodelist="$SELECTED_NODELIST_SRUN" \ - --kill-on-bad-exit=1 \ - --signal=TERM@30 \ - --unbuffered \ +run_amd_multinode_after_preflight \ + "$SELECTED_NODELIST_SRUN" "$NUM_NODES" \ + "$DI_REPO_DIR/benchmarks/multi_node/amd_utils/preflight_node.sh" \ + "$CONT_FILTER" "$SKIP_GPU_SANITY" \ bash -lc " set -eo pipefail @@ -675,23 +674,6 @@ else fi fi # end: if ENGINE == atom-disagg -# Pre-clean (idempotent): stop then force-remove so GPU VRAM is released -# before the drain gate. stop-only left containers in Created/Exited state -# on some nodes. -\$DOCKER_CMD ps -aq --filter \"$CONT_FILTER\" | xargs -r \$DOCKER_CMD rm -f || true -\$DOCKER_CMD ps -aq | xargs -r \$DOCKER_CMD stop -t 15 || true -\$DOCKER_CMD ps -aq | xargs -r \$DOCKER_CMD rm -f || true -sleep 2 - -# GPU drain gate: fail fast on leftover VRAM use instead of OOMing in model -# load ~15 min later. Reuses wait_for_amd_gpu_clean from benchmark_lib.sh. -if [[ \"${SKIP_GPU_SANITY}\" == \"1\" ]]; then - echo \"[INFO] SKIP_GPU_SANITY=1 set; skipping GPU pre-flight drain check\" -else - # Unset so benchmark_lib.sh's unrelated agentic KV_OFFLOADING check doesn't exit 1 here. - bash -c \"unset IS_AGENTIC SCENARIO_TYPE; source $DI_REPO_DIR/benchmarks/benchmark_lib.sh && wait_for_amd_gpu_clean\" -fi - # Start vLLM external router container on node 0 if [[ \"$ENGINE\" == \"vllm-disagg\" && \"$ROUTER_TYPE\" == \"vllm-router\" && \"\$SLURM_PROCID\" == \"0\" ]]; then \$DOCKER_CMD rm -f \"$ROUTER_CONT_NAME\" 2>/dev/null || true @@ -775,6 +757,7 @@ echo \"[rank 0] Main container exited (rc=\$DOCKER_EXIT_CODE). Stopping vllm-rou \$DOCKER_CMD rm -f \"$ROUTER_CONT_NAME\" 2>/dev/null || true exit \$DOCKER_EXIT_CODE " +SERVER_SRUN_RC=$? if [[ "${KEEP_CONTAINERS}" != "1" ]]; then srun --nodelist="$SELECTED_NODELIST_SRUN" bash -c 'eval "$DOCKER_CMD_DETECT"; $DOCKER_CMD rm -f '"$DOCKER_CONT_NAME"' '"$CLIENT_CONT_NAME"' 2>/dev/null || true' @@ -786,3 +769,29 @@ if [[ "${KEEP_CONTAINERS}" != "1" ]]; then ' fi fi + +# /run_logs is backed by each compute node's local /tmp, so the node-0 copy +# performed by the engine launcher cannot see prefill/decode logs written on +# other nodes. Collect after the server step and container cleanup so failed +# runs also include shutdown output. KEEP_CONTAINERS=1 retains a snapshot of +# any containers left running for debugging. +# Use sudo because the container-created source and the existing node-0 +# destination can be root-owned. Restore ownership after the fan-in so a +# subsequent runner job can clean the workspace normally. +SHARED_JOB_LOGS="${BENCHMARK_LOGS_DIR}/logs/slurm_job-${SLURM_JOB_ID}" +if ! srun --nodelist="$SELECTED_NODELIST_SRUN" \ + --nodes="$NUM_NODES" --ntasks="$NUM_NODES" --ntasks-per-node=1 \ + bash "$DI_REPO_DIR/benchmarks/multi_node/amd_utils/stage_node_logs.sh" \ + "/tmp/slurm_job-${SLURM_JOB_ID}" "$SHARED_JOB_LOGS"; then + echo "[logs][ERROR] failed to stage logs from one or more Slurm nodes" >&2 + if [[ "$SERVER_SRUN_RC" -eq 0 ]]; then + SERVER_SRUN_RC=1 + fi +fi + +if [[ -d "$SHARED_JOB_LOGS" ]]; then + sudo chown -R "$(id -u):$(id -g)" "$SHARED_JOB_LOGS" 2>/dev/null || true + chmod -R a+rwX "$SHARED_JOB_LOGS" 2>/dev/null || true +fi + +exit "$SERVER_SRUN_RC" diff --git a/benchmarks/multi_node/amd_utils/models.yaml b/benchmarks/multi_node/amd_utils/models.yaml index f8507cf0f0..53666ce487 100644 --- a/benchmarks/multi_node/amd_utils/models.yaml +++ b/benchmarks/multi_node/amd_utils/models.yaml @@ -354,9 +354,21 @@ DeepSeek-R1-0528-MXFP4-v2: DeepSeek-V4-Pro-AgentX: &DeepSeek-V4-Pro-AgentX base_flags: "--enable-deepseek-v4-fp4-indexer --watchdog-timeout 3600 --load-balance-method round_robin --kv-cache-dtype fp8_e4m3 --attention-backend dsv4 --page-size 256 --swa-full-tokens-ratio 0.1 --enforce-shared-experts-fusion --tool-call-parser deepseekv4 --reasoning-parser deepseek-v4 --disaggregation-transfer-backend mori --tokenizer-worker-num 8 --stream-interval 20 --log-level info --log-level-http error" - dp_flags: "--enable-dp-attention --swa-full-tokens-ratio 0.15 --enable-dp-attention-local-control-broadcast" + # --enable-dp-lm-head is required by SGLang for DSpark under DP attention; it + # is harmless for the EAGLE/MTP arms, so it stays unconditional here rather + # than needing a second DP flag string. + dp_flags: "--enable-dp-attention --enable-dp-lm-head --swa-full-tokens-ratio 0.15 --enable-dp-attention-local-control-broadcast" ep_flags: "--ep-dispatch-algorithm fake --moe-a2a-backend mori --deepep-mode normal" mtp_flags: "--speculative-algorithm EAGLE --speculative-eagle-topk 1" + # DSpark draft head, selected when the sweep sets spec-decoding: draft_model. + # Kept alongside mtp_flags rather than replacing it, so recipes that stay on + # spec-decoding: mtp keep the EAGLE arm untouched. The -0813 checkpoint + # bundles the draft head (dspark_block_size / dspark_markov_rank / + # dspark_target_layer_ids in config.json), so --speculative-draft-model-path + # defaults to --model-path and no separate draft checkpoint is needed. + # server_sglang.sh appends the block size and the verify window from + # DECODE_MTP_SIZE (= gamma); unlike EAGLE, num-steps is pinned to 1. + dspark_flags: "--speculative-algorithm DSPARK --speculative-eagle-topk 1" prefill: disable_radix_cache: false disable_cuda_graph: true @@ -384,8 +396,11 @@ DeepSeek-V4-Pro-AgentX: &DeepSeek-V4-Pro-AgentX max_running_requests: "BENCH_MAX_CONC_VALUE*2" cuda_graph_bs_range: "1-BENCH_MAX_CONC_VALUE*2" -# Pro-0813 retains EAGLE 3-1-4 for PD compatibility. Synthetic acceptance is -# checkpoint-specific in server_sglang.sh: thinking-on, length 3 uses AL 3.01. +# Pro-0813 serves the PD path with DSPARK (spec-decoding: draft_model), which +# measured clean across c4-c256; the EAGLE 3-1-4 arm remains available through +# spec-decoding: mtp. Synthetic acceptance is checkpoint-specific in +# server_sglang.sh and does not depend on which of the two runs: thinking-on, +# draft length 3 uses AL 3.01. DeepSeek-V4-Pro-0813-AgentX: *DeepSeek-V4-Pro-AgentX DeepSeek-V4-Pro-DI: diff --git a/benchmarks/multi_node/amd_utils/preflight_node.sh b/benchmarks/multi_node/amd_utils/preflight_node.sh new file mode 100644 index 0000000000..34759bab89 --- /dev/null +++ b/benchmarks/multi_node/amd_utils/preflight_node.sh @@ -0,0 +1,31 @@ +#!/usr/bin/env bash +set -eo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +source "$SCRIPT_DIR/../../benchmark_lib.sh" --validation-only +check_env_vars DOCKER_CMD_DETECT DI_REPO_DIR SLURM_JOB_ID +CONT_FILTER="$1" +SKIP_GPU_SANITY="$2" +check_env_vars CONT_FILTER SKIP_GPU_SANITY + +preflight_node() { + eval "$DOCKER_CMD_DETECT" + + # Preserve the existing pre-clean scope and ordering. Moving it into this + # separate Slurm step prevents one node starting while another still drains. + $DOCKER_CMD ps -aq --filter "$CONT_FILTER" | xargs -r $DOCKER_CMD rm -f || true + $DOCKER_CMD ps -aq | xargs -r $DOCKER_CMD stop -t 15 || true + $DOCKER_CMD ps -aq | xargs -r $DOCKER_CMD rm -f || true + sleep 2 + + if [[ "$SKIP_GPU_SANITY" == "1" ]]; then + echo "[INFO] SKIP_GPU_SANITY=1 set; skipping GPU pre-flight drain check" + else + # Avoid benchmark-only agentic initialization on the host, as before. + bash -c 'unset IS_AGENTIC SCENARIO_TYPE; source "$DI_REPO_DIR/benchmarks/benchmark_lib.sh" && wait_for_amd_gpu_clean' + fi +} + +NODE_LOG_DIR="/tmp/slurm_job-${SLURM_JOB_ID}" +mkdir -p "$NODE_LOG_DIR" +preflight_node 2>&1 | tee "$NODE_LOG_DIR/preflight_$(hostname).log" diff --git a/benchmarks/multi_node/amd_utils/server_sglang.sh b/benchmarks/multi_node/amd_utils/server_sglang.sh index b44ea1a11f..39ab32be20 100755 --- a/benchmarks/multi_node/amd_utils/server_sglang.sh +++ b/benchmarks/multi_node/amd_utils/server_sglang.sh @@ -23,6 +23,12 @@ export BENCH_MAX_CONC_VALUE source $SGLANG_WS_PATH/setup_deps.sh source $SGLANG_WS_PATH/env.sh +# Install before starting UMBP or serving processes. Early readiness failures must +# close the same owned groups as normal completion, including orphaned workers. +SGLANG_OWNED_PGIDS=() +UMBP_SA_PID="" +trap 'exit_after_background_process_cleanup "$?" 30 5 "$UMBP_SA_PID" "${SGLANG_OWNED_PGIDS[@]}"' EXIT + host_ip=$(ip route get 1.1.1.1 | awk '/src/ {print $7}') host_name=$(hostname) @@ -100,6 +106,7 @@ def parse_range(cuda_range, default_start, default_end): # Output shell variables print(f'MODEL_BASE_FLAGS=\"{m.get(\"base_flags\", \"\")}\"') print(f'MODEL_MTP_FLAGS=\"{m.get(\"mtp_flags\", \"\")}\"') +print(f'MODEL_DSPARK_FLAGS=\"{m.get(\"dspark_flags\", \"\")}\"') print(f'MODEL_DP_FLAGS=\"{m.get(\"dp_flags\", \"\")}\"') print(f'MODEL_EP_FLAGS=\"{m.get(\"ep_flags\", \"\")}\"') @@ -381,8 +388,30 @@ build_server_config() { local ep_config="" local specific_config="" + # Speculative-decoding config (only if a draft length is set). + # + # DECODE_MTP_SIZE carries the draft length for BOTH algorithms, but the two + # spend it differently: + # EAGLE/MTP -- num-steps = draft length, i.e. that many sequential draft + # forward passes, each producing one token. + # DSPARK -- one draft pass emits a whole block, so num-steps is + # pinned to 1 and the draft length becomes the block size + # (gamma). Passing gamma as num-steps here would ask for + # gamma sequential DSpark passes instead of one gamma-token + # block. + # The verify window (num-draft-tokens = draft length + 1) is the same for + # both, which is also what makes the MORI decode dispatch scaling + # (x (DECODE_MTP_SIZE + 1)) correct for DSpark without further change. if [ "$decode_mtp_size" -gt 0 ]; then - mtp_config="${MODEL_MTP_FLAGS} --speculative-num-steps ${decode_mtp_size} --speculative-num-draft-tokens $((decode_mtp_size + 1))" + if [[ "${SPEC_DECODING:-}" == "draft_model" ]]; then + # MODEL_DSPARK_FLAGS is validated at the call site, not here: this + # function is only ever invoked inside $( ), where an exit would + # terminate the subshell and leave the caller with an empty config + # rather than aborting the launch. + mtp_config="${MODEL_DSPARK_FLAGS} --speculative-dspark-block-size ${decode_mtp_size} --speculative-num-steps 1 --speculative-num-draft-tokens $((decode_mtp_size + 1))" + else + mtp_config="${MODEL_MTP_FLAGS} --speculative-num-steps ${decode_mtp_size} --speculative-num-draft-tokens $((decode_mtp_size + 1))" + fi fi if [[ "$enable_dp" == "true" ]]; then @@ -439,6 +468,15 @@ build_server_config() { echo "$full_config" } +# Validate the DSpark path before building either config. This has to happen at +# top level: build_server_config only ever runs inside $( ), so an exit there +# would kill the subshell and hand the caller an empty config string instead of +# stopping the launch. +if [[ "$DECODE_MTP_SIZE" -gt 0 ]] && [[ "${SPEC_DECODING:-}" == "draft_model" ]] && [[ -z "${MODEL_DSPARK_FLAGS// }" ]]; then + echo "FATAL: SPEC_DECODING=draft_model but model '${MODEL_NAME}' has no dspark_flags in models.yaml." >&2 + exit 1 +fi + PREFILL_SERVER_CONFIG=$(build_server_config "prefill" "$MODEL_NAME" "$PREFILL_TP_SIZE" "$PREFILL_ENABLE_EP" "$PREFILL_ENABLE_DP" "$DECODE_MTP_SIZE") DECODE_SERVER_CONFIG=$(build_server_config "decode" "$MODEL_NAME" "$DECODE_TP_SIZE" "$DECODE_ENABLE_EP" "$DECODE_ENABLE_DP" "$DECODE_MTP_SIZE") @@ -654,7 +692,6 @@ elif [[ "$KV_OFFLOADING" != "none" && "$KV_OFFLOAD_BACKEND" == umbp-linker* ]]; "$UMBP_SA_BIN" "$UMBP_STANDALONE_ADDRESS" > "$UMBP_SA_LOG" 2>&1 & UMBP_SA_PID=$! echo "[UMBP] standalone server PID: $UMBP_SA_PID" - trap '[[ -n "${UMBP_SA_PID:-}" ]] && kill "$UMBP_SA_PID" 2>/dev/null || true' EXIT # Three waits, all bounded by wall time rather than by a guess at how # fast this node is. Bind time for a 549 GB tier measured 120 s on @@ -927,8 +964,8 @@ if [ "$NODE_RANK" -eq 0 ]; then > >(tee /run_logs/slurm_job-${SLURM_JOB_ID}/prefill_${host_name}.log >/dev/null) 2>&1 & set +x prefill0_pid=$! - prefill0_pgid=$(ps -o pgid= -p "$prefill0_pid" 2>/dev/null | tr -d ' ') - : "${prefill0_pgid:=$prefill0_pid}" + prefill0_pgid=$prefill0_pid + SGLANG_OWNED_PGIDS+=("$prefill0_pgid") fi echo "Waiting for all prefill and decode servers to be up . . ." @@ -985,8 +1022,8 @@ if [ "$NODE_RANK" -eq 0 ]; then fi set +x proxy_pid=$! - proxy_pgid=$(ps -o pgid= -p "$proxy_pid" 2>/dev/null | tr -d ' ') - : "${proxy_pgid:=$proxy_pid}" + proxy_pgid=$proxy_pid + SGLANG_OWNED_PGIDS+=("$proxy_pgid") HEALTH_BARRIER_CMD="python3 $SGLANG_WS_PATH/sync.py barrier \ --node-ips ${NODE0_ADDR} \ @@ -1082,6 +1119,7 @@ if [ "$NODE_RANK" -eq 0 ]; then IS_AGENTIC_RUN=1 fi + BENCHMARK_EXIT_CODE=0 if [[ "${EVAL_ONLY}" == "true" ]]; then echo "EVAL_ONLY mode: skipping throughput benchmark" elif [[ "$DRY_RUN" -eq 1 ]]; then @@ -1156,10 +1194,12 @@ print(json.dumps(json.loads(sys.stdin.read())))' <<<"$_val")" || { --entrypoint "" \ "${CLIENT_IMAGE}" \ bash -lc "cd /workspace/benchmarks/multi_node/amd_utils && bash trace_replay.sh /models ${MODEL_NAME} \"${BENCH_MAX_CONCURRENCY}\" /run_logs/slurm_job-${SLURM_JOB_ID}" + BENCHMARK_EXIT_CODE=$? set +x else set -x eval "$BENCH_CMD" + BENCHMARK_EXIT_CODE=$? set +x fi @@ -1183,6 +1223,8 @@ print(json.dumps(json.loads(sys.stdin.read())))' <<<"$_val")" || { # Must run from repo root so infx/evals/gsm8k.yaml resolves pushd /workspace + # Match the disaggregation router launched above. + export PORT=30000 source /workspace/benchmarks/benchmark_lib.sh # CONC must be exported before run_eval so meta_env.json matches validate_scores.py. @@ -1204,9 +1246,9 @@ print(json.dumps(json.loads(sys.stdin.read())))' <<<"$_val")" || { # arrive via Docker -e flags from job.slurm. if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: run_eval --port 30000 (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS}, ctx=${EVAL_MAX_MODEL_LEN:-auto})" + echo "DRY RUN: run_eval --port ${PORT} (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS}, ctx=${EVAL_MAX_MODEL_LEN:-auto})" else - run_eval --port 30000 + run_eval --port "$PORT" eval_rc=$? if [[ $eval_rc -ne 0 ]]; then @@ -1247,22 +1289,11 @@ print(json.dumps(json.loads(sys.stdin.read())))' <<<"$_val")" || { echo "Copied results to $LOGS_OUTPUT/slurm_job-${SLURM_JOB_ID}" fi - echo "Killing the proxy server and prefill server" - - if [[ "$DRY_RUN" -eq 0 ]]; then - # Group-kill the router (setsid at launch): the python launcher has usually - # exited after spawning the Rust worker, which reparents to init but stays in - # this group; kill $proxy_pid alone misses it and :30000 stays open. - kill -TERM -"${proxy_pgid:-$proxy_pid}" 2>/dev/null || true - # Group-kill the prefill tree so TP-scheduler children release the tee pipe - # and the container can exit. - kill -TERM -"${prefill0_pgid:-$prefill0_pid}" 2>/dev/null || true - fi - - if [[ "${EVAL_FAILED:-0}" -eq 1 ]]; then - echo "ERROR: eval failed; exiting node-0 with rc=1" - exit 1 + node_exit_status=$BENCHMARK_EXIT_CODE + if [[ "${EVAL_FAILED:-0}" -eq 1 && "$node_exit_status" -eq 0 ]]; then + node_exit_status=1 fi + exit "$node_exit_status" elif [ "$NODE_RANK" -gt 0 ] && [ "$NODE_RANK" -lt "$NODE_OFFSET" ]; then echo "${host_name}:${host_ip} is Prefill Node (Model: ${MODEL_NAME})" @@ -1306,8 +1337,8 @@ elif [ "$NODE_RANK" -gt 0 ] && [ "$NODE_RANK" -lt "$NODE_OFFSET" ]; then > >(tee /run_logs/slurm_job-${SLURM_JOB_ID}/prefill_${host_name}.log >/dev/null) 2>&1 & set +x prefill_pid=$! - prefill_pgid=$(ps -o pgid= -p "$prefill_pid" 2>/dev/null | tr -d ' ') - : "${prefill_pgid:=$prefill_pid}" + prefill_pgid=$prefill_pid + SGLANG_OWNED_PGIDS+=("$prefill_pgid") fi echo "Waiting for proxy server to be up..." @@ -1337,8 +1368,7 @@ elif [ "$NODE_RANK" -gt 0 ] && [ "$NODE_RANK" -lt "$NODE_OFFSET" ]; then echo "Killing the rank $NODE_RANK prefill server" if [[ "$DRY_RUN" -eq 0 ]]; then - # Group-kill so TP-scheduler children release the tee pipe and the container exits. - kill -TERM -"${prefill_pgid:-$prefill_pid}" 2>/dev/null || true + exit 0 fi else @@ -1393,7 +1423,7 @@ else if [[ -n "$DSV4_GOLDEN_AL" ]]; then DECODE_SIM_ACC_ENV="SGLANG_SIMULATE_ACC_LEN=${DSV4_GOLDEN_AL} SGLANG_SIMULATE_ACC_METHOD=match-expected SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token" else - echo "WARNING: agentic MTP run (model=${MODEL_NAME}, DECODE_MTP_SIZE=${DECODE_MTP_SIZE}) has no golden AL wired in server_sglang.sh -- falling back to real (unsimulated, non-representative) acceptance. Add a case in server_sglang.sh and golden_al_distribution/ before shipping this arm. See golden_al_distribution/README.md." >&2 + echo "WARNING: agentic spec-decoding run (model=${MODEL_NAME}, algorithm=${SPEC_DECODING:-mtp}, DECODE_MTP_SIZE=${DECODE_MTP_SIZE}) has no golden AL wired in server_sglang.sh -- falling back to real (unsimulated, non-representative) acceptance. Add a case in server_sglang.sh and golden_al_distribution/ before shipping this arm. See golden_al_distribution/README.md." >&2 fi fi fi @@ -1425,8 +1455,8 @@ else set +x decode_pid=$! - decode_pgid=$(ps -o pgid= -p "$decode_pid" 2>/dev/null | tr -d ' ') - : "${decode_pgid:=$decode_pid}" + decode_pgid=$decode_pid + SGLANG_OWNED_PGIDS+=("$decode_pgid") fi echo "Waiting for proxy server to be up..." @@ -1455,8 +1485,7 @@ else echo "Killing the rank $RANK decode server" if [[ "$DRY_RUN" -eq 0 ]]; then - # Group-kill so TP-scheduler children release the tee pipe and the container exits. - kill -TERM -"${decode_pgid:-$decode_pid}" 2>/dev/null || true + exit 0 fi fi diff --git a/benchmarks/multi_node/amd_utils/stage_node_logs.sh b/benchmarks/multi_node/amd_utils/stage_node_logs.sh new file mode 100755 index 0000000000..f1e67814e4 --- /dev/null +++ b/benchmarks/multi_node/amd_utils/stage_node_logs.sh @@ -0,0 +1,23 @@ +#!/usr/bin/env bash + +set -eo pipefail + +if [[ $# -ne 2 ]]; then + echo "Usage: $0 " >&2 + exit 2 +fi + +SOURCE_LOGS=$1 +SHARED_LOGS=$2 + +if [[ ! -d "$SOURCE_LOGS" ]]; then + echo "[logs][ERROR] no node-local logs found on $(hostname): $SOURCE_LOGS" >&2 + exit 1 +fi + +# Server containers create the source tree as root, and node 0 may have already +# created the shared destination as root. The Slurm nodes provide passwordless +# sudo for the same Docker lifecycle used by job.slurm. +sudo mkdir -p "$SHARED_LOGS" +sudo cp -r "$SOURCE_LOGS"/. "$SHARED_LOGS"/ +echo "[logs] staged $(hostname):$SOURCE_LOGS -> $SHARED_LOGS" diff --git a/benchmarks/runtime_settings.sh b/benchmarks/runtime_settings.sh index 0b04efe9ba..a369898f8b 100644 --- a/benchmarks/runtime_settings.sh +++ b/benchmarks/runtime_settings.sh @@ -32,4 +32,4 @@ export SGLANG_TORCH_PROFILER_DIR='/workspace' export VLLM_TORCH_PROFILER_DIR='/workspace' # Explicitly forward these settings across container boundaries. -export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY SWEBENCH_EXPECTED_INSTANCES SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_WATCHDOG_POLL SWEBENCH_SANDBOX_SWEEP SWEBENCH_SKIP_SCORE SWEBENCH_EVAL_TIMEOUT SWEBENCH_SCORE_TIMEOUT SWEBENCH_MAX_WORKERS EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR" +export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY SWEBENCH_EXPECTED_INSTANCES SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_WATCHDOG_POLL SWEBENCH_SANDBOX_SWEEP SWEBENCH_SKIP_SCORE SWEBENCH_EVAL_TIMEOUT SWEBENCH_SCORE_TIMEOUT SWEBENCH_MAX_WORKERS EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_EXPERIMENTAL_FAST AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER REQUIRE_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR" diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index e233f73564..f291bce549 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -1193,8 +1193,15 @@ minimaxm3-fp8-mi325x-vllm-agentic-mtp: search-space: - { tp: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 2, 4, 8, 10, 12, 14, 16, 18] } -dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: - image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260911 +dsv4-fp4-mi355x-sglang-disagg-agentic-umbp-dspark: + # Renamed from dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp when this arm + # moved from EAGLE/MTP to DSpark; earlier perf-changelog entries are recorded + # under the old key. + # 20260913 rather than 20260911: it is the first tag carrying the DSpark + # optimizations, which is the whole point of this arm. CLIENT_IMAGE stays at + # 20260907 -- that container only runs the load generator, so the + # server-side DSpark work does not reach it. + image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913 model: deepseek-ai/DeepSeek-V4-Pro-0813 model-prefix: dsv4 runner: cluster:mi355x-amds @@ -1207,7 +1214,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: agentic-coding: - dram-utilization: 0.80 search-space: - - spec-decoding: "mtp" + - spec-decoding: "draft_model" conc-list: [ 4 ] kv-offloading: none prefill: @@ -1226,7 +1233,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: additional-settings: - "DECODE_NODES=1" - "DECODE_MTP_SIZE=3" - - spec-decoding: "mtp" + - spec-decoding: "draft_model" conc-list: [ 16 ] kv-offloading: none prefill: @@ -1245,7 +1252,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: additional-settings: - "DECODE_NODES=1" - "DECODE_MTP_SIZE=3" - - spec-decoding: "mtp" + - spec-decoding: "draft_model" conc-list: [ 32, 48 ] kv-offloading: dram kv-offload-backend: { name: hicache } @@ -1266,7 +1273,7 @@ dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: additional-settings: - "DECODE_NODES=1" - "DECODE_MTP_SIZE=3" - - spec-decoding: "mtp" + - spec-decoding: "draft_model" conc-list: [ 128, 192, 256 ] kv-offloading: dram kv-offload-backend: { name: umbp-linker } diff --git a/docs/architecture.md b/docs/architecture.md index a462f7215b..9e4b5c743a 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -277,7 +277,7 @@ The collector and reusable-artifact validator share format recognition, concurre Agentic throughput jobs have a different contract. They validate AIPerf output with [`infx/results/agentic/validate_agentic_result.py`](../infx/results/agentic/validate_agentic_result.py), upload an aggregate `bmk_agentic_` artifact, and upload the raw `agentic_` sibling containing trace-replay material. InferenceX-app pairs those siblings by their shared suffix. Agentic eval-only jobs follow the eval output contract instead and do not require a throughput result. -Server logs and GPU metrics are diagnostic side artifacts. They are uploaded with `always()` so a failed run can still be investigated. Their presence does not turn a failed benchmark into a valid result. +Server logs and GPU metrics are diagnostic side artifacts. They are uploaded with `always()` so a failed run can still be investigated. Their presence does not turn a failed benchmark into a valid result. On the AMD Slurm fleet, `/run_logs` is node-local; after the server step finishes, `job.slurm` merges the closed log tree from every allocated node into shared storage so the diagnostic artifact includes prefill and decode logs from the full deployment. ## Stage 6: artifact collection and handoff diff --git a/docs/architecture_zh.md b/docs/architecture_zh.md index 3e8fde840f..377a62cba4 100644 --- a/docs/architecture_zh.md +++ b/docs/architecture_zh.md @@ -277,7 +277,7 @@ rows = build_rows(raw_eval, metadata, source="eval_job/results.json") 智能体吞吐量作业采用不同的契约。它们使用 [`infx/results/agentic/validate_agentic_result.py`](../infx/results/agentic/validate_agentic_result.py) 验证 AIPerf 输出,上传聚合的 `bmk_agentic_` 工件,并上传包含追踪重放材料的原始 `agentic_` 同级工件。InferenceX-app 通过它们共享的后缀对这些同级工件进行配对。智能体仅评测作业改为遵循评测输出契约,不要求吞吐量结果。 -服务器日志和 GPU 指标是诊断辅助工件。它们通过 `always()` 上传,因此失败的运行仍可供调查。它们的存在不会将失败的基准测试转变为有效结果。 +服务器日志和 GPU 指标是诊断辅助工件。它们通过 `always()` 上传,因此失败的运行仍可供调查。它们的存在不会将失败的基准测试转变为有效结果。在 AMD Slurm 机群上,`/run_logs` 是节点本地目录;服务器步骤结束后,`job.slurm` 会把每个已分配节点上已经关闭的日志树合并到共享存储中,使诊断工件包含整个部署的 Prefill 和 Decode 日志。 ## 阶段 6:工件收集与交接 diff --git a/docs/recovery-results-procedures.md b/docs/recovery-results-procedures.md index 0d140d4c84..8b8ac153ac 100644 --- a/docs/recovery-results-procedures.md +++ b/docs/recovery-results-procedures.md @@ -425,3 +425,28 @@ Remaining durable fix: ``` This evidence is the completion gate. “Workflow green” without artifact identity, source/merge identity, and ingest counts is not a verified result recovery. + +### AMD multi-node SGLang teardown + +On exit, including a failed startup/readiness check, the AMD SGLang launcher sends +TERM only to its recorded `setsid` process groups. Normal completion stages results +before this cleanup. It allows 30 seconds for graceful +exit, then sends KILL to surviving groups and checks for exit for another five +seconds. This handles orphaned or TERM-resistant workers that otherwise hold log +pipes open. These cleanup deadlines do not change profiling, evaluation, or server +readiness deadlines. A failed client retains its exit status; unresolved cleanup +fails an otherwise successful node. Kernel-blocked processes may still require +separately authorized node repair. Do not change or discard completed metrics to +work around teardown failures. A single EXIT handler owns group cleanup and the +existing UMBP standalone PID cleanup; the latter still runs if group cleanup fails. + +### AMD multi-node GPU preflight coordination + +The Slurm launcher completes Docker pre-clean and the existing GPU VRAM drain +check on every selected node in a separate Slurm step before launching any server +container. A failed preflight prevents the serving step; it does not consume a +healthy peer's container-readiness deadline. Node-local `preflight_.log` +files are included in the normal log fan-in, including failures. The VRAM threshold, +15-minute GPU guard, and container/server readiness deadlines remain unchanged. +This coordination prevents a peer-barrier race; it does not repair a GPU driver +that fails to reclaim memory. The existing Docker pre-clean scope is unchanged. diff --git a/docs/recovery-results-procedures_zh.md b/docs/recovery-results-procedures_zh.md index df3ed5e045..1427077bb6 100644 --- a/docs/recovery-results-procedures_zh.md +++ b/docs/recovery-results-procedures_zh.md @@ -425,3 +425,23 @@ Remaining durable fix: ``` 这些证据就是完成关卡。如果没有制品身份、source/merge 身份和摄取数量,仅仅“工作流绿色”并不代表结果恢复已经验证。 + +### AMD 多节点 SGLang 清理 + +退出时(包括启动或就绪检查失败),AMD SGLang 启动器仅向其记录的 `setsid` +进程组发送 TERM,等待最多 30 秒。正常完成时,先暂存结果再进行清理。随后向仍存活的进程组发送 KILL,再等待最多 +5 秒并检查退出状态。这可以清理已成为孤儿进程或忽略 TERM 的工作进程,避免其 +持续占用日志管道。这些清理期限不会改变性能采集、评估或服务器就绪检查的期限。 +客户端失败时保留原退出码;若客户端成功但清理仍未完成,则节点任务失败。 +内核阻塞的进程仍可能需要另行授权的节点修复。不要为绕过清理失败而修改或丢弃 +已完成的指标。单一 EXIT 处理器统一负责进程组清理和现有 UMBP 独立进程 PID +清理;即使进程组清理失败,后者仍会执行。 + +### AMD 多节点 GPU 预检协调 + +Slurm 启动器先在独立步骤中完成所有选定节点的 Docker 预清理和现有 GPU VRAM +回收检查,然后才启动服务器容器。任一节点预检失败都会阻止服务步骤启动,不会 +消耗健康节点等待容器就绪的期限。节点本地的 `preflight_.log` 文件 +通过常规日志汇总流程收集,包括失败日志。VRAM 阈值、15 分钟 GPU 检查期限及 +容器和服务器就绪期限均保持不变。这一协调消除了节点间等待的竞态,但无法修复 +不能回收显存的 GPU 驱动。现有 Docker 预清理范围保持不变。 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index f987183523..487f46dc69 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8003,3 +8003,18 @@ - "Adds glm5.2/fp8 routing to runners/launch_b200-nscale-compat.sh (MODEL_PATH default /scratch/models/GLM-5.2-FP8, SRT_SLURM_MODEL_PREFIX glm5.2-fp8), which previously hard-failed with 'Unsupported model prefix/precision' for this model. The script's completeness-checked, flock-serialized hf download stages the checkpoint on the day-zero run if it is not already on the cluster." - "Re-pin from lmsysorg/sglang:nightly-dev-cu13-20260907-30705c00 to lmsysorg/sglang:nightly-dev-cu13-20260908-20ca564b (2026-09-08 cu13 dev nightly, digest sha256:9a352a35c973a2357372e85f3bcb5388b6b3c46c1329165987260f3b089647dc; Docker Hub last pushed 2026-09-08T01:40:59Z, tag commit sgl-project/sglang@20ca564b). The 2026-09-07 build carries an unguarded kv_index_translator.translate_dcp_read_ids call on the DSA fp8 KV read path that the EAGLE draft backend never binds, so GLM-5.2 MTP runs crash intermittently with AttributeError (observed on the MI355X FP8 sibling in run 34173459478 after 74 minutes of serving). sgl-project/sglang#38318 (merged 2026-09-07T20:03Z) adds the None guard and is six commits behind 20ca564b." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2863 + +- config-keys: + - dsv4-fp4-mi355x-sglang-disagg-agentic-umbp-dspark + scenario-type: + - agentic-coding + description: + - "Switch all four arms from EAGLE (spec-decoding: mtp) to DSpark (spec-decoding: draft_model) at DECODE_MTP_SIZE=3, and bump the image from lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260911 to -20260913, the first tag carrying the DSpark optimizations. CLIENT_IMAGE stays at 20260907 (load generator only). Topology, concurrency list, HiCache/UMBP settings, and precision are unchanged." + - "DeepSeek-V4-Pro-AgentX in models.yaml gains dspark_flags (--speculative-algorithm DSPARK --speculative-eagle-topk 1) alongside mtp_flags rather than replacing it, so recipes remaining on spec-decoding: mtp keep their EAGLE arm unchanged; dp_flags gains --enable-dp-lm-head, required by SGLang for DSpark under DP attention and inert for EAGLE. The -0813 checkpoint bundles the draft head, so no separate draft checkpoint is staged." + - "server_sglang.sh builds DSpark flags as --speculative-dspark-block-size DECODE_MTP_SIZE --speculative-num-steps 1 --speculative-num-draft-tokens (DECODE_MTP_SIZE + 1). DECODE_MTP_SIZE is the draft length for both algorithms but EAGLE spends it as sequential draft passes while DSpark emits one block, so num-steps is pinned to 1. The verify window is identical for both, so the MORI decode dispatch scaling by (DECODE_MTP_SIZE + 1) is unchanged. A model set to draft_model without dspark_flags now fails hard instead of silently serving EAGLE." + - "Golden AL is unchanged: the AgentX curve is keyed on checkpoint, thinking mode, and draft length, not on the algorithm, so DeepSeek-V4-Pro-0813 at draft length 3 keeps AL 3.01 from golden_al_distribution/dsv4-pro-0813-dspark.yaml (thinking_on)." + - "将四条臂全部从 EAGLE(spec-decoding: mtp)切换为 DSpark(spec-decoding: draft_model),DECODE_MTP_SIZE=3;镜像由 lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260911 升级到 -20260913(首个带 DSpark 优化的标签),CLIENT_IMAGE 保持 20260907(仅负载生成器)。拓扑、并发列表、HiCache/UMBP 设置与精度均不变。models.yaml 中 DeepSeek-V4-Pro-AgentX 新增 dspark_flags(与 mtp_flags 并列而非替换,因此仍用 mtp 的配方 EAGLE 分支不受影响),dp_flags 增加 --enable-dp-lm-head。server_sglang.sh 按 SPEC_DECODING == draft_model 分流:DSpark 一次前向出一整块,故 num-steps 钉死为 1,草稿长度转为 --speculative-dspark-block-size;验证窗口(长度+1)与 EAGLE 相同,所以 MORI 解码 dispatch 的 ×(DECODE_MTP_SIZE+1) 缩放无需改动。黄金 AL 不变:按检查点而非算法选表,DeepSeek-V4-Pro-0813 在草稿长度 3 仍为 AL 3.01。" + - "Rename the config key from dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp to dsv4-fp4-mi355x-sglang-disagg-agentic-umbp-dspark to reflect the DSpark algorithm and the UMBP-linker arm; earlier entries for this recipe are recorded under the old key. No other field changes with the rename." + - "配置键由 dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp 改名为 dsv4-fp4-mi355x-sglang-disagg-agentic-umbp-dspark,以反映 DSpark 算法与 UMBP-linker 分支;该配方此前的条目仍记录在旧键名下。改名不伴随任何其他字段变化。" + - "Collect completed logs from every Slurm node so both prefill and decode artifacts are uploaded; forward workflow-owned AgentX fast-mode and power inputs, and supply the existing router port to evaluation." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3188