Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
18 commits
Select commit Hold shift + click to select a range
6a95aa6
feat: enable DSpark for DeepSeek-V4-Pro-0813 AgentX disagg on MI355X
Duyi-Wang Sep 16, 2026
68df08e
chore: record the DSpark switch in perf-changelog
Duyi-Wang Sep 16, 2026
c68184d
refactor: rename the MI355X DSV4 disagg AgentX key to -umbp-dspark
Duyi-Wang Sep 16, 2026
64bf729
fix: abort the launch when draft_model is set without dspark_flags
Duyi-Wang Sep 16, 2026
b6a1e24
Merge branch 'main' into dspark-dsv4-mi355-agentx-disagg
ichbinblau Sep 16, 2026
0bf6ff2
fix(amd): forward required client inputs and collect all node logs
cquil11 Sep 16, 2026
3430744
chore: merge main and preserve performance changelog entries
cquil11 Sep 16, 2026
3ba3a9c
Merge branch 'main' into dspark-dsv4-mi355-agentx-disagg
billishyahao Sep 16, 2026
2d02d1f
Merge branch 'main' into dspark-dsv4-mi355-agentx-disagg
ichbinblau Sep 17, 2026
cbcaed5
fix lint
billishyahao Sep 17, 2026
7c39244
chore: merge main and preserve performance changelog entries
cquil11 Sep 17, 2026
e112620
fix: bound teardown of owned AMD SGLang process groups
cquil11 Sep 16, 2026
3450090
fix: coordinate AMD node preflight before server startup
cquil11 Sep 17, 2026
3773163
fix: clean owned SGLang processes on startup failure
cquil11 Sep 17, 2026
298f835
chore: sync latest main while preserving AMD sweep source
cquil11 Sep 17, 2026
3b4751c
fix: exclude failed AgentX rows from reusable artifact identities
cquil11 Sep 17, 2026
f753326
Revert "fix: exclude failed AgentX rows from reusable artifact identi…
cquil11 Sep 17, 2026
f4a3fba
test: remove added AMD lifecycle and log staging tests
cquil11 Sep 17, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
109 changes: 109 additions & 0 deletions benchmarks/benchmark_lib.sh
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,115 @@ check_env_vars() {
fi
}

# Report live members of explicitly owned process groups. Zombies cannot hold
# output pipes open. Do not use leader liveness: a router can orphan its workers.
_background_process_groups_alive() {
local groups=" $* "
local listing
listing=$(ps -eo pgid=,stat=) || return 1
awk -v groups="$groups" '
index(groups, " " $1 " ") && $2 !~ /^[ZX]/ { alive[$1] = 1 }
END { for (group in alive) print group }
' <<< "$listing"
}

# Called only after benchmark/eval work ends. Preserve its exit status while
# bounding teardown of the setsid groups recorded by the launcher. Grace periods
# are explicit arguments, independent of benchmark duration and server readiness.
stop_background_process_groups() {
local work_status="$1" term_grace="$2" kill_grace="$3"
shift 3
local pgid own_pgid remaining deadline cleanup_status=0
local groups=("$@")
if [[ ! "$work_status" =~ ^[0-9]+$ || ! "$term_grace" =~ ^[0-9]+$ || ! "$kill_grace" =~ ^[0-9]+$ ]]; then
echo "ERROR: invalid process-group cleanup status or grace period" >&2
return 1
fi
if ! own_pgid=$(ps -o pgid= -p "$$"); then
if [[ "$work_status" -ne 0 ]]; then return "$work_status"; fi
return 1
fi
own_pgid="${own_pgid//[[:space:]]/}"
for pgid in "${groups[@]}"; do
if [[ ! "$pgid" =~ ^[1-9][0-9]*$ || "$pgid" -le 1 || "$pgid" == "$own_pgid" ]]; then
echo "ERROR: refusing unsafe process-group cleanup: '$pgid'" >&2
if [[ "$work_status" -ne 0 ]]; then return "$work_status"; fi
return 1
fi
done
if [[ ${#groups[@]} -eq 0 ]]; then return "$work_status"; fi

echo "Stopping owned process groups: ${groups[*]}"
for pgid in "${groups[@]}"; do
kill -TERM -- "-$pgid" 2>/dev/null || true
done
deadline=$((SECONDS + term_grace))
while true; do
remaining=$(_background_process_groups_alive "${groups[@]}") || { cleanup_status=1; break; }
[[ -n "$remaining" && $SECONDS -lt $deadline ]] || break
sleep 1
done
if [[ -n "$remaining" ]]; then
echo "TERM grace expired; force-stopping owned process groups: $remaining"
for pgid in $remaining; do
kill -KILL -- "-$pgid" 2>/dev/null || true
done
deadline=$((SECONDS + kill_grace))
while true; do
remaining=$(_background_process_groups_alive "${groups[@]}") || { cleanup_status=1; break; }
[[ -n "$remaining" && $SECONDS -lt $deadline ]] || break
sleep 1
done
if [[ -n "$remaining" ]]; then
echo "ERROR: process groups still alive after KILL grace: $remaining" >&2
cleanup_status=1
fi
fi
if [[ "$work_status" -ne 0 ]]; then return "$work_status"; fi
return "$cleanup_status"
}

# EXIT handler for launchers that own explicit setsid groups and optionally one
# auxiliary daemon PID. Disable this handler before exiting to avoid recursion.
# Run auxiliary cleanup even when group cleanup fails, preserving the work code.
exit_after_background_process_cleanup() {
local work_status="$1" term_grace="$2" kill_grace="$3" auxiliary_pid="$4"
shift 4
local final_status
trap - EXIT
if stop_background_process_groups "$work_status" "$term_grace" "$kill_grace" "$@"; then
final_status=0
else
final_status=$?
fi
if [[ -n "$auxiliary_pid" ]]; then
kill "$auxiliary_pid" 2>/dev/null || true
fi
exit "$final_status"
}

# Finish preflight on every allocated node before any server container starts its
# peer-readiness deadline. A failed node prevents the entire serving step.
run_amd_multinode_after_preflight() {
local nodelist="$1" node_count="$2" preflight_script="$3"
local container_filter="$4" skip_gpu_sanity="$5"
shift 5
local preflight_rc
if srun --nodelist="$nodelist" \
--nodes="$node_count" --ntasks="$node_count" --ntasks-per-node=1 \
--kill-on-bad-exit=1 --unbuffered \
bash "$preflight_script" "$container_filter" "$skip_gpu_sanity"; then
echo "[preflight] all nodes ready; launching server containers"
else
preflight_rc=$?
echo "[preflight][ERROR] node preflight failed; no server containers launched" >&2
return "$preflight_rc"
fi
srun --nodelist="$nodelist" \
--nodes="$node_count" --ntasks="$node_count" --ntasks-per-node=1 \
--kill-on-bad-exit=1 --signal=TERM@30 --unbuffered "$@"
}

# Launchers may load only input validation, without benchmark initialization.
if [[ "${1-}" == "--validation-only" ]]; then
return 0
Expand Down
53 changes: 31 additions & 22 deletions benchmarks/multi_node/amd_utils/job.slurm
Original file line number Diff line number Diff line change
Expand Up @@ -583,11 +583,10 @@ if [[ -n "${CLIENT_IMAGE:-}" ]]; then
srun --nodelist="$SELECTED_NODELIST_SRUN" bash -c 'eval "$DOCKER_CMD_DETECT"; $DOCKER_CMD pull '"$CLIENT_IMAGE"' >/dev/null 2>&1 || true' 2>/dev/null || true
fi

srun \
--nodelist="$SELECTED_NODELIST_SRUN" \
--kill-on-bad-exit=1 \
--signal=TERM@30 \
--unbuffered \
run_amd_multinode_after_preflight \
"$SELECTED_NODELIST_SRUN" "$NUM_NODES" \
"$DI_REPO_DIR/benchmarks/multi_node/amd_utils/preflight_node.sh" \
"$CONT_FILTER" "$SKIP_GPU_SANITY" \
bash -lc "
set -eo pipefail

Expand Down Expand Up @@ -675,23 +674,6 @@ else
fi
fi # end: if ENGINE == atom-disagg

# Pre-clean (idempotent): stop then force-remove so GPU VRAM is released
# before the drain gate. stop-only left containers in Created/Exited state
# on some nodes.
\$DOCKER_CMD ps -aq --filter \"$CONT_FILTER\" | xargs -r \$DOCKER_CMD rm -f || true
\$DOCKER_CMD ps -aq | xargs -r \$DOCKER_CMD stop -t 15 || true
\$DOCKER_CMD ps -aq | xargs -r \$DOCKER_CMD rm -f || true
sleep 2

# GPU drain gate: fail fast on leftover VRAM use instead of OOMing in model
# load ~15 min later. Reuses wait_for_amd_gpu_clean from benchmark_lib.sh.
if [[ \"${SKIP_GPU_SANITY}\" == \"1\" ]]; then
echo \"[INFO] SKIP_GPU_SANITY=1 set; skipping GPU pre-flight drain check\"
else
# Unset so benchmark_lib.sh's unrelated agentic KV_OFFLOADING check doesn't exit 1 here.
bash -c \"unset IS_AGENTIC SCENARIO_TYPE; source $DI_REPO_DIR/benchmarks/benchmark_lib.sh && wait_for_amd_gpu_clean\"
fi

# Start vLLM external router container on node 0
if [[ \"$ENGINE\" == \"vllm-disagg\" && \"$ROUTER_TYPE\" == \"vllm-router\" && \"\$SLURM_PROCID\" == \"0\" ]]; then
\$DOCKER_CMD rm -f \"$ROUTER_CONT_NAME\" 2>/dev/null || true
Expand Down Expand Up @@ -775,6 +757,7 @@ echo \"[rank 0] Main container exited (rc=\$DOCKER_EXIT_CODE). Stopping vllm-rou
\$DOCKER_CMD rm -f \"$ROUTER_CONT_NAME\" 2>/dev/null || true
exit \$DOCKER_EXIT_CODE
"
SERVER_SRUN_RC=$?

if [[ "${KEEP_CONTAINERS}" != "1" ]]; then
srun --nodelist="$SELECTED_NODELIST_SRUN" bash -c 'eval "$DOCKER_CMD_DETECT"; $DOCKER_CMD rm -f '"$DOCKER_CONT_NAME"' '"$CLIENT_CONT_NAME"' 2>/dev/null || true'
Expand All @@ -786,3 +769,29 @@ if [[ "${KEEP_CONTAINERS}" != "1" ]]; then
'
fi
fi

# /run_logs is backed by each compute node's local /tmp, so the node-0 copy
# performed by the engine launcher cannot see prefill/decode logs written on
# other nodes. Collect after the server step and container cleanup so failed
# runs also include shutdown output. KEEP_CONTAINERS=1 retains a snapshot of
# any containers left running for debugging.
# Use sudo because the container-created source and the existing node-0
# destination can be root-owned. Restore ownership after the fan-in so a
# subsequent runner job can clean the workspace normally.
SHARED_JOB_LOGS="${BENCHMARK_LOGS_DIR}/logs/slurm_job-${SLURM_JOB_ID}"
if ! srun --nodelist="$SELECTED_NODELIST_SRUN" \
--nodes="$NUM_NODES" --ntasks="$NUM_NODES" --ntasks-per-node=1 \
bash "$DI_REPO_DIR/benchmarks/multi_node/amd_utils/stage_node_logs.sh" \
"/tmp/slurm_job-${SLURM_JOB_ID}" "$SHARED_JOB_LOGS"; then
echo "[logs][ERROR] failed to stage logs from one or more Slurm nodes" >&2
if [[ "$SERVER_SRUN_RC" -eq 0 ]]; then
SERVER_SRUN_RC=1
fi
fi

if [[ -d "$SHARED_JOB_LOGS" ]]; then
sudo chown -R "$(id -u):$(id -g)" "$SHARED_JOB_LOGS" 2>/dev/null || true
chmod -R a+rwX "$SHARED_JOB_LOGS" 2>/dev/null || true
fi

exit "$SERVER_SRUN_RC"
21 changes: 18 additions & 3 deletions benchmarks/multi_node/amd_utils/models.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -354,9 +354,21 @@ DeepSeek-R1-0528-MXFP4-v2:

DeepSeek-V4-Pro-AgentX: &DeepSeek-V4-Pro-AgentX
base_flags: "--enable-deepseek-v4-fp4-indexer --watchdog-timeout 3600 --load-balance-method round_robin --kv-cache-dtype fp8_e4m3 --attention-backend dsv4 --page-size 256 --swa-full-tokens-ratio 0.1 --enforce-shared-experts-fusion --tool-call-parser deepseekv4 --reasoning-parser deepseek-v4 --disaggregation-transfer-backend mori --tokenizer-worker-num 8 --stream-interval 20 --log-level info --log-level-http error"
dp_flags: "--enable-dp-attention --swa-full-tokens-ratio 0.15 --enable-dp-attention-local-control-broadcast"
# --enable-dp-lm-head is required by SGLang for DSpark under DP attention; it
# is harmless for the EAGLE/MTP arms, so it stays unconditional here rather
# than needing a second DP flag string.
dp_flags: "--enable-dp-attention --enable-dp-lm-head --swa-full-tokens-ratio 0.15 --enable-dp-attention-local-control-broadcast"
ep_flags: "--ep-dispatch-algorithm fake --moe-a2a-backend mori --deepep-mode normal"
mtp_flags: "--speculative-algorithm EAGLE --speculative-eagle-topk 1"
# DSpark draft head, selected when the sweep sets spec-decoding: draft_model.
# Kept alongside mtp_flags rather than replacing it, so recipes that stay on
# spec-decoding: mtp keep the EAGLE arm untouched. The -0813 checkpoint
# bundles the draft head (dspark_block_size / dspark_markov_rank /
# dspark_target_layer_ids in config.json), so --speculative-draft-model-path
# defaults to --model-path and no separate draft checkpoint is needed.
# server_sglang.sh appends the block size and the verify window from
# DECODE_MTP_SIZE (= gamma); unlike EAGLE, num-steps is pinned to 1.
dspark_flags: "--speculative-algorithm DSPARK --speculative-eagle-topk 1"
prefill:
disable_radix_cache: false
disable_cuda_graph: true
Expand Down Expand Up @@ -384,8 +396,11 @@ DeepSeek-V4-Pro-AgentX: &DeepSeek-V4-Pro-AgentX
max_running_requests: "BENCH_MAX_CONC_VALUE*2"
cuda_graph_bs_range: "1-BENCH_MAX_CONC_VALUE*2"

# Pro-0813 retains EAGLE 3-1-4 for PD compatibility. Synthetic acceptance is
# checkpoint-specific in server_sglang.sh: thinking-on, length 3 uses AL 3.01.
# Pro-0813 serves the PD path with DSPARK (spec-decoding: draft_model), which
# measured clean across c4-c256; the EAGLE 3-1-4 arm remains available through
# spec-decoding: mtp. Synthetic acceptance is checkpoint-specific in
# server_sglang.sh and does not depend on which of the two runs: thinking-on,
# draft length 3 uses AL 3.01.
DeepSeek-V4-Pro-0813-AgentX: *DeepSeek-V4-Pro-AgentX

DeepSeek-V4-Pro-DI:
Expand Down
31 changes: 31 additions & 0 deletions benchmarks/multi_node/amd_utils/preflight_node.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
#!/usr/bin/env bash
set -eo pipefail

SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../../benchmark_lib.sh" --validation-only
check_env_vars DOCKER_CMD_DETECT DI_REPO_DIR SLURM_JOB_ID
CONT_FILTER="$1"
SKIP_GPU_SANITY="$2"
check_env_vars CONT_FILTER SKIP_GPU_SANITY

preflight_node() {
eval "$DOCKER_CMD_DETECT"

# Preserve the existing pre-clean scope and ordering. Moving it into this
# separate Slurm step prevents one node starting while another still drains.
$DOCKER_CMD ps -aq --filter "$CONT_FILTER" | xargs -r $DOCKER_CMD rm -f || true
$DOCKER_CMD ps -aq | xargs -r $DOCKER_CMD stop -t 15 || true
$DOCKER_CMD ps -aq | xargs -r $DOCKER_CMD rm -f || true
sleep 2

if [[ "$SKIP_GPU_SANITY" == "1" ]]; then
echo "[INFO] SKIP_GPU_SANITY=1 set; skipping GPU pre-flight drain check"
else
# Avoid benchmark-only agentic initialization on the host, as before.
bash -c 'unset IS_AGENTIC SCENARIO_TYPE; source "$DI_REPO_DIR/benchmarks/benchmark_lib.sh" && wait_for_amd_gpu_clean'
fi
}

NODE_LOG_DIR="/tmp/slurm_job-${SLURM_JOB_ID}"
mkdir -p "$NODE_LOG_DIR"
preflight_node 2>&1 | tee "$NODE_LOG_DIR/preflight_$(hostname).log"
Loading