Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 27 additions & 0 deletions benchmarks/multi_node/amd_utils/job.slurm
Original file line number Diff line number Diff line change
Expand Up @@ -775,6 +775,7 @@ echo \"[rank 0] Main container exited (rc=\$DOCKER_EXIT_CODE). Stopping vllm-rou
\$DOCKER_CMD rm -f \"$ROUTER_CONT_NAME\" 2>/dev/null || true
exit \$DOCKER_EXIT_CODE
"
SERVER_SRUN_RC=$?

if [[ "${KEEP_CONTAINERS}" != "1" ]]; then
srun --nodelist="$SELECTED_NODELIST_SRUN" bash -c 'eval "$DOCKER_CMD_DETECT"; $DOCKER_CMD rm -f '"$DOCKER_CONT_NAME"' '"$CLIENT_CONT_NAME"' 2>/dev/null || true'
Expand All @@ -786,3 +787,29 @@ if [[ "${KEEP_CONTAINERS}" != "1" ]]; then
'
fi
fi

# /run_logs is backed by each compute node's local /tmp, so the node-0 copy
# performed by the engine launcher cannot see prefill/decode logs written on
# other nodes. Collect after the server step and container cleanup so failed
# runs also include shutdown output. KEEP_CONTAINERS=1 retains a snapshot of
# any containers left running for debugging.
# Use sudo because the container-created source and the existing node-0
# destination can be root-owned. Restore ownership after the fan-in so a
# subsequent runner job can clean the workspace normally.
SHARED_JOB_LOGS="${BENCHMARK_LOGS_DIR}/logs/slurm_job-${SLURM_JOB_ID}"
if ! srun --nodelist="$SELECTED_NODELIST_SRUN" \
--nodes="$NUM_NODES" --ntasks="$NUM_NODES" --ntasks-per-node=1 \
bash "$DI_REPO_DIR/benchmarks/multi_node/amd_utils/stage_node_logs.sh" \
"/tmp/slurm_job-${SLURM_JOB_ID}" "$SHARED_JOB_LOGS"; then
echo "[logs][ERROR] failed to stage logs from one or more Slurm nodes" >&2
if [[ "$SERVER_SRUN_RC" -eq 0 ]]; then
SERVER_SRUN_RC=1
fi
fi

if [[ -d "$SHARED_JOB_LOGS" ]]; then
sudo chown -R "$(id -u):$(id -g)" "$SHARED_JOB_LOGS" 2>/dev/null || true
chmod -R a+rwX "$SHARED_JOB_LOGS" 2>/dev/null || true
fi

exit "$SERVER_SRUN_RC"
6 changes: 4 additions & 2 deletions benchmarks/multi_node/amd_utils/server_sglang.sh
Original file line number Diff line number Diff line change
Expand Up @@ -1183,6 +1183,8 @@ print(json.dumps(json.loads(sys.stdin.read())))' <<<"$_val")" || {
# Must run from repo root so infx/evals/gsm8k.yaml resolves
pushd /workspace

# Match the disaggregation router launched above.
export PORT=30000
source /workspace/benchmarks/benchmark_lib.sh

# CONC must be exported before run_eval so meta_env.json matches validate_scores.py.
Expand All @@ -1204,9 +1206,9 @@ print(json.dumps(json.loads(sys.stdin.read())))' <<<"$_val")" || {
# arrive via Docker -e flags from job.slurm.

if [[ "$DRY_RUN" -eq 1 ]]; then
echo "DRY RUN: run_eval --port 30000 (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS}, ctx=${EVAL_MAX_MODEL_LEN:-auto})"
echo "DRY RUN: run_eval --port ${PORT} (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS}, ctx=${EVAL_MAX_MODEL_LEN:-auto})"
else
run_eval --port 30000
run_eval --port "$PORT"
eval_rc=$?

if [[ $eval_rc -ne 0 ]]; then
Expand Down
23 changes: 23 additions & 0 deletions benchmarks/multi_node/amd_utils/stage_node_logs.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
#!/usr/bin/env bash

set -eo pipefail

if [[ $# -ne 2 ]]; then
echo "Usage: $0 <node-local-log-dir> <shared-log-dir>" >&2
exit 2
fi

SOURCE_LOGS=$1
SHARED_LOGS=$2

if [[ ! -d "$SOURCE_LOGS" ]]; then
echo "[logs][ERROR] no node-local logs found on $(hostname): $SOURCE_LOGS" >&2
exit 1
fi

# Server containers create the source tree as root, and node 0 may have already
# created the shared destination as root. The Slurm nodes provide passwordless
# sudo for the same Docker lifecycle used by job.slurm.
sudo mkdir -p "$SHARED_LOGS"
sudo cp -r "$SOURCE_LOGS"/. "$SHARED_LOGS"/
echo "[logs] staged $(hostname):$SOURCE_LOGS -> $SHARED_LOGS"
2 changes: 1 addition & 1 deletion benchmarks/runtime_settings.sh
Original file line number Diff line number Diff line change
Expand Up @@ -32,4 +32,4 @@ export SGLANG_TORCH_PROFILER_DIR='/workspace'
export VLLM_TORCH_PROFILER_DIR='/workspace'

# Explicitly forward these settings across container boundaries.
export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY SWEBENCH_EXPECTED_INSTANCES SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_WATCHDOG_POLL SWEBENCH_SANDBOX_SWEEP SWEBENCH_SKIP_SCORE SWEBENCH_EVAL_TIMEOUT SWEBENCH_SCORE_TIMEOUT SWEBENCH_MAX_WORKERS EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR"
export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY SWEBENCH_EXPECTED_INSTANCES SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_WATCHDOG_POLL SWEBENCH_SANDBOX_SWEEP SWEBENCH_SKIP_SCORE SWEBENCH_EVAL_TIMEOUT SWEBENCH_SCORE_TIMEOUT SWEBENCH_MAX_WORKERS EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_EXPERIMENTAL_FAST AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER REQUIRE_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR"
2 changes: 1 addition & 1 deletion configs/amd-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -1195,7 +1195,7 @@ minimaxm3-fp8-mi325x-vllm-agentic-mtp:

dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp:
image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260911
model: deepseek-ai/DeepSeek-V4-Pro-0813
model: deepseek-ai/DeepSeek-V4-Pro
model-prefix: dsv4
runner: cluster:mi355x-amds
precision: fp4
Expand Down
2 changes: 1 addition & 1 deletion docs/architecture.md
Original file line number Diff line number Diff line change
Expand Up @@ -277,7 +277,7 @@ The collector and reusable-artifact validator share format recognition, concurre

Agentic throughput jobs have a different contract. They validate AIPerf output with [`infx/results/agentic/validate_agentic_result.py`](../infx/results/agentic/validate_agentic_result.py), upload an aggregate `bmk_agentic_<suffix>` artifact, and upload the raw `agentic_<suffix>` sibling containing trace-replay material. InferenceX-app pairs those siblings by their shared suffix. Agentic eval-only jobs follow the eval output contract instead and do not require a throughput result.

Server logs and GPU metrics are diagnostic side artifacts. They are uploaded with `always()` so a failed run can still be investigated. Their presence does not turn a failed benchmark into a valid result.
Server logs and GPU metrics are diagnostic side artifacts. They are uploaded with `always()` so a failed run can still be investigated. Their presence does not turn a failed benchmark into a valid result. On the AMD Slurm fleet, `/run_logs` is node-local; after the server step finishes, `job.slurm` merges the closed log tree from every allocated node into shared storage so the diagnostic artifact includes prefill and decode logs from the full deployment.

## Stage 6: artifact collection and handoff

Expand Down
2 changes: 1 addition & 1 deletion docs/architecture_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -277,7 +277,7 @@ rows = build_rows(raw_eval, metadata, source="eval_job/results.json")

智能体吞吐量作业采用不同的契约。它们使用 [`infx/results/agentic/validate_agentic_result.py`](../infx/results/agentic/validate_agentic_result.py) 验证 AIPerf 输出,上传聚合的 `bmk_agentic_<suffix>` 工件,并上传包含追踪重放材料的原始 `agentic_<suffix>` 同级工件。InferenceX-app 通过它们共享的后缀对这些同级工件进行配对。智能体仅评测作业改为遵循评测输出契约,不要求吞吐量结果。

服务器日志和 GPU 指标是诊断辅助工件。它们通过 `always()` 上传,因此失败的运行仍可供调查。它们的存在不会将失败的基准测试转变为有效结果。
服务器日志和 GPU 指标是诊断辅助工件。它们通过 `always()` 上传,因此失败的运行仍可供调查。它们的存在不会将失败的基准测试转变为有效结果。在 AMD Slurm 机群上,`/run_logs` 是节点本地目录;服务器步骤结束后,`job.slurm` 会把每个已分配节点上已经关闭的日志树合并到共享存储中,使诊断工件包含整个部署的 Prefill 和 Decode 日志。

## 阶段 6:工件收集与交接

Expand Down
10 changes: 10 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7923,3 +7923,13 @@
- "Bump image to lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260915."
- "Switch HiCache defaults to --hicache-io-backend kernel and --hicache-mem-layout page_first (from direct / page_first_direct). Ratio 1.5 and write_through are unchanged."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3118

- config-keys:
- dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp
scenario-type:
- agentic-coding
description:
- "Rerun the MI355X disaggregated MTP sweep with the original DeepSeek-V4-Pro checkpoint, whose native MTP weights match the existing EAGLE recipe. The existing checkpoint selector restores thinking-on golden AL 2.49 for three draft tokens; evals use real acceptance."
- "Collect completed server logs from every Slurm node into the shared artifact directory after the benchmark server step exits, including decode-node logs."
- "Pass the existing router port to evaluation and forward workflow-owned AgentX fast-mode and power requirements across container boundaries."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3170
61 changes: 61 additions & 0 deletions utils/test_amd_node_log_staging.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
from __future__ import annotations

import os
import subprocess
from pathlib import Path

REPO_ROOT = Path(__file__).resolve().parents[1]
STAGE_SCRIPT = REPO_ROOT / "benchmarks/multi_node/amd_utils/stage_node_logs.sh"


def _stub_sudo(tmp_path: Path) -> dict[str, str]:
bin_dir = tmp_path / "bin"
bin_dir.mkdir()
sudo = bin_dir / "sudo"
sudo.write_text('#!/bin/sh\nexec "$@"\n')
sudo.chmod(0o755)
return {**os.environ, "PATH": f"{bin_dir}:{os.environ['PATH']}"}


def test_stage_node_logs_merges_prefill_and_decode_nodes(tmp_path: Path) -> None:
prefill = tmp_path / "prefill-node"
decode = tmp_path / "decode-node"
shared = tmp_path / "shared"
prefill.mkdir()
decode.mkdir()
(prefill / "prefill_host-a.log").write_text("prefill output\n")
(prefill / "server_host-a.log").write_text("frontend output\n")
(decode / "decode_host-b.log").write_text("decode output\n")
(decode / "server_host-b.log").write_text("decode wrapper output\n")
env = _stub_sudo(tmp_path)

for node_logs in (prefill, decode):
subprocess.run(
["bash", str(STAGE_SCRIPT), str(node_logs), str(shared)],
check=True,
env=env,
)

assert sorted(path.name for path in shared.iterdir()) == [
"decode_host-b.log",
"prefill_host-a.log",
"server_host-a.log",
"server_host-b.log",
]
assert (shared / "decode_host-b.log").read_text() == "decode output\n"


def test_stage_node_logs_rejects_a_node_without_logs(tmp_path: Path) -> None:
missing = tmp_path / "missing"
shared = tmp_path / "shared"

completed = subprocess.run(
["bash", str(STAGE_SCRIPT), str(missing), str(shared)],
check=False,
capture_output=True,
text=True,
)

assert completed.returncode == 1
assert "no node-local logs found" in completed.stderr
assert not shared.exists()
Loading