From b5ddd3cf829c80d4b3992a603f6d7a2af93bb0d8 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Tue, 15 Sep 2026 23:57:08 -0500 Subject: [PATCH 1/5] fix(amd): restore original DeepSeek-V4-Pro for native MTP --- configs/amd-master.yaml | 2 +- perf-changelog.yaml | 8 ++++++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index bfcf0467cf..cfd5b34386 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -1195,7 +1195,7 @@ minimaxm3-fp8-mi325x-vllm-agentic-mtp: dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp: image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260911 - model: deepseek-ai/DeepSeek-V4-Pro-0813 + model: deepseek-ai/DeepSeek-V4-Pro model-prefix: dsv4 runner: cluster:mi355x-amds precision: fp4 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 910fd6fa7e..97af6fb3cd 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -7914,3 +7914,11 @@ - "Update SGLang image for the resident arm from lmsysorg/sglang:nightly-dev-cu13-20260907-30705c00 and the HiCache arm from lmsysorg/sglang:v0.5.19-cu130 to lmsysorg/sglang:nightly-dev-cu13-20260914-4358a161 (2026-09-14 upstream cu13 nightly, sgl-project/sglang@4358a161). The 2026-09-15 nightly (nightly-dev-cu13-20260915-8874c51a) cannot start any HiCache arm: sgl-project/sglang#35233 (merged 2026-09-14T22:49Z) made the host pool import get_device_accessible_ptr from a kernel wheel that image does not ship, so every kv-offloading dram point died at startup (runs 35016956476, 35016965515); sgl-project/sglang#39516 fixes it for the 2026-09-16 build, and 09-14 is the newest nightly that predates the regression. NEXTN MTP settings, HiCache arms, and the grid are unchanged" - "将 SGLang 镜像(驻留分支自 lmsysorg/sglang:nightly-dev-cu13-20260907-30705c00,HiCache 分支自 lmsysorg/sglang:v0.5.19-cu130)更新为 lmsysorg/sglang:nightly-dev-cu13-20260914-4358a161(2026-09-14 上游 cu13 nightly,sgl-project/sglang@4358a161)。2026-09-15 nightly(nightly-dev-cu13-20260915-8874c51a)的所有 HiCache 分支无法启动:sgl-project/sglang#35233(2026-09-14T22:49Z 合入)令主机池从镜像未附带的内核 wheel 导入 get_device_accessible_ptr,所有 kv-offloading dram 点在启动时失败(运行 35016956476、35016965515);sgl-project/sglang#39516 已修复并将随 2026-09-16 构建发布,09-14 是早于该回归的最新 nightly。NEXTN MTP 设置、HiCache 分支及网格保持不变" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3149 + +- config-keys: + - dsv4-fp4-mi355x-sglang-disagg-agentic-hicache-mtp + scenario-type: + - agentic-coding + description: + - "Rerun the MI355X disaggregated MTP sweep with the original DeepSeek-V4-Pro checkpoint, whose native MTP weights match the existing EAGLE recipe. The existing checkpoint selector restores thinking-on golden AL 2.49 for three draft tokens; evals use real acceptance." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX From 9dbb593dc338313cb0807222eaf8099b48e45ec0 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Tue, 15 Sep 2026 23:57:51 -0500 Subject: [PATCH 2/5] chore: link native MTP sweep changelog to PR 3170 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 97af6fb3cd..8f95a05208 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -7921,4 +7921,4 @@ - agentic-coding description: - "Rerun the MI355X disaggregated MTP sweep with the original DeepSeek-V4-Pro checkpoint, whose native MTP weights match the existing EAGLE recipe. The existing checkpoint selector restores thinking-on golden AL 2.49 for three draft tokens; evals use real acceptance." - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3170 From 140820dfbbb240c0a5c13a6c7d7004d46b9e78bf Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Wed, 16 Sep 2026 00:11:35 -0500 Subject: [PATCH 3/5] fix(amd): collect server logs from every Slurm node --- benchmarks/multi_node/amd_utils/job.slurm | 27 ++++++++ .../multi_node/amd_utils/stage_node_logs.sh | 23 +++++++ docs/architecture.md | 2 +- docs/architecture_zh.md | 2 +- perf-changelog.yaml | 1 + utils/test_amd_node_log_staging.py | 61 +++++++++++++++++++ 6 files changed, 114 insertions(+), 2 deletions(-) create mode 100755 benchmarks/multi_node/amd_utils/stage_node_logs.sh create mode 100644 utils/test_amd_node_log_staging.py diff --git a/benchmarks/multi_node/amd_utils/job.slurm b/benchmarks/multi_node/amd_utils/job.slurm index c12535776c..bb151f2039 100755 --- a/benchmarks/multi_node/amd_utils/job.slurm +++ b/benchmarks/multi_node/amd_utils/job.slurm @@ -775,6 +775,7 @@ echo \"[rank 0] Main container exited (rc=\$DOCKER_EXIT_CODE). Stopping vllm-rou \$DOCKER_CMD rm -f \"$ROUTER_CONT_NAME\" 2>/dev/null || true exit \$DOCKER_EXIT_CODE " +SERVER_SRUN_RC=$? if [[ "${KEEP_CONTAINERS}" != "1" ]]; then srun --nodelist="$SELECTED_NODELIST_SRUN" bash -c 'eval "$DOCKER_CMD_DETECT"; $DOCKER_CMD rm -f '"$DOCKER_CONT_NAME"' '"$CLIENT_CONT_NAME"' 2>/dev/null || true' @@ -786,3 +787,29 @@ if [[ "${KEEP_CONTAINERS}" != "1" ]]; then ' fi fi + +# /run_logs is backed by each compute node's local /tmp, so the node-0 copy +# performed by the engine launcher cannot see prefill/decode logs written on +# other nodes. Collect after the server step and container cleanup so failed +# runs also include shutdown output. KEEP_CONTAINERS=1 retains a snapshot of +# any containers left running for debugging. +# Use sudo because the container-created source and the existing node-0 +# destination can be root-owned. Restore ownership after the fan-in so a +# subsequent runner job can clean the workspace normally. +SHARED_JOB_LOGS="${BENCHMARK_LOGS_DIR}/logs/slurm_job-${SLURM_JOB_ID}" +if ! srun --nodelist="$SELECTED_NODELIST_SRUN" \ + --nodes="$NUM_NODES" --ntasks="$NUM_NODES" --ntasks-per-node=1 \ + bash "$DI_REPO_DIR/benchmarks/multi_node/amd_utils/stage_node_logs.sh" \ + "/tmp/slurm_job-${SLURM_JOB_ID}" "$SHARED_JOB_LOGS"; then + echo "[logs][ERROR] failed to stage logs from one or more Slurm nodes" >&2 + if [[ "$SERVER_SRUN_RC" -eq 0 ]]; then + SERVER_SRUN_RC=1 + fi +fi + +if [[ -d "$SHARED_JOB_LOGS" ]]; then + sudo chown -R "$(id -u):$(id -g)" "$SHARED_JOB_LOGS" 2>/dev/null || true + chmod -R a+rwX "$SHARED_JOB_LOGS" 2>/dev/null || true +fi + +exit "$SERVER_SRUN_RC" diff --git a/benchmarks/multi_node/amd_utils/stage_node_logs.sh b/benchmarks/multi_node/amd_utils/stage_node_logs.sh new file mode 100755 index 0000000000..f1e67814e4 --- /dev/null +++ b/benchmarks/multi_node/amd_utils/stage_node_logs.sh @@ -0,0 +1,23 @@ +#!/usr/bin/env bash + +set -eo pipefail + +if [[ $# -ne 2 ]]; then + echo "Usage: $0 " >&2 + exit 2 +fi + +SOURCE_LOGS=$1 +SHARED_LOGS=$2 + +if [[ ! -d "$SOURCE_LOGS" ]]; then + echo "[logs][ERROR] no node-local logs found on $(hostname): $SOURCE_LOGS" >&2 + exit 1 +fi + +# Server containers create the source tree as root, and node 0 may have already +# created the shared destination as root. The Slurm nodes provide passwordless +# sudo for the same Docker lifecycle used by job.slurm. +sudo mkdir -p "$SHARED_LOGS" +sudo cp -r "$SOURCE_LOGS"/. "$SHARED_LOGS"/ +echo "[logs] staged $(hostname):$SOURCE_LOGS -> $SHARED_LOGS" diff --git a/docs/architecture.md b/docs/architecture.md index a462f7215b..9e4b5c743a 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -277,7 +277,7 @@ The collector and reusable-artifact validator share format recognition, concurre Agentic throughput jobs have a different contract. They validate AIPerf output with [`infx/results/agentic/validate_agentic_result.py`](../infx/results/agentic/validate_agentic_result.py), upload an aggregate `bmk_agentic_` artifact, and upload the raw `agentic_` sibling containing trace-replay material. InferenceX-app pairs those siblings by their shared suffix. Agentic eval-only jobs follow the eval output contract instead and do not require a throughput result. -Server logs and GPU metrics are diagnostic side artifacts. They are uploaded with `always()` so a failed run can still be investigated. Their presence does not turn a failed benchmark into a valid result. +Server logs and GPU metrics are diagnostic side artifacts. They are uploaded with `always()` so a failed run can still be investigated. Their presence does not turn a failed benchmark into a valid result. On the AMD Slurm fleet, `/run_logs` is node-local; after the server step finishes, `job.slurm` merges the closed log tree from every allocated node into shared storage so the diagnostic artifact includes prefill and decode logs from the full deployment. ## Stage 6: artifact collection and handoff diff --git a/docs/architecture_zh.md b/docs/architecture_zh.md index 3e8fde840f..377a62cba4 100644 --- a/docs/architecture_zh.md +++ b/docs/architecture_zh.md @@ -277,7 +277,7 @@ rows = build_rows(raw_eval, metadata, source="eval_job/results.json") 智能体吞吐量作业采用不同的契约。它们使用 [`infx/results/agentic/validate_agentic_result.py`](../infx/results/agentic/validate_agentic_result.py) 验证 AIPerf 输出,上传聚合的 `bmk_agentic_` 工件,并上传包含追踪重放材料的原始 `agentic_` 同级工件。InferenceX-app 通过它们共享的后缀对这些同级工件进行配对。智能体仅评测作业改为遵循评测输出契约,不要求吞吐量结果。 -服务器日志和 GPU 指标是诊断辅助工件。它们通过 `always()` 上传,因此失败的运行仍可供调查。它们的存在不会将失败的基准测试转变为有效结果。 +服务器日志和 GPU 指标是诊断辅助工件。它们通过 `always()` 上传,因此失败的运行仍可供调查。它们的存在不会将失败的基准测试转变为有效结果。在 AMD Slurm 机群上,`/run_logs` 是节点本地目录;服务器步骤结束后,`job.slurm` 会把每个已分配节点上已经关闭的日志树合并到共享存储中,使诊断工件包含整个部署的 Prefill 和 Decode 日志。 ## 阶段 6:工件收集与交接 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 64910d5fcf..c712e38a6a 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -7930,4 +7930,5 @@ - agentic-coding description: - "Rerun the MI355X disaggregated MTP sweep with the original DeepSeek-V4-Pro checkpoint, whose native MTP weights match the existing EAGLE recipe. The existing checkpoint selector restores thinking-on golden AL 2.49 for three draft tokens; evals use real acceptance." + - "Collect completed server logs from every Slurm node into the shared artifact directory after the benchmark server step exits, including decode-node logs." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3170 diff --git a/utils/test_amd_node_log_staging.py b/utils/test_amd_node_log_staging.py new file mode 100644 index 0000000000..ead7c7c853 --- /dev/null +++ b/utils/test_amd_node_log_staging.py @@ -0,0 +1,61 @@ +from __future__ import annotations + +import os +import subprocess +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[1] +STAGE_SCRIPT = REPO_ROOT / "benchmarks/multi_node/amd_utils/stage_node_logs.sh" + + +def _stub_sudo(tmp_path: Path) -> dict[str, str]: + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + sudo = bin_dir / "sudo" + sudo.write_text('#!/bin/sh\nexec "$@"\n') + sudo.chmod(0o755) + return {**os.environ, "PATH": f"{bin_dir}:{os.environ['PATH']}"} + + +def test_stage_node_logs_merges_prefill_and_decode_nodes(tmp_path: Path) -> None: + prefill = tmp_path / "prefill-node" + decode = tmp_path / "decode-node" + shared = tmp_path / "shared" + prefill.mkdir() + decode.mkdir() + (prefill / "prefill_host-a.log").write_text("prefill output\n") + (prefill / "server_host-a.log").write_text("frontend output\n") + (decode / "decode_host-b.log").write_text("decode output\n") + (decode / "server_host-b.log").write_text("decode wrapper output\n") + env = _stub_sudo(tmp_path) + + for node_logs in (prefill, decode): + subprocess.run( + ["bash", str(STAGE_SCRIPT), str(node_logs), str(shared)], + check=True, + env=env, + ) + + assert sorted(path.name for path in shared.iterdir()) == [ + "decode_host-b.log", + "prefill_host-a.log", + "server_host-a.log", + "server_host-b.log", + ] + assert (shared / "decode_host-b.log").read_text() == "decode output\n" + + +def test_stage_node_logs_rejects_a_node_without_logs(tmp_path: Path) -> None: + missing = tmp_path / "missing" + shared = tmp_path / "shared" + + completed = subprocess.run( + ["bash", str(STAGE_SCRIPT), str(missing), str(shared)], + check=False, + capture_output=True, + text=True, + ) + + assert completed.returncode == 1 + assert "no node-local logs found" in completed.stderr + assert not shared.exists() From 6b53c2826c4c59858d011a4a5890759e6afc0520 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Wed, 16 Sep 2026 01:21:15 -0500 Subject: [PATCH 4/5] fix(amd): pass the router port to SGLang evaluation --- benchmarks/multi_node/amd_utils/server_sglang.sh | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/benchmarks/multi_node/amd_utils/server_sglang.sh b/benchmarks/multi_node/amd_utils/server_sglang.sh index b44ea1a11f..d00018f5bc 100755 --- a/benchmarks/multi_node/amd_utils/server_sglang.sh +++ b/benchmarks/multi_node/amd_utils/server_sglang.sh @@ -1183,6 +1183,8 @@ print(json.dumps(json.loads(sys.stdin.read())))' <<<"$_val")" || { # Must run from repo root so infx/evals/gsm8k.yaml resolves pushd /workspace + # Match the disaggregation router launched above. + export PORT=30000 source /workspace/benchmarks/benchmark_lib.sh # CONC must be exported before run_eval so meta_env.json matches validate_scores.py. @@ -1204,9 +1206,9 @@ print(json.dumps(json.loads(sys.stdin.read())))' <<<"$_val")" || { # arrive via Docker -e flags from job.slurm. if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: run_eval --port 30000 (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS}, ctx=${EVAL_MAX_MODEL_LEN:-auto})" + echo "DRY RUN: run_eval --port ${PORT} (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS}, ctx=${EVAL_MAX_MODEL_LEN:-auto})" else - run_eval --port 30000 + run_eval --port "$PORT" eval_rc=$? if [[ $eval_rc -ne 0 ]]; then From 351a9d28d72f9b50fba7584a755e0da05ec41815 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Wed, 16 Sep 2026 03:02:26 -0500 Subject: [PATCH 5/5] fix(amd): forward required AgentX client policy inputs --- benchmarks/runtime_settings.sh | 2 +- perf-changelog.yaml | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/benchmarks/runtime_settings.sh b/benchmarks/runtime_settings.sh index 0b04efe9ba..a369898f8b 100644 --- a/benchmarks/runtime_settings.sh +++ b/benchmarks/runtime_settings.sh @@ -32,4 +32,4 @@ export SGLANG_TORCH_PROFILER_DIR='/workspace' export VLLM_TORCH_PROFILER_DIR='/workspace' # Explicitly forward these settings across container boundaries. -export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY SWEBENCH_EXPECTED_INSTANCES SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_WATCHDOG_POLL SWEBENCH_SANDBOX_SWEEP SWEBENCH_SKIP_SCORE SWEBENCH_EVAL_TIMEOUT SWEBENCH_SCORE_TIMEOUT SWEBENCH_MAX_WORKERS EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR" +export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY SWEBENCH_EXPECTED_INSTANCES SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_WATCHDOG_POLL SWEBENCH_SANDBOX_SWEEP SWEBENCH_SKIP_SCORE SWEBENCH_EVAL_TIMEOUT SWEBENCH_SCORE_TIMEOUT SWEBENCH_MAX_WORKERS EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_EXPERIMENTAL_FAST AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER REQUIRE_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR" diff --git a/perf-changelog.yaml b/perf-changelog.yaml index c712e38a6a..e560adf1e2 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -7931,4 +7931,5 @@ description: - "Rerun the MI355X disaggregated MTP sweep with the original DeepSeek-V4-Pro checkpoint, whose native MTP weights match the existing EAGLE recipe. The existing checkpoint selector restores thinking-on golden AL 2.49 for three draft tokens; evals use real acceptance." - "Collect completed server logs from every Slurm node into the shared artifact directory after the benchmark server step exits, including decode-node logs." + - "Pass the existing router port to evaluation and forward workflow-owned AgentX fast-mode and power requirements across container boundaries." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3170