From 92b995ec79e724fd1ac3292918de7dcb58599fc3 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Mon, 21 Sep 2026 17:33:40 -0700 Subject: [PATCH 1/2] feat: drive GB200 AgentX power from resolved recipes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 通过实际生效的 SRT 配方接入 GB200 AgentX 功耗,保留 required-power、测量窗口和模型运行差异,并为 DSV4 添加普通配方接入与离线行为回归。 --- .../vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml | 13 + .../vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml | 13 + configs/nvidia-master.yaml | 2 + docs/configuration-procedures.md | 27 ++ docs/configuration-procedures_zh.md | 22 + infx/srt_slurm/synthetic_acceptance.py | 83 +++- perf-changelog.yaml | 30 ++ runners/launch_gb200-nv.sh | 85 ++-- runners/slurm_utils.sh | 56 +++ utils/test_gb200_recipe_power.py | 386 ++++++++++++++++++ 10 files changed, 651 insertions(+), 66 deletions(-) create mode 100644 utils/test_gb200_recipe_power.py diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml index 9c604d726b..77c728547e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-dep8-mtp.yaml @@ -96,7 +96,20 @@ roles: sbatch_directives: {cpus-per-task: "144", mem: "0"} srun_options: {container-remap-root: ""} +telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + concurrencies: [1] type: custom command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh env: diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml index 845cbf966f..9d1e0eaa77 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml @@ -106,7 +106,20 @@ sbatch_directives: srun_options: container-remap-root: "" +telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + collector_join_timeout_seconds: 12 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 + benchmark: + concurrencies: [1] type: custom command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh env: diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 9cecfd0c92..afca0ed5c7 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -5702,6 +5702,8 @@ dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-agg: scenarios: agentic-coding: - search-space: + # Both recipes require PowerX; the launcher supplies each matrix concurrency + # to the native collector and the shared AgentX benchmark window writer. # Match the B200 TP8 latency points for direct normalized-interactivity # comparison, then use the public vLLM GB200 DEP8 topology for throughput. - spec-decoding: mtp diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index de9f1cda5c..d1d6bb293b 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -183,6 +183,33 @@ Mapping source: [`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchm Do not ship one side alone. `srtctl` reads the recipe, while matrix generation reads the master config. Recipe-only changes can mislabel results. Master-only changes do not alter the deployed recipe. +### GB200 AgentX measured power + +GB200 SRT AgentX uses the selected recipe, after native selector expansion and +caller overrides, to enable PowerX. Use the existing `telemetry.enabled: true`, +`telemetry.required: true`, `storage_subdir: power`, and `dcgm_exporter` block +(`container_image: dcgm-exporter`); no model-specific power branch is needed. +The benchmark must use `bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh` +with `INFMAX_CONTAINER_WORKSPACE: /infmax-workspace`, `RESULT_DIR: /logs/agentic`, +and `IS_MULTINODE: "true"`. SRT validates the head-client clock, topology and +sampling settings. Model paths, serving arguments, quantization and mounts retain +their existing runtime configuration. + +The launcher installs the pinned submodule runtime before inspecting the recipe, +then provisions the exporter and records the same producer SHA for collection. +Each AgentX or power-enabled matrix job must select exactly one recipe (including an indexed zip +selector); a multi-recipe selection fails before submission. Native overrides +supply matrix `CONC_LIST` to both the benchmark and `benchmark.concurrencies`, +without rewriting recipe YAML. Enabled AgentX power also requires the shared +window writer and strict result adapter; it cannot be made successful by disabling +required telemetry. Eval-only jobs retain normal eval behavior and do not publish +throughput power results. + +The DSV4 GB200 vLLM TP8/DEP8 aggregate recipes demonstrate ordinary recipe-only +power enrollment. Local routing tests do not qualify live sampling: the selected +PR sweep still needs complete device/window evidence, failed-collection artifact +retention, evals, and downstream ingest/API/page verification. + ## Register an llm-d recipe Sources: [`benchmarks/llm-d/README.md`](../benchmarks/llm-d/README.md), [`benchmarks/multi_node/llm-d/README.md`](../benchmarks/multi_node/llm-d/README.md), [`llm-d-recipes/`](../benchmarks/multi_node/llm-d-recipes/), and the current [`llmd-vllm` benchmark wrapper](../benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh). diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index b2e0ce2d60..901ed32ef6 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -160,6 +160,28 @@ B200 Nscale 的 GLM-5.1 可用 `MODEL_PATH` 指定已有共享权重,覆盖默 不得只提交一侧:`srtctl` 读取配方,而矩阵生成读取主配置。仅改配方可能给结果贴错标签;仅改主配置不会改变实际部署的配方。 +### GB200 AgentX 实测功耗 + +GB200 SRT AgentX 根据原生 selector 展开及调用方 overrides 后的实际配方启用 +PowerX。使用现有的 `telemetry.enabled: true`、`telemetry.required: true`、 +`storage_subdir: power` 及 `dcgm_exporter` 配置(`container_image: dcgm-exporter`), +无需添加模型专属功耗分支。benchmark 必须使用 +`bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh`,并设置 +`INFMAX_CONTAINER_WORKSPACE: /infmax-workspace`、`RESULT_DIR: /logs/agentic` +和 `IS_MULTINODE: "true"`。SRT 校验 head client 时钟、拓扑与采样设置。 +模型路径、服务参数、量化和挂载保留各自现有的运行配置。 + +launcher 先安装固定 submodule runtime,再检查配方、准备 exporter,记录同一份 +producer SHA 供采集使用。每个 AgentX 或启用功耗的矩阵 job 必须只选中一个配方(可用带索引的 zip +selector);多配方选择会在提交前失败。原生 overrides 将矩阵 `CONC_LIST` 同时 +传给 benchmark 和 `benchmark.concurrencies`,不重写配方 YAML。启用 AgentX +功耗后必须使用共享窗口写入器与严格结果适配器,不能通过关闭 required telemetry +制造成功。eval-only 保留正常评估行为,不发布吞吐测量功耗结果。 + +DSV4 GB200 vLLM TP8/DEP8 aggregate 配方示范了普通配方的功耗接入。 +本地路由测试不代表真实采样通过:所选 PR sweep 仍需完整设备/窗口证据、 +采集失败后的产物保留、eval,以及下游 ingest/API/页面验收。 + ## 注册 llm-d 配方 来源:[`benchmarks/llm-d/README.md`](../benchmarks/llm-d/README.md)、[`benchmarks/multi_node/llm-d/README.md`](../benchmarks/multi_node/llm-d/README.md)、[`llm-d-recipes/`](../benchmarks/multi_node/llm-d-recipes/) 和当前 [`llmd-vllm` 基准 wrapper](../benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh)。 diff --git a/infx/srt_slurm/synthetic_acceptance.py b/infx/srt_slurm/synthetic_acceptance.py index 95d06b7763..d15d3dab34 100644 --- a/infx/srt_slurm/synthetic_acceptance.py +++ b/infx/srt_slurm/synthetic_acceptance.py @@ -9,6 +9,7 @@ import math import os import re +import shlex import subprocess import sys from collections.abc import Mapping @@ -230,15 +231,10 @@ def plan_commands( from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides path, _, selector = config.partition(":") - raw = yaml.safe_load(Path(path).read_text()) - if not isinstance(raw, dict): - raise ValueError("Recipe must be a mapping") + raw, caller_overrides = recipe_with_overrides(path, arguments) parser = argparse.ArgumentParser(add_help=False, allow_abbrev=False) parser.add_argument("--set", action="append") parser.add_argument("--unset", action="append") - existing, _ = parser.parse_known_args(arguments) - caller_overrides = parse_overrides(existing.set, existing.unset) - apply_overrides_to_recipe(raw, caller_overrides) commands = [] for variant, recipe in selected_recipes(raw, selector or None): arguments_to_add = build_overrides(recipe, framework, environment, golden_dir=golden_dir) @@ -260,17 +256,90 @@ def plan_commands( return commands +def recipe_with_overrides(path: str, arguments: list[str]) -> tuple[dict[str, Any], list[Any]]: + """Use the same native caller-override ordering for inspection and submission.""" + from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides + + raw = yaml.safe_load(Path(path).read_text()) + if not isinstance(raw, dict): + raise ValueError("Recipe must be a mapping") + parser = argparse.ArgumentParser(add_help=False, allow_abbrev=False) + parser.add_argument("--set", action="append") + parser.add_argument("--unset", action="append") + existing, _ = parser.parse_known_args(arguments) + overrides = parse_overrides(existing.set, existing.unset) + apply_overrides_to_recipe(raw, overrides) + return raw, overrides + + +def inspect_power(config: str, arguments: list[str], environment: Mapping[str, str]) -> str: + """Check the existing GB200 power artifact/window contract before submission.""" + from marshmallow import ValidationError + from srtctl.core.config import expand_engine_config_defaults, resolve_config_with_defaults + from srtctl.core.schema import SrtConfig + + path, _, selector = config.partition(":") + raw, _ = recipe_with_overrides(path, arguments) + variants = selected_recipes(raw, selector or None) + if len(variants) != 1: + # Preserve existing non-AgentX, non-power multi-variant submissions. + # Their lifecycle is separate from the single-job PowerX result contract. + if ( + variants + and environment["IS_AGENTIC"] != "1" + and all(not recipe.get("telemetry", {}).get("enabled", False) for _, recipe in variants) + ): + return "none" + raise ValueError("GB200 launcher requires exactly one selected recipe per job") + recipe = variants[0][1] + if not recipe.get("telemetry", {}).get("enabled", False): + return "none" + resolved = resolve_config_with_defaults(recipe, None) + expand_engine_config_defaults(resolved) + try: + typed = SrtConfig.Schema().load(resolved) + except ValidationError as error: + raise ValueError(f"Invalid power recipe: {error}") from error + if not typed.telemetry.enabled: + return "none" + if typed.telemetry.dcgm_exporter is None: + raise ValueError("GB200 PowerX requires telemetry.dcgm_exporter") + if typed.telemetry.dcgm_exporter.container_image != "dcgm-exporter": + raise ValueError("GB200 PowerX requires the dcgm-exporter container alias") + if typed.telemetry.storage_subdir != "power": + raise ValueError("GB200 PowerX requires telemetry.storage_subdir: power") + if environment["IS_AGENTIC"] != "1": + return "dcgm" + benchmark = typed.benchmark + if ( + benchmark.type != "custom" + or shlex.split(benchmark.command or "") + != ["bash", "/infmax-workspace/benchmarks/multi_node/agentic_srt.sh"] + or benchmark.env.get("RESULT_DIR") != "/logs/agentic" + or benchmark.env.get("INFMAX_CONTAINER_WORKSPACE") != "/infmax-workspace" + or benchmark.env.get("IS_MULTINODE") != "true" + ): + raise ValueError("AgentX PowerX requires the shared agentic_srt.sh window/result contract") + if not typed.telemetry.required: + raise ValueError("AgentX PowerX requires telemetry.required: true") + return "agentx" + + def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--inspect-power", action="store_true") parser.add_argument("config") parser.add_argument("framework") parser.add_argument("arguments", nargs=argparse.REMAINDER) args = parser.parse_args(argv) arguments = args.arguments[1:] if args.arguments[:1] == ["--"] else args.arguments try: + if args.inspect_power: + print(inspect_power(args.config, arguments, os.environ)) + return 0 commands = plan_commands(args.config, args.framework, arguments, os.environ) except (OSError, KeyError, ValueError, TypeError, yaml.YAMLError) as error: - print(f"ERROR: golden acceptance: {error}", file=sys.stderr) + print(f"ERROR: SRT recipe: {error}", file=sys.stderr) return 1 for command in commands: result = subprocess.run(command, check=False) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 86eb8dbb63..c6256c77a9 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8587,3 +8587,33 @@ - "将镜像从已失效的 nightly-eed1f3d0 重新固定到 vllm/vllm-openai-rocm:nightly-rocm100-3df4ae153eb385e27b52f26c81f8edb9e20b9984(与 MI355X 臂在 SemiAnalysisAI/InferenceX#3326 中所用相同 commit 的 ROCm 10.0 nightly 渠道构建);该镜像包含 vllm-project/vllm#57491,将两处 is_cuda() 判定放宽为 is_cuda_alike(),使 engram_config 在 gfx942 上可解析;由于 cpu_offload 现通过 VLLM_PLE_CPU_OFFLOAD 默认开启,配方按 TP 显式设置该值" - "新增 --no-swa-bounded-replay:vllm-project/vllm#56227 在两个 pin 之间将 SWA bounded replay 默认开启,其依赖的 window clamp 在 ROCm sparse SWA 路径中缺失,曾使所有 gfx950 数据点以 HSA_STATUS_ERROR_MEMORY_FAULT 崩溃;gfx942 使用同一路径。prefix caching 保持开启" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3337 + +- config-keys: + - dsr1-fp4-gb200-dynamo-trt + - dsr1-fp8-gb200-dynamo-trt + - dsr1-fp8-gb200-dynamo-sglang + - dsr1-fp4-gb200-dynamo-sglang + - qwen3.5-fp8-gb200-dynamo-sglang + - qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp + - minimaxm3-fp4-gb200-dynamo-vllm-agentic-agg-mtp + - minimaxm3-fp4-gb200-dynamo-vllm-agentic-disagg-mtp + - minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-agg + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-disagg + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-agg + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-disagg + - kimik3-fp4-gb200-dynamo-vllm-agentic + - kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-dcp16-agg + - kimik3-fp4-gb200-dynamo-vllm-agentic-mooncake-dcp16-agg + - kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2 + - glm5.2-fp4-gb200-dynamo-sglang-agentic-agg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-disagg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg + - qwen3.5-fp8-gb200-dynamo-sglang-mtp + description: + - Resolve GB200 SRT power from the selected recipe and native overrides; retain + required telemetry, window identity and failure propagation. + - Enable required measured power for both DSV4 GB200 vLLM AgentX aggregate recipes + without a model-specific power branch. + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX diff --git a/runners/launch_gb200-nv.sh b/runners/launch_gb200-nv.sh index 60dd034fcd..c180f078bd 100755 --- a/runners/launch_gb200-nv.sh +++ b/runners/launch_gb200-nv.sh @@ -264,52 +264,6 @@ NGINX_SQUASH_FILE="${SQUASH_DIR}/$(echo "$NGINX_IMAGE" | sed 's/[\/:@#]/_/g').sq import_squash "$SQUASH_FILE" "$IMAGE" import_squash "$NGINX_SQUASH_FILE" "$NGINX_IMAGE" -# The power lane is on iff the resolved recipe carries an enabled dcgm-power -# telemetry block. Read the workspace mirror; it overlays the srt-slurm clone later. -USES_DCGM_POWER=0 -_RECIPE_REL="${CONFIG_FILE%%:*}" -_RECIPE_SRC="$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/${_RECIPE_REL#recipes/}" -# Scoped match: a stray "enabled: true" outside the telemetry block must not flip the lane. -if [[ -n "$CONFIG_FILE" && -f "$_RECIPE_SRC" ]] && awk ' - /^telemetry:/ { t = 1; next } - t && /^[^ ]/ { t = 0 } - t && /^ dcgm_exporter:/ { p = 1 } - t && /^ enabled: true$/ { e = 1 } - END { exit !(p && e) } -' "$_RECIPE_SRC"; then - USES_DCGM_POWER=1 -fi - -USES_AGENTX_POWER=0 -if [[ "$USES_DCGM_POWER" == "1" && "$IS_AGENTIC" == "1" ]]; then - if [[ "$MODEL_PREFIX" == "glm5.2" && "$PRECISION" == "fp4" && - "$FRAMEWORK" == "dynamo-sglang" && - "$_RECIPE_REL" == "recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml" ]]; then - USES_AGENTX_POWER=1 - elif [[ "$MODEL_PREFIX" == "kimik3" && "$PRECISION" == "fp4" && - "$FRAMEWORK" == "dynamo-vllm" && - "$_RECIPE_REL" == recipes/kimik3/vllm/gb200-fp4/agentx/* ]]; then - USES_AGENTX_POWER=1 - else - echo "Error: AgentX dcgm-power requires the GLM-5.2 aggregate or supported Kimi-K3 recipe" >&2 - exit 1 - fi -fi -if [[ "$USES_DCGM_POWER" == "1" && "$FRAMEWORK" != "dynamo-sglang" && "$USES_AGENTX_POWER" != "1" ]]; then - echo "Error: dcgm-power requires dynamo-sglang or the supported Kimi-K3 AgentX route" >&2 - exit 1 -fi - -if [[ "$USES_DCGM_POWER" == "1" ]]; then - DCGM_EXPORTER_IMAGE="nvcr.io/nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless" - DCGM_EXPORTER_SQSH="${SQUASH_DIR}/$(echo "$DCGM_EXPORTER_IMAGE" | sed 's/[\/:@#]/_/g').sqsh" - import_squash "$DCGM_EXPORTER_SQSH" "$DCGM_EXPORTER_IMAGE" - test -r "$DCGM_EXPORTER_SQSH" || { echo "Error: DCGM exporter squash not readable: $DCGM_EXPORTER_SQSH" >&2; exit 1; } - unsquashfs -l "$DCGM_EXPORTER_SQSH" > /dev/null || { echo "Error: DCGM exporter squash invalid: $DCGM_EXPORTER_SQSH" >&2; exit 1; } - sha256sum "$DCGM_EXPORTER_SQSH" > "$GITHUB_WORKSPACE/exporter-image.sha256" -fi - - export ISL="$ISL" export OSL="$OSL" @@ -379,11 +333,8 @@ if [ -d "$SRT_REPO_DIR" ]; then rm -rf "$SRT_REPO_DIR" fi -# This checkpoint is staged on compute-node NVMe for the power lane. -if [[ "$USES_DCGM_POWER" == "1" && "$FRAMEWORK" == "dynamo-sglang" && "$MODEL_PREFIX" == "dsv4" && "$IS_AGENTIC" != "1" ]]; then - export MODEL_PATH="/mnt/numa1/models/DeepSeek-V4-Pro" -fi -setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" "$USES_DCGM_POWER" || exit 1 +# Power inspection needs this exact runtime's selector/override implementation. +setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" 0 || exit 1 echo "Installing srtctl..." curl -LsSf https://astral.sh/uv/install.sh | sh @@ -405,6 +356,23 @@ if ! command -v srtctl &> /dev/null; then exit 1 fi +prepare_gb200_srt_power "$CONFIG_FILE" "$FRAMEWORK" || exit 1 + +# This checkpoint is staged on compute-node NVMe for the non-AgentX power lane. +if [[ "$USES_DCGM_POWER" == "1" && "$FRAMEWORK" == "dynamo-sglang" && "$MODEL_PREFIX" == "dsv4" && "$IS_AGENTIC" != "1" ]]; then + export MODEL_PATH="/mnt/numa1/models/DeepSeek-V4-Pro" +fi +if [[ "$USES_DCGM_POWER" == "1" ]]; then + cp "$GITHUB_WORKSPACE/srt-slurm-sha.txt" "$GITHUB_WORKSPACE/power-producer-sha.txt" || exit 1 + DCGM_EXPORTER_IMAGE="nvcr.io/nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless" + DCGM_EXPORTER_SQSH="${SQUASH_DIR}/$(echo "$DCGM_EXPORTER_IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + import_squash "$DCGM_EXPORTER_SQSH" "$DCGM_EXPORTER_IMAGE" + test -r "$DCGM_EXPORTER_SQSH" || { echo "Error: DCGM exporter squash not readable: $DCGM_EXPORTER_SQSH" >&2; exit 1; } + unsquashfs -l "$DCGM_EXPORTER_SQSH" > /dev/null || { echo "Error: DCGM exporter squash invalid: $DCGM_EXPORTER_SQSH" >&2; exit 1; } + sha256sum "$DCGM_EXPORTER_SQSH" > "$GITHUB_WORKSPACE/exporter-image.sha256" +fi + + echo "Configs available at: $SRT_REPO_DIR/" SRTCTL_ROOT="${GITHUB_WORKSPACE}/srt-slurm" @@ -485,13 +453,7 @@ if command -v squeue >/dev/null 2>&1; then sleep 5 done fi -sed -i "s/^name:.*/name: \"${SRT_SLURM_JOB_NAME}\"/" "$CONFIG_PATH" - -if [[ "$USES_AGENTX_POWER" == "1" ]]; then - read -r -a POWER_CONCURRENCIES <<< "$CONC_LIST" - python3 "$GITHUB_WORKSPACE/runners/inject_srt_power_concurrencies.py" \ - "$CONFIG_PATH" "${POWER_CONCURRENCIES[@]}" || exit 1 -fi +SRTCTL_RECIPE_ARGS+=(--set "name=$SRT_SLURM_JOB_NAME") # sbatch's --export=ALL would carry VIRTUAL_ENV into job_script_minimal.j2, # whose `uv run` then dies with "Broken symlink at .venv/bin/python3" because @@ -516,8 +478,13 @@ if [[ "$FRAMEWORK" == "dynamo-sglang" ]]; then fi # srtctl gives RUNNER_NAME precedence over config.name; override it for the # submission so the #SBATCH job name keeps the namespace used above. -SRTCTL_OUTPUT=$(RUNNER_NAME="$SRT_SLURM_JOB_NAME" apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_EVAL_ARGS[@]}" "${SRTCTL_APPLY_ARGS[@]}" 2>&1) +SRTCTL_OUTPUT=$(RUNNER_NAME="$SRT_SLURM_JOB_NAME" apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_RECIPE_ARGS[@]}" "${SRTCTL_APPLY_ARGS[@]}" 2>&1) +SRTCTL_RC=$? echo "$SRTCTL_OUTPUT" +if [[ "$SRTCTL_RC" != "0" ]]; then + cancel_submitted_srt_jobs "$SRTCTL_OUTPUT" + exit "$SRTCTL_RC" +fi JOB_ID=$(echo "$SRTCTL_OUTPUT" | grep -oP '✅ Job \K[0-9]+' || echo "$SRTCTL_OUTPUT" | grep -oP 'Job \K[0-9]+') diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 67a6eac780..135febdce7 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -127,6 +127,62 @@ apply_srt_recipe() { "$config" "$framework" -- "$@" } +prepare_gb200_srt_power() { + check_env_vars IS_AGENTIC EVAL_ONLY + # Keep inspection and submission on the same native overrides. The benchmark + # client and the collector must expect the same matrix concurrency windows. + local config="$1" framework="$2" concurrency power_mode concurrency_json + SRTCTL_RECIPE_ARGS=("${SRTCTL_EVAL_ARGS[@]}") + if [[ "$IS_AGENTIC" == "1" ]]; then + check_env_vars CONC_LIST + local -a power_concurrencies + read -r -a power_concurrencies <<< "$CONC_LIST" + for concurrency in "${power_concurrencies[@]}"; do + [[ "$concurrency" =~ ^[1-9][0-9]*$ ]] || { + echo "Error: invalid AgentX concurrency: $concurrency" >&2 + return 1 + } + done + concurrency_json=$(IFS=,; echo "[${power_concurrencies[*]}]") + SRTCTL_RECIPE_ARGS+=( + --set "benchmark.concurrencies=$concurrency_json" + --set "benchmark.env.CONC_LIST=\"${power_concurrencies[*]}\"" + ) + fi + power_mode=$(PYTHONPATH="$INFERENCEX_SLURM_UTILS_DIR/..${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.srt_slurm.synthetic_acceptance --inspect-power \ + "$config" "$framework" -- "${SRTCTL_RECIPE_ARGS[@]}") || return 1 + USES_DCGM_POWER=0 + USES_AGENTX_POWER=0 + case "$power_mode" in + agentx) + USES_DCGM_POWER=1 + USES_AGENTX_POWER=1 + SRTCTL_RECIPE_ARGS+=( + --set 'benchmark.env.ENABLE_AGENTX_POWER="1"' + --set 'benchmark.env.REQUIRE_POWER="1"' + ) + ;; + dcgm) + USES_DCGM_POWER=1 + if [[ "$framework" != "dynamo-sglang" ]]; then + echo "Error: non-AgentX dcgm-power requires dynamo-sglang" >&2 + return 1 + fi + ;; + none) ;; + *) echo "Error: unknown recipe power mode: $power_mode" >&2; return 1 ;; + esac +} + +# A later submission can fail after an earlier variant acquired an allocation. +cancel_submitted_srt_jobs() { + local job_id + while read -r job_id; do + [[ -n "$job_id" ]] && scancel "$job_id" 2>/dev/null || true + done < <(printf '%s\n' "$1" | sed -nE 's/.*Job ([0-9]+).*/\1/p' | sort -u) +} + slurm_job_is_active() { local job_id="$1" squeue -j "$job_id" --noheader 2>/dev/null | grep -q "$job_id" diff --git a/utils/test_gb200_recipe_power.py b/utils/test_gb200_recipe_power.py new file mode 100644 index 0000000000..00604a4c7f --- /dev/null +++ b/utils/test_gb200_recipe_power.py @@ -0,0 +1,386 @@ +"""Exercise GB200 recipe power routing and submission through the shell entrypoints.""" + +from __future__ import annotations + +import copy +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest +import yaml + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) + +from srtctl.core.config import generate_override_configs # noqa: E402 +from srtctl.core.overrides import ( # noqa: E402 + apply_overrides_to_recipe, + parse_overrides, +) + + +@pytest.fixture +def dsv4_recipe() -> dict: + """Use the supported serving recipe, with a controlled telemetry contract.""" + path = ROOT / ( + "benchmarks/multi_node/srt-slurm-recipes/" + "dsv4/vllm/gb200-fp4/agentx/agg-tp8-mtp.yaml" + ) + recipe = yaml.safe_load(path.read_text()) + recipe["telemetry"] = { + "enabled": True, + "required": True, + "collect_interval_ms": 1000, + "storage_subdir": "power", + "dcgm_exporter": {"container_image": "dcgm-exporter", "port": 9401}, + } + recipe["benchmark"]["concurrencies"] = [99] + recipe["benchmark"]["env"]["CONC_LIST"] = "99" + return recipe + + +def _executable(path: Path, source: str) -> None: + path.write_text(f"#!{sys.executable}\n{source}") + path.chmod(0o755) + + +def _submit(tmp_path: Path, recipe: dict, *, selector="", overrides=(), **environment): + recipe_path = tmp_path / "recipe with spaces.yaml" + original = yaml.safe_dump(recipe) + recipe_path.write_text(original) + _executable( + tmp_path / "srtctl", + "import json,os,sys\n" + "with open(os.environ['SUBMISSIONS'], 'a') as handle:\n" + " handle.write(json.dumps(sys.argv[1:])+'\\n')\n" + "print('✅ Job 12345 submitted')\n" + "sys.exit(int(os.environ.get('SUBMISSION_RC', '0')))\n", + ) + result = subprocess.run( + [ + "bash", + "-c", + 'source "$1" || exit $?; config="$2"; shift 2; ' + 'SRTCTL_EVAL_ARGS+=("$@"); ' + 'prepare_gb200_srt_power "$config" dynamo-vllm || exit $?; ' + 'printf \'{"dcgm":%s,"agentx":%s}\\n\' ' + '"$USES_DCGM_POWER" "$USES_AGENTX_POWER" > "$LANE"; ' + 'apply_srt_recipe "$config" dynamo-vllm ' + '-f "$config" --tags "ordinary AgentX submission" "${SRTCTL_RECIPE_ARGS[@]}"', + "bash", + str(ROOT / "runners/slurm_utils.sh"), + f"{recipe_path}{':' + selector if selector else ''}", + *overrides, + ], + env={ + **os.environ, + "PATH": f"{tmp_path}:{Path(sys.executable).parent}:{os.environ['PATH']}", + "PYTHONPATH": os.pathsep.join( + [str(ROOT), str(ROOT / "utils/srt-slurm/src")] + ), + "MODEL_PREFIX": "dsv4", + "IS_AGENTIC": "1", + "EVAL_ONLY": "false", + "SPEC_DECODING": "mtp", + "THINKING_MODE": "thinking_on", + "CONC_LIST": "1 4", + "SUBMISSIONS": str(tmp_path / "submissions.jsonl"), + "LANE": str(tmp_path / "lane.json"), + **environment, + }, + cwd=tmp_path, + text=True, + capture_output=True, + check=False, + ) + assert recipe_path.read_text() == original + recorded = tmp_path / "submissions.jsonl" + commands = ( + [json.loads(line) for line in recorded.read_text().splitlines()] + if recorded.exists() + else [] + ) + lane_path = tmp_path / "lane.json" + lane = json.loads(lane_path.read_text()) if lane_path.exists() else None + return result, commands, lane + + +def _resolved_submission(raw: dict, command: list[str]) -> dict: + """Decode recorded native CLI arguments with the pinned SRT implementation.""" + recipe = copy.deepcopy(raw) + sets = [command[i + 1] for i, value in enumerate(command[:-1]) if value == "--set"] + unsets = [ + command[i + 1] for i, value in enumerate(command[:-1]) if value == "--unset" + ] + apply_overrides_to_recipe(recipe, parse_overrides(sets, unsets)) + config = command[command.index("--file") + 1] + _, _, selector = config.partition(":") + if "base" in recipe: + return generate_override_configs(recipe, selector=selector or None)[0][1] + return recipe + + +def test_dsv4_normal_submission_injects_matrix_power_without_changing_serving( + tmp_path, dsv4_recipe +): + result, commands, lane = _submit(tmp_path, dsv4_recipe) + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": 1, "agentx": 1} + assert len(commands) == 1 + resolved = _resolved_submission(dsv4_recipe, commands[0]) + assert resolved["benchmark"]["concurrencies"] == [1, 4] + assert resolved["benchmark"]["env"]["CONC_LIST"] == "1 4" + assert resolved["benchmark"]["env"]["REQUIRE_POWER"] == "1" + assert resolved["benchmark"]["env"]["ENABLE_AGENTX_POWER"] == "1" + assert resolved["telemetry"]["required"] is True + for field in ("model", "identity", "resources", "frontend", "engine"): + assert resolved[field] == dsv4_recipe[field] + before = copy.deepcopy(dsv4_recipe["roles"]) + after = copy.deepcopy(resolved["roles"]) + for role in before: + original_spec = json.loads(before[role]["args"].pop("speculative-config")) + generated_spec = json.loads(after[role]["args"].pop("speculative-config")) + assert generated_spec.pop("rejection_sample_method") == "synthetic" + assert generated_spec.pop("synthetic_acceptance_length") > 1 + assert generated_spec == original_spec + assert after == before + + +@pytest.mark.parametrize( + "selector, expected", + [ + ("base", 0), + ("override_power", 1), + ("zip_override_power[0]", 0), + ("zip_override_power[1]", 1), + ], +) +def test_native_selector_controls_the_power_lane( + tmp_path, dsv4_recipe, selector, expected +): + dsv4_recipe["telemetry"]["enabled"] = False + raw = { + "schema": 2, + "base": dsv4_recipe, + "override_power": {"telemetry": {"enabled": True}}, + "zip_override_power": { + "name": ["disabled", "enabled"], + "telemetry": {"enabled": [False, True]}, + }, + } + result, commands, lane = _submit(tmp_path, raw, selector=selector) + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": expected, "agentx": expected} + assert len(commands) == 1 + resolved = _resolved_submission(raw, commands[0]) + assert resolved["telemetry"]["enabled"] is bool(expected) + assert resolved["benchmark"]["concurrencies"] == [1, 4] + + +@pytest.mark.parametrize( + "overrides", [("--set", "telemetry.enabled=false"), ("--unset", "telemetry")] +) +def test_native_caller_override_disables_power(tmp_path, dsv4_recipe, overrides): + result, commands, lane = _submit(tmp_path, dsv4_recipe, overrides=overrides) + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": 0, "agentx": 0} + resolved = _resolved_submission(dsv4_recipe, commands[0]) + assert not resolved.get("telemetry", {}).get("enabled", False) + assert "REQUIRE_POWER" not in resolved["benchmark"]["env"] + + +def test_eval_uses_real_verification_with_the_same_recipe_contract( + tmp_path, dsv4_recipe +): + result, commands, lane = _submit(tmp_path, dsv4_recipe, EVAL_ONLY="true") + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": 1, "agentx": 1} + resolved = _resolved_submission(dsv4_recipe, commands[0]) + assert resolved["roles"] == dsv4_recipe["roles"] + assert resolved["benchmark"]["concurrencies"] == [1, 4] + + +def test_multiple_selected_recipes_fail_before_submission(tmp_path, dsv4_recipe): + raw = { + "schema": 2, + "base": dsv4_recipe, + "zip_override_power": {"name": ["first", "second"]}, + } + result, commands, lane = _submit(tmp_path, raw, selector="zip_override_power") + assert result.returncode != 0 + assert "exactly one selected recipe" in result.stderr + assert commands == [] + assert lane is None + + +def test_non_agentx_disabled_telemetry_preserves_multiple_submissions( + tmp_path, dsv4_recipe +): + dsv4_recipe["telemetry"]["enabled"] = False + raw = { + "schema": 2, + "base": dsv4_recipe, + "zip_override_legacy": {"name": ["first", "second"]}, + } + result, commands, lane = _submit( + tmp_path, raw, selector="zip_override_legacy", IS_AGENTIC="0" + ) + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": 0, "agentx": 0} + assert len(commands) == 2 + resolved = [_resolved_submission(raw, command) for command in commands] + assert [recipe["name"] for recipe in resolved] == ["first", "second"] + for recipe in resolved: + assert recipe["telemetry"]["enabled"] is False + assert recipe["benchmark"]["concurrencies"] == [99] + assert recipe["roles"] == dsv4_recipe["roles"] + + +def test_non_agentx_mixed_telemetry_fails_before_any_submission(tmp_path, dsv4_recipe): + raw = { + "schema": 2, + "base": dsv4_recipe, + "zip_override_mixed": { + "name": ["disabled", "enabled"], + "telemetry": {"enabled": [False, True]}, + }, + } + result, commands, lane = _submit( + tmp_path, raw, selector="zip_override_mixed", IS_AGENTIC="0" + ) + assert result.returncode != 0 + assert "exactly one selected recipe" in result.stderr + assert commands == [] + assert lane is None + + +@pytest.mark.parametrize("concurrencies", ["1 1", "0 4", "1 nan"]) +def test_invalid_matrix_concurrencies_fail_before_submission( + tmp_path, dsv4_recipe, concurrencies +): + result, commands, _ = _submit(tmp_path, dsv4_recipe, CONC_LIST=concurrencies) + assert result.returncode != 0 + assert commands == [] + + +@pytest.mark.parametrize( + "overrides, reason", + [ + (("--set", "telemetry.required=false"), "telemetry.required"), + (("--set", 'telemetry.storage_subdir="other"'), "storage_subdir"), + (("--set", 'benchmark.env.RESULT_DIR="/logs/other"'), "window/result contract"), + ], +) +def test_incompatible_power_contract_fails_before_submission( + tmp_path, dsv4_recipe, overrides, reason +): + result, commands, _ = _submit(tmp_path, dsv4_recipe, overrides=overrides) + assert result.returncode != 0 + assert reason in result.stderr + assert commands == [] + + +def test_submission_failure_is_returned_even_with_a_job_id(tmp_path, dsv4_recipe): + result, commands, lane = _submit(tmp_path, dsv4_recipe, SUBMISSION_RC="7") + assert result.returncode == 7 + assert "Job 12345" in result.stdout + assert len(commands) == 1 + assert lane == {"dcgm": 1, "agentx": 1} + + +@pytest.mark.parametrize( + "job_status, adapter_rc, expected_rc", + [("COMPLETED|0:0", 9, 9), ("FAILED|1:0", 0, 1)], +) +def test_collection_retains_results_and_audits_when_a_collaborator_fails( + tmp_path, job_status, adapter_rc, expected_rc +): + """Check shell failure propagation; the adapter here is deliberately a stub.""" + source, workspace, logs = [ + tmp_path / name for name in ("source", "workspace", "logs") + ] + for path in (source, workspace, logs): + path.mkdir() + for concurrency in (1, 4): + (source / f"point_conc{concurrency}.json").write_text( + json.dumps({"conc": concurrency}) + ) + _executable(tmp_path / "sacct", f"print('12345|{job_status}')\n") + _executable( + tmp_path / "python3", + "import json,os,sys\nfrom pathlib import Path\n" + "args=sys.argv[1:]\n" + "directory=Path(args[args.index('--result-dir')+1]); directory.mkdir(parents=True,exist_ok=True)\n" + "code=int(os.environ['ADAPTER_RC']) if directory.name=='conc_1' else 0\n" + "(directory/'power_validation.json').write_text(json.dumps({'stub_exit_code':code}))\n" + "sys.exit(code)\n", + ) + result = subprocess.run( + [ + "bash", + "-c", + 'source "$1"; collect_agentic_power_results 12345 "$2" "$3" "$4" point producer-sha 1 4', + "bash", + str(ROOT / "runners/slurm_utils.sh"), + str(logs), + str(source), + str(workspace), + ], + env={ + **os.environ, + "PATH": f"{tmp_path}:{os.environ['PATH']}", + "ADAPTER_RC": str(adapter_rc), + }, + cwd=tmp_path, + text=True, + capture_output=True, + check=False, + ) + assert result.returncode == expected_rc, result.stderr + assert ( + logs / "power/native-job-status.txt" + ).read_text().strip() == f"12345|{job_status}" + for concurrency, expected_adapter_rc in ((1, adapter_rc), (4, 0)): + assert json.loads( + (workspace / f"point_conc{concurrency}.json").read_text() + ) == {"conc": concurrency} + assert json.loads( + (logs / f"agentic/conc_{concurrency}/power_validation.json").read_text() + ) == {"stub_exit_code": expected_adapter_rc} + + +def test_failed_submission_cleanup_cancels_only_returned_job_ids(tmp_path): + _executable( + tmp_path / "scancel", + "import json,os,sys\n" + "with open(os.environ['CANCELLED'], 'a') as f:\n" + " f.write(json.dumps(sys.argv[1:])+'\\n')\n" + "sys.exit(1)\n", + ) + output = "✅ Job 12345 submitted\n✅ Job 12346 submitted\nJob 12345\nERROR: next submission failed\n" + result = subprocess.run( + [ + "bash", + "-c", + 'source "$1"; cancel_submitted_srt_jobs "$2"; exit 7', + "bash", + str(ROOT / "runners/slurm_utils.sh"), + output, + ], + env={ + **os.environ, + "PATH": f"{tmp_path}:{os.environ['PATH']}", + "CANCELLED": str(tmp_path / "cancelled.jsonl"), + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 7 + assert [ + json.loads(line) + for line in (tmp_path / "cancelled.jsonl").read_text().splitlines() + ] == [["12345"], ["12346"]] From 879b50170e66832d9afdafbf24a7ec324aebf049 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Mon, 21 Sep 2026 22:43:54 -0700 Subject: [PATCH 2/2] chore: link GB200 recipe power changelog to PR 3358 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将本次 GB200 配方功耗改动追加条目的占位链接更新为 PR #3358,保留历史条目和运行逻辑。 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index c6256c77a9..3eaf1de395 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8616,4 +8616,4 @@ required telemetry, window identity and failure propagation. - Enable required measured power for both DSV4 GB200 vLLM AgentX aggregate recipes without a model-specific power branch. - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3358