diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 24f958976e..885a34a220 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -35,6 +35,8 @@ git submodule update --init To upgrade, fetch and check out the desired commit inside the relevant submodule, then commit the updated submodule pointer in InferenceX. Benchmark workflows already initialize submodules. Slurm launchers make a local Git clone for each job so recipe staging and runtime writes do not modify the submodule, and record the actual commit for result provenance. NVIDIA setup clones locally; TileRT setup fetches its pinned fork commit over the network. +B200, GB200, and GB300 resolve the job checkout's actual Hatch package version on local disk before installing it. This avoids repeated `git describe --dirty` scans of shared storage. The launcher preserves that version for both login and compute editable builds, uses a Python 3.12 login environment, and stops on environment or installation failures. The version is retained in `srt-slurm-version.txt` and the uploaded `srt-setup.log`; it supplements the pinned `srt-slurm-sha.txt` and does not replace source or patch provenance. Native compute jobs without a preserved version keep their existing version lookup. + Single-node fixed-sequence recipes use NVIDIA upstream srt-slurm. ATOM recipes use the native `atomesh` frontend with one aggregate worker and `enable_multiple_frontends: false`. The router's pinned official image belongs in @@ -160,6 +162,20 @@ Sources: [`configs/CONFIGS.md`](../configs/CONFIGS.md), [`validation.py`](../uti Fixed-sequence `8192/1024` scenarios may set `require-power: true` to opt into validated measured power. The matrix passes this flag to standard sweeps and manual E2E throughput jobs; eval-only and AgentX rows do not inherit it. Omit the field to preserve existing behavior. Enable it only alongside the corresponding runtime and result adapter, then qualify the complete selected scope. +For native AgentX on B200 Nscale, GB200 and GB300, enable `telemetry.enabled` and +`telemetry.required` in the SRT recipe, use `storage_subdir: power` and the +`dcgm-exporter` container alias. The launchers inspect the selected recipe after +native overrides, require the shared `agentic_srt.sh` window/result contract and +forward the matrix concurrency to both replay and power validation. An invalid +enabled contract fails before submission. Eval-only jobs retain real verification +and do not produce performance power results. + +Qualification requires the strict power adapter to accept the recorded serving +window, all participating GPUs identified by hostname and UUID, samples bracketing +both boundaries and no adjacent sample gap over three seconds. Retain the raw +telemetry, window and configuration audit plus producer SHA. Recipe enablement or +passing local checks alone does not qualify measured power or publish results. + ## Register and set up a runner Setup source: [`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNER_SETUP.md). Config source: [`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners). diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 83fd0bee97..4405e0bb8a 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -35,6 +35,8 @@ git submodule update --init 升级时,在对应子模块中获取并检出目标提交,再将更新后的子模块指针提交到 InferenceX。基准测试工作流已配置为自动初始化子模块。Slurm 启动器为每个作业创建本地 Git 克隆,避免配方准备和运行时写入修改子模块,并记录实际提交以供结果溯源。NVIDIA 启动器使用本地克隆;TileRT 启动器通过网络获取固定的分支提交。 +B200、GB200 和 GB300 在本地磁盘上解析作业检出副本的实际 Hatch 包版本,再安装运行环境,避免在共享存储上反复执行 `git describe --dirty` 扫描。启动器为登录节点和计算节点的可编辑构建保留同一版本,登录环境使用 Python 3.12;环境创建或安装失败时立即停止。版本写入 `srt-slurm-version.txt` 和上传的 `srt-setup.log`,作为固定 `srt-slurm-sha.txt` 的补充,不替代源码或补丁来源记录。没有保留版本文件的原生计算作业继续使用原有版本解析方式。 + 单节点固定序列长度配方使用 NVIDIA 上游 srt-slurm。ATOM 配方使用原生 `atomesh` frontend、一个聚合 worker,并设置 `enable_multiple_frontends: false`。旧版基准 worker 镜像不包含 AToMesh,因此通过 `frontend.container_image` 单独固定路由器的官方镜像。 @@ -111,6 +113,16 @@ STP(Single Token Prediction,单 Token 预测)是每次前向传播生成 固定序列 `8192/1024` 场景可设置 `require-power: true`,要求经过验证的实测功耗。矩阵将此标记传递给标准 sweep 和手动 E2E 吞吐作业;eval-only 和 AgentX 行不继承该标记。省略此字段可保留现有行为。仅在对应 runtime 和结果适配器同时交付时启用,然后验证完整选定范围。 +对于 B200 Nscale、GB200 和 GB300 的原生 AgentX,在 SRT 配方中启用 +`telemetry.enabled` 和 `telemetry.required`,设置 `storage_subdir: power`,并使用 +`dcgm-exporter` 容器别名。launcher 在应用原生覆盖项后检查选定配方,要求使用共享 +`agentic_srt.sh` 的测量窗口和结果契约,并将矩阵并发同时传给回放和功耗验证。 +已启用但不符合契约的配置会在提交前失败。eval-only 作业保留真实验证,不生成性能功耗结果。 + +验收要求严格功耗适配器接受记录的服务窗口,按 hostname 和 UUID 标识所有参与 GPU, +采样覆盖窗口两端,且相邻样本间隔不超过三秒。保留原始遥测、窗口、配置审计和 +producer SHA。仅启用配方或通过本地检查不代表实测功耗合格,也不代表结果已发布。 + ## 注册并设置 runner 设置来源:[`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNER_SETUP.md)。配置来源:[`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners)。 diff --git a/infx/srt_slurm/synthetic_acceptance.py b/infx/srt_slurm/synthetic_acceptance.py index be7232eb12..f921a9e071 100644 --- a/infx/srt_slurm/synthetic_acceptance.py +++ b/infx/srt_slurm/synthetic_acceptance.py @@ -9,6 +9,7 @@ import math import os import re +import shlex import subprocess import sys from collections.abc import Mapping @@ -241,15 +242,10 @@ def plan_commands( from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides path, _, selector = config.partition(":") - raw = yaml.safe_load(Path(path).read_text()) - if not isinstance(raw, dict): - raise ValueError("Recipe must be a mapping") + raw, caller_overrides = recipe_with_overrides(path, arguments) parser = argparse.ArgumentParser(add_help=False, allow_abbrev=False) parser.add_argument("--set", action="append") parser.add_argument("--unset", action="append") - existing, _ = parser.parse_known_args(arguments) - caller_overrides = parse_overrides(existing.set, existing.unset) - apply_overrides_to_recipe(raw, caller_overrides) commands = [] for variant, recipe in selected_recipes(raw, selector or None): arguments_to_add = build_overrides(recipe, framework, environment, golden_dir=golden_dir) @@ -271,17 +267,90 @@ def plan_commands( return commands +def recipe_with_overrides(path: str, arguments: list[str]) -> tuple[dict[str, Any], list[Any]]: + """Use the same native caller-override ordering for inspection and submission.""" + from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides + + raw = yaml.safe_load(Path(path).read_text()) + if not isinstance(raw, dict): + raise ValueError("Recipe must be a mapping") + parser = argparse.ArgumentParser(add_help=False, allow_abbrev=False) + parser.add_argument("--set", action="append") + parser.add_argument("--unset", action="append") + existing, _ = parser.parse_known_args(arguments) + overrides = parse_overrides(existing.set, existing.unset) + apply_overrides_to_recipe(raw, overrides) + return raw, overrides + + +def inspect_power(config: str, arguments: list[str], environment: Mapping[str, str]) -> str: + """Check the native power artifact/window contract before submission.""" + from marshmallow import ValidationError + from srtctl.core.config import expand_engine_config_defaults, resolve_config_with_defaults + from srtctl.core.schema import SrtConfig + + path, _, selector = config.partition(":") + raw, _ = recipe_with_overrides(path, arguments) + variants = selected_recipes(raw, selector or None) + if len(variants) != 1: + # Preserve existing non-AgentX, non-power multi-variant submissions. + # Their lifecycle is separate from the single-job PowerX result contract. + if ( + variants + and environment["IS_AGENTIC"] != "1" + and all(not recipe.get("telemetry", {}).get("enabled", False) for _, recipe in variants) + ): + return "none" + raise ValueError("SRT power launcher requires exactly one selected recipe per job") + recipe = variants[0][1] + if not recipe.get("telemetry", {}).get("enabled", False): + return "none" + resolved = resolve_config_with_defaults(recipe, None) + expand_engine_config_defaults(resolved) + try: + typed = SrtConfig.Schema().load(resolved) + except ValidationError as error: + raise ValueError(f"Invalid power recipe: {error}") from error + if not typed.telemetry.enabled: + return "none" + if typed.telemetry.dcgm_exporter is None: + raise ValueError("SRT PowerX requires telemetry.dcgm_exporter") + if typed.telemetry.dcgm_exporter.container_image != "dcgm-exporter": + raise ValueError("SRT PowerX requires the dcgm-exporter container alias") + if typed.telemetry.storage_subdir != "power": + raise ValueError("SRT PowerX requires telemetry.storage_subdir: power") + if environment["IS_AGENTIC"] != "1": + return "dcgm" + benchmark = typed.benchmark + if ( + benchmark.type != "custom" + or shlex.split(benchmark.command or "") + != ["bash", "/infmax-workspace/benchmarks/multi_node/agentic_srt.sh"] + or benchmark.env.get("RESULT_DIR") != "/logs/agentic" + or benchmark.env.get("INFMAX_CONTAINER_WORKSPACE") != "/infmax-workspace" + or benchmark.env.get("IS_MULTINODE") != "true" + ): + raise ValueError("AgentX PowerX requires the shared agentic_srt.sh window/result contract") + if not typed.telemetry.required: + raise ValueError("AgentX PowerX requires telemetry.required: true") + return "agentx" + + def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--inspect-power", action="store_true") parser.add_argument("config") parser.add_argument("framework") parser.add_argument("arguments", nargs=argparse.REMAINDER) args = parser.parse_args(argv) arguments = args.arguments[1:] if args.arguments[:1] == ["--"] else args.arguments try: + if args.inspect_power: + print(inspect_power(args.config, arguments, os.environ)) + return 0 commands = plan_commands(args.config, args.framework, arguments, os.environ) except (OSError, KeyError, ValueError, TypeError, yaml.YAMLError) as error: - print(f"ERROR: golden acceptance: {error}", file=sys.stderr) + print(f"ERROR: SRT recipe: {error}", file=sys.stderr) return 1 for command in commands: result = subprocess.run(command, check=False) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 89f152cacc..657e148517 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8887,3 +8887,92 @@ - "Add DeepSeek-V4-Pro-0813 golden AL for draft lengths 4, 5, 7 and 8 (3.36 / 3.61 / 3.73 / 3.47)." - "Agentic PD router: pin --decode-policy round_robin so decode no longer inherits the prefill --policy (consistent_hashing)." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3256 + +- config-keys: + - dsr1-fp4-b200-dynamo-trt + - dsr1-fp8-b200-dynamo-trt + - dsr1-fp4-b300-dynamo-trt + - dsr1-fp8-b300-dynamo-trt + - kimik3-fp4-h200-vllm-agentic-latency + - kimik3-fp4-h200-vllm-agentic-balanced + - kimik3-fp4-h200-vllm-agentic-simple + - dsv4-fp8-h200-dynamo-sglang-agentic-agg + - dsr1-fp8-h200-dynamo-trt + - dsr1-fp8-h100-dynamo-trt + - dsr1-fp8-h100-dynamo-sglang + - dsr1-fp4-gb200-dynamo-trt + - dsr1-fp8-gb200-dynamo-trt + - dsr1-fp8-gb200-dynamo-sglang + - dsr1-fp8-gb300-dynamo-sglang + - dsr1-fp4-gb200-dynamo-sglang + - dsr1-fp4-gb300-dynamo-trt + - dsr1-fp4-gb300-dynamo-sglang + - dsr1-fp8-gb300-dynamo-trt + - dsr1-fp8-h200-dynamo-sglang + - dsr1-fp4-b200-dynamo-sglang + - dsr1-fp8-b200-dynamo-sglang + - dsr1-fp8-b200-dynamo-sglang-mtp + - dsr1-fp4-b200-dynamo-sglang-mtp + - qwen3.5-fp8-gb200-dynamo-sglang + - qwen3.5-fp4-gb300-dynamo-sglang + - qwen3.5-fp4-gb300-dynamo-trt + - qwen3.5-fp4-gb300-dynamo-trt-mtp + - qwen3.5-fp4-gb300-dynamo-trt-agentic-disagg + - qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg + - qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg + - qwen3.5-fp4-gb300-dynamo-sglang-agentic-pp-pareto + - qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg-pareto + - dsv4-fp4-gb300-dynamo-vllm-agentic + - qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp + - minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg + - minimaxm3-fp4-gb200-dynamo-vllm-agentic-agg-mtp + - minimaxm3-fp4-gb200-dynamo-vllm-agentic-disagg-mtp + - minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-agg + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-disagg + - dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-agg + - dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-agg + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-disagg + - kimik3-fp4-gb200-dynamo-vllm-agentic + - kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-dcp16-agg + - kimik3-fp4-gb200-dynamo-vllm-agentic-mooncake-dcp16-agg + - dsv4-fp4-gb300-dynamo-trt-agentx + - kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2 + - dsv4-fp4-gb300-dynamo-sglang-agentic-agg + - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg + - qwen3.5-fp8-gb300-dynamo-sglang + - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-2p2d + - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-1p1d-hicache + - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-agg + - glm5.2-fp4-b200-dynamo-sglang-agentic-agg + - glm5.2-fp4-b200-dynamo-sglang-agentic-disagg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-agg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-disagg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg + - glm5.2-fp4-gb300-dynamo-sglang-agentic-agg + - glm5.2-fp4-gb300-dynamo-sglang-agentic-disagg + - glm5.2-fp4-gb300-dynamo-trt-agentic-disagg-mtp + - qwen3.5-fp8-gb200-dynamo-sglang-mtp + - kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg + - kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-agg + - kimik3-fp4-b200-dynamo-vllm-agentic-dspark + - qwen3.5-fp8-gb300-dynamo-sglang-mtp + - glm5.1-fp8-b200-tilert-agentic + - dsv4-fp4-b200-dynamo-sglang-agentic-agg + - dsv4-fp4-b200-dynamo-sglang-agentic-disagg + - qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp + - qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg + - qwen3.5-fp8-mi355x-sglang-disagg + - qwen3.5-fp4-mi355x-sglang-disagg + - dsr1-fp8-mi355x-sglang-disagg + - dsr1-fp8-mi355x-sglang-disagg-mtp + - dsr1-fp4-mi355x-sglang-disagg + - dsr1-fp4-mi355x-sglang-disagg-8k1k-mtp + - dsr1-fp4-mi355x-sglang-disagg-mtp + description: + - "Resolve required AgentX power from selected native recipes on B200, GB200 and GB300; preserve replay concurrency and strict power audits." + - "Resolve srtctl package versions on local disk for those pools and stop native compute setup on dependency-sync failure; shared runtime scope is retained for qualification." + - "B200、GB200 和 GB300 根据选定原生配方启用必需功耗,保留回放并发和严格功耗审计;在本地磁盘解析包版本,原生计算节点依赖安装失败即停止。验收范围保留共享 runtime 的影响。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3429 diff --git a/runners/launch_b200-nscale-slurm.sh b/runners/launch_b200-nscale-slurm.sh index a041b9f403..adeca41a83 100755 --- a/runners/launch_b200-nscale-slurm.sh +++ b/runners/launch_b200-nscale-slurm.sh @@ -241,56 +241,44 @@ run_native_srt_lane() { # Importing the vLLM image over this cluster's shared home can take a while. SQUASH_LOCK_TIMEOUT=3600 - USES_DCGM_POWER=0 - USES_AGENTX_POWER=0 - _POWER_CONFIG_FILE="${CONFIG_FILE:-}" - if [[ "${EVAL_ONLY}" == "true" && -n "${EVAL_CONFIG_FILE:-}" ]]; then - _POWER_CONFIG_FILE="$EVAL_CONFIG_FILE" - fi - _RECIPE_REL="${_POWER_CONFIG_FILE%%:*}" - _RECIPE_SRC="$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/${_RECIPE_REL#recipes/}" - if [[ -n "$_POWER_CONFIG_FILE" && -f "$_RECIPE_SRC" ]] && awk ' - /^telemetry:/ { t = 1; next } - t && /^[^ ]/ { t = 0 } - t && /^ dcgm_exporter:/ { p = 1 } - t && /^ enabled: true$/ { e = 1 } - END { exit !(p && e) } - ' "$_RECIPE_SRC"; then - USES_DCGM_POWER=1 - fi - if [[ "$USES_DCGM_POWER" == "1" && "$IS_AGENTIC" == "1" && - "$MODEL_PREFIX" == "kimik3" && "$PRECISION" == "fp4" && "$FRAMEWORK" == "dynamo-vllm" ]]; then - USES_AGENTX_POWER=1 - elif [[ "$USES_DCGM_POWER" == "1" && ( - "${IS_AGENTIC}" == "1" || - "$PRECISION" != "fp4" || - ( "$MODEL_PREFIX" == "dsv4" && "$FRAMEWORK" != "dynamo-sglang" && "$FRAMEWORK" != "dynamo-vllm" ) || - "$MODEL_PREFIX" != "dsv4" - ) ]]; then - echo "Error: B200 nscale dcgm-power requires a supported fixed-sequence lane or Kimi-K3 AgentX vLLM" >&2 - exit 1 + if [[ "$EVAL_ONLY" == "true" && -n "${EVAL_CONFIG_FILE:-}" ]]; then + CONFIG_FILE="$EVAL_CONFIG_FILE" fi + check_env_vars CONFIG_FILE export SERVED_MODEL_NAME=$MODEL echo "Preparing job-local srt-slurm checkout..." SRT_REPO_DIR="srt-slurm" rm -rf "$SRT_REPO_DIR" - setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" "$USES_DCGM_POWER" || exit 1 + setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" 0 || exit 1 echo "Installing srtctl..." export UV_INSTALL_DIR="$GITHUB_WORKSPACE/.local/bin" curl -LsSf https://astral.sh/uv/install.sh | sh export PATH="$UV_INSTALL_DIR:$PATH" - uv venv --quiet "$GITHUB_WORKSPACE/.venv" - source "$GITHUB_WORKSPACE/.venv/bin/activate" - uv pip install --quiet -e . + install_srt_slurm "$GITHUB_WORKSPACE/.venv" || exit 1 if ! command -v srtctl &> /dev/null; then echo "Error: Failed to install srtctl" >&2 exit 1 fi + # TileRT uses the retained legacy SRT API without native DCGM telemetry. + USES_DCGM_POWER=0 + USES_AGENTX_POWER=0 + SRTCTL_RECIPE_ARGS=("${SRTCTL_EVAL_ARGS[@]}") + if [[ "$FRAMEWORK" != "tilert" ]]; then + prepare_srt_power "$CONFIG_FILE" "$FRAMEWORK" || exit 1 + fi + if [[ "$USES_DCGM_POWER" == "1" && "$USES_AGENTX_POWER" != "1" && ( + "$PRECISION" != "fp4" || "$MODEL_PREFIX" != "dsv4" || + ( "$FRAMEWORK" != "dynamo-sglang" && "$FRAMEWORK" != "dynamo-vllm" ) + ) ]]; then + echo "Error: unsupported B200 fixed-sequence DCGM lane" >&2 + exit 1 + fi + NGINX_IMAGE="nginx:1.27.4" ensure_writable_squash_dir @@ -352,7 +340,7 @@ run_native_srt_lane() { echo "Generated srtslurm.yaml:" cat srtslurm.yaml - run_srt_setup ARCH=x86_64 + run_srt_setup ARCH=x86_64 || exit 1 # Read by srt-slurm's post-benchmark eval. export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" @@ -384,7 +372,7 @@ run_native_srt_lane() { sed -i 's/^ max_attempts: [0-9]*/ max_attempts: 720/' "$CONFIG_PATH" fi - if [[ "$USES_DCGM_POWER" == "1" ]]; then + if [[ "$USES_DCGM_POWER" == "1" && "$USES_AGENTX_POWER" != "1" ]]; then read -r -a POWER_CONCURRENCIES <<< "$CONC_LIST" python "$GITHUB_WORKSPACE/runners/inject_srt_power_concurrencies.py" \ "$CONFIG_PATH" "${POWER_CONCURRENCIES[@]}" || exit 1 @@ -398,7 +386,7 @@ run_native_srt_lane() { SRTCTL_PREFLIGHT_ARGS+=(--no-preflight) fi - SRTCTL_OUTPUT=$(apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_EVAL_ARGS[@]}" -f "$CONFIG_FILE" "${SRTCTL_PREFLIGHT_ARGS[@]}" --tags "b200,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" 2>&1) + SRTCTL_OUTPUT=$(apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_RECIPE_ARGS[@]}" -f "$CONFIG_FILE" "${SRTCTL_PREFLIGHT_ARGS[@]}" --tags "b200,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" 2>&1) echo "$SRTCTL_OUTPUT" JOB_ID=$(echo "$SRTCTL_OUTPUT" | grep -oP '✅ Job \K[0-9]+' || echo "$SRTCTL_OUTPUT" | grep -oP 'Job \K[0-9]+') @@ -537,35 +525,10 @@ run_multinode_srt() { exit 1 fi - USES_DCGM_POWER=0 - USES_AGENTX_POWER=0 - _POWER_CONFIG_FILE="${CONFIG_FILE:-}" - if [[ "${EVAL_ONLY}" == "true" && -n "${EVAL_CONFIG_FILE:-}" ]]; then - _POWER_CONFIG_FILE="$EVAL_CONFIG_FILE" - fi - _RECIPE_REL="${_POWER_CONFIG_FILE%%:*}" - _RECIPE_SRC="$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/${_RECIPE_REL#recipes/}" - if [[ -n "$_POWER_CONFIG_FILE" && -f "$_RECIPE_SRC" ]] && awk ' - /^telemetry:/ { t = 1; next } - t && /^[^ ]/ { t = 0 } - t && /^ dcgm_exporter:/ { p = 1 } - t && /^ enabled: true$/ { e = 1 } - END { exit !(p && e) } - ' "$_RECIPE_SRC"; then - USES_DCGM_POWER=1 - fi - if [[ "$USES_DCGM_POWER" == "1" && "$IS_AGENTIC" == "1" && - "$MODEL_PREFIX" == "qwen3.5" && "$PRECISION" == "fp8" && "$FRAMEWORK" == "dynamo-sglang" ]]; then - USES_AGENTX_POWER=1 - elif [[ "$USES_DCGM_POWER" == "1" && ( - "${IS_AGENTIC}" == "1" || - "$MODEL_PREFIX" != "dsv4" || - "$PRECISION" != "fp4" || - "$FRAMEWORK" != "dynamo-vllm" - ) ]]; then - echo "Error: B200 Nscale dcgm-power requires fixed-sequence DSV4 FP4 dynamo-vllm or Qwen3.5 FP8 AgentX dynamo-sglang" >&2 - exit 1 + if [[ "$EVAL_ONLY" == "true" && -n "${EVAL_CONFIG_FILE:-}" ]]; then + CONFIG_FILE="$EVAL_CONFIG_FILE" fi + check_env_vars CONFIG_FILE export SERVED_MODEL_NAME=$MODEL @@ -576,22 +539,28 @@ run_multinode_srt() { rm -rf "$SRT_REPO_DIR" fi - setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" "$USES_DCGM_POWER" || exit 1 + setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" 0 || exit 1 echo "Installing srtctl..." export UV_INSTALL_DIR="$GITHUB_WORKSPACE/.local/bin" curl -LsSf https://astral.sh/uv/install.sh | sh export PATH="$UV_INSTALL_DIR:$PATH" - uv venv --quiet "$GITHUB_WORKSPACE/.venv" - source "$GITHUB_WORKSPACE/.venv/bin/activate" - uv pip install --quiet -e . + install_srt_slurm "$GITHUB_WORKSPACE/.venv" || exit 1 if ! command -v srtctl &> /dev/null; then echo "Error: Failed to install srtctl" exit 1 fi + prepare_srt_power "$CONFIG_FILE" "$FRAMEWORK" || exit 1 + if [[ "$USES_DCGM_POWER" == "1" && "$USES_AGENTX_POWER" != "1" && ( + "$MODEL_PREFIX" != "dsv4" || "$PRECISION" != "fp4" || "$FRAMEWORK" != "dynamo-vllm" + ) ]]; then + echo "Error: unsupported B200 fixed-sequence DCGM lane" >&2 + exit 1 + fi + NGINX_IMAGE="nginx:1.27.4" # Set by runners/runtime_settings.sh. check_env_vars B200_SQUASH_DIR B200_SQUASH_LOCK_TIMEOUT @@ -642,7 +611,7 @@ run_multinode_srt() { echo "Generated srtslurm.yaml:" cat srtslurm.yaml - run_srt_setup ARCH=x86_64 + run_srt_setup ARCH=x86_64 || exit 1 # Read by srt-slurm's post-benchmark eval. export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" @@ -676,7 +645,7 @@ run_multinode_srt() { SRTCTL_PREFLIGHT_ARGS+=(--no-preflight) fi - SRTCTL_OUTPUT=$(apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_EVAL_ARGS[@]}" -f "$CONFIG_FILE" "${SRTCTL_PREFLIGHT_ARGS[@]}" --tags "b200,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" 2>&1) + SRTCTL_OUTPUT=$(apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_RECIPE_ARGS[@]}" -f "$CONFIG_FILE" "${SRTCTL_PREFLIGHT_ARGS[@]}" --tags "b200,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" 2>&1) echo "$SRTCTL_OUTPUT" JOB_ID=$(echo "$SRTCTL_OUTPUT" | grep -oP '✅ Job \K[0-9]+' || echo "$SRTCTL_OUTPUT" | grep -oP 'Job \K[0-9]+') diff --git a/runners/launch_gb200-nv.sh b/runners/launch_gb200-nv.sh index f26da84aff..5149f37f4f 100755 --- a/runners/launch_gb200-nv.sh +++ b/runners/launch_gb200-nv.sh @@ -273,50 +273,6 @@ NGINX_SQUASH_FILE="${SQUASH_DIR}/$(echo "$NGINX_IMAGE" | sed 's/[\/:@#]/_/g').sq import_squash "$SQUASH_FILE" "$IMAGE" import_squash "$NGINX_SQUASH_FILE" "$NGINX_IMAGE" -# The power lane is on iff the resolved recipe carries an enabled dcgm-power -# telemetry block. Read the workspace mirror; it overlays the srt-slurm clone later. -USES_DCGM_POWER=0 -_RECIPE_REL="${CONFIG_FILE%%:*}" -_RECIPE_SRC="$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/${_RECIPE_REL#recipes/}" -# Scoped match: a stray "enabled: true" outside the telemetry block must not flip the lane. -if [[ -n "$CONFIG_FILE" && -f "$_RECIPE_SRC" ]] && awk ' - /^telemetry:/ { t = 1; next } - t && /^[^ ]/ { t = 0 } - t && /^ dcgm_exporter:/ { p = 1 } - t && /^ enabled: true$/ { e = 1 } - END { exit !(p && e) } -' "$_RECIPE_SRC"; then - USES_DCGM_POWER=1 -fi - -USES_AGENTX_POWER=0 -if [[ "$USES_DCGM_POWER" == "1" && "$IS_AGENTIC" == "1" ]]; then - if [[ "$MODEL_PREFIX" == "glm5.2" && "$PRECISION" == "fp4" && - "$FRAMEWORK" == "dynamo-sglang" && - "$_RECIPE_REL" == "recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml" ]]; then - USES_AGENTX_POWER=1 - elif [[ "$MODEL_PREFIX" == "kimik3" && "$PRECISION" == "fp4" && - "$FRAMEWORK" == "dynamo-vllm" && - "$_RECIPE_REL" == recipes/kimik3/vllm/gb200-fp4/agentx/* ]]; then - USES_AGENTX_POWER=1 - else - echo "Error: AgentX dcgm-power requires the GLM-5.2 aggregate or supported Kimi-K3 recipe" >&2 - exit 1 - fi -fi -if [[ "$USES_DCGM_POWER" == "1" && "$FRAMEWORK" != "dynamo-sglang" && "$USES_AGENTX_POWER" != "1" ]]; then - echo "Error: dcgm-power requires dynamo-sglang or the supported Kimi-K3 AgentX route" >&2 - exit 1 -fi - -if [[ "$USES_DCGM_POWER" == "1" ]]; then - DCGM_EXPORTER_IMAGE="nvcr.io/nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless" - DCGM_EXPORTER_SQSH="${SQUASH_DIR}/$(echo "$DCGM_EXPORTER_IMAGE" | sed 's/[\/:@#]/_/g').sqsh" - import_squash "$DCGM_EXPORTER_SQSH" "$DCGM_EXPORTER_IMAGE" - test -r "$DCGM_EXPORTER_SQSH" || { echo "Error: DCGM exporter squash not readable: $DCGM_EXPORTER_SQSH" >&2; exit 1; } - unsquashfs -l "$DCGM_EXPORTER_SQSH" > /dev/null || { echo "Error: DCGM exporter squash invalid: $DCGM_EXPORTER_SQSH" >&2; exit 1; } - sha256sum "$DCGM_EXPORTER_SQSH" > "$GITHUB_WORKSPACE/exporter-image.sha256" -fi export ISL="$ISL" @@ -388,32 +344,38 @@ if [ -d "$SRT_REPO_DIR" ]; then rm -rf "$SRT_REPO_DIR" fi -# This checkpoint is staged on compute-node NVMe for the power lane. -if [[ "$USES_DCGM_POWER" == "1" && "$FRAMEWORK" == "dynamo-sglang" && "$MODEL_PREFIX" == "dsv4" && "$IS_AGENTIC" != "1" ]]; then - export MODEL_PATH="/mnt/numa1/models/DeepSeek-V4-Pro" -fi -setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" "$USES_DCGM_POWER" || exit 1 +setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" 0 || exit 1 echo "Installing srtctl..." curl -LsSf https://astral.sh/uv/install.sh | sh source $HOME/.local/bin/env -# On watchtower compute nodes inherit the activated .venv through the shared-FS -# SRT_REPO_DIR; a uv-managed python under a head-node-only path leaves -# .venv/bin/python3 a broken symlink there, so pin /usr/bin/python3. -if uses_watchtower_shared_fs && [[ -x /usr/bin/python3 ]]; then - uv venv --quiet --seed --python /usr/bin/python3 -else - uv venv --quiet --seed -fi -source .venv/bin/activate -uv pip install --quiet -e . +install_srt_slurm .venv || exit 1 if ! command -v srtctl &> /dev/null; then echo "Error: Failed to install srtctl" exit 1 fi +prepare_srt_power "$CONFIG_FILE" "$FRAMEWORK" || exit 1 +if [[ "$USES_DCGM_POWER" == "1" && "$USES_AGENTX_POWER" != "1" && "$FRAMEWORK" != "dynamo-sglang" ]]; then + echo "Error: non-AgentX dcgm-power requires dynamo-sglang" >&2 + exit 1 +fi + +if [[ "$USES_DCGM_POWER" == "1" && "$FRAMEWORK" == "dynamo-sglang" && "$MODEL_PREFIX" == "dsv4" && "$IS_AGENTIC" != "1" ]]; then + export MODEL_PATH="/mnt/numa1/models/DeepSeek-V4-Pro" +fi +if [[ "$USES_DCGM_POWER" == "1" ]]; then + DCGM_EXPORTER_IMAGE="nvcr.io/nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless" + DCGM_EXPORTER_SQSH="${SQUASH_DIR}/$(echo "$DCGM_EXPORTER_IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + import_squash "$DCGM_EXPORTER_SQSH" "$DCGM_EXPORTER_IMAGE" + test -r "$DCGM_EXPORTER_SQSH" || { echo "Error: DCGM exporter squash not readable: $DCGM_EXPORTER_SQSH" >&2; exit 1; } + unsquashfs -l "$DCGM_EXPORTER_SQSH" > /dev/null || { echo "Error: DCGM exporter squash invalid: $DCGM_EXPORTER_SQSH" >&2; exit 1; } + sha256sum "$DCGM_EXPORTER_SQSH" > "$GITHUB_WORKSPACE/exporter-image.sha256" +fi + + echo "Configs available at: $SRT_REPO_DIR/" SRTCTL_ROOT="${GITHUB_WORKSPACE}/srt-slurm" @@ -495,7 +457,7 @@ if command -v squeue >/dev/null 2>&1; then fi sed -i "s/^name:.*/name: \"${SRT_SLURM_JOB_NAME}\"/" "$CONFIG_PATH" -if [[ "$USES_DCGM_POWER" == "1" ]]; then +if [[ "$USES_DCGM_POWER" == "1" && "$USES_AGENTX_POWER" != "1" ]]; then read -r -a POWER_CONCURRENCIES <<< "$CONC_LIST" python3 "$GITHUB_WORKSPACE/runners/inject_srt_power_concurrencies.py" \ "$CONFIG_PATH" "${POWER_CONCURRENCIES[@]}" || exit 1 @@ -524,7 +486,7 @@ if [[ "$FRAMEWORK" == "dynamo-sglang" ]]; then fi # srtctl gives RUNNER_NAME precedence over config.name; override it for the # submission so the #SBATCH job name keeps the namespace used above. -SRTCTL_OUTPUT=$(RUNNER_NAME="$SRT_SLURM_JOB_NAME" apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_EVAL_ARGS[@]}" "${SRTCTL_APPLY_ARGS[@]}" 2>&1) +SRTCTL_OUTPUT=$(RUNNER_NAME="$SRT_SLURM_JOB_NAME" apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_RECIPE_ARGS[@]}" "${SRTCTL_APPLY_ARGS[@]}" 2>&1) echo "$SRTCTL_OUTPUT" JOB_ID=$(echo "$SRTCTL_OUTPUT" | grep -oP '✅ Job \K[0-9]+' || echo "$SRTCTL_OUTPUT" | grep -oP 'Job \K[0-9]+') diff --git a/runners/launch_gb300-nv.sh b/runners/launch_gb300-nv.sh index 2e5d4069dc..ecb6427804 100644 --- a/runners/launch_gb300-nv.sh +++ b/runners/launch_gb300-nv.sh @@ -141,49 +141,6 @@ fi import_squash "$NGINX_SQUASH_FILE" "$NGINX_IMAGE" -# A recipe opts into the power lane via an enabled dcgm-power telemetry block. -# The srt-slurm checkout does not exist yet, so read the workspace mirror; -# recipes that exist only upstream stay non-power. -USES_DCGM_POWER=0 -_RECIPE_REL="${CONFIG_FILE%%:*}" -_RECIPE_SRC="$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/${_RECIPE_REL#recipes/}" -# Scoped match: a stray "enabled: true" outside the telemetry block must not flip the lane. -if [[ -n "$CONFIG_FILE" && -f "$_RECIPE_SRC" ]] && awk ' - /^telemetry:/ { t = 1; next } - t && /^[^ ]/ { t = 0 } - t && /^ dcgm_exporter:/ { p = 1 } - t && /^ enabled: true$/ { e = 1 } - END { exit !(p && e) } -' "$_RECIPE_SRC"; then - USES_DCGM_POWER=1 -fi - -USES_AGENTX_POWER=0 -if [[ "$USES_DCGM_POWER" == "1" && "$IS_AGENTIC" == "1" && - "$MODEL_PREFIX" == "kimik3" && "$PRECISION" == "fp4" && - "$FRAMEWORK" == "dynamo-vllm" && - "$_RECIPE_REL" == recipes/kimik3/vllm/*/agentx/* ]]; then - USES_AGENTX_POWER=1 -fi -if [[ "$USES_DCGM_POWER" == "1" && "$FRAMEWORK" != "dynamo-sglang" && "$USES_AGENTX_POWER" != "1" ]]; then - echo "Error: dcgm-power requires dynamo-sglang or the supported Kimi-K3 AgentX route" >&2 - exit 1 -fi - - -if [[ "$USES_DCGM_POWER" == "1" ]]; then - DCGM_EXPORTER_IMAGE="nvcr.io/nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless" - # enroot resolves bare paths against Docker Hub; nvcr.io pulls need the registry# form - DCGM_EXPORTER_ENROOT_REF="${DCGM_EXPORTER_IMAGE/nvcr.io\//nvcr.io#}" - DCGM_EXPORTER_SQSH="/data/home/sa-shared/gharunners/squash/$(echo "$DCGM_EXPORTER_IMAGE" | sed 's/[\/:@#]/_/g').sqsh" - # import_squash does not re-validate a fresh import, so check explicitly on - # a compute node (login node is x86, nodes aarch64). - import_squash "$DCGM_EXPORTER_SQSH" "$DCGM_EXPORTER_ENROOT_REF" - test -r "$DCGM_EXPORTER_SQSH" || { echo "Error: DCGM exporter squash not readable: $DCGM_EXPORTER_SQSH" >&2; exit 1; } - srun --account="$SLURM_ACCOUNT" --partition="$SLURM_PARTITION" --exclusive --time=30 bash -c "unsquashfs -l \"$DCGM_EXPORTER_SQSH\" > /dev/null" || { echo "Error: DCGM exporter squash invalid: $DCGM_EXPORTER_SQSH" >&2; exit 1; } - sha256sum "$DCGM_EXPORTER_SQSH" > "$GITHUB_WORKSPACE/exporter-image.sha256" -fi - if [[ "$EVAL_ONLY" == "true" && -n "${EVAL_CONFIG_FILE:-}" ]]; then CONFIG_FILE="$EVAL_CONFIG_FILE" echo "EVAL_ONLY=true: selecting real-verification recipe $CONFIG_FILE" @@ -199,7 +156,7 @@ check_env_vars GITHUB_RUN_ID GITHUB_RUN_ATTEMPT SRT_REPO_DIR="${GITHUB_WORKSPACE}/srt-slurm-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}-${RUN_KEY}" rm -rf "$SRT_REPO_DIR" -setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" "$USES_DCGM_POWER" || exit 1 +setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" 0 || exit 1 if [[ "$FRAMEWORK" == "dynamo-trt" && "$MODEL_PREFIX" == "dsv4" ]]; then SRT_SLURM_MODEL_PREFIX="deepseek-ai/DeepSeek-V4-Pro" @@ -221,17 +178,32 @@ export PATH="$UV_INSTALL_DIR:$PATH" check_env_vars GITHUB_RUN_ID GITHUB_RUN_ATTEMPT VENV_DIR="${GITHUB_WORKSPACE}/.venv-srt-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}-${RUN_KEY}" rm -rf "$VENV_DIR" -# --seed installs pip; srtctl's prefetch-ai-dynamo-wheel.sh (recipes with -# dynamo.wheel) otherwise fails with "No module named pip". -uv venv --quiet --seed "$VENV_DIR" -source "$VENV_DIR/bin/activate" -uv pip install --quiet -e . +install_srt_slurm "$VENV_DIR" || exit 1 if ! command -v srtctl &> /dev/null; then echo "Error: Failed to install srtctl" exit 1 fi +prepare_srt_power "$CONFIG_FILE" "$FRAMEWORK" || exit 1 +if [[ "$USES_DCGM_POWER" == "1" && "$USES_AGENTX_POWER" != "1" && "$FRAMEWORK" != "dynamo-sglang" ]]; then + echo "Error: non-AgentX dcgm-power requires dynamo-sglang" >&2 + exit 1 +fi + +if [[ "$USES_DCGM_POWER" == "1" ]]; then + DCGM_EXPORTER_IMAGE="nvcr.io/nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless" + # enroot resolves bare paths against Docker Hub; nvcr.io pulls need the registry# form + DCGM_EXPORTER_ENROOT_REF="${DCGM_EXPORTER_IMAGE/nvcr.io\//nvcr.io#}" + DCGM_EXPORTER_SQSH="/data/home/sa-shared/gharunners/squash/$(echo "$DCGM_EXPORTER_IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + # import_squash does not re-validate a fresh import, so check explicitly on + # a compute node (login node is x86, nodes aarch64). + import_squash "$DCGM_EXPORTER_SQSH" "$DCGM_EXPORTER_ENROOT_REF" + test -r "$DCGM_EXPORTER_SQSH" || { echo "Error: DCGM exporter squash not readable: $DCGM_EXPORTER_SQSH" >&2; exit 1; } + srun --account="$SLURM_ACCOUNT" --partition="$SLURM_PARTITION" --exclusive --time=30 bash -c "unsquashfs -l \"$DCGM_EXPORTER_SQSH\" > /dev/null" || { echo "Error: DCGM exporter squash invalid: $DCGM_EXPORTER_SQSH" >&2; exit 1; } + sha256sum "$DCGM_EXPORTER_SQSH" > "$GITHUB_WORKSPACE/exporter-image.sha256" +fi + echo "Configs available at: $SRT_REPO_DIR/" SRTCTL_ROOT="${SRT_REPO_DIR}" @@ -250,7 +222,7 @@ write_srt_cluster_config gb300-nv srtslurm.yaml "$USES_DCGM_POWER" \ echo "Generated srtslurm.yaml:" cat srtslurm.yaml -run_srt_setup ARCH=aarch64 +run_srt_setup ARCH=aarch64 || exit 1 # Read by srt-slurm's post-benchmark eval. export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" @@ -271,7 +243,7 @@ sed -i "s/^name:.*/name: \"${RUNNER_NAME}\"/" "$CONFIG_PATH" # Throughput recipes opt into synthetic acceptance via the master config; # eval-only jobs strip it so tokens get real target-model verification. -if [[ "$USES_DCGM_POWER" == "1" ]]; then +if [[ "$USES_DCGM_POWER" == "1" && "$USES_AGENTX_POWER" != "1" ]]; then read -r -a POWER_CONCURRENCIES <<< "$CONC_LIST" python3 "$GITHUB_WORKSPACE/runners/inject_srt_power_concurrencies.py" \ "$CONFIG_PATH" "${POWER_CONCURRENCIES[@]}" || exit 1 @@ -287,7 +259,7 @@ if [[ "$IS_AGENTIC" == "1" || ( "$MODEL_PREFIX" == "qwen3.5" && "$PRECISION" == SRTCTL_APPLY_ARGS+=(--no-preflight) fi -SRTCTL_OUTPUT=$(apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_EVAL_ARGS[@]}" "${SRTCTL_APPLY_ARGS[@]}" 2>&1) +SRTCTL_OUTPUT=$(apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_RECIPE_ARGS[@]}" "${SRTCTL_APPLY_ARGS[@]}" 2>&1) echo "$SRTCTL_OUTPUT" JOB_ID=$(echo "$SRTCTL_OUTPUT" | grep -oP '✅ Job \K[0-9]+' || echo "$SRTCTL_OUTPUT" | grep -oP 'Job \K[0-9]+') diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 68aedf6a25..87c76e5b6a 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -94,6 +94,43 @@ PYENV cp -R "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/configs/." configs/ || return 1 } +# Resolve Hatch's actual version on local disk: git describe --dirty otherwise +# scans the shared filesystem during every editable build, including on compute. +# Call before venv creation to avoid copying build caches. +srt_slurm_version() ( + local version_root + version_root=$(mktemp -d /tmp/infx-srt-version.XXXXXX) || exit 1 + trap 'rm -rf "$version_root"' EXIT + cp -R . "$version_root/source" || exit 1 + cd "$version_root/source" || exit 1 + env -u SETUPTOOLS_SCM_PRETEND_VERSION -u SETUPTOOLS_SCM_PRETEND_VERSION_FOR_SRTCTL \ + -u VCS_VERSIONING_PRETEND_VERSION -u VCS_VERSIONING_PRETEND_VERSION_FOR_SRTCTL \ + uv tool run --python 3.12 --from hatchling --with hatch-vcs hatchling version +) + +install_srt_slurm() { + if [[ $# -ne 1 || -z "$1" ]]; then + echo "Usage: install_srt_slurm venv_path" >&2 + return 1 + fi + check_env_vars GITHUB_WORKSPACE + local version + version=$(srt_slurm_version) || return 1 + if [[ -z "$version" || "$version" == *$'\n'* ]]; then + echo "Error: srt-slurm did not produce one package version" >&2 + return 1 + fi + printf '%s\n' "$version" > .infx-srt-version || return 1 + cp .infx-srt-version "$GITHUB_WORKSPACE/srt-slurm-version.txt" || return 1 + export SETUPTOOLS_SCM_PRETEND_VERSION_FOR_SRTCTL="$version" + printf 'srtctl package version: %s\n' "$version" >> "$GITHUB_WORKSPACE/srt-setup.log" || return 1 + # Login and compute use separate venvs. InferenceX's power adapter needs 3.12; + # compute creates its own architecture-compatible Python through native uv. + uv venv --quiet --seed --python 3.12 "$1" || return 1 + source "$1/bin/activate" || return 1 + uv pip install --quiet -e . +} + # Keep installer output in the artifacts, but print diagnostics on failure. run_srt_setup() { check_env_vars GITHUB_WORKSPACE @@ -150,6 +187,54 @@ apply_srt_recipe() { "$config" "$framework" -- "$@" } +prepare_srt_power() { + check_env_vars IS_AGENTIC EVAL_ONLY + # Keep inspection and submission on the same native overrides. The benchmark + # client and the collector must expect the same matrix concurrency windows. + local config="$1" framework="$2" concurrency power_mode concurrency_json + SRTCTL_RECIPE_ARGS=("${SRTCTL_EVAL_ARGS[@]}") + if [[ "$IS_AGENTIC" == "1" ]]; then + check_env_vars CONC_LIST + local -a power_concurrencies + read -r -a power_concurrencies <<< "$CONC_LIST" + for concurrency in "${power_concurrencies[@]}"; do + [[ "$concurrency" =~ ^[1-9][0-9]*$ ]] || { + echo "Error: invalid AgentX concurrency: $concurrency" >&2 + return 1 + } + done + concurrency_json=$(IFS=,; echo "[${power_concurrencies[*]}]") + SRTCTL_RECIPE_ARGS+=( + --set "benchmark.concurrencies=$concurrency_json" + --set "benchmark.env.CONC_LIST=\"${power_concurrencies[*]}\"" + ) + fi + power_mode=$(PYTHONPATH="$INFERENCEX_SLURM_UTILS_DIR/..${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.srt_slurm.synthetic_acceptance --inspect-power \ + "$config" "$framework" -- "${SRTCTL_RECIPE_ARGS[@]}") || return 1 + USES_DCGM_POWER=0 + USES_AGENTX_POWER=0 + case "$power_mode" in + agentx) + USES_DCGM_POWER=1 + USES_AGENTX_POWER=1 + SRTCTL_RECIPE_ARGS+=( + --set 'benchmark.env.ENABLE_AGENTX_POWER="1"' + --set 'benchmark.env.REQUIRE_POWER="1"' + ) + ;; + dcgm) + USES_DCGM_POWER=1 + ;; + none) ;; + *) echo "Error: unknown recipe power mode: $power_mode" >&2; return 1 ;; + esac + if [[ "$USES_DCGM_POWER" == "1" ]]; then + check_env_vars GITHUB_WORKSPACE + cp "$GITHUB_WORKSPACE/srt-slurm-sha.txt" "$GITHUB_WORKSPACE/power-producer-sha.txt" || return 1 + fi +} + # One native submission per fixed-sequence matrix point, shared across Slurm pools. launch_srt_single_node() { set -eo pipefail diff --git a/runners/srt-slurm/patches/README.md b/runners/srt-slurm/patches/README.md index cb619d0506..2f2b08cd93 100644 --- a/runners/srt-slurm/patches/README.md +++ b/runners/srt-slurm/patches/README.md @@ -2,8 +2,9 @@ `setup_srt_slurm()` in [`runners/slurm_utils.sh`](../../slurm_utils.sh) applies every `*.patch` here to the job's srt-slurm clone after checking out the pinned submodule. TileRT jobs use the fork checkout and skip these patches. -Each patch is a temporary fix for an open upstream PR. When the PR merges and the submodule pin includes it, delete the patch and its row. +Patches are temporary fixes pending upstream integration. When the corresponding change merges and the submodule pin includes it, delete the patch and its row. A local candidate without an upstream PR is identified explicitly below. | Patch | Upstream PR | Fix | |-------|-------------|-----| | `504-post-eval-srun-options.patch` | [NVIDIA/srt-slurm#504](https://github.com/NVIDIA/srt-slurm/pull/504) | Forward recipe `srun_options` (e.g. `container-writable`) to post-eval steps | +| `local-version-compute-setup.patch` | Local candidate; upstream PR pending | Give Hatch the distribution name for its scoped version override, reuse the locally derived package version on compute nodes, and stop if compute environment synchronization fails | diff --git a/runners/srt-slurm/patches/local-version-compute-setup.patch b/runners/srt-slurm/patches/local-version-compute-setup.patch new file mode 100644 index 0000000000..4da5d49301 --- /dev/null +++ b/runners/srt-slurm/patches/local-version-compute-setup.patch @@ -0,0 +1,40 @@ +--- a/pyproject.toml ++++ b/pyproject.toml +@@ -55,6 +55,8 @@ + source = "vcs" + + [tool.hatch.version.raw-options] ++# Honor the distribution-scoped version preserved by the submitting launcher. ++dist_name = "srtctl" + # 2.0.0 at the tag, 2.0.0.post3+g three commits later (not a guessed 2.0.1.dev3). + version_scheme = "post-release" + fallback_version = "0.0.0+unknown" +--- a/src/srtctl/templates/job_script_minimal.j2 ++++ b/src/srtctl/templates/job_script_minimal.j2 +@@ -139,6 +139,17 @@ + exit 1 + fi + ++# InferenceX resolves the actual Hatch VCS version on local disk before staging ++# this job. Reuse it instead of scanning Lustre for git describe --dirty. ++if [[ -f "${SRTCTL_SOURCE}/.infx-srt-version" ]]; then ++ read -r SETUPTOOLS_SCM_PRETEND_VERSION_FOR_SRTCTL < "${SRTCTL_SOURCE}/.infx-srt-version" ++ if [[ -z "$SETUPTOOLS_SCM_PRETEND_VERSION_FOR_SRTCTL" ]]; then ++ echo "ERROR: staged srtctl package version is empty" ++ exit 1 ++ fi ++ export SETUPTOOLS_SCM_PRETEND_VERSION_FOR_SRTCTL ++fi ++ + echo "Using uv $(uv --version)..." + + if [ -n "${DYNAMO_WHEEL_NAME:-}" ] || [ "${SRTCTL_PREFETCH_AI_DYNAMO:-0}" = "1" ]; then +@@ -170,7 +181,7 @@ + # "Directory not empty" / "Failed to spawn 'python -m'" / missing CACHEDIR.TAG + # errors. The flock serializes only the venv init; the lock is released + # before the actual sweep so jobs still run in parallel. +-flock -x "${SRTCTL_SOURCE}/.venv-compute.lock" uv sync --python 3.12 --no-dev ++flock -x "${SRTCTL_SOURCE}/.venv-compute.lock" uv sync --python 3.12 --no-dev || exit $? + uv run --python 3.12 --no-dev --no-sync -m srtctl.cli.do_sweep "${OUTPUT_DIR}/{{ runtime_config_filename }}"{% if serve_only %} --serve-only{% endif %} + EXIT_CODE=$? + set -e # Re-enable exit-on-error diff --git a/utils/test_srt_install.py b/utils/test_srt_install.py new file mode 100644 index 0000000000..a684c82448 --- /dev/null +++ b/utils/test_srt_install.py @@ -0,0 +1,155 @@ +"""Run the login installer and rendered compute setup without GPUs or Slurm.""" + +import os +import shutil +import subprocess +from pathlib import Path + +import pytest +from jinja2 import Environment, FileSystemLoader + +ROOT = Path(__file__).resolve().parents[1] + + +def executable(path, text): + path.write_text("#!/bin/bash\n" + text) + path.chmod(0o755) + + +def test_installer_resolves_version_off_shared_disk_and_preserves_edits(tmp_path): + checkout = tmp_path / "shared-checkout" + checkout.mkdir() + (checkout / "tracked").write_text("original\n") + subprocess.run(["git", "init", "-q", str(checkout)], check=True) + subprocess.run(["git", "add", "tracked"], cwd=checkout, check=True) + subprocess.run( + ["git", "-c", "user.name=Test", "-c", "user.email=test@example.com", "commit", "-qm", "fixture"], + cwd=checkout, check=True, + ) + subprocess.run(["git", "tag", "v2.7.0"], cwd=checkout, check=True) + (checkout / "tracked").write_text("patched\n") + binaries = tmp_path / "bin" + binaries.mkdir() + executable(binaries / "git", ''' +if [[ "$PWD" == "$SHARED_CHECKOUT" && "$1" == describe ]]; then + echo 'git describe timed out on shared storage' >&2 + exit 124 +fi +exec "$REAL_GIT" "$@" +''') + executable(binaries / "uv", ''' +case "$1 $2" in + 'tool run') + git describe --dirty --tags --long > "$GITHUB_WORKSPACE/describe" + rc=$? + (( rc == 0 )) || exit "$rc" + [[ "$(cat tracked)" == patched ]] || exit 40 + printf '%s' "$PWD" > "$GITHUB_WORKSPACE/version-source" + echo '2.7.0.post0+dirty' + ;; + 'venv --quiet') + mkdir -p "${@: -1}/bin" + printf '%s\n' '# activated fixture' > "${@: -1}/bin/activate" + ;; + 'pip install') + printf '%s\n' "$SETUPTOOLS_SCM_PRETEND_VERSION_FOR_SRTCTL" > "$GITHUB_WORKSPACE/installed-version" + ;; + *) exit 41 ;; +esac +''') + env = { + **os.environ, "PATH": f"{binaries}:{os.environ['PATH']}", + "GITHUB_WORKSPACE": str(tmp_path), "SHARED_CHECKOUT": str(checkout), + "REAL_GIT": shutil.which("git"), + } + result = subprocess.run( + ["bash", "-c", 'source "$1"; install_srt_slurm "$2"', "bash", + str(ROOT / "runners/slurm_utils.sh"), str(tmp_path / "venv")], + cwd=checkout, env=env, capture_output=True, text=True, + ) + assert result.returncode == 0, result.stderr + assert (tmp_path / "installed-version").read_text() == "2.7.0.post0+dirty\n" + assert (checkout / ".infx-srt-version").read_text() == "2.7.0.post0+dirty\n" + assert (tmp_path / "srt-slurm-version.txt").read_text() == "2.7.0.post0+dirty\n" + assert (tmp_path / "describe").read_text().rstrip().endswith("-dirty") + assert not Path((tmp_path / "version-source").read_text()).exists() + assert (checkout / "tracked").read_text() == "patched\n" + + +@pytest.mark.parametrize("failure", ["version", "empty-version", "venv", "install"]) +def test_install_failure_stops_before_submission(tmp_path, failure): + binaries = tmp_path / "bin" + binaries.mkdir() + executable(binaries / "uv", ''' +case "$1 $2" in + 'tool run') + [[ "$FAILURE" != version ]] || exit 42 + [[ "$FAILURE" != empty-version ]] || exit 0 + echo 2.7.0 + ;; + 'venv --quiet') + [[ "$FAILURE" != venv ]] || exit 43 + mkdir -p "${@: -1}/bin" + echo '# fixture' > "${@: -1}/bin/activate" + ;; + 'pip install') exit 44 ;; + *) exit 45 ;; +esac +''') + checkout = tmp_path / "checkout" + checkout.mkdir() + result = subprocess.run( + ["bash", "-c", 'source "$1"; install_srt_slurm "$2" || exit $?; touch "$3"', "bash", + str(ROOT / "runners/slurm_utils.sh"), str(tmp_path / "venv"), str(tmp_path / "submitted")], + cwd=checkout, env={**os.environ, "PATH": f"{binaries}:{os.environ['PATH']}", + "GITHUB_WORKSPACE": str(tmp_path), "FAILURE": failure}, + capture_output=True, text=True, + ) + assert result.returncode != 0 + assert not (tmp_path / "submitted").exists() + assert (checkout / ".infx-srt-version").exists() == (failure in {"venv", "install"}) + + +def compute_script(tmp_path): + """Apply the shipped patch and render the actual upstream job template.""" + template_root = tmp_path / "src/srtctl/templates" + template_root.mkdir(parents=True) + shutil.copy(ROOT / "utils/srt-slurm/src/srtctl/templates/job_script_minimal.j2", template_root) + shutil.copy(ROOT / "utils/srt-slurm/pyproject.toml", tmp_path) + subprocess.run( + ["git", "apply", str(ROOT / "runners/srt-slurm/patches/local-version-compute-setup.patch")], + cwd=tmp_path, check=True, capture_output=True, + ) + rendered = Environment(loader=FileSystemLoader(template_root)).get_template("job_script_minimal.j2").render( + srtctl_source=str(tmp_path), output_base=str(tmp_path / "outputs"), + config_environment={}, sbatch_directives={}, runtime_config_filename="runtime.yaml", + ) + script = tmp_path / "job.sh" + script.write_text(rendered) + return script + + +@pytest.mark.parametrize("sync_status", [0, 38]) +def test_compute_uses_staged_version_and_propagates_sync_failure(tmp_path, sync_status): + script = compute_script(tmp_path) + (tmp_path / ".infx-srt-version").write_text("2.7.0.post1+g123abcd\n") + binaries = tmp_path / "bin" + binaries.mkdir() + executable(binaries / "flock", 'shift 2\nexec "$@"\n') + executable(binaries / "uv", ''' +case "$1" in + --version) echo fixture ;; + sync) + echo "$SETUPTOOLS_SCM_PRETEND_VERSION_FOR_SRTCTL" > "$SRTCTL_SOURCE_DIR/sync-version" + exit "$SYNC_STATUS" + ;; + run) touch "$SRTCTL_SOURCE_DIR/ran-orchestrator" ;; + *) exit 46 ;; +esac +''') + env = {**os.environ, "SLURM_JOB_ID": "42", "SYNC_STATUS": str(sync_status)} + env.pop("SETUPTOOLS_SCM_PRETEND_VERSION_FOR_SRTCTL", None) + result = subprocess.run(["bash", str(script)], env=env, capture_output=True, text=True) + assert result.returncode == sync_status, result.stdout + assert (tmp_path / "sync-version").read_text() == "2.7.0.post1+g123abcd\n" + assert (tmp_path / "ran-orchestrator").exists() == (sync_status == 0) diff --git a/utils/test_srt_recipe_power.py b/utils/test_srt_recipe_power.py new file mode 100644 index 0000000000..7cc9ee2e7d --- /dev/null +++ b/utils/test_srt_recipe_power.py @@ -0,0 +1,367 @@ +"""Exercise native recipe power routing and submission through the shell entrypoints.""" + +from __future__ import annotations + +import copy +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest +import yaml + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) + +from srtctl.core.config import generate_override_configs # noqa: E402 +from srtctl.core.overrides import ( # noqa: E402 + apply_overrides_to_recipe, + parse_overrides, +) + + +@pytest.fixture +def agentx_recipe() -> dict: + """A controlled custom benchmark with two physical GPU roles.""" + return { + "schema": 2, + "name": "controlled-agentx", + "model": {"path": "weights", "container": "serving", "precision": "fp4"}, + "resources": {"gpu_type": "gb200", "gpus_per_node": 4}, + "engine": "vllm", + "roles": { + "prefill": {"nodes": 1, "workers": 1, "gpus": 4, "args": {}}, + "decode": { + "nodes": 1, + "workers": 1, + "gpus": 4, + "args": { + "speculative-config": json.dumps( + {"method": "mtp", "num_speculative_tokens": 3} + ) + }, + }, + }, + "telemetry": { + "enabled": True, + "required": True, + "collect_interval_ms": 1000, + "storage_subdir": "power", + "dcgm_exporter": {"container_image": "dcgm-exporter", "port": 9401}, + }, + "benchmark": { + "type": "custom", + "concurrencies": [99], + "command": "bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh", + "env": { + "RESULT_DIR": "/logs/agentic", + "INFMAX_CONTAINER_WORKSPACE": "/infmax-workspace", + "IS_MULTINODE": "true", + "CONC_LIST": "99", + }, + }, + } + + +def _executable(path: Path, source: str) -> None: + path.write_text(f"#!{sys.executable}\n{source}") + path.chmod(0o755) + + +def _submit( + tmp_path: Path, + recipe: dict, + *, + selector="", + overrides=(), + framework="dynamo-vllm", + **environment, +): + (tmp_path / "srt-slurm-sha.txt").write_text("a" * 40 + "\n") + recipe_path = tmp_path / "recipe with spaces.yaml" + original = yaml.safe_dump(recipe) + recipe_path.write_text(original) + _executable( + tmp_path / "srtctl", + "import json,os,sys\n" + "with open(os.environ['SUBMISSIONS'], 'a') as handle:\n" + " handle.write(json.dumps(sys.argv[1:])+'\\n')\n" + "print('✅ Job 12345 submitted')\n" + "sys.exit(int(os.environ.get('SUBMISSION_RC', '0')))\n", + ) + result = subprocess.run( + [ + "bash", + "-c", + 'source "$1" || exit $?; config="$2"; framework="$3"; shift 3; ' + 'SRTCTL_EVAL_ARGS+=("$@"); ' + 'prepare_srt_power "$config" "$framework" || exit $?; ' + 'printf \'{"dcgm":%s,"agentx":%s}\\n\' ' + '"$USES_DCGM_POWER" "$USES_AGENTX_POWER" > "$LANE"; ' + 'apply_srt_recipe "$config" "$framework" ' + '-f "$config" --tags "ordinary AgentX submission" "${SRTCTL_RECIPE_ARGS[@]}"', + "bash", + str(ROOT / "runners/slurm_utils.sh"), + f"{recipe_path}{':' + selector if selector else ''}", + framework, + *overrides, + ], + env={ + **os.environ, + "PATH": f"{tmp_path}:{Path(sys.executable).parent}:{os.environ['PATH']}", + "PYTHONPATH": os.pathsep.join( + [str(ROOT), str(ROOT / "utils/srt-slurm/src")] + ), + "GITHUB_WORKSPACE": str(tmp_path), + "MODEL_PREFIX": "dsv4", + "IS_AGENTIC": "1", + "EVAL_ONLY": "false", + "SPEC_DECODING": "mtp", + "THINKING_MODE": "thinking_on", + "CONC_LIST": "1 4", + "SUBMISSIONS": str(tmp_path / "submissions.jsonl"), + "LANE": str(tmp_path / "lane.json"), + **environment, + }, + cwd=tmp_path, + text=True, + capture_output=True, + check=False, + ) + assert recipe_path.read_text() == original + recorded = tmp_path / "submissions.jsonl" + commands = ( + [json.loads(line) for line in recorded.read_text().splitlines()] + if recorded.exists() + else [] + ) + lane_path = tmp_path / "lane.json" + lane = json.loads(lane_path.read_text()) if lane_path.exists() else None + return result, commands, lane + + +def _resolved_submission(raw: dict, command: list[str]) -> dict: + """Decode recorded native CLI arguments with the pinned SRT implementation.""" + recipe = copy.deepcopy(raw) + sets = [command[i + 1] for i, value in enumerate(command[:-1]) if value == "--set"] + unsets = [ + command[i + 1] for i, value in enumerate(command[:-1]) if value == "--unset" + ] + apply_overrides_to_recipe(recipe, parse_overrides(sets, unsets)) + config = command[command.index("--file") + 1] + _, _, selector = config.partition(":") + if "base" in recipe: + return generate_override_configs(recipe, selector=selector or None)[0][1] + return recipe + + +def test_normal_submission_injects_matrix_power_without_changing_serving( + tmp_path, agentx_recipe +): + result, commands, lane = _submit(tmp_path, agentx_recipe) + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": 1, "agentx": 1} + assert len(commands) == 1 + resolved = _resolved_submission(agentx_recipe, commands[0]) + assert resolved["benchmark"]["concurrencies"] == [1, 4] + assert resolved["benchmark"]["env"]["CONC_LIST"] == "1 4" + assert resolved["benchmark"]["env"]["REQUIRE_POWER"] == "1" + assert resolved["benchmark"]["env"]["ENABLE_AGENTX_POWER"] == "1" + assert resolved["telemetry"]["required"] is True + assert (tmp_path / "power-producer-sha.txt").read_text() == "a" * 40 + "\n" + for field in ("model", "resources", "engine"): + assert resolved[field] == agentx_recipe[field] + before = copy.deepcopy(agentx_recipe["roles"]) + after = copy.deepcopy(resolved["roles"]) + for role in ("decode",): + original_spec = json.loads(before[role]["args"].pop("speculative-config")) + generated_spec = json.loads(after[role]["args"].pop("speculative-config")) + assert generated_spec.pop("rejection_sample_method") == "synthetic" + assert generated_spec.pop("synthetic_acceptance_length") > 1 + assert generated_spec == original_spec + assert after == before + + +@pytest.mark.parametrize( + "selector, expected", + [ + ("base", 0), + ("override_power", 1), + ("zip_override_power[0]", 0), + ("zip_override_power[1]", 1), + ], +) +def test_native_selector_controls_the_power_lane( + tmp_path, agentx_recipe, selector, expected +): + agentx_recipe["telemetry"]["enabled"] = False + raw = { + "schema": 2, + "base": agentx_recipe, + "override_power": {"telemetry": {"enabled": True}}, + "zip_override_power": { + "name": ["disabled", "enabled"], + "telemetry": {"enabled": [False, True]}, + }, + } + result, commands, lane = _submit(tmp_path, raw, selector=selector) + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": expected, "agentx": expected} + assert len(commands) == 1 + resolved = _resolved_submission(raw, commands[0]) + assert resolved["telemetry"]["enabled"] is bool(expected) + assert resolved["benchmark"]["concurrencies"] == [1, 4] + + +@pytest.mark.parametrize( + "overrides", [("--set", "telemetry.enabled=false"), ("--unset", "telemetry")] +) +def test_native_caller_override_disables_power(tmp_path, agentx_recipe, overrides): + result, commands, lane = _submit(tmp_path, agentx_recipe, overrides=overrides) + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": 0, "agentx": 0} + resolved = _resolved_submission(agentx_recipe, commands[0]) + assert not resolved.get("telemetry", {}).get("enabled", False) + assert "REQUIRE_POWER" not in resolved["benchmark"]["env"] + + +def test_eval_uses_real_verification_with_the_same_recipe_contract( + tmp_path, agentx_recipe +): + result, commands, lane = _submit(tmp_path, agentx_recipe, EVAL_ONLY="true") + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": 1, "agentx": 1} + resolved = _resolved_submission(agentx_recipe, commands[0]) + assert resolved["roles"] == agentx_recipe["roles"] + assert resolved["benchmark"]["concurrencies"] == [1, 4] + + +def test_multiple_selected_recipes_fail_before_submission(tmp_path, agentx_recipe): + raw = { + "schema": 2, + "base": agentx_recipe, + "zip_override_power": {"name": ["first", "second"]}, + } + result, commands, lane = _submit(tmp_path, raw, selector="zip_override_power") + assert result.returncode != 0 + assert "exactly one selected recipe" in result.stderr + assert commands == [] + assert lane is None + + +def test_non_agentx_disabled_telemetry_preserves_multiple_submissions( + tmp_path, agentx_recipe +): + agentx_recipe["telemetry"]["enabled"] = False + raw = { + "schema": 2, + "base": agentx_recipe, + "zip_override_legacy": {"name": ["first", "second"]}, + } + result, commands, lane = _submit( + tmp_path, raw, selector="zip_override_legacy", IS_AGENTIC="0" + ) + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": 0, "agentx": 0} + assert len(commands) == 2 + resolved = [_resolved_submission(raw, command) for command in commands] + assert [recipe["name"] for recipe in resolved] == ["first", "second"] + for recipe in resolved: + assert recipe["telemetry"]["enabled"] is False + assert recipe["benchmark"]["concurrencies"] == [99] + assert recipe["roles"] == agentx_recipe["roles"] + + +def test_non_agentx_mixed_telemetry_fails_before_any_submission( + tmp_path, agentx_recipe +): + raw = { + "schema": 2, + "base": agentx_recipe, + "zip_override_mixed": { + "name": ["disabled", "enabled"], + "telemetry": {"enabled": [False, True]}, + }, + } + result, commands, lane = _submit( + tmp_path, raw, selector="zip_override_mixed", IS_AGENTIC="0" + ) + assert result.returncode != 0 + assert "exactly one selected recipe" in result.stderr + assert commands == [] + assert lane is None + + +@pytest.mark.parametrize("concurrencies", ["1 1", "0 4", "1 nan"]) +def test_invalid_matrix_concurrencies_fail_before_submission( + tmp_path, agentx_recipe, concurrencies +): + result, commands, _ = _submit(tmp_path, agentx_recipe, CONC_LIST=concurrencies) + assert result.returncode != 0 + assert commands == [] + + +@pytest.mark.parametrize( + "overrides, reason", + [ + (("--set", "telemetry.required=false"), "telemetry.required"), + (("--set", 'telemetry.storage_subdir="other"'), "storage_subdir"), + (("--set", 'benchmark.env.RESULT_DIR="/logs/other"'), "window/result contract"), + ], +) +def test_incompatible_power_contract_fails_before_submission( + tmp_path, agentx_recipe, overrides, reason +): + result, commands, _ = _submit(tmp_path, agentx_recipe, overrides=overrides) + assert result.returncode != 0 + assert reason in result.stderr + assert commands == [] + + +def test_submission_failure_is_returned_even_with_a_job_id(tmp_path, agentx_recipe): + result, commands, lane = _submit(tmp_path, agentx_recipe, SUBMISSION_RC="7") + assert result.returncode == 7 + assert "Job 12345" in result.stdout + assert len(commands) == 1 + assert lane == {"dcgm": 1, "agentx": 1} + + +@pytest.mark.parametrize("disaggregated", [False, True]) +def test_sglang_power_admission_preserves_physical_role_topology( + tmp_path, agentx_recipe, disaggregated +): + agentx_recipe["engine"] = "sglang" + worker = { + "nodes": 2, + "workers": 1, + "gpus": 8, + "env": {"SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE": "0"}, + "args": {"tensor-parallel-size": 8, "expert-parallel-size": 1}, + } + if disaggregated: + agentx_recipe["roles"] = { + "prefill": worker, + "decode": { + "nodes": 4, + "workers": 4, + "gpus": 4, + "args": {"tensor-parallel-size": 4}, + }, + } + else: + agentx_recipe["roles"] = {"agg": worker} + result, commands, lane = _submit( + tmp_path, + agentx_recipe, + framework="dynamo-sglang", + MODEL_PREFIX="new-model", + SPEC_DECODING="none", + ) + assert result.returncode == 0, result.stderr + assert lane == {"dcgm": 1, "agentx": 1} + resolved = _resolved_submission(agentx_recipe, commands[0]) + assert resolved["roles"] == agentx_recipe["roles"] + assert resolved["benchmark"]["env"]["REQUIRE_POWER"] == "1" + assert resolved["benchmark"]["concurrencies"] == [1, 4] diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 979f286daf..dc20fd0f85 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -278,7 +278,9 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, # Only external executables are stubbed; run the real pool launcher, shared # setup/profile/acceptance helpers, binder, and artifact collection. scripts = { - "git": 'if [[ "$1" == clone ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', + "git": 'while [[ "$1" == -c || "$1" == -C ]]; do shift 2; done; ' + 'case "$1" in clone) mkdir -p "${@: -1}/configs";; ' + 'rev-parse) echo test-commit;; *) exit 1;; esac', "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', "make": '[[ "$TEST_FAILURE" == bootstrap ]] && exit 13; mkdir -p bin; touch bin/uv', "squeue": '[[ "$TEST_FAILURE" == submission || "$TEST_FAILURE" == agentic ]] && echo "42"; exit 0',