From 16f5a08db8020d5ea4fce1fef9a0a6866d5bc507 Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Thu, 1 Oct 2026 07:40:15 +0000 Subject: [PATCH 1/2] feat(srt): bind planned srt-slurm points to their recipes before dispatch MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit run-sweep, e2e-tests and profile now pipe their matrix through infx.srt_slurm.preflight after benchmark_schema. A single-node point must select exactly one variant through the runtime's select_recipe. Every CONFIG_FILE and EVAL_CONFIG_FILE recipe a multi-node point can launch must select at least one variant, and its worker containers must resolve to the images the launcher stages: the point's image or a cluster alias, and PREFILL_IMAGE for a TileRT prefill role. A TileRT frontend or client must run one of those images, and a benchmark client that names another tag of the point's image is reported. Recipe paths must stay inside their recipe trees. A mismatch fails setup before any canary or benchmark job is dispatched. In e2e-tests and profile, the validator and its srtctl come from the trusted tree, which reads the measured tree only as data, and the check always runs. require_launcher already rejects revisions older than the Python launcher, which are also the revisions whose runner config this tooling cannot read. run-sweep、e2e-tests 和 profile 现在在 benchmark_schema 之后把 matrix 交给 infx.srt_slurm.preflight 检查。单节点点必须通过运行时的 select_recipe 恰好 匹配一个 variant。多节点点可能启动的每个 CONFIG_FILE 和 EVAL_CONFIG_FILE 配方必须至少选中一个 variant,其 worker container 必须解析到 launcher 暂存 的镜像:该点的镜像或 cluster alias;TileRT 的 prefill role 则必须是 PREFILL_IMAGE。TileRT 的 frontend 和客户端必须使用这两个镜像之一;benchmark 客户端如果写了该点镜像的另一个 tag,会被报告。配方路径必须位于各自的配方 目录内。任何不一致都会让 setup 在派发 canary 或 benchmark 任务之前失败。 在 e2e-tests 和 profile 中,validator 及其 srtctl 都来自可信代码树,被测代码 树只作为数据读取,并且检查始终运行。require_launcher 已经拒绝早于 Python launcher 的版本,而本工具无法读取 runner 配置的正是这些版本。 Co-authored-by: Cursor --- .github/workflows/e2e-tests.yml | 6 + .github/workflows/profile.yml | 6 + .github/workflows/run-sweep.yml | 7 + inferencex-e2e/infx/srt_slurm/preflight.py | 314 ++++++++++++++++++ .../infx/tests/srt_slurm/test_preflight.py | 304 +++++++++++++++++ 5 files changed, 637 insertions(+) create mode 100644 inferencex-e2e/infx/srt_slurm/preflight.py create mode 100644 inferencex-e2e/infx/tests/srt_slurm/test_preflight.py diff --git a/.github/workflows/e2e-tests.yml b/.github/workflows/e2e-tests.yml index 34acf88ae4..a406d743c9 100644 --- a/.github/workflows/e2e-tests.yml +++ b/.github/workflows/e2e-tests.yml @@ -319,6 +319,12 @@ jobs: CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | env PYTHONPATH="$PRIORITY_ROOT" \ uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \ python -P -m infx.workflows.benchmark_schema) + # srtctl comes from the trusted tree; the measured tree is only read as data. + git -C "$PRIORITY_ROOT" submodule update --init utils/srt-slurm + CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | env PYTHONPATH="$PRIORITY_ROOT:$PRIORITY_ROOT/utils/srt-slurm/src" \ + uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \ + --with marshmallow --with marshmallow-dataclass --with ruamel.yaml --with requests \ + python -P -m infx.srt_slurm.preflight --root "$MEASURED_ROOT") score_matrix() { local family="$1" env PYTHONPATH="$PRIORITY_ROOT" uv run --no-project --exclude-newer PT12H --python 3.12 --with pyyaml \ diff --git a/.github/workflows/profile.yml b/.github/workflows/profile.yml index c506bc02c9..3face6903f 100644 --- a/.github/workflows/profile.yml +++ b/.github/workflows/profile.yml @@ -107,6 +107,12 @@ jobs: uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \ python -P "${GENERATOR[@]}" \ test-config --config-files "$INPUTS_CONFIG_FILE" --config-keys "$INPUTS_CONFIG_KEY" --conc "$INPUTS_CONC") + # srtctl comes from the trusted tree; the measured tree is only read as data. + git -C "$PRIORITY_ROOT" submodule update --init utils/srt-slurm + CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | env PYTHONPATH="$PRIORITY_ROOT:$PRIORITY_ROOT/utils/srt-slurm/src" \ + uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \ + --with marshmallow --with marshmallow-dataclass --with ruamel.yaml --with requests \ + python -P -m infx.srt_slurm.preflight --root "$MEASURED_ROOT") CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | env PYTHONPATH="$PRIORITY_ROOT" uv run --no-project --exclude-newer PT12H --python 3.12 --with pyyaml \ python -P -m infx.workflows.ci_priority \ --policy "${PRIORITY_ROOT}/configs/ci-priority.yaml" \ diff --git a/.github/workflows/run-sweep.yml b/.github/workflows/run-sweep.yml index 6f182d4abf..754e38716c 100644 --- a/.github/workflows/run-sweep.yml +++ b/.github/workflows/run-sweep.yml @@ -282,6 +282,9 @@ jobs: echo "criteria=$criteria" >> "$GITHUB_OUTPUT" echo "Priority criteria: $criteria" + - name: Initialize srt-slurm for recipe preflight + run: git submodule update --init inferencex-e2e/utils/srt-slurm + - id: setup env: PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} @@ -308,6 +311,10 @@ jobs: CONFIG_JSON=$("${CMD[@]}") CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \ python -m infx.workflows.benchmark_schema --plan) + CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | env PYTHONPATH="${PYTHONPATH}:${PWD}/utils/srt-slurm/src" \ + uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \ + --with marshmallow --with marshmallow-dataclass --with ruamel.yaml --with requests \ + python -m infx.srt_slurm.preflight) CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | uv run --no-project --exclude-newer PT12H --python 3.12 --with pyyaml \ python -m infx.workflows.ci_priority \ --event-name "${GITHUB_EVENT_NAME}" \ diff --git a/inferencex-e2e/infx/srt_slurm/preflight.py b/inferencex-e2e/infx/srt_slurm/preflight.py new file mode 100644 index 0000000000..fedcb20037 --- /dev/null +++ b/inferencex-e2e/infx/srt_slurm/preflight.py @@ -0,0 +1,314 @@ +"""Bind every planned srt-slurm point to the recipe variants it would launch, before dispatch. + +Sweep entry points pipe their matrix through this check after ``infx.workflows.benchmark_schema``. +A single-node point must select exactly one variant through the runtime's ``select_recipe``, +which compares the variant's ``model.container`` with the point's image. Every multi-node recipe a +point can launch (``CONFIG_FILE`` and ``EVAL_CONFIG_FILE``) must select at least one variant, and +each variant's worker containers must resolve to an image the launcher stages for the job: the +point's image, or ``PREFILL_IMAGE`` for a TileRT prefill role. A TileRT frontend or client must +run one of those images, and any other benchmark client that names another tag of the point's +image is stale, because srtctl would pull that literal instead. The matrix is echoed unchanged on +success; any problem exits non-zero so no benchmark job is dispatched. +""" + +from __future__ import annotations + +import argparse +import json +import sys +from collections.abc import Iterator, Mapping +from pathlib import Path +from typing import Any + +import yaml + +from infx.clusters import CLUSTER_LABEL_PREFIX, RunnerInventory, load_inventory +from infx.clusters.slurm import slurm_settings +from infx.launch.drivers.srt.config import pyxis_spelling +from infx.launch.drivers.srt.recipe import RECIPES_MIRROR, recipe_mirror_path +from infx.srt_slurm.single_node import select_recipe +from infx.srt_slurm.synthetic_acceptance import selected_recipes + +E2E_ROOT = Path(__file__).resolve().parents[2] +SINGLE_NODE_RECIPES = Path("benchmarks/single_node/srt-slurm-recipes") +RECIPE_SETTINGS = ("CONFIG_FILE", "EVAL_CONFIG_FILE") + + +def matrix_points(matrix: Any) -> Iterator[Mapping[str, Any]]: + """Every benchmark point in a sweep plan or a flat generated matrix.""" + if isinstance(matrix, Mapping): + if "image" in matrix and ("srt-recipe" in matrix or "prefill" in matrix): + yield matrix + return + for value in matrix.values(): + yield from matrix_points(value) + elif isinstance(matrix, list): + for item in matrix: + yield from matrix_points(item) + + +def workflow_text(value: Any) -> str: + """A matrix value as a GitHub expression renders it into an environment variable.""" + if value is None: + return "" + if isinstance(value, bool): + return "true" if value else "false" + return str(value) + + +def single_node_environment(point: Mapping[str, Any]) -> dict[str, str]: + """The recipe-binding inputs ``benchmark-tmpl.yml`` exports for a single-node point.""" + agentic = point.get("scenario-type") == "agentic-coding" + tp, pp, pcp = (int(point.get(name, 1)) for name in ("tp", "pp", "pcp-size")) + return { + "FRAMEWORK": str(point["framework"]), + "MODEL": str(point["model"]), + "IMAGE": str(point["image"]), + "PRECISION": str(point["precision"]), + "TP": str(tp), + "PP_SIZE": str(pp), + "DCP_SIZE": workflow_text(point.get("dcp-size", 1)), + "PCP_SIZE": str(pcp), + "EP_SIZE": workflow_text(point["ep"]), + "DP_ATTENTION": workflow_text(point.get("dp-attn", False)), + "CONC": workflow_text(point["conc"]), + "SPEC_DECODING": str(point["spec-decoding"]), + "GPU_COUNT": str(tp * pp * pcp), + "IS_AGENTIC": "1" if agentic else "0", + "KV_OFFLOADING": workflow_text(point.get("kv-offloading")) if agentic else "", + "TOTAL_CPU_DRAM_GB": workflow_text(point.get("total-cpu-dram-gb")) if agentic else "0", + "ISL": "0" if agentic else workflow_text(point["isl"]), + "OSL": "0" if agentic else workflow_text(point["osl"]), + "RANDOM_RANGE_RATIO": "0.8", + } + + +def docker_spelling(image: str) -> str: + """``registry/path`` for a pyxis ``registry#path`` image, else ``image``.""" + host, separator, path = image.partition("#") + return f"{host}/{path}" if separator else image + + +def repository(image: str) -> str: + """``image`` without its tag or digest, in ``registry/path`` spelling.""" + head, _, last = docker_spelling(image).partition("@")[0].rpartition("/") + name = last.partition(":")[0] + return f"{head}/{name}" if head else name + + +def point_label(point: Mapping[str, Any]) -> str: + """The point as its benchmark job is named, with the eval suffix the job title carries.""" + suffix = ( + " | eval-only" if point.get("eval-only") else " | eval" if point.get("run-eval") else "" + ) + return f"{point.get('exp-name', '')} on {point.get('runner', '')}{suffix}" + + +def container_fields(config: Mapping[str, Any]) -> list[tuple[str, Any]]: + """The ``(field, value)`` pairs naming the images a variant's workers, frontend and client run.""" + fields: list[tuple[str, Any]] = [ + ("model.container", (config.get("model") or {}).get("container")) + ] + for role, spec in (config.get("roles") or {}).items(): + if isinstance(spec, Mapping) and "container" in spec: + fields.append((f"roles.{role}.container", spec["container"])) + for block in ("frontend", "benchmark"): + section = config.get(block) + if isinstance(section, Mapping) and "container_image" in section: + fields.append((f"{block}.container_image", section["container_image"])) + return fields + + +def container_problems( + config: Mapping[str, Any], image: str, aliases: frozenset[str], prefill: str | None +) -> list[str]: + """Containers of one variant that would not run the images the launcher stages for it. + + Workers run the point's image (directly or through a cluster alias), except a TileRT prefill + role, which runs ``PREFILL_IMAGE``. A TileRT frontend and client run one of those two images. + Other frontends may pin an image of their own, but a benchmark client that names another tag + of the point's image is stale. + """ + resolved = {image, pyxis_spelling(image), *aliases} + problems = [] + for field, value in container_fields(config): + if value is None: + if field == "model.container": + problems.append(f"{field} is not set") + elif field == "roles.prefill.container" and prefill is not None: + if value != prefill: + problems.append(f"{field} {value!r} is not PREFILL_IMAGE {prefill}") + elif field == "model.container" or field.startswith("roles."): + if value not in resolved: + known = ", ".join(sorted(aliases)) or "none" + problems.append( + f"{field} {value!r} does not resolve to {image} (container aliases: {known})" + ) + elif prefill is not None: + if value not in resolved and value != prefill: + problems.append(f"{field} {value!r} is neither {image} nor PREFILL_IMAGE {prefill}") + elif ( + field == "benchmark.container_image" + and value not in resolved + and repository(str(value)) == repository(image) + ): + problems.append( + f"{field} {value!r} is another tag of {image}; " + "srtctl would pull it instead of the staged image" + ) + return problems + + +def check_single_node(point: Mapping[str, Any], root: Path) -> list[str]: + reference = str(point["srt-recipe"]) + path, _, selector = reference.partition(":") + recipe = (root / path).resolve() + if not recipe.is_relative_to((root / SINGLE_NODE_RECIPES).resolve()): + return [f"srt-recipe={reference}: not inside {SINGLE_NODE_RECIPES}"] + config = str(recipe) + (f":{selector}" if selector else "") + try: + _, variant = select_recipe(config, single_node_environment(point)) + except (OSError, KeyError, TypeError, ValueError, yaml.YAMLError) as exc: + return [f"srt-recipe={reference}: {exc}"] + problems = container_problems(variant, str(point["image"]), frozenset(), None) + return [f"srt-recipe={reference}: {problem}" for problem in problems] + + +def point_settings(point: Mapping[str, Any]) -> dict[str, str]: + """NAME=value additional-settings in the order ``benchmark-multinode-tmpl.yml`` exports them.""" + settings: dict[str, str] = {} + for role in ("prefill", "decode"): + for setting in (point.get(role) or {}).get("additional-settings") or []: + name, _, value = str(setting).partition("=") + settings[name] = value + return settings + + +def container_aliases(inventory: RunnerInventory, runner: str) -> frozenset[str]: + """Aliases the srt-slurm config maps to the job image on every cluster ``runner`` reaches.""" + if runner.startswith(CLUSTER_LABEL_PREFIX): + cluster_ids = {runner.removeprefix(CLUSTER_LABEL_PREFIX)} + elif runner in inventory.labels: + cluster_ids = {inventory.cluster_for(name).id for name in inventory.labels[runner]} + else: + # Some master entries schedule on a bare cluster id rather than a runners.yaml label. + cluster_ids = {runner} + aliases: frozenset[str] | None = None + for cluster_id in sorted(cluster_ids): + cluster = inventory.clusters.get(cluster_id) + srt = slurm_settings(cluster).srt_slurm if cluster is not None else None + names = frozenset(srt.container_aliases) if srt is not None else frozenset() + aliases = names if aliases is None else aliases & names + return aliases or frozenset() + + +def check_multi_node(point: Mapping[str, Any], root: Path, inventory: RunnerInventory) -> list[str]: + image = str(point["image"]) + aliases = container_aliases(inventory, str(point.get("runner", ""))) + settings = point_settings(point) + prefill = None + if point.get("framework") == "tilert": + prefill = settings.get("PREFILL_IMAGE") or None + if prefill is None: + return ["TileRT needs a PREFILL_IMAGE setting for its prefill role"] + problems = [] + mirror = (root / RECIPES_MIRROR).resolve() + for name in RECIPE_SETTINGS: + reference = settings.get(name) + if not reference: + continue + path = recipe_mirror_path(root, reference).resolve() + if not reference.startswith("recipes/") or not path.is_relative_to(mirror): + problems.append(f"{name}={reference}: not a recipes/ path inside {RECIPES_MIRROR}") + continue + selector = reference.partition(":")[2] or None + try: + recipe = yaml.safe_load(path.read_text()) + if not isinstance(recipe, Mapping): + problems.append(f"{name}={reference}: recipe is not a mapping") + continue + variants = selected_recipes(recipe, selector) + except (OSError, TypeError, ValueError, yaml.YAMLError) as exc: + problems.append(f"{name}={reference}: {exc}") + continue + if not variants: + problems.append( + f"{name}={reference}: selects no variant, so srtctl would submit nothing" + ) + for variant, config in variants: + label = f"{name}={reference}" + ("" if variant in (None, selector) else f" ({variant})") + found = container_problems(config, image, aliases, prefill) + declared = ((config.get("identity") or {}).get("container") or {}).get("image") + if declared is not None and docker_spelling(str(declared)) != docker_spelling(image): + found.append(f"identity.container.image {declared!r} is not {image}") + problems.extend(f"{label}: {problem}" for problem in found) + return problems + + +def check_matrix( + matrix: Any, root: Path, inventory: RunnerInventory | None +) -> dict[str, list[str]]: + """Each problem found, mapped to the points it affects.""" + problems: dict[str, list[str]] = {} + for point in matrix_points(matrix): + if "prefill" in point: + found = ( + check_multi_node(point, root, inventory) + if inventory is not None + else ["multi-node points need a runner inventory"] + ) + elif point.get("srt-recipe"): + found = check_single_node(point, root) + else: + continue + label = point_label(point) + for problem in found: + points = problems.setdefault(problem, []) + if label not in points: + points.append(label) + return problems + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--root", + type=Path, + default=E2E_ROOT, + help="inferencex-e2e tree whose recipes the matrix names", + ) + parser.add_argument( + "--runner-config", type=Path, help="runner inventory (default: ROOT/configs/runners.yaml)" + ) + args = parser.parse_args() + raw = sys.stdin.read() + matrix = json.loads(raw) + inventory = None + if any("prefill" in point for point in matrix_points(matrix)): + runner_config = args.runner_config or args.root / "configs/runners.yaml" + try: + inventory = load_inventory(runner_config) + except (OSError, ValueError, yaml.YAMLError) as exc: + print( + f"srt-slurm recipe preflight cannot read {runner_config} " + f"with this tooling's runner schema: {exc}", + file=sys.stderr, + ) + sys.exit(1) + problems = check_matrix(matrix, args.root, inventory) + if problems: + affected = len({label for points in problems.values() for label in points}) + print( + f"srt-slurm recipe preflight found {len(problems)} problem(s) affecting {affected} point(s):", + file=sys.stderr, + ) + for problem, points in problems.items(): + print(f" {problem}", file=sys.stderr) + for label in points: + print(f" - {label}", file=sys.stderr) + sys.exit(1) + sys.stdout.write(raw) + + +if __name__ == "__main__": + main() diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_preflight.py b/inferencex-e2e/infx/tests/srt_slurm/test_preflight.py new file mode 100644 index 0000000000..1dd6d85802 --- /dev/null +++ b/inferencex-e2e/infx/tests/srt_slurm/test_preflight.py @@ -0,0 +1,304 @@ +"""Recipe preflight over controlled matrices, recipes, and runner inventories.""" + +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest +import yaml + +from infx.clusters import load_inventory +from infx.srt_slurm.preflight import check_matrix + +ROOT = Path(__file__).resolve().parents[3] +sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) + +SINGLE = "benchmarks/single_node/srt-slurm-recipes" +MULTI = "benchmarks/multi_node/srt-slurm-recipes" +INVENTORY = { + "labels": {"cluster:c": ["c_0"], "pool": ["c_0"]}, + "clusters": { + "c": { + "gpus-per-node": 8, "arch": "x86_64", "scheduler": "slurm", "models": {"entries": {}}, + "slurm": { + "partition": "batch", "exclusive": True, + "srt-slurm": {"network-interface": "", "container-aliases": ["dynamo-sglang"]}, + }, + }, + }, +} # fmt: skip + + +def single_node_recipe(container: str, ep: int) -> dict: + return { + "engine": "sglang", + "resources": {"gpus_per_node": 8}, + "model": {"path": "hf:test/model", "container": container, "precision": "fp8"}, + "roles": {"agg": {"nodes": 1, "workers": 1, "gpus": 8, "args": { + "tensor-parallel-size": 8, "expert-parallel-size": ep, + }}}, + "benchmark": {"type": "custom", "command": "bash /bench/srt_agentic.sh", "env": {"MODEL": "test/model"}}, + } # fmt: skip + + +def single_node_point(recipe: str, image: str, ep: int = 1) -> dict: + return { + "exp-name": f"ep{ep}", "runner": "cluster:c", "srt-recipe": recipe, "image": image, + "model": "test/model", "framework": "sglang", "precision": "fp8", "tp": 8, "ep": ep, + "pp": 1, "dcp-size": 1, "pcp-size": 1, "dp-attn": False, "conc": 4, "spec-decoding": "none", + "scenario-type": "agentic-coding", "kv-offloading": "none", "total-cpu-dram-gb": 0, + } # fmt: skip + + +def multi_node_point(*settings: str, image: str = "img:1", runner: str = "cluster:c", + framework: str = "dynamo-sglang") -> dict: # fmt: skip + return { + "exp-name": "disagg", "runner": runner, "image": image, "framework": framework, + "prefill": {"additional-settings": list(settings)}, "decode": {"additional-settings": []}, + } # fmt: skip + + +def write_yaml(root: Path, relative: str, data: dict) -> str: + path = root / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(yaml.safe_dump(data)) + return relative + + +def multi_node_recipe(root: Path, container: str, identity: str | None = None) -> str: + recipe = {"base": {"name": "m", "model": {"path": "hf:t/m", "container": "img:1", "precision": "fp8"}}, + "override_run": {"model": {"container": container}}} # fmt: skip + if identity is not None: + recipe["base"]["identity"] = {"container": {"image": identity}} + write_yaml(root, f"{MULTI}/m/recipe.yaml", recipe) + return "CONFIG_FILE=recipes/m/recipe.yaml:override_run" + + +def tilert_recipe(root: Path, decode: str, prefill: str) -> str: + write_yaml(root, f"{MULTI}/t/recipe.yaml", { + "name": "t", "model": {"path": "hf:t/m", "container": decode, "precision": "fp8"}, + "roles": {"prefill": {"container": prefill}, "decode": {"container": decode}}, + "frontend": {"container_image": decode}, "benchmark": {"container_image": prefill}, + }) # fmt: skip + return "CONFIG_FILE=recipes/t/recipe.yaml" + + +def run_cli(root: Path, runner_config: Path, matrix: object) -> subprocess.CompletedProcess: + env = {**os.environ, "PYTHONPATH": f"{ROOT}{os.pathsep}{ROOT / 'utils/srt-slurm/src'}"} + command = [sys.executable, "-m", "infx.srt_slurm.preflight", "--root", str(root), + "--runner-config", str(runner_config)] # fmt: skip + return subprocess.run( + command, input=json.dumps(matrix), capture_output=True, text=True, env=env, check=False + ) + + +def test_cli_echoes_a_consistent_matrix_and_rejects_a_one_sided_image_update(tmp_path): + recipe = write_yaml( + tmp_path, f"{SINGLE}/a.yaml", {"base": single_node_recipe("img:2", 1), "override_c4": {}} + ) + inventory = tmp_path / "runners.yaml" + inventory.write_text(yaml.safe_dump(INVENTORY)) + + good = {"single_node": {"agentic": [single_node_point(recipe, "img:2")]}} + passed = run_cli(tmp_path, inventory, good) + assert (passed.returncode, passed.stdout) == (0, json.dumps(good)) + + throughput = single_node_point(recipe, "img:3") + failed = run_cli( + tmp_path, inventory, [throughput, {**throughput, "run-eval": True, "eval-only": True}] + ) + assert (failed.returncode, failed.stdout) == (1, "") + report = failed.stderr.splitlines() + assert report[0] == "srt-slurm recipe preflight found 1 problem(s) affecting 2 point(s):" + assert report[1].startswith(f" srt-recipe={SINGLE}/a.yaml: ") + assert "Single-node SRT image: recipe/matrix 'img:2' != 'img:3'" in report[1] + assert report[2:] == [" - ep1 on cluster:c", " - ep1 on cluster:c | eval-only"] + + +def test_cli_reads_the_runner_inventory_only_for_multi_node_points(tmp_path): + single = write_yaml(tmp_path, f"{SINGLE}/a.yaml", single_node_recipe("img:1", 1)) + inventory = tmp_path / "runners.yaml" + inventory.write_text(yaml.safe_dump({**INVENTORY, "retired-section": {}})) + + assert run_cli(tmp_path, inventory, [single_node_point(single, "img:1")]).returncode == 0 + multi = multi_node_point(multi_node_recipe(tmp_path, "img:1")) + failed = run_cli(tmp_path, inventory, [single_node_point(single, "img:1"), multi]) + assert failed.returncode == 1 + assert failed.stderr.startswith( + f"srt-slurm recipe preflight cannot read {inventory} with this tooling's runner schema: " + ) + + +def test_shared_recipe_rejects_variant_images_swapped_between_its_master_keys(tmp_path): + def recipe_with(base_image: str, override_image: str) -> str: + return write_yaml(tmp_path, f"{SINGLE}/shared.yaml", { + "base": single_node_recipe(base_image, 8), + "override_ep8": {}, + "override_ep1": {"model": {"container": override_image}, + "roles": {"agg": {"args": {"expert-parallel-size": 1}}}}, + }) # fmt: skip + + matrix = [single_node_point(f"{SINGLE}/shared.yaml", "img:a", ep=8), + single_node_point(f"{SINGLE}/shared.yaml", "img:b", ep=1)] # fmt: skip + inventory = load_inventory(INVENTORY) + recipe_with("img:a", "img:b") + assert check_matrix(matrix, tmp_path, inventory) == {} + recipe_with("img:b", "img:a") + problems = check_matrix(matrix, tmp_path, inventory) + assert list(problems.values()) == [["ep8 on cluster:c"], ["ep1 on cluster:c"]] + + +def test_eval_recipe_is_checked_alongside_the_throughput_recipe(tmp_path): + throughput = multi_node_recipe(tmp_path, "img:1") + write_yaml(tmp_path, f"{MULTI}/m/eval.yaml", + {"name": "e", "model": {"path": "hf:t/m", "container": "img:0", "precision": "fp8"}}) # fmt: skip + point = multi_node_point(throughput, "EVAL_CONFIG_FILE=recipes/m/eval.yaml") + assert check_matrix([point], tmp_path, load_inventory(INVENTORY)) == { + "EVAL_CONFIG_FILE=recipes/m/eval.yaml: model.container 'img:0' does not resolve to img:1 " + "(container aliases: dynamo-sglang)": ["disagg on cluster:c"] + } + + +def test_override_recipe_without_overrides_selects_nothing_unless_base_is_named(tmp_path): + write_yaml(tmp_path, f"{MULTI}/b.yaml", + {"base": {"name": "b", "model": {"path": "hf:t/m", "container": "img:1", "precision": "fp8"}}}) # fmt: skip + inventory = load_inventory(INVENTORY) + assert list( + check_matrix([multi_node_point("CONFIG_FILE=recipes/b.yaml")], tmp_path, inventory) + ) == ["CONFIG_FILE=recipes/b.yaml: selects no variant, so srtctl would submit nothing"] + assert ( + check_matrix([multi_node_point("CONFIG_FILE=recipes/b.yaml:base")], tmp_path, inventory) + == {} + ) + + +@pytest.mark.parametrize( + ("image", "container", "resolves"), + [ + ("img:1", "dynamo-sglang", True), + ("img:1", "dynamo-sglan", False), + ("nvcr.io/nvidia/x:1", "nvcr.io#nvidia/x:1", True), + ("img:1", "img:2", False), + ], +) +def test_multi_node_container_must_be_a_cluster_alias_or_the_point_image( + tmp_path, image, container, resolves +): + point = multi_node_point(multi_node_recipe(tmp_path, container), image=image) + assert (check_matrix([point], tmp_path, load_inventory(INVENTORY)) == {}) is resolves + + +@pytest.mark.parametrize( + ("image", "identity", "matches"), + [ + ("nvcr.io#nvidia/x:1", "nvcr.io/nvidia/x:1", True), + ("nvcr.io#nvidia/x:1", "nvcr.io/nvidia/x:2", False), + ], +) +def test_identity_image_accepts_either_registry_spelling_only(tmp_path, image, identity, matches): + point = multi_node_point(multi_node_recipe(tmp_path, "dynamo-sglang", identity), image=image) + assert (check_matrix([point], tmp_path, load_inventory(INVENTORY)) == {}) is matches + + +@pytest.mark.parametrize("runner", ["cluster:c", "pool", "c"]) +def test_aliases_follow_cluster_labels_runner_pools_and_bare_cluster_ids(tmp_path, runner): + point = multi_node_point(multi_node_recipe(tmp_path, "dynamo-sglang"), runner=runner) + assert check_matrix([point], tmp_path, load_inventory(INVENTORY)) == {} + + +@pytest.mark.parametrize( + ("image", "prefill_image", "stale"), + [ + ("dec:1", "pre:1", []), + ( + "dec:2", + "pre:1", + ["frontend.container_image", "model.container", "roles.decode.container"], + ), + ("dec:1", "pre:2", ["benchmark.container_image", "roles.prefill.container"]), + ], +) +def test_tilert_roles_follow_the_decode_image_and_prefill_image( + tmp_path, image, prefill_image, stale +): + point = multi_node_point(tilert_recipe(tmp_path, "dec:1", "pre:1"), f"PREFILL_IMAGE={prefill_image}", + image=image, framework="tilert") # fmt: skip + problems = check_matrix([point], tmp_path, load_inventory(INVENTORY)) + assert sorted(problem.split(": ", 1)[1].split(" ", 1)[0] for problem in problems) == stale + + +def test_tilert_point_without_prefill_image_is_reported(tmp_path): + point = multi_node_point( + tilert_recipe(tmp_path, "dec:1", "pre:1"), image="dec:1", framework="tilert" + ) + assert list(check_matrix([point], tmp_path, load_inventory(INVENTORY))) == [ + "TileRT needs a PREFILL_IMAGE setting for its prefill role" + ] + + +@pytest.mark.parametrize( + ("block", "image", "client", "stale"), + [ + ("benchmark", "img:2", "img:2", False), + ("benchmark", "img:2", "img:1", True), + ("benchmark", "img:2", "nginx", False), + ("benchmark", "nvcr.io/nvidia/x:1", "nvcr.io#nvidia/x:1", False), + ("benchmark", "nvcr.io/nvidia/x:1", "nvcr.io/nvidia/x@sha256:0", True), + ("frontend", "img:2", "img@sha256:0", False), + ], +) +def test_benchmark_client_must_not_name_another_tag_of_the_point_image( + tmp_path, block, image, client, stale +): + single = single_node_recipe(image, 1) + single.setdefault(block, {})["container_image"] = client + recipe = write_yaml(tmp_path, f"{SINGLE}/a.yaml", single) + write_yaml(tmp_path, f"{MULTI}/m.yaml", { + "name": "m", "model": {"path": "hf:t/m", "container": image, "precision": "fp8"}, + block: {"container_image": client}, + }) # fmt: skip + matrix = [ + single_node_point(recipe, image), + multi_node_point("CONFIG_FILE=recipes/m.yaml", image=image), + ] + problems = check_matrix(matrix, tmp_path, load_inventory(INVENTORY)) + assert list(problems.values()) == ( + [["ep1 on cluster:c"], ["disagg on cluster:c"]] if stale else [] + ) + + +@pytest.mark.parametrize( + "point", + [ + lambda root: single_node_point(str(root / "elsewhere/r.yaml"), "img:1"), + lambda root: single_node_point(f"{SINGLE}/../../../elsewhere/r.yaml", "img:1"), + lambda root: multi_node_point(f"CONFIG_FILE=recipes/{root}/elsewhere/r.yaml"), + lambda root: multi_node_point("CONFIG_FILE=recipes/../../../elsewhere/r.yaml"), + lambda root: multi_node_point("CONFIG_FILE=elsewhere/r.yaml"), + ], + ids=["single-absolute", "single-parent", "multi-absolute", "multi-parent", "multi-unprefixed"], +) +def test_recipe_paths_must_stay_inside_their_recipe_trees(tmp_path, point): + write_yaml(tmp_path, "elsewhere/r.yaml", single_node_recipe("img:1", 1)) + [problem] = check_matrix([point(tmp_path)], tmp_path, load_inventory(INVENTORY)) + assert problem.endswith((f"not inside {SINGLE}", f"not a recipes/ path inside {MULTI}")) + + +def test_variant_without_a_model_container_is_reported(tmp_path): + write_yaml( + tmp_path, f"{MULTI}/n.yaml", {"name": "n", "model": {"path": "hf:t/m", "precision": "fp8"}} + ) + point = multi_node_point("CONFIG_FILE=recipes/n.yaml") + assert list(check_matrix([point], tmp_path, load_inventory(INVENTORY))) == [ + "CONFIG_FILE=recipes/n.yaml: model.container is not set" + ] + + +def test_missing_recipe_is_reported_instead_of_raised(tmp_path): + point = multi_node_point("CONFIG_FILE=recipes/absent.yaml") + [(problem, points)] = check_matrix([point], tmp_path, load_inventory(INVENTORY)).items() + assert problem.startswith("CONFIG_FILE=recipes/absent.yaml: ") + assert "No such file" in problem + assert points == ["disagg on cluster:c"] From 5930300772cccef7235e015ec4a0f8dedea1ce87 Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Fri, 2 Oct 2026 06:02:44 +0000 Subject: [PATCH 2/2] fix(srt): require the CONFIG_FILE the launcher selects for multi-node points MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The srt-slurm launcher (lanes.config_file) launches EVAL_CONFIG_FILE only for an eval-only point that sets it; every other multi-node point needs a non-empty CONFIG_FILE and otherwise fails only on the GPU runner. The preflight now reports such points before dispatch, and still checks every recipe a point references. srt-slurm launcher(lanes.config_file)只会为设置了 EVAL_CONFIG_FILE 的 eval-only 点启动该配方;其他多节点点都需要非空的 CONFIG_FILE,否则要到 GPU runner 上才会失败。preflight 现在会在派发前报告这类点,并继续检查点 所引用的每个配方。 Co-authored-by: Cursor --- inferencex-e2e/infx/srt_slurm/preflight.py | 11 ++++++-- .../infx/tests/srt_slurm/test_preflight.py | 26 +++++++++++++++++++ 2 files changed, 35 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/infx/srt_slurm/preflight.py b/inferencex-e2e/infx/srt_slurm/preflight.py index fedcb20037..21f8f1587a 100644 --- a/inferencex-e2e/infx/srt_slurm/preflight.py +++ b/inferencex-e2e/infx/srt_slurm/preflight.py @@ -2,8 +2,9 @@ Sweep entry points pipe their matrix through this check after ``infx.workflows.benchmark_schema``. A single-node point must select exactly one variant through the runtime's ``select_recipe``, -which compares the variant's ``model.container`` with the point's image. Every multi-node recipe a -point can launch (``CONFIG_FILE`` and ``EVAL_CONFIG_FILE``) must select at least one variant, and +which compares the variant's ``model.container`` with the point's image. A multi-node point must +set ``CONFIG_FILE``, unless it is eval-only and sets ``EVAL_CONFIG_FILE``, the recipe the launcher +then selects. Every multi-node recipe a point can launch must select at least one variant, and each variant's worker containers must resolve to an image the launcher stages for the job: the point's image, or ``PREFILL_IMAGE`` for a TileRT prefill role. A TileRT frontend or client must run one of those images, and any other benchmark client that names another tag of the point's @@ -212,6 +213,12 @@ def check_multi_node(point: Mapping[str, Any], root: Path, inventory: RunnerInve if prefill is None: return ["TileRT needs a PREFILL_IMAGE setting for its prefill role"] problems = [] + if not settings.get("CONFIG_FILE") and not ( + point.get("eval-only") and settings.get("EVAL_CONFIG_FILE") + ): + problems.append( + "CONFIG_FILE is not set; only an eval-only point may launch its EVAL_CONFIG_FILE instead" + ) mirror = (root / RECIPES_MIRROR).resolve() for name in RECIPE_SETTINGS: reference = settings.get(name) diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_preflight.py b/inferencex-e2e/infx/tests/srt_slurm/test_preflight.py index 1dd6d85802..bca1eab3ce 100644 --- a/inferencex-e2e/infx/tests/srt_slurm/test_preflight.py +++ b/inferencex-e2e/infx/tests/srt_slurm/test_preflight.py @@ -161,6 +161,32 @@ def test_eval_recipe_is_checked_alongside_the_throughput_recipe(tmp_path): } +@pytest.mark.parametrize( + ("eval_only", "settings", "launches"), + [ + (False, [], False), + (False, ["EVAL_CONFIG_FILE=recipes/m/eval.yaml"], False), + (True, ["EVAL_CONFIG_FILE=recipes/m/eval.yaml"], True), + (True, [], False), + (True, ["CONFIG_FILE=recipes/m/eval.yaml"], True), + ], + ids=["throughput-none", "throughput-eval-only-recipe", "eval-only-eval-recipe", "eval-only-none", + "eval-only-config-recipe"], +) # fmt: skip +def test_multi_node_point_needs_the_recipe_its_launcher_selects( + tmp_path, eval_only, settings, launches +): + write_yaml(tmp_path, f"{MULTI}/m/eval.yaml", + {"name": "e", "model": {"path": "hf:t/m", "container": "img:1", "precision": "fp8"}}) # fmt: skip + point = {**multi_node_point(*settings), "eval-only": eval_only} + label = "disagg on cluster:c" + (" | eval-only" if eval_only else "") + missing = ( + "CONFIG_FILE is not set; only an eval-only point may launch its EVAL_CONFIG_FILE instead" + ) + expected = {} if launches else {missing: [label]} + assert check_matrix([point], tmp_path, load_inventory(INVENTORY)) == expected + + def test_override_recipe_without_overrides_selects_nothing_unless_base_is_named(tmp_path): write_yaml(tmp_path, f"{MULTI}/b.yaml", {"base": {"name": "b", "model": {"path": "hf:t/m", "container": "img:1", "precision": "fp8"}}}) # fmt: skip