From c9165a7a27fb0757d8b1c8494820b468b8ff1685 Mon Sep 17 00:00:00 2001 From: Yichao Zhu Date: Wed, 30 Sep 2026 09:54:11 +0800 Subject: [PATCH 1/4] feat(launch): integrate native Kimi-K3 PD with the Python launcher MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 迁移 Kimi-K3 PD 所需的集群资源声明和 recipe 镜像准备,沿用 main 的任务提交、取消与产物收集。 --- inferencex-e2e/configs/runners.yaml | 12 +++- .../docs/configuration-procedures.md | 2 + .../docs/configuration-procedures_zh.md | 2 + .../infx/launch/backends/slurm/squash.py | 7 ++- .../infx/launch/drivers/srt/config.py | 28 +++++++-- .../infx/launch/drivers/srt/lanes.py | 3 + .../infx/launch/drivers/srt/recipe.py | 30 ++++++++++ .../infx/srt_slurm/recipe_images.py | 60 +++++++++++++++++++ .../infx/tests/launch/fake_slurm.py | 31 ++++++---- .../infx/tests/launch/test_squash_images.py | 23 +++++++ .../infx/tests/launch/test_srt_config.py | 43 ++++++++++++- .../infx/tests/launch/test_srt_driver.py | 50 ++++++++++++++++ .../tests/launch/test_srt_recipe_edits.py | 43 +++++++++++++ 13 files changed, 313 insertions(+), 21 deletions(-) create mode 100644 inferencex-e2e/infx/srt_slurm/recipe_images.py diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index 7cd6fa2651..122c13b33c 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -844,6 +844,7 @@ clusters: MORI_RDMA_TC: "104" models: entries: + Kimi-K3: {root: it-share-data, dir: Kimi-K3} DeepSeek-V4-Pro-0813: {root: it-share-data, dir: DeepSeek-V4-Pro-0813} GLM-5.3: {root: it-share-data, dir: GLM-5.3} Qwen3.5-397B-A17B-FP8: {root: it-share-data, dir: Qwen3.5-397B-A17B-FP8} @@ -859,13 +860,19 @@ clusters: shared-hf-hub-cache: {path: /it-share/hf-hub-cache} aiperf-cache: {path: /it-share/aiperf-cache} it-share-data: {path: /it-share/data} + k3-draft: {path: /it-share/data/Inferact-Kimi-K3-DSpark} squash: dir: /it-share/gharunners2/srt-slurm/containers visibility: shared import: compute - multi-node-import: false + multi-node-import: true + import-step-args: [--cpus-per-task=4] srt-slurm: network-interface: eno0 + env: + NCCL_IB_HCA: rdma0,rdma1,rdma2,rdma3,rdma4,rdma5,rdma6,rdma7 + NCCL_SOCKET_IFNAME: eno0 + GLOO_SOCKET_IFNAME: eno0 single-node-time-limit: 500 gpus-per-node-directive: true segment-directive: false @@ -878,11 +885,14 @@ clusters: nodes: all volume-mounts: shared-hf-hub-cache: /hf_hub_cache/hub + k3-draft: /models/Inferact-Kimi-K3-DSpark mounts: /dev/kfd: /dev/kfd /dev/dri: /dev/dri + /dev/infiniband: /dev/infiniband /it-share/hf_home: /it-share/hf_home extra: visible_devices_env: ROCR_VISIBLE_DEVICES default_gpu_exporter: null nginx_raise_ulimit: false + default_bash_preamble: ulimit -l unlimited diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 4e8a7d0d23..1ac592f020 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -105,6 +105,8 @@ and preserve resources used by other jobs. ## Procedure index +The Kimi-K3 native PD lane resolves the selected recipe before image provisioning, requires its worker image to match the matrix, and stages the recipe's frontend image through the same backend. This keeps router images recipe-owned and rejects mismatched inputs before an import allocation. Its staged target model, draft mount, fabric devices, worker network environment and memlock preamble are declared in the cluster record; image imports, submission, cancellation and artifact collection use the shared Python launcher. + 1. [Prepare a worktree](#prepare-a-worktree) 2. [Add a model + hardware recipe](#add-a-model--hardware-recipe) 3. [Change a master config](#change-a-master-config) diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index efbf9347ba..90d8e8cb13 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -87,6 +87,8 @@ PowerX 严格校验。现有的 Tachometer 1000 ms / 功耗 exporter 100 ms 采 ## 规程索引 +Kimi-K3 原生 PD 路径在准备镜像前解析所选 recipe,要求 worker 镜像与矩阵一致,并通过同一后端准备 recipe 指定的 frontend 镜像。Router 镜像仍由 recipe 管理;输入不一致会在镜像导入任务申请资源前失败。已部署的目标模型、draft 挂载、网络设备、worker 网络环境及 memlock 前置命令由集群记录声明;镜像导入、任务提交、取消及产物收集沿用共享 Python launcher。 + 1. [准备 worktree](#准备-worktree) 2. [添加模型 + 硬件配方](#添加模型--硬件配方) 3. [修改主配置](#修改主配置) diff --git a/inferencex-e2e/infx/launch/backends/slurm/squash.py b/inferencex-e2e/infx/launch/backends/slurm/squash.py index a1d0a41172..47a4c1ee73 100644 --- a/inferencex-e2e/infx/launch/backends/slurm/squash.py +++ b/inferencex-e2e/infx/launch/backends/slurm/squash.py @@ -68,7 +68,8 @@ def enroot_uri(image: str) -> str: Enroot 3.x cannot parse ``tag@digest``, so a pinned image becomes ``registry#repository:digest`` (the digest is immutable, so the tag is dropped). - Pyxis-style ``registry#repo`` input is read as ``registry/repo``. + Pyxis-style ``registry#repo`` input is read as ``registry/repo``. Docker Hub's + public aliases resolve to its registry API, not the docker.io website. """ image = image.replace("#", "/", 1) without_digest, digest = image, "" @@ -79,8 +80,10 @@ def enroot_uri(image: str) -> str: registry, repository = first, without_digest.split("/", 1)[1] else: registry, repository = "registry-1.docker.io", without_digest + if registry in {"docker.io", "index.docker.io"}: + registry = "registry-1.docker.io" if not digest: - if registry == "registry-1.docker.io": + if registry == "registry-1.docker.io" and image == repository: return f"docker://{image}" return f"docker://{registry}#{repository}" directory, _, name = repository.rpartition("/") diff --git a/inferencex-e2e/infx/launch/drivers/srt/config.py b/inferencex-e2e/infx/launch/drivers/srt/config.py index 569156c4c2..6388a0d6d1 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/config.py +++ b/inferencex-e2e/infx/launch/drivers/srt/config.py @@ -17,8 +17,8 @@ from infx.clusters.slurm import slurm_settings from infx.launch.context import LaunchError -from infx.launch.drivers.srt.lanes import srt_time_limit -from infx.launch.drivers.srt.recipe import HEALTH_ATTEMPTS +from infx.launch.drivers.srt.lanes import config_file, srt_time_limit +from infx.launch.drivers.srt.recipe import HEALTH_ATTEMPTS, recipe_images if TYPE_CHECKING: from infx.clusters import Cluster @@ -185,13 +185,21 @@ def create_volume_mounts(run: SrtRun) -> None: def lane_mounts(run: SrtRun, lane: SrtLane) -> list[tuple[str, str]]: - """The (host, container) mounts the lane adds for this request; their hosts are created.""" + """Request-selected mounts; only writable cache directories are created on the launcher. + + Read-only assets may be files present only on compute nodes. Leave their paths and + permissions untouched; the container runtime validates them on the allocated host. + """ mounts: list[tuple[str, str]] = [] for mount in lane.mounts: if mount.when(run.request): host = volume_path(run.cluster, mount.volume) - _create_dir(host, world_writable=mount.world_writable) - mounts.append((str(host), mount.target or str(host))) + target = mount.target or str(host) + if mount.read_only: + target += ":ro" + else: + _create_dir(host, world_writable=mount.world_writable) + mounts.append((str(host), target)) if run.request.framework == "tilert": mounts.append((str(run.workspace), "/infmax-workspace")) return mounts @@ -206,6 +214,11 @@ def write_lane_config( ) -> None: """Stage a multi-node job's images, create its mounts, and write its srtslurm.yaml.""" backend, request = run.backend, run.request + images = ( + recipe_images(run, checkout, config_file(request)) + if lane.stage_recipe_images is not None and lane.stage_recipe_images(request) + else [request.image] + ) container = backend.stage_image( request.image, framework=request.framework, model_prefix=request.model_prefix ).reference @@ -215,6 +228,11 @@ def write_lane_config( else None ) containers: dict[str, str] = {} + for image in images[1:]: + staged = backend.stage_image( + image, framework=request.framework, model_prefix=request.model_prefix + ).reference + containers[image] = containers[pyxis_spelling(image)] = staged if request.framework == "tilert": prefill_image = request.env["PREFILL_IMAGE"] containers[prefill_image] = backend.stage_image(prefill_image).reference diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index 8c71413de1..b39508a725 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -23,6 +23,7 @@ class LaneMount: volume: str target: str | None = None world_writable: bool = False + read_only: bool = False @dataclass(frozen=True) @@ -41,6 +42,7 @@ class SrtLane: time_limit: str | None = None long_time_limit: str | None = None long_time: Match | None = None + stage_recipe_images: Match | None = None _DYNAMO = any_of("dynamo-sglang", "dynamo-trt", "dynamo-vllm") @@ -104,6 +106,7 @@ class SrtLane: LaneMount(Match(), "aiperf-cache", "/aiperf_mmap_cache"), LaneMount(Match(frameworks=any_of("tilert")), "it-share-data", "/models"), ), + stage_recipe_images=Match(any_of("kimik3"), frameworks=any_of("vllm-disagg")), eval_unsets=( "roles.prefill.args.ep-dispatch-algorithm", "roles.decode.args.ep-dispatch-algorithm", diff --git a/inferencex-e2e/infx/launch/drivers/srt/recipe.py b/inferencex-e2e/infx/launch/drivers/srt/recipe.py index c03eda1734..aad7c5aaa6 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/recipe.py +++ b/inferencex-e2e/infx/launch/drivers/srt/recipe.py @@ -7,6 +7,7 @@ from __future__ import annotations import fnmatch +import json import re from collections.abc import Sequence from pathlib import Path @@ -14,10 +15,13 @@ import yaml +from infx.launch import proc from infx.launch.context import LaunchError if TYPE_CHECKING: + from infx.launch.drivers.srt.checkout import Checkout from infx.launch.drivers.srt.lanes import SrtLane + from infx.launch.drivers.srt.run import SrtRun from infx.launch.request import LaunchRequest RECIPES_MIRROR = Path("benchmarks/multi_node/srt-slurm-recipes") @@ -37,6 +41,32 @@ def recipe_mirror_path(workspace: Path, config_file: str) -> Path: return workspace / RECIPES_MIRROR / recipe_relpath(config_file).removeprefix("recipes/") +def recipe_images(run: SrtRun, checkout: Checkout, config_file: str) -> list[str]: + """Resolve images in the installed srtctl environment, not the launcher's interpreter.""" + path = recipe_mirror_path(run.workspace, config_file) + selector = config_file.partition(":")[2] + config = f"{path}:{selector}" if selector else str(path) + argv = [ + str(checkout.venv / "bin/python"), "-m", "infx.srt_slurm.recipe_images", + config, run.request.image, + ] # fmt: skip + try: + result = proc.run(argv, env=run.env, cwd=checkout.root, capture=True) + if result.returncode: + raise ValueError(f"resolver exited {result.returncode}: {result.stderr.strip()}") + images = json.loads(result.stdout) + if ( + not isinstance(images, list) + or not images + or images[0] != run.request.image + or not all(isinstance(image, str) and image for image in images) + ): + raise ValueError("resolver did not return the expected image list") + return images + except (OSError, ValueError) as error: + raise LaunchError(f"recipe image resolution failed for {config_file}: {error}") from error + + def rename_job(text: str, name: str) -> str: """Set the top-level ``name:``, the job name srtctl submits.""" return re.sub(r"(?m)^name:.*$", lambda _: f'name: "{name}"', text) diff --git a/inferencex-e2e/infx/srt_slurm/recipe_images.py b/inferencex-e2e/infx/srt_slurm/recipe_images.py new file mode 100644 index 0000000000..4dc88a66d5 --- /dev/null +++ b/inferencex-e2e/infx/srt_slurm/recipe_images.py @@ -0,0 +1,60 @@ +"""Resolve recipe images with the job-local srtctl, before importing any containers.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +import yaml + +from infx.srt_slurm.synthetic_acceptance import selected_recipes + + +def resolve_images(config: str, expected_worker: str) -> list[str]: + """Use native override expansion to validate and deduplicate one recipe's images.""" + path, _, selector = config.partition(":") + try: + raw = yaml.safe_load(Path(path).read_text()) + if not isinstance(raw, dict): + raise ValueError("recipe must be a mapping") + selected = selected_recipes(raw, selector or None) + if len(selected) != 1: + raise ValueError("image provisioning requires exactly one selected recipe") + recipe = selected[0][1] + worker = recipe["model"]["container"] + if worker != expected_worker: + raise ValueError("recipe model.container must match the matrix IMAGE") + frontend_config = recipe.get("frontend", {}) + if not isinstance(frontend_config, dict): + raise TypeError("frontend must be a mapping") + frontend = frontend_config.get("container_image") + images = [worker, *([frontend] if frontend is not None else [])] + if any( + not isinstance(image, str) or not image or any(c.isspace() for c in image) + for image in images + ): + raise ValueError("container identities must be non-empty strings without whitespace") + return list(dict.fromkeys(images)) + except (OSError, KeyError, TypeError, ValueError, yaml.YAMLError) as error: + raise ValueError(f"invalid recipe images for {config}: {error}") from error + + +def main() -> int: + """Print a JSON image list; failed resolution must stop the launcher before import.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("config") + parser.add_argument("expected_worker") + args = parser.parse_args() + try: + images = resolve_images(args.config, args.expected_worker) + except ValueError as error: + print(error, file=sys.stderr) + return 1 + print(json.dumps(images)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/inferencex-e2e/infx/tests/launch/fake_slurm.py b/inferencex-e2e/infx/tests/launch/fake_slurm.py index 851a62b06d..48403e2d72 100644 --- a/inferencex-e2e/infx/tests/launch/fake_slurm.py +++ b/inferencex-e2e/infx/tests/launch/fake_slurm.py @@ -28,13 +28,6 @@ *" init "*) mkdir -p "${@: -1}" ;; *" rev-parse HEAD"*) echo "$FAKE_SRT_COMMIT" ;; esac -""", - "uv": r""" -printf '%s\n' "$*" >> "$FAKE_LOG_DIR/uv.log" -if [[ "$1" == venv ]]; then - dest="${@: -1}"; mkdir -p "$dest/bin" - printf '#!/bin/bash\nexec "%s" "$@"\n' "$FAKE_PYTHON" > "$dest/bin/python"; chmod +x "$dest/bin/python" -fi """, "make": r""" printf '%s\n' "$*" >> "$FAKE_LOG_DIR/make.log" @@ -91,6 +84,24 @@ "rsync": r"""printf '%s\n' "$*" >> "$FAKE_LOG_DIR/rsync.log" """, } +_UV = r""" +import os, pathlib, subprocess, sys, sysconfig +argv = sys.argv[1:] +with open(os.path.join(os.environ["FAKE_LOG_DIR"], "uv.log"), "a") as handle: + handle.write(" ".join(argv) + "\n") +if argv[0] == "venv": + subprocess.run([sys.executable, "-m", "venv", "--without-pip", argv[-1]], check=True) +elif argv[:2] == ["pip", "install"]: + python = argv[argv.index("--python") + 1] + site = subprocess.check_output( + [python, "-c", "import sysconfig; print(sysconfig.get_path('purelib'))"], text=True + ).strip() + # Ordinary dependencies are shared, but srtctl is visible only in the job venv. + pathlib.Path(site, "fixture-dependencies.pth").write_text( + sysconfig.get_path("purelib") + "\n" + os.environ["FAKE_SRT_SOURCE"] + "/src\n" + ) +""" + _SRTCTL = r""" import json, os, pathlib, sys import yaml @@ -163,7 +174,7 @@ def install_fakes(directory: Path) -> Path: binary = directory / name binary.write_text(f"#!/bin/bash\n{body.strip()}\n") binary.chmod(0o755) - for name, body in {"srtctl": _SRTCTL, "sbatch": _SBATCH}.items(): + for name, body in {"uv": _UV, "srtctl": _SRTCTL, "sbatch": _SBATCH}.items(): binary = directory / name binary.write_text(f"#!{sys.executable}\n{body.lstrip()}") binary.chmod(0o755) @@ -241,12 +252,12 @@ def base_env(*, fakes: Path, logs: Path, workspace: Path, sandbox: Path) -> dict } env.update( PATH=f"{fakes}{os.pathsep}{Path(sys.executable).parent}{os.pathsep}/usr/bin{os.pathsep}/bin", - PYTHONPATH=f"{ROOT}{os.pathsep}{ROOT / 'utils/srt-slurm/src'}", + PYTHONPATH=str(ROOT), HOME=str(sandbox / "home"), USER="runner", GITHUB_WORKSPACE=str(workspace), FAKE_LOG_DIR=str(logs), - FAKE_PYTHON=sys.executable, + FAKE_SRT_SOURCE=str(ROOT / "utils/srt-slurm"), FAKE_SRT_COMMIT="0123456789abcdef0123456789abcdef01234567", ENROOT_IMPORT_TIME_LIMIT="10", SALLOC_TIME_LIMIT="10", diff --git a/inferencex-e2e/infx/tests/launch/test_squash_images.py b/inferencex-e2e/infx/tests/launch/test_squash_images.py index 1103521352..ddb2b54d0f 100644 --- a/inferencex-e2e/infx/tests/launch/test_squash_images.py +++ b/inferencex-e2e/infx/tests/launch/test_squash_images.py @@ -191,6 +191,29 @@ def test_enroot_uri_normalization(image, uri): assert enroot_uri(image) == uri +@pytest.mark.parametrize("registry", ["docker.io", "index.docker.io", "registry-1.docker.io"]) +@pytest.mark.parametrize("separator", ["/", "#"]) +@pytest.mark.parametrize(("repository", "resolved"), [ + ("team/image:dev", "team/image:dev"), + ("team/image:dev@sha256:" + "e" * 64, "team/image:sha256:" + "e" * 64), + ("nginx@sha256:" + "e" * 64, "library/nginx:sha256:" + "e" * 64), +]) # fmt: skip +def test_docker_hub_aliases_use_the_registry_api(registry, separator, repository, resolved): + assert enroot_uri(f"{registry}{separator}{repository}") == ( + f"docker://registry-1.docker.io#{resolved}" + ) + + +@pytest.mark.parametrize("mode", ["submit-host", "compute"]) +def test_digest_pinned_docker_hub_alias_reaches_the_importer(tools, tmp_path, mode): + _, logs = tools + image = "docker.io/team/router@sha256:" + "e" * 64 + squash = ensure_image(image, policy(tmp_path, mode), job=Job("7") if mode == "compute" else None) + [imported] = calls(logs["enroot"]) + assert imported["argv"][-1] == "docker://registry-1.docker.io#team/router:sha256:" + "e" * 64 + assert Path(squash).read_text() == VALID + + def test_squash_locations_override_the_cache_field_by_field(): squash = SquashCache.model_validate({ "dir": "/cache", "import": "compute", diff --git a/inferencex-e2e/infx/tests/launch/test_srt_config.py b/inferencex-e2e/infx/tests/launch/test_srt_config.py index 193281657e..23b5b6147a 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_config.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_config.py @@ -13,16 +13,19 @@ from infx.launch.backends.slurm import SlurmBackend from infx.launch.context import Launch from infx.launch.drivers.srt.checkout import Checkout, compute_workspace -from infx.launch.drivers.srt.config import SrtJob, pyxis_spelling, render, write +from infx.launch.drivers.srt.config import SrtJob, lane_mounts, pyxis_spelling, render, write +from infx.launch.drivers.srt.lanes import LaneMount, SrtLane from infx.launch.drivers.srt.recipe import HEALTH_ATTEMPTS, prepare_recipe from infx.launch.drivers.srt.run import SrtRun from infx.launch.lifecycle import Lifecycle -from infx.launch.policy import LaunchPath +from infx.launch.policy import LaunchPath, Match, any_of from infx.launch.request import SrtRequest ROOT = Path(__file__).resolve().parents[3] sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) from srtctl.core.config import resolve_config_with_defaults # noqa: E402 +from srtctl.core.schema import ClusterConfig # noqa: E402 +from srtctl.core.slurm import get_container_mounts_str # noqa: E402 def cluster(slurm: dict | None = None, srt: dict | None = None, entries: dict | None = None) -> Cluster: @@ -59,11 +62,15 @@ def test_the_profile_renders_its_facts_and_mounts_a_volume_at_a_second_target(): }, "volume-mounts": {"hub": "/hf_hub_cache/hub"}, "mounts": {"/dev/kfd": "/dev/kfd"}, - "extra": {"visible_devices_env": "ROCR_VISIBLE_DEVICES", "default_gpu_exporter": None}, + "extra": {"visible_devices_env": "ROCR_VISIBLE_DEVICES", "default_gpu_exporter": None, + "default_bash_preamble": "ulimit -l unlimited"}, }, entries={"Model-A": {"root": "data", "dir": "Model-A"}}, ) # fmt: skip config = render(record, job(mounts=[("/share/hub", "/mnt/hf_hub_cache/")], single_node=True)) + native = ClusterConfig.Schema().load(config) + assert native.default_bash_preamble == "ulimit -l unlimited" + assert native.network_interface == "eno0" assert config["default_mounts"] == { "/share/hub": "/hf_hub_cache/hub", "/dev/kfd": "/dev/kfd", @@ -88,6 +95,36 @@ def test_a_host_directory_cannot_be_mounted_at_three_targets(): render(record, job(mounts=[("/share/hub", "/a"), ("/share/hub/", "/b")])) +@pytest.mark.parametrize("asset_exists", [False, True]) +def test_read_only_lane_assets_are_forwarded_without_creating_or_chmodding_them( + tmp_path, asset_exists +): + asset = tmp_path / "provider.so" + if asset_exists: + asset.write_bytes(b"host-owned asset") + asset.chmod(0o440) + record = cluster(slurm={"volumes": { + "provider": {"path": str(asset), "visibility": "node-local"}, + }}) + lane = SrtLane(mounts=(LaneMount( + Match(frameworks=any_of("vllm-disagg")), "provider", read_only=True, + ),)) + run = SimpleNamespace(cluster=record, request=SimpleNamespace(framework="vllm-disagg")) + rendered = render(record, job(mounts=lane_mounts(run, lane))) + native = ClusterConfig.Schema().load(rendered) + command = get_container_mounts_str({ + Path(host): Path(target) for host, target in native.default_mounts.items() + }) + assert command == f"{asset}:{asset}:ro" + if asset_exists: + assert asset.read_bytes() == b"host-owned asset" + assert asset.stat().st_mode & 0o777 == 0o440 + else: + assert not asset.exists() + run.request.framework = "sglang" + assert lane_mounts(run, lane) == [] + + def test_node_exclusions_cpus_and_image_aliases_are_rendered(): record = cluster( slurm={"exclude": ["node-1", "node-2"], "cpus-per-task": 192, diff --git a/inferencex-e2e/infx/tests/launch/test_srt_driver.py b/inferencex-e2e/infx/tests/launch/test_srt_driver.py index 176f8cb71a..9d4e423307 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_driver.py @@ -226,6 +226,56 @@ def launch_here(monkeypatch, env: dict[str, str], config: Path, cwd: Path) -> in return main(["--runner-config", str(config), "run"]) +@pytest.mark.parametrize("worker", ["test:tag", "mismatched:tag"]) +def test_multinode_recipe_images_are_validated_before_import_and_frontend_is_staged( + harness, worker +): + recipe = yaml.safe_load(LANE_RECIPE) + router = "docker.io/example/router@sha256:" + "d" * 64 + recipe["frontend"] = {"container_image": router} + env = lane_env( + harness, + "lab-a", + yaml.safe_dump({"base": recipe, "override_image": {"model": {"container": worker}}}), + RUNNER_NAME="lab-a_00", + FRAMEWORK="vllm-disagg", + MODEL_PREFIX="kimik3", + MODEL="org/Model", + PRECISION="fp4", + CONFIG_FILE="recipes/test/lane.yaml:override_image", + ) + # A fresh launcher must not inherit test_srt_config's in-process srtctl imports. + completed = subprocess.run( + [sys.executable, "-c", """ +from infx.launch.drivers.srt import lanes +from infx.launch.drivers.srt.lanes import SrtLane +from infx.launch.policy import LaunchPath, Match +from infx.launch.__main__ import main +lanes.SRT_LANES[("lab-a", LaunchPath.SRT_MULTI)] = SrtLane(stage_recipe_images=Match()) +raise SystemExit(main()) +""", "--runner-config", str(lab_config(harness.tmp)), "run"], + env=env, cwd=harness.workspace, capture_output=True, text=True, timeout=120, + ) + rc = completed.returncode + imported = [line.split()[-1] for line in lines(harness.logs, "enroot")] + if worker != "test:tag": + assert rc == 1 + assert "must match" in completed.stderr + assert imported == [] + assert srtctl_calls(harness.logs) == [] + return + assert rc == 0, completed.stdout + completed.stderr + assert imported == [ + "docker://test:tag", + "docker://registry-1.docker.io#example/router:sha256:" + "d" * 64, + ] + config = srtslurm(harness.workspace) + assert config["containers"]["test:tag"].endswith("test_tag.sqsh") + assert config["containers"][router].endswith("docker.io_example_router_sha256_" + "d" * 64 + ".sqsh") + assert config["containers"][router] == config["containers"][router.replace("/", "#", 1)] + assert (harness.workspace / "multinode_server_logs.tar.gz").is_file() + + @pytest.mark.parametrize("cluster_id", LABS) def test_multinode_lane_stages_workflow_artifacts(harness, monkeypatch, cluster_id): lab = LABS[cluster_id] diff --git a/inferencex-e2e/infx/tests/launch/test_srt_recipe_edits.py b/inferencex-e2e/infx/tests/launch/test_srt_recipe_edits.py index 8cdf1aba3e..98373139db 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_recipe_edits.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_recipe_edits.py @@ -1,15 +1,19 @@ """Edits the multi-node lanes apply to the staged recipe copy.""" +from pathlib import Path + import pytest import yaml from infx.launch.drivers.srt.recipe import ( + RECIPES_MIRROR, add_dist_timeout, inject_concurrencies, parse_concurrencies, raise_health_attempts, rename_job, ) +from infx.srt_slurm.recipe_images import resolve_images RECIPE = """name: "upstream" roles: @@ -76,3 +80,42 @@ def test_conc_list_must_be_canonical_positive_integers(): for bad in ("", "08", "0", "-4", "4.0", "4 4", "+4"): with pytest.raises(ValueError): parse_concurrencies(bad) + + +def test_recipe_images_resolve_the_selected_override_and_deduplicate(tmp_path, monkeypatch): + monkeypatch.syspath_prepend(str(Path(__file__).resolve().parents[3] / "utils/srt-slurm/src")) + path = tmp_path / RECIPES_MIRROR / "images.yaml" + path.parent.mkdir(parents=True) + path.write_text( + yaml.safe_dump( + { + "base": { + "model": {"container": "worker:base"}, + "frontend": {"container_image": "router:v1"}, + }, + "override_changed": {"model": {"container": "worker:v2"}}, + "override_shared": {"frontend": {"container_image": "worker:base"}}, + } + ) + ) + assert resolve_images(f"{path}:override_changed", "worker:v2") == [ + "worker:v2", + "router:v1", + ] + assert resolve_images(f"{path}:override_shared", "worker:base") == [ + "worker:base" + ] + with pytest.raises(ValueError, match="exactly one"): + resolve_images(str(path), "worker:base") + with pytest.raises(ValueError, match="must match"): + resolve_images(f"{path}:override_changed", "worker:base") + + +@pytest.mark.parametrize("frontend", [{"container_image": "bad image"}, "not-a-mapping"]) +def test_recipe_images_reject_invalid_frontend_identities(tmp_path, frontend, monkeypatch): + monkeypatch.syspath_prepend(str(Path(__file__).resolve().parents[3] / "utils/srt-slurm/src")) + path = tmp_path / RECIPES_MIRROR / "images.yaml" + path.parent.mkdir(parents=True) + path.write_text(yaml.safe_dump({"model": {"container": "worker:v1"}, "frontend": frontend})) + with pytest.raises(ValueError, match="invalid recipe images"): + resolve_images(str(path), "worker:v1") From b54ebe9cac2adca3eba9e9f863a98c608c992830 Mon Sep 17 00:00:00 2001 From: Yichao Zhu Date: Wed, 30 Sep 2026 09:56:06 +0800 Subject: [PATCH 2/4] feat(config): add official-nightly Kimi-K3 FP32 PD MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pin the validated official ROCm nightly in both master and native recipe. Keep serving configuration separate from removable engine and discovery backports. 在主配置与原生 recipe 中固定已验证的官方 ROCm nightly;运行配置与可删除的 engine、discovery 补丁分层。 --- .../mi355x-fp4/agentx/disagg-variants.yaml | 222 ++++++++++++++++++ inferencex-e2e/configs/amd-master.yaml | 87 +++++++ .../docs/configuration-procedures.md | 2 + .../docs/configuration-procedures_zh.md | 2 + inferencex-e2e/docs/k3-pd-native.md | 33 +++ inferencex-e2e/docs/k3-pd-native_zh.md | 33 +++ inferencex-e2e/perf-changelog.yaml | 11 + 7 files changed, 390 insertions(+) create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml create mode 100644 inferencex-e2e/docs/k3-pd-native.md create mode 100644 inferencex-e2e/docs/k3-pd-native_zh.md diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml new file mode 100644 index 0000000000..dfa1a703ad --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml @@ -0,0 +1,222 @@ +# Native Kimi-K3 FP32 SSM PD on an immutable official ROCm nightly. +base: + schema: 2 + name: kimik3-fp4-mi355x-vllm-disagg-agentic + model: + path: Kimi-K3 + container: vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19 + precision: fp4 + slurm: + time_limit: '08:00:00' + sbatch_directives: + mem: '0' + srun_options: + container-remap-root: '' + container-writable: '' + mem: '0' + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: vllm-router + enable_multiple_frontends: false + orchestrator_placement: head + container_image: docker.io/vllm/vllm-router@sha256:1fbf06701abce8c8cd459414e472d63999c05b201baffc62e00727c10583f5ea + args: + policy: consistent_hash + prefill-policy: consistent_hash + decode-policy: consistent_hash + log-level: info + observability: + enabled: false + tachometer: + enabled: false + health_check: + interval_seconds: 10 + max_attempts: 90 + engine: + type: vllm + connector: moriio + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + args: &common_args + served-model-name: moonshotai/Kimi-K3 + trust-remote-code: true + tensor-parallel-size: 8 + decode-context-parallel-size: 1 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + moe-backend: auto + load-format: fastsafetensors + gpu-memory-utilization: 0.90 + language-model-only: true + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + max-model-len: 1048576 + stream-interval: 10 + enable-prefix-caching: true + prefix-match-unit: 128 + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: float32 + block-size: 128 + attention-backend: ROCM_AITER_MLA + attention-config: '{"mla_prefill_backend":"ROCM_AITER_FA","use_prefill_query_quantization":true}' + # The native srt driver supplies the measured probabilistic DSpark4 AL for + # throughput and leaves real block rejection for eval. Draft unchanged. + speculative-config: '{"model":"/models/Inferact-Kimi-K3-DSpark","method":"dspark","num_speculative_tokens":4,"attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"MoRIIOConnector","kv_role":"kv_producer","kv_load_failure_policy":"fail","kv_connector_extra_config":{"backend":"rdma"}},{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1799000000000,"lazy_offload":false}}]}}' + env: &common_env + VLLM_USE_V1: '1' + VLLM_ROCM_USE_AITER: '1' + VLLM_ROCM_USE_AITER_MOE_SITUV2_A8W4: '1' + AITER_SITUV2_A8W4: '1' + AITER_BF16_FP8_MOE_BOUND: '0' + VLLM_ROCM_AITER_MLA_ASM_PADDING: asm + VLLM_ROCM_AITER_MLA_DCP_VERIFY: asm + SAFETENSORS_FAST_GPU: '1' + VLLM_SSM_CONV_STATE_LAYOUT: DS + VLLM_KV_CACHE_LAYOUT: HND + NCCL_DMABUF_ENABLE: '0' + HSA_ENABLE_IPC_MODE_LEGACY: '1' + HIP_FORCE_DEV_KERNARG: '1' + PYTHONHASHSEED: '42' + PREFIX_CACHING_HASH_ALGO: sha256 + VLLM_USE_DIRECT_DCP_A2A: '0' + VLLM_USE_DIRECT_DCP_Q_GATHER: '0' + VLLM_USE_DIRECT_DCP_KV_GATHER: '0' + VLLM_ALLOW_DCP_FULL_CUDAGRAPH: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1200' + PYTHONNOUSERSITE: '1' + VLLM_SERVER_DEV_MODE: '0' + VLLM_DISABLE_REQUEST_ID_RANDOMIZATION: '1' + TORCH_NCCL_BLOCKING_WAIT: '0' + NCCL_BLOCKING_WAIT: '0' + MORI_IO_SQ_BACKOFF_TIMEOUT_US: '50000' + MORI_IO_QP_MAX_SEND_WR: '16384' + MORI_IO_QP_MAX_CQE: '32768' + MORI_IO_QP_MAX_SGE: '2' + MORI_IO_TC_DISABLE: '0' + decode: + nodes: 1 + workers: 1 + gpus: 8 + args: + <<: *common_args + kv-transfer-config: '{"kv_connector":"MoRIIOConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_connector_extra_config":{"backend":"rdma"}}' + env: + <<: *common_env + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + benchmark: + type: custom + client_placement: head + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + MODEL: moonshotai/Kimi-K3 + PORT: '8000' + IS_MULTINODE: 'true' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1799' + AIPERF_FAILED_REQUEST_THRESHOLD: '0.01' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.01' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + +override_fp32_c1: + roles: + prefill: + args: + max-num-seqs: 2 + max-num-batched-tokens: 16384 + speculative-config: '{"model":"/models/Inferact-Kimi-K3-DSpark","method":"dspark","num_speculative_tokens":7,"attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MoRIIOConnector","kv_role":"kv_producer","kv_load_failure_policy":"fail","kv_connector_extra_config":{"backend":"rdma"}}' + compilation-config: '{"mode":3,"cudagraph_mode":"PIECEWISE","max_cudagraph_capture_size":128,"cudagraph_capture_sizes":[1,16,32,64,128],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + env: + VLLM_USE_BREAKABLE_CUDAGRAPH: '1' + decode: + args: + max-num-seqs: 2 + max-num-batched-tokens: 512 + speculative-config: '{"model":"/models/Inferact-Kimi-K3-DSpark","method":"dspark","num_speculative_tokens":7,"attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":16,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + benchmark: + concurrencies: [1] + env: + KV_OFFLOADING: none + TOTAL_CPU_DRAM_GB: '0' + +override_fp32_c10: + roles: + prefill: + args: + max-num-seqs: 20 + max-num-batched-tokens: 8192 + compilation-config: '{"mode":3,"cudagraph_mode":"PIECEWISE","max_cudagraph_capture_size":128,"cudagraph_capture_sizes":[1,16,32,64,128],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + env: + VLLM_USE_BREAKABLE_CUDAGRAPH: '1' + decode: + args: + max-num-seqs: 20 + max-num-batched-tokens: 512 + compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":80,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + benchmark: + concurrencies: [10] + +override_fp32_c24: + roles: + prefill: + args: + decode-context-parallel-size: 8 + max-num-seqs: 48 + max-num-batched-tokens: 8192 + compilation-config: '{"mode":3,"cudagraph_mode":"NONE","max_cudagraph_capture_size":0,"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + env: + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + decode: + nodes: 2 + workers: 2 + args: + decode-context-parallel-size: 8 + max-num-seqs: 24 + max-num-batched-tokens: 512 + compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":120,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,70,80,90,100,110,120],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + benchmark: + concurrencies: [24] + +override_fp32_c48: &fp32_c48 + roles: &fp32_c48_roles + prefill: + args: + decode-context-parallel-size: 8 + max-num-seqs: 96 + max-num-batched-tokens: 8192 + compilation-config: '{"mode":3,"cudagraph_mode":"NONE","max_cudagraph_capture_size":0,"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + env: + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + # Short-tail segmented MLA can require ~4.1 GB of non-reclaimable scratch. + HSA_NO_SCRATCH_RECLAIM: '0' + decode: &fp32_c48_decode + args: &fp32_c48_decode_args + decode-context-parallel-size: 8 + max-num-seqs: 96 + max-num-batched-tokens: 512 + compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":240,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128,130,140,150,160,170,180,190,200,210,220,230,240],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + benchmark: + concurrencies: [48] + +override_fp32_c48_1p2d: + <<: *fp32_c48 + roles: + <<: *fp32_c48_roles + decode: + <<: *fp32_c48_decode + nodes: 2 + workers: 2 + args: + <<: *fp32_c48_decode_args + max-num-seqs: 48 diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index bb63c07808..f510584484 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -1526,3 +1526,90 @@ dsv41flash-fp4-mi355x-sglang-agentic-dspark: - dram-utilization: 0.60 search-space: - { tp: 4, ep: 4, dp-attn: false, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/mi355x-fp4-mtp/agentic.yaml } + +# Native Kimi-K3 PD with FP32 SSM state. +kimik3-fp4-mi355x-vllm-disagg-agentic: + image: vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19 + model: moonshotai/Kimi-K3 + model-prefix: kimik3 + runner: cluster:mi355x-amds + precision: fp4 + framework: vllm-disagg + router: { name: vllm-router, version: nightly-20260913-83944c4 } + kv-p2p-transfer: moriio + multinode: true + disagg: true + scenarios: + agentic-coding: + - dram-utilization: 0.6 + search-space: + - spec-decoding: mtp + conc-list: [1] + kv-offloading: none + prefill: + num-worker: 1 + tp: 8 + dcp-size: 1 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c1 + decode: { num-worker: 1, tp: 8, dcp-size: 1, ep: 1, dp-attn: false } + - spec-decoding: mtp + conc-list: [10] + kv-offloading: dram + kv-offload-backend: { name: vllm-simple } + prefill: + num-worker: 1 + tp: 8 + dcp-size: 1 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c10 + - TOTAL_CPU_DRAM_GB=1799 + decode: { num-worker: 1, tp: 8, dcp-size: 1, ep: 1, dp-attn: false } + - spec-decoding: mtp + conc-list: [24] + kv-offloading: dram + kv-offload-backend: { name: vllm-simple } + prefill: + num-worker: 1 + tp: 8 + dcp-size: 8 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c24 + - TOTAL_CPU_DRAM_GB=1799 + decode: { num-worker: 2, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } + - spec-decoding: mtp + conc-list: [48] + kv-offloading: dram + kv-offload-backend: { name: vllm-simple } + prefill: + num-worker: 1 + tp: 8 + dcp-size: 8 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c48 + # Recipe enables P-only scratch reclaim; D and GMU0.90 are unchanged. + - TOTAL_CPU_DRAM_GB=1799 + decode: { num-worker: 1, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } + - spec-decoding: mtp + conc-list: [48] + kv-offloading: dram + kv-offload-backend: { name: vllm-simple } + prefill: + num-worker: 1 + tp: 8 + dcp-size: 8 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c48_1p2d + # Inherits the same P-only scratch-reclaim setting as 1P1D c48. + - TOTAL_CPU_DRAM_GB=1799 + decode: { num-worker: 2, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 1ac592f020..20485c5995 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -107,6 +107,8 @@ and preserve resources used by other jobs. The Kimi-K3 native PD lane resolves the selected recipe before image provisioning, requires its worker image to match the matrix, and stages the recipe's frontend image through the same backend. This keeps router images recipe-owned and rejects mismatched inputs before an import allocation. Its staged target model, draft mount, fabric devices, worker network environment and memlock preamble are declared in the cluster record; image imports, submission, cancellation and artifact collection use the shared Python launcher. +The shared Slurm image resolver maps explicit Docker Hub hosts (`docker.io`, `index.docker.io`, and `registry-1.docker.io`) to Enroot's `registry-1.docker.io#repository` registry endpoint, for both slash and `#` input forms. Digest pins are preserved even when a tag is also present. Import validation must exercise the resolved reference against the registry with a cold, task-private cache; a cached squash or mocked import does not validate that boundary. + 1. [Prepare a worktree](#prepare-a-worktree) 2. [Add a model + hardware recipe](#add-a-model--hardware-recipe) 3. [Change a master config](#change-a-master-config) diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 90d8e8cb13..aa902074e9 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -89,6 +89,8 @@ PowerX 严格校验。现有的 Tachometer 1000 ms / 功耗 exporter 100 ms 采 Kimi-K3 原生 PD 路径在准备镜像前解析所选 recipe,要求 worker 镜像与矩阵一致,并通过同一后端准备 recipe 指定的 frontend 镜像。Router 镜像仍由 recipe 管理;输入不一致会在镜像导入任务申请资源前失败。已部署的目标模型、draft 挂载、网络设备、worker 网络环境及 memlock 前置命令由集群记录声明;镜像导入、任务提交、取消及产物收集沿用共享 Python launcher。 +共享 Slurm 镜像解析器将显式 Docker Hub 主机名(`docker.io`、`index.docker.io` 和 `registry-1.docker.io`)统一映射为 Enroot 的 `registry-1.docker.io#repository` 仓库端点,同时支持斜杠和 `#` 两种输入形式。即使输入还带有 tag,也会保留固定 digest。导入验证必须使用任务独享的空缓存,将解析后的引用交给真实仓库;命中已有 squash 缓存或模拟导入不能证明这一衔接有效。 + 1. [准备 worktree](#准备-worktree) 2. [添加模型 + 硬件配方](#添加模型--硬件配方) 3. [修改主配置](#修改主配置) diff --git a/inferencex-e2e/docs/k3-pd-native.md b/inferencex-e2e/docs/k3-pd-native.md new file mode 100644 index 0000000000..05ed0a79aa --- /dev/null +++ b/inferencex-e2e/docs/k3-pd-native.md @@ -0,0 +1,33 @@ +# Native Kimi-K3 PD on MI355X + +**English** | [中文](k3-pd-native_zh.md) + +This configuration uses InferenceX's Python launcher and native srt-slurm orchestration for Kimi-K3 prefill/decode disaggregation. The master config and selected recipe own benchmark topology and tuning; there is no alternate Bash launcher. + +## Configuration ownership + +- `configs/amd-master.yaml` selects `kimik3-fp4-mi355x-vllm-disagg-agentic` and supplies matrix identity and result metadata. +- `benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml` owns roles, graph settings, concurrency, the worker and router images, and FP32 SSM state. Target weights remain MXFP4, KV is FP8, and GPU memory utilization is 0.90. +- The draft checkpoint runs through the upstream DSpark path without weight conversion. Throughput uses InferenceX's measured golden acceptance selection; eval uses real verification. The latency variant uses DSpark K7 without CPU offload; other variants use K4 with prefill SimpleCPUOffload. +- `configs/runners.yaml` owns staged models, the draft mount, fabric devices, worker network settings, memlock and image-import policy. The named srt lane enables recipe-image staging for this workload. +- `infx/launch/` owns imports, setup, submission, cancellation and result preservation. Recipe images are resolved with the job-local srt-slurm Python environment, not the launcher interpreter. Worker-image disagreement is rejected before image import, and the selected recipe's router image is staged using the existing backend. + +The high-concurrency prefill configuration retains `HSA_NO_SCRATCH_RECLAIM=0`; decode is unchanged. That setting is an explicit runtime policy, not a temporary source patch. No BF16 SSM, workspace-development, READ-credit/QP or router-algorithm changes are introduced. + +## Official image and temporary integration + +The worker uses the official `vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d` image, pinned to amd64 digest `sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19`. The master and recipe use exactly the same identity. This replaces the custom worker build without a shared-MR modification. + +That nightly contains [vLLM#57700](https://github.com/vllm-project/vllm/pull/57700). Discovery templates also require [srt-slurm#508](https://github.com/NVIDIA/srt-slurm/pull/508), which is not in the pinned srt-slurm revision. Required engine backports and image/provider compatibility prerequisites belong in a separate, removable debug commit, not in the framework or configuration commits. + +The temporary changes are not merge-ready engine policy. Remove them once their upstream dependencies ship: + +1. Resolve an official ROCm nightly that contains the required engine fixes, update both worker-image references, and remove the corresponding temporary setup and patch assets after qualification. +2. Independently advance the srt-slurm submodule to a revision containing PR #508, then remove the temporary job-local cherry-pick and its tests. Updating the worker image does not update srt-slurm. +3. Append the performance changelog and validate the new source/image pair through the normal smoke, sweep and eval gates. Removing temporary patches makes the source stack clean; it does not replace runtime qualification or review. + +## Validation scope + +The Python-launcher migration is covered by behavior tests of selected images, pre-import rejection, real launch entrypoints with external scheduler/install commands stubbed, result staging, failure propagation and cancellation. Registry/source checks establish image identity and upstream inclusion, not device/provider compatibility or RDMA stability. + +The earlier [c48 run](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36556488243) used a different harness revision and custom image. It is historical evidence only, not qualification of this official-nightly candidate. No new GPU run, throughput, correctness or fully graceful shutdown result is claimed here. diff --git a/inferencex-e2e/docs/k3-pd-native_zh.md b/inferencex-e2e/docs/k3-pd-native_zh.md new file mode 100644 index 0000000000..19814c8ef9 --- /dev/null +++ b/inferencex-e2e/docs/k3-pd-native_zh.md @@ -0,0 +1,33 @@ +# MI355X 上的原生 Kimi-K3 PD + +[English](k3-pd-native.md) | **中文** + +此配置通过 InferenceX 的 Python launcher 和原生 srt-slurm 编排运行 Kimi-K3 prefill/decode 分离。主配置与所选 recipe 管理基准拓扑和调优,不增加另一套 Bash launcher。 + +## 配置归属 + +- `configs/amd-master.yaml` 选择 `kimik3-fp4-mi355x-vllm-disagg-agentic`,提供矩阵标识和结果元数据。 +- `benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml` 管理角色、图配置、并发、worker/router 镜像及 FP32 SSM 状态。目标权重仍为 MXFP4,KV 为 FP8,GPU memory utilization 为 0.90。 +- Draft checkpoint 使用上游 DSpark 路径,不转换权重精度。吞吐由 InferenceX 自动选择实测 golden acceptance,eval 保持真实校验。低延迟配置使用 DSpark K7 且无 CPU offload;其余配置使用 K4 与 prefill SimpleCPUOffload。 +- `configs/runners.yaml` 管理已部署模型、draft 挂载、网络设备、worker 网络环境、memlock 和镜像导入策略。具名 srt 路径为该工作负载启用 recipe 镜像准备。 +- `infx/launch/` 管理导入、安装、提交、取消和结果保留。Recipe 镜像由任务私有的 srt-slurm Python 环境解析,不在 launcher 解释器中导入依赖。Worker 镜像不一致会在导入前失败,所选 recipe 的 router 镜像也通过现有后端准备。 + +高并发 prefill 保留 `HSA_NO_SCRATCH_RECLAIM=0`,decode 不变。这是显式运行策略,不是临时源码补丁。不引入 BF16 SSM、workspace 开发、READ-credit/QP 或 router 算法改动。 + +## 官方镜像与临时集成 + +Worker 使用官方 `vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d`,固定 amd64 digest 为 `sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19`。主配置和 recipe 使用完全相同的镜像标识,替换原自定义构建,不携带 shared-MR 修改。 + +该 nightly 已包含 [vLLM#57700](https://github.com/vllm-project/vllm/pull/57700)。Discovery 模板还依赖尚未进入 srt-slurm pin 的 [srt-slurm#508](https://github.com/NVIDIA/srt-slurm/pull/508)。必需的 engine backport 和镜像/provider 兼容前置放在单独、可删除的 debug commit 中,不混入框架或配置提交。 + +这些临时修改不符合直接合入的引擎补丁政策;待上游发布后删除: + +1. 确认官方 ROCm nightly 包含必需的 engine 修复,同步更新 worker 镜像的两处引用,验收后删除对应临时 setup 和 patch 文件。 +2. 独立升级 srt-slurm 子模块到包含 PR #508 的版本,删除临时的任务内 cherry-pick 及其测试。替换 worker 镜像不会升级 srt-slurm。 +3. 追加性能 changelog,并按正常 smoke、sweep、eval 门槛验证新的源码与镜像组合。删除临时补丁只能让源码栈干净,不能替代运行验收和 review。 + +## 验证范围 + +Python launcher 迁移的行为测试覆盖镜像选择、导入前拒绝错误输入、使用外部调度/安装桩运行真实启动入口、结果归档、失败传播及取消。镜像仓库和源码检查证明镜像身份与上游能力包含关系,不证明设备/provider 兼容性或 RDMA 稳定性。 + +此前的 [c48 run](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36556488243) 使用不同 harness 版本和自定义镜像,仅作历史记录,不是本次官方 nightly 候选的验收证据。此处不宣称新的 GPU 实测、吞吐、精度或完全优雅退出结果。 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index db5eaa6965..44dfa3983f 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9211,3 +9211,14 @@ - "Add --attention-config '{\"indexer_kv_dtype\":\"mxfp4\",\"indexer_sparse_logits\":true}' and --block-size 128 to enable the vllm-project/vllm#58671 ROCm paged MXFP4 sparse-logits indexer, replacing the dense fp8 indexer path. A live A/B test (TP2 c16, matched 900s window, vllm-project/vllm#58208 reverted via vllm-project/vllm#59125 so the dense fallback doesn't crash) measured +14.7/+14.9% p50/p90 interactivity and -9.5/-10.8% p50/p90 e2e latency over the dense path, with throughput/GPU unchanged." - "Drop c128 from both TP2 and TP4. Neither c128 point was on the Pareto frontier in #3555's run 36528242520: TP2 c64 dominated both (P90 E2EL 58 s against 185 s and 85 s, at 111k against 85k and 79k total tok/s/GPU)." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3571 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Add native Kimi-K3 PD through the Python launcher with FP32 SSM, FP8 KV, GMU0.90 and upstream DSpark verification. Preserve the recipe-owned concurrency and topology variants." + - "Use official ROCm nightly 36768d1bfd39094681cdbc8cb37d4b31c0729c89 at amd64 digest sha256:50b4f2aec13ffcb11ed846fec049d72250a9c2421eb1ad24c95a5210d46cee4c instead of the custom image; omit shared-MR and the candidate provider bind. Runtime qualification of this image is pending." + - "通过 Python launcher 增加原生 Kimi-K3 PD,使用 FP32 SSM、FP8 KV、GMU0.90 和上游 DSpark 校验,保留 recipe 管理的并发与拓扑配置。" + - "使用固定 digest 的官方 ROCm nightly 替换自定义镜像,不携带 shared-MR 或候选 provider bind;新镜像仍待运行验收。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 From c04a832b253d41a3540710ec6c8ab9c0edddfd15 Mon Sep 17 00:00:00 2001 From: Yichao Zhu Date: Fri, 2 Oct 2026 15:12:40 +0800 Subject: [PATCH 3/4] debug: isolate READ zeroing and synthetic draft compatibility MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将临时 engine 栈收束为固定 #59164 与仅用于 synthetic 的 draft gather 回退,移除 EOF;保留独立的 #508 和 provider 兼容层。 --- .../configs/k3-moriio-debug.sh | 34 +++++++ .../configs/k3-moriio-sync-zeroing.patch | 92 +++++++++++++++++++ .../k3-synthetic-unproposed-drafts.patch | 32 +++++++ .../mi355x-fp4/agentx/disagg-variants.yaml | 2 + inferencex-e2e/configs/runners.yaml | 2 + .../docs/configuration-procedures.md | 4 + .../docs/configuration-procedures_zh.md | 4 + inferencex-e2e/docs/k3-pd-native.md | 46 ++++++---- inferencex-e2e/docs/k3-pd-native_zh.md | 46 ++++++---- .../infx/launch/drivers/srt/checkout.py | 18 ++++ .../infx/launch/drivers/srt/lanes.py | 6 ++ .../infx/tests/launch/test_k3_moriio_setup.py | 91 ++++++++++++++++++ .../infx/tests/launch/test_srt_debug_prs.py | 46 ++++++++++ inferencex-e2e/perf-changelog.yaml | 80 ++++++++++++++++ 14 files changed, 471 insertions(+), 32 deletions(-) create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-debug.sh create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-sync-zeroing.patch create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-synthetic-unproposed-drafts.patch create mode 100644 inferencex-e2e/infx/tests/launch/test_k3_moriio_setup.py create mode 100644 inferencex-e2e/infx/tests/launch/test_srt_debug_prs.py diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-debug.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-debug.sh new file mode 100644 index 0000000000..f81cef4912 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-debug.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +set -eo pipefail + +apply_verified_patch() { + local package_root="$1" patch_file="$2" checksum="$3" + printf '%s %s\n' "$checksum" "$patch_file" | sha256sum --check --status + git -C "$package_root" apply --check --include='vllm/*' "$patch_file" + git -C "$package_root" apply --include='vllm/*' "$patch_file" +} + +main() { + # Native srt-slurm runs this preamble in both worker and router containers. + # Only the standalone router image is exempt; a worker with missing vLLM must fail. + if command -v vllm-router >/dev/null 2>&1 && ! command -v vllm >/dev/null 2>&1; then + echo 'Standalone vLLM Router image: no engine backport required' + exit 0 + fi + + # Keep the pinned #59164 backport plus the synthetic-only + # compatibility experiment in the installed wheel; no compiled replacement. + package_root=$(python3 -c 'import importlib.util; from pathlib import Path; spec = importlib.util.find_spec("vllm"); assert spec and spec.submodule_search_locations, "vLLM package not found"; print(Path(next(iter(spec.submodule_search_locations))).parent)') + # Pin the latest reviewed #59164 head; its runtime diff is equivalent to + # the previously validated d5e6faa9 extraction. + zeroing_patch="$(dirname -- "${BASH_SOURCE[0]}")/k3-moriio-sync-zeroing.patch" + apply_verified_patch "$package_root" "$zeroing_patch" 3da3746e85d53e4a3b17062b4113475d31a86cc07418e87ef5b4bb0c906cad20 + echo 'Applied vLLM#59164 at c5b1350f1f2bf10a128127a9b85e93d5f9f18e62' + synthetic_patch="$(dirname -- "${BASH_SOURCE[0]}")/k3-synthetic-unproposed-drafts.patch" + apply_verified_patch "$package_root" "$synthetic_patch" 33a6f792b13d39705a50562ca037a1d3c49dc054a8bdd8539fd8a154667f39df + echo 'Applied synthetic-only pre-#58784 draft gathering; real verification retains placeholder rejection' +} + +if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then + main "$@" +fi diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-sync-zeroing.patch b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-sync-zeroing.patch new file mode 100644 index 0000000000..83e2e1d6cc --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-sync-zeroing.patch @@ -0,0 +1,92 @@ +diff --git a/vllm/distributed/kv_transfer/kv_connector/v1/base.py b/vllm/distributed/kv_transfer/kv_connector/v1/base.py +index 6df9d1cfb..192b4ea91 100644 +--- a/vllm/distributed/kv_transfer/kv_connector/v1/base.py ++++ b/vllm/distributed/kv_transfer/kv_connector/v1/base.py +@@ -543,6 +543,16 @@ class KVConnectorBase_V1(ABC): + """ + pass + ++ def get_sync_load_block_ids(self, request: "Request") -> list[int]: ++ """Return blocks whose synchronous load replaces worker zeroing. ++ ++ Called after update_state_after_alloc. Each returned block must be ++ fully initialized before forward consumes it. A failed load must abort ++ forward instead of recomputing with uninitialized blocks. ++ Defaults to no blocks. ++ """ ++ return [] ++ + @abstractmethod + def build_connector_meta( + self, scheduler_output: SchedulerOutput +diff --git a/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py b/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py +index b482e6a8c..a6d7ffedc 100644 +--- a/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py ++++ b/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py +@@ -303,6 +303,24 @@ class MoRIIOConnector(KVConnectorBase_V1, SupportsHMA): + assert self.connector_scheduler is not None + self.connector_scheduler.on_new_request(request) + ++ def get_sync_load_block_ids(self, request: "Request") -> list[int]: ++ scheduler = self.connector_scheduler ++ if ( ++ self.mode != MoRIIOMode.READ ++ or not self.kv_transfer_config.is_kv_consumer ++ or scheduler is None ++ or not scheduler._has_mamba ++ or self._vllm_config.cache_config.get_resolved_kv_cache_layout().name ++ not in ("LBHNC", "LBNHC") ++ ): ++ return [] ++ pending = scheduler._reqs_need_recv.get(request.request_id) ++ if pending is None: ++ return [] ++ # Hybrid READ fills these entire attention pages and aborts on failure. ++ # The scheduler excludes these IDs only from newly allocated page zeroing. ++ return pending[1][0] ++ + def build_connector_meta( + self, + scheduler_output: SchedulerOutput, +diff --git a/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py b/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py +index 73b6b0ccf..5cb75e17a 100644 +--- a/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py ++++ b/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py +@@ -463,6 +463,12 @@ class MultiConnector(KVConnectorBase_V1, SupportsHMA): + for c in self._connectors: + c.on_new_request(request) + ++ def get_sync_load_block_ids(self, request: "Request") -> list[int]: ++ chosen = self._requests_to_connector.get(request.request_id) ++ if chosen is None: ++ return [] ++ return self._connectors[chosen].get_sync_load_block_ids(request) ++ + def build_connector_meta( + self, scheduler_output: SchedulerOutput + ) -> MultiKVConnectorMetadata: +diff --git a/vllm/v1/core/sched/scheduler.py b/vllm/v1/core/sched/scheduler.py +index fcb1421b1..a86e0810a 100644 +--- a/vllm/v1/core/sched/scheduler.py ++++ b/vllm/v1/core/sched/scheduler.py +@@ -349,7 +349,7 @@ class Scheduler(SchedulerInterface): + + self.has_mamba_layers = kv_cache_config.has_mamba_layers + self.needs_kv_cache_zeroing = kv_cache_config.needs_kv_cache_zeroing +- # Blocks that async KV loads will overwrite this step, skipped from ++ # Blocks that KV loads will overwrite this step, skipped from + # zeroing since the zeroing could race the out-of-band write. + self._skip_zero_block_ids: set[int] = set() + self.need_mamba_block_aligned_split = ( +@@ -1295,6 +1295,11 @@ class Scheduler(SchedulerInterface): + if num_external_computed_tokens > 0: + # load_kv_async is False here + has_sync_kv_loads = True ++ if self.needs_kv_cache_zeroing: ++ assert self.connector is not None ++ self._skip_zero_block_ids.update( ++ self.connector.get_sync_load_block_ids(request) ++ ) + if self.log_stats: + request.record_event( + EngineCoreEventType.SCHEDULED, scheduled_timestamp diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-synthetic-unproposed-drafts.patch b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-synthetic-unproposed-drafts.patch new file mode 100644 index 0000000000..0e51482e5d --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-synthetic-unproposed-drafts.patch @@ -0,0 +1,32 @@ +diff --git a/vllm/v1/worker/gpu/spec_decode/rejection_sampler.py b/vllm/v1/worker/gpu/spec_decode/rejection_sampler.py +--- a/vllm/v1/worker/gpu/spec_decode/rejection_sampler.py ++++ b/vllm/v1/worker/gpu/spec_decode/rejection_sampler.py +@@ -365,14 +365,20 @@ class RejectionSampler: + # that num_nans is computed before applying penalties and temperature. + num_nans = get_num_nans(logits) if self.sampler.compute_nans else None + +- draft_sampled, pos = gather_draft_sampled( +- input_batch.input_ids, +- input_batch.positions, +- input_batch.logits_indices, +- input_batch.expanded_idx_mapping, +- input_batch.expanded_local_pos, +- self.sampler.req_states.prefill_len.gpu, +- ) ++ if self.synthetic_conditional_rates is not None: ++ # Benchmark compatibility: restore pre-#58784 synthetic gathering. ++ # Real verification must still reject never-proposed draft slots. ++ draft_sampled = input_batch.input_ids[input_batch.logits_indices] ++ pos = input_batch.positions[input_batch.logits_indices] ++ else: ++ draft_sampled, pos = gather_draft_sampled( ++ input_batch.input_ids, ++ input_batch.positions, ++ input_batch.logits_indices, ++ input_batch.expanded_idx_mapping, ++ input_batch.expanded_local_pos, ++ self.sampler.req_states.prefill_len.gpu, ++ ) + + max_num_logprobs = self.sampler.sampling_states.max_num_logprobs( + input_batch.idx_mapping_np diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml index dfa1a703ad..6319298c83 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml @@ -2,6 +2,8 @@ base: schema: 2 name: kimik3-fp4-mi355x-vllm-disagg-agentic + # Temporary #59164 and K3 EOF backports; remove as qualified nightlies ship them. + setup_script: k3-moriio-debug.sh model: path: Kimi-K3 container: vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19 diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index 122c13b33c..6e96de0714 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -861,6 +861,8 @@ clusters: aiperf-cache: {path: /it-share/aiperf-cache} it-share-data: {path: /it-share/data} k3-draft: {path: /it-share/data/Inferact-Kimi-K3-DSpark} + # Selected only by the Kimi-K3 vLLM PD lane; never replace libibverbs core. + ionic-provider: {path: /usr/lib/x86_64-linux-gnu/libibverbs/libionic-rdmav34.so, visibility: node-local} squash: dir: /it-share/gharunners2/srt-slurm/containers visibility: shared diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 20485c5995..10a56bc5b2 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -107,6 +107,10 @@ and preserve resources used by other jobs. The Kimi-K3 native PD lane resolves the selected recipe before image provisioning, requires its worker image to match the matrix, and stages the recipe's frontend image through the same backend. This keeps router images recipe-owned and rejects mismatched inputs before an import allocation. Its staged target model, draft mount, fabric devices, worker network environment and memlock preamble are declared in the cluster record; image imports, submission, cancellation and artifact collection use the shared Python launcher. +The Kimi-K3 vLLM debug setup uses digest-pinned official ROCm nightly `ac68c3087215e0a4f3cdfa218508c6aada57235d`, which already includes #57700, with only #59164 READ-zeroing and the K3 streaming-EOF runtime subset. It omits the heartbeat and other #58968 feature groups. The pinned revisions, tested scope and independent removal conditions are maintained in [Native Kimi-K3 PD](k3-pd-native.md). Serving settings, automatic DSpark acceptance and benchmark criteria are unchanged. + +The Kimi-K3 vLLM PD lane also selects the cluster's node-local `ionic-provider` volume as a read-only single-file mount. This is a temporary compatibility prerequisite for the pinned image's Ionic userspace provider and the host kernel ABI, independent of shared-MR. Other models and frameworks do not select it. Read-only lane assets are neither created nor chmodded on the launch host; the container runtime validates them on the allocated compute node. Keep the image's `libibverbs` core and all unrelated providers unchanged. Remove the volume and its lane entry only after the replacement image, without the mount, enumerates and opens/closes every expected RDMA device on the target fleet. Device-open success is not MR, traffic, throughput, or shutdown qualification. + The shared Slurm image resolver maps explicit Docker Hub hosts (`docker.io`, `index.docker.io`, and `registry-1.docker.io`) to Enroot's `registry-1.docker.io#repository` registry endpoint, for both slash and `#` input forms. Digest pins are preserved even when a tag is also present. Import validation must exercise the resolved reference against the registry with a cold, task-private cache; a cached squash or mocked import does not validate that boundary. 1. [Prepare a worktree](#prepare-a-worktree) diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index aa902074e9..7adaa82687 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -89,6 +89,10 @@ PowerX 严格校验。现有的 Tachometer 1000 ms / 功耗 exporter 100 ms 采 Kimi-K3 原生 PD 路径在准备镜像前解析所选 recipe,要求 worker 镜像与矩阵一致,并通过同一后端准备 recipe 指定的 frontend 镜像。Router 镜像仍由 recipe 管理;输入不一致会在镜像导入任务申请资源前失败。已部署的目标模型、draft 挂载、网络设备、worker 网络环境及 memlock 前置命令由集群记录声明;镜像导入、任务提交、取消及产物收集沿用共享 Python launcher。 +Kimi-K3 vLLM debug setup 使用按 digest 固定且已包含 #57700 的官方 ROCm nightly `ac68c3087215e0a4f3cdfa218508c6aada57235d`,仅叠加 #59164 READ-zeroing 和 K3 streaming-EOF 运行时子集,不携带心跳或 #58968 其他特性组。固定版本、已验证范围及独立删除条件统一记录于[原生 Kimi-K3 PD](k3-pd-native_zh.md)。Serving 设置、自动选择的 DSpark acceptance 及 benchmark 判据不变。 + +Kimi-K3 vLLM PD 路径还会选择集群声明的节点本地 `ionic-provider` 卷,以只读方式挂载单个文件。这是固定镜像内的 Ionic 用户态 provider 与主机内核 ABI 之间的临时兼容前置条件,与 shared-MR 无关;其他模型和框架不选择此挂载。Launcher 不会在提交主机上创建只读资产或修改其权限,由容器运行时在已分配的计算节点上验证。保留镜像原有的 `libibverbs` 核心库及其他 provider。只有替换镜像在不挂载 provider 的情况下,能在目标集群枚举并成功打开、关闭全部预期 RDMA 设备,才可移除该卷及路径选择项。设备打开成功不能代替 MR、实际传输、吞吐或退出验证。 + 共享 Slurm 镜像解析器将显式 Docker Hub 主机名(`docker.io`、`index.docker.io` 和 `registry-1.docker.io`)统一映射为 Enroot 的 `registry-1.docker.io#repository` 仓库端点,同时支持斜杠和 `#` 两种输入形式。即使输入还带有 tag,也会保留固定 digest。导入验证必须使用任务独享的空缓存,将解析后的引用交给真实仓库;命中已有 squash 缓存或模拟导入不能证明这一衔接有效。 1. [准备 worktree](#准备-worktree) diff --git a/inferencex-e2e/docs/k3-pd-native.md b/inferencex-e2e/docs/k3-pd-native.md index 05ed0a79aa..a7e953d3d2 100644 --- a/inferencex-e2e/docs/k3-pd-native.md +++ b/inferencex-e2e/docs/k3-pd-native.md @@ -2,32 +2,46 @@ **English** | [中文](k3-pd-native_zh.md) -This configuration uses InferenceX's Python launcher and native srt-slurm orchestration for Kimi-K3 prefill/decode disaggregation. The master config and selected recipe own benchmark topology and tuning; there is no alternate Bash launcher. +This draft integrates Kimi-K3 prefill/decode disaggregation through InferenceX's shared Python launcher and native srt-slurm lifecycle. There is no alternate launcher or router algorithm. ## Configuration ownership - `configs/amd-master.yaml` selects `kimik3-fp4-mi355x-vllm-disagg-agentic` and supplies matrix identity and result metadata. -- `benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml` owns roles, graph settings, concurrency, the worker and router images, and FP32 SSM state. Target weights remain MXFP4, KV is FP8, and GPU memory utilization is 0.90. -- The draft checkpoint runs through the upstream DSpark path without weight conversion. Throughput uses InferenceX's measured golden acceptance selection; eval uses real verification. The latency variant uses DSpark K7 without CPU offload; other variants use K4 with prefill SimpleCPUOffload. -- `configs/runners.yaml` owns staged models, the draft mount, fabric devices, worker network settings, memlock and image-import policy. The named srt lane enables recipe-image staging for this workload. -- `infx/launch/` owns imports, setup, submission, cancellation and result preservation. Recipe images are resolved with the job-local srt-slurm Python environment, not the launcher interpreter. Worker-image disagreement is rejected before image import, and the selected recipe's router image is staged using the existing backend. +- `benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml` owns the topology, worker/router images and serving settings: MXFP4 target weights, FP32 SSM, FP8 KV and GMU 0.90. +- The existing variants remain 1P1D c1/c10/c48 and 1P2D c24/c48. The latency variant uses DSpark K7 without CPU offload; the others use K4 with prefill SimpleCPUOffload. Draft weights and precision are unchanged. Throughput uses automatic golden acceptance selection; eval uses real verification. +- `configs/runners.yaml` declares staged models, draft/fabric mounts, worker network settings, memlock and image-import policy. `infx/launch/` uses the job-local srt-slurm Python to resolve recipe images, rejects worker-image disagreement before import, stages the recipe-owned router image, and retains shared submission, cancellation and result collection. -The high-concurrency prefill configuration retains `HSA_NO_SCRATCH_RECLAIM=0`; decode is unchanged. That setting is an explicit runtime policy, not a temporary source patch. No BF16 SSM, workspace-development, READ-credit/QP or router-algorithm changes are introduced. +High-concurrency prefill retains `HSA_NO_SCRATCH_RECLAIM=0`. Existing graph modes, transport settings and 1% request-error gates are unchanged. No BF16 SSM, workspace-development, shared-MR, QP/credit implementation or client cancellation patch is added. -## Official image and temporary integration +## Pinned runtime and removable debug layer -The worker uses the official `vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d` image, pinned to amd64 digest `sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19`. The master and recipe use exactly the same identity. This replaces the custom worker build without a shared-MR modification. +The worker image is `vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19`. Master and recipe use the same immutable identity. This image already includes vLLM #57700; no backport of that PR remains. -That nightly contains [vLLM#57700](https://github.com/vllm-project/vllm/pull/57700). Discovery templates also require [srt-slurm#508](https://github.com/NVIDIA/srt-slurm/pull/508), which is not in the pinned srt-slurm revision. Required engine backports and image/provider compatibility prerequisites belong in a separate, removable debug commit, not in the framework or configuration commits. +The temporary engine setup applies two checked-in Python runtime patches: synchronous READ zeroing and synthetic-only draft gathering. It verifies SHA256 and `git apply --check` before applying each, without replacing compiled extensions or downloading a moving PR head. The parser is the unmodified nightly implementation; no EOF backport is applied. A standalone router skips engine setup; a worker with missing vLLM or an incompatible patch fails startup. -The temporary changes are not merge-ready engine policy. Remove them once their upstream dependencies ship: +| Dependency | Pinned source | Purpose | +| --- | --- | --- | +| [vLLM #59164](https://github.com/vllm-project/vllm/pull/59164) | `c5b1350f1f2bf10a128127a9b85e93d5f9f18e62` | Exclude synchronous READ destinations from newly allocated KV-page zeroing. | +| Synthetic draft-gathering experiment | `k3-synthetic-unproposed-drafts.patch`, against the pinned nightly | Restore pre-#58784 input gathering only when synthetic acceptance is active. Normal/block verification still rejects never-proposed slots. This is benchmark compatibility, not a general correctness fix. | +| [srt-slurm #508](https://github.com/NVIDIA/srt-slurm/pull/508) | tested revision `51cee8904a0b402a834887a26008adb79b8cd26b` | Bind discovery topology inside connector templates in the matching job's disposable checkout. | -1. Resolve an official ROCm nightly that contains the required engine fixes, update both worker-image references, and remove the corresponding temporary setup and patch assets after qualification. -2. Independently advance the srt-slurm submodule to a revision containing PR #508, then remove the temporary job-local cherry-pick and its tests. Updating the worker image does not update srt-slurm. -3. Append the performance changelog and validate the new source/image pair through the normal smoke, sweep and eval gates. Removing temporary patches makes the source stack clean; it does not replace runtime qualification or review. +READ-zeroing patch hash: `3da3746e85d53e4a3b17062b4113475d31a86cc07418e87ef5b4bb0c906cad20`. The setup does not include EOF, heartbeat, FULL-context, draft-fence or transfer-ownership patches from #58968. -## Validation scope +Synthetic experiment hash: `33a6f792b13d39705a50562ca037a1d3c49dc054a8bdd8539fd8a154667f39df`. The automatic golden acceptance curve, draft model/count, parser and error gates are unchanged. This restores historical treatment of bootstrap placeholders, which can change the first visible token and subsequent generation. Its results require separate interpretation from ordinary model-quality evaluation and are not yet qualified as performance-equivalent. -The Python-launcher migration is covered by behavior tests of selected images, pre-import rejection, real launch entrypoints with external scheduler/install commands stubbed, result staging, failure propagation and cancellation. Registry/source checks establish image identity and upstream inclusion, not device/provider compatibility or RDMA stability. +The debug layer also retains the cluster-declared, read-only single-file `ionic-provider` mount. This is image/kernel ABI compatibility, not shared-MR. Keep the image's libibverbs core and unrelated providers unchanged. The srt-slurm dependency URL, submodule pin and AIPerf source are not replaced by this integration. -The earlier [c48 run](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36556488243) used a different harness revision and custom image. It is historical evidence only, not qualification of this official-nightly candidate. No new GPU run, throughput, correctness or fully graceful shutdown result is claimed here. +## Removal conditions + +1. Once a qualified official worker image contains #59164, update master and recipe together and remove that patch/application step. Remove the separate synthetic experiment once its compatibility question is resolved; it is not an upstream fix awaiting inclusion. Delete the worker setup script, recipe reference and its tests once no runtime patch remains. +2. Independently update the shared srt-slurm pin to a revision containing #508, then remove the job-local cherry-pick and its focused tests. A worker-image update does not upgrade srt-slurm. +3. Remove the provider mount only after the replacement image opens all expected RDMA devices without it on the target fleet. Device enumeration alone does not qualify RDMA traffic. +4. Append the performance changelog and complete the applicable smoke, sweep and eval gates before merge. These temporary engine patches remain a draft debug integration, not an exception to upstream-image policy. + +## Evidence and remaining qualification + +EOF is excluded from the current candidate. Earlier EOF regression and throughput results remain historical evidence for a different parser stack, not qualification of this candidate. + +The matched [1P1D c48 one-hour run](https://github.com/billishyahao/InferenceMINI/actions/runs/36847087330) used the same immutable worker image, #59164 and byte-identical EOF runtime. Warmup: 531 valid / 0 empty; profile: 4040 valid / 4 empty (0.0989%). Submission, result export and native cleanup completed. This is supporting runtime evidence from a separately adapted harness, not a sweep on this PR's current commit or an accuracy certification. + +The earlier [InferenceX full-feature control](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36800732704) used a broader stack. The later #59164-only [InferenceX run](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36814088271) was cancelled while queued, without a GPU runner. Neither qualifies the consolidated candidate. The minimal combination still needs current-PR 1P2D, sweep/eval and loaded-service teardown qualification; omission of other independently identified fixes is not proof that their failure modes are impossible. diff --git a/inferencex-e2e/docs/k3-pd-native_zh.md b/inferencex-e2e/docs/k3-pd-native_zh.md index 19814c8ef9..9e277aab44 100644 --- a/inferencex-e2e/docs/k3-pd-native_zh.md +++ b/inferencex-e2e/docs/k3-pd-native_zh.md @@ -2,32 +2,46 @@ [English](k3-pd-native.md) | **中文** -此配置通过 InferenceX 的 Python launcher 和原生 srt-slurm 编排运行 Kimi-K3 prefill/decode 分离。主配置与所选 recipe 管理基准拓扑和调优,不增加另一套 Bash launcher。 +此 draft 通过 InferenceX 共享 Python launcher 和原生 srt-slurm 生命周期接入 Kimi-K3 prefill/decode 分离,不增加另一套 launcher 或 router 算法。 ## 配置归属 - `configs/amd-master.yaml` 选择 `kimik3-fp4-mi355x-vllm-disagg-agentic`,提供矩阵标识和结果元数据。 -- `benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml` 管理角色、图配置、并发、worker/router 镜像及 FP32 SSM 状态。目标权重仍为 MXFP4,KV 为 FP8,GPU memory utilization 为 0.90。 -- Draft checkpoint 使用上游 DSpark 路径,不转换权重精度。吞吐由 InferenceX 自动选择实测 golden acceptance,eval 保持真实校验。低延迟配置使用 DSpark K7 且无 CPU offload;其余配置使用 K4 与 prefill SimpleCPUOffload。 -- `configs/runners.yaml` 管理已部署模型、draft 挂载、网络设备、worker 网络环境、memlock 和镜像导入策略。具名 srt 路径为该工作负载启用 recipe 镜像准备。 -- `infx/launch/` 管理导入、安装、提交、取消和结果保留。Recipe 镜像由任务私有的 srt-slurm Python 环境解析,不在 launcher 解释器中导入依赖。Worker 镜像不一致会在导入前失败,所选 recipe 的 router 镜像也通过现有后端准备。 +- `benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml` 管理拓扑、worker/router 镜像及 serving 设置:MXFP4 目标权重、FP32 SSM、FP8 KV 和 GMU 0.90。 +- 保留现有 1P1D c1/c10/c48 与 1P2D c24/c48。低延迟配置使用 DSpark K7 且不做 CPU offload;其他配置使用 K4 和 prefill SimpleCPUOffload。不改变 draft 权重或精度;吞吐自动选择 golden acceptance,eval 使用真实校验。 +- `configs/runners.yaml` 声明已部署模型、draft/fabric 挂载、worker 网络设置、memlock 和镜像导入策略。`infx/launch/` 使用任务私有 srt-slurm Python 解析 recipe 镜像,导入前拒绝 worker 镜像不一致,准备 recipe 指定的 router 镜像,并沿用共享提交、取消和结果收集。 -高并发 prefill 保留 `HSA_NO_SCRATCH_RECLAIM=0`,decode 不变。这是显式运行策略,不是临时源码补丁。不引入 BF16 SSM、workspace 开发、READ-credit/QP 或 router 算法改动。 +高并发 prefill 保留 `HSA_NO_SCRATCH_RECLAIM=0`。原有图模式、传输设置及 1% 请求错误门槛不变。不加入 BF16 SSM、workspace 开发、shared-MR、QP/credit 实现或客户端取消补丁。 -## 官方镜像与临时集成 +## 固定运行栈与可删除 debug 层 -Worker 使用官方 `vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d`,固定 amd64 digest 为 `sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19`。主配置和 recipe 使用完全相同的镜像标识,替换原自定义构建,不携带 shared-MR 修改。 +Worker 镜像为 `vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19`,master 和 recipe 使用完全相同的不可变标识。镜像已包含 vLLM #57700,不再携带该 PR 的 backport。 -该 nightly 已包含 [vLLM#57700](https://github.com/vllm-project/vllm/pull/57700)。Discovery 模板还依赖尚未进入 srt-slurm pin 的 [srt-slurm#508](https://github.com/NVIDIA/srt-slurm/pull/508)。必需的 engine backport 和镜像/provider 兼容前置放在单独、可删除的 debug commit 中,不混入框架或配置提交。 +临时 engine setup 只应用两个随源码交付的 Python 运行时补丁:同步 READ 清零保护和仅用于 synthetic 的 draft gather 回退。分别校验 SHA256 和 `git apply --check`,不替换编译扩展,也不下载浮动 PR head。Parser 使用 nightly 原实现,不携带 EOF 回补。独立 router 跳过 engine setup;worker 缺少 vLLM 或补丁不兼容时终止启动。 -这些临时修改不符合直接合入的引擎补丁政策;待上游发布后删除: +| 依赖 | 固定来源 | 作用 | +| --- | --- | --- | +| [vLLM #59164](https://github.com/vllm-project/vllm/pull/59164) | `c5b1350f1f2bf10a128127a9b85e93d5f9f18e62` | 从新分配 KV 页的清零列表中排除同步 READ 目标。 | +| Synthetic draft gathering 实验 | 针对固定 nightly 的 `k3-synthetic-unproposed-drafts.patch` | 仅在启用 synthetic acceptance 时恢复 #58784 之前的输入 gather;普通/block 验证仍拒绝未提出的槽位。这是 benchmark 兼容实验,不是通用正确性修复。 | +| [srt-slurm #508](https://github.com/NVIDIA/srt-slurm/pull/508) | 已测版本 `51cee8904a0b402a834887a26008adb79b8cd26b` | 在匹配任务的临时 checkout 中补齐 connector 模板内的 discovery 拓扑绑定。 | -1. 确认官方 ROCm nightly 包含必需的 engine 修复,同步更新 worker 镜像的两处引用,验收后删除对应临时 setup 和 patch 文件。 -2. 独立升级 srt-slurm 子模块到包含 PR #508 的版本,删除临时的任务内 cherry-pick 及其测试。替换 worker 镜像不会升级 srt-slurm。 -3. 追加性能 changelog,并按正常 smoke、sweep、eval 门槛验证新的源码与镜像组合。删除临时补丁只能让源码栈干净,不能替代运行验收和 review。 +READ zeroing 补丁哈希为 `3da3746e85d53e4a3b17062b4113475d31a86cc07418e87ef5b4bb0c906cad20`。不加入 #58968 的 EOF、心跳、FULL-context、draft-fence 或传输所有权补丁。 -## 验证范围 +Synthetic 实验补丁哈希为 `33a6f792b13d39705a50562ca037a1d3c49dc054a8bdd8539fd8a154667f39df`。自动 golden acceptance 曲线、draft 模型/数量、parser 与错误门槛不变。该实验恢复首次调度占位槽的历史处理方式,可能改变首次可见 token 和后续生成;结果应与普通模型质量评估区分,尚未证明性能口径等价。 -Python launcher 迁移的行为测试覆盖镜像选择、导入前拒绝错误输入、使用外部调度/安装桩运行真实启动入口、结果归档、失败传播及取消。镜像仓库和源码检查证明镜像身份与上游能力包含关系,不证明设备/provider 兼容性或 RDMA 稳定性。 +Debug 层还保留集群声明的单文件只读 `ionic-provider` 挂载,用于镜像与内核 ABI 兼容,与 shared-MR 无关。镜像内 libibverbs 核心库和其他 provider 不变。本集成不替换 srt-slurm 仓库 URL、子模块 pin 或 AIPerf 源码。 -此前的 [c48 run](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36556488243) 使用不同 harness 版本和自定义镜像,仅作历史记录,不是本次官方 nightly 候选的验收证据。此处不宣称新的 GPU 实测、吞吐、精度或完全优雅退出结果。 +## 删除条件 + +1. 官方 worker 镜像包含 #59164 并通过验收后,同步更新 master/recipe,删除对应补丁及应用步骤。Synthetic 兼容问题定案后单独删除该实验,它不是等待上游合入的修复。全部运行时补丁移除后,再删除 worker setup、recipe 引用及其测试。 +2. 独立升级共享 srt-slurm pin 到包含 #508 的版本,再删除任务内 cherry-pick 及专项测试。更换 worker 镜像不会升级 srt-slurm。 +3. 只有替换镜像在目标平台不挂载 provider 也能打开全部预期 RDMA 设备,才删除兼容挂载。设备枚举不能代替 RDMA 流量验收。 +4. 追加性能 changelog,并在合入前完成适用 smoke、sweep 和 eval。临时 engine 补丁仍属于 draft debug 集成,不代表获得上游镜像政策豁免。 + +## 证据与剩余验收 + +当前候选不带 EOF。此前 EOF 回归和吞吐结果仅作为另一套 parser 栈的历史证据,不代表当前候选已通过验证。 + +配对 [1P1D c48 一小时作业](https://github.com/billishyahao/InferenceMINI/actions/runs/36847087330) 使用相同不可变 worker 镜像、#59164 和字节一致的 EOF 运行时代码。Warmup 为 531 valid / 0 empty;profile 为 4040 valid / 4 empty(0.0989%),完成提交、结果导出及原生收尾。这是另一套适配 harness 的运行支持证据,不是本 PR 当前 commit 的 sweep 或精度认证。 + +此前 [InferenceX 完整特性对照](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36800732704) 使用更广的补丁栈;后续仅带 #59164 的 [InferenceX 作业](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36814088271) 在排队且未获得 GPU runner 时取消,均不构成本次收束候选的验收。最小组合仍需当前 PR 的 1P2D、sweep/eval 和已加载服务收尾验证;省略其他已定位修复不等于证明其故障不可能发生。 diff --git a/inferencex-e2e/infx/launch/drivers/srt/checkout.py b/inferencex-e2e/infx/launch/drivers/srt/checkout.py index 822ef43e3b..1912774cc0 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/checkout.py +++ b/inferencex-e2e/infx/launch/drivers/srt/checkout.py @@ -24,6 +24,14 @@ UV_INSTALLER = "https://astral.sh/uv/install.sh" +# Temporary debug layer: delete after the srt-slurm pin includes NVIDIA/srt-slurm#508. +SRT_DEBUG_PRS = { + ("kimik3", "vllm-disagg"): ( + ("https://github.com/NVIDIA/srt-slurm.git", "51cee8904a0b402a834887a26008adb79b8cd26b"), + ), +} + + @dataclass(frozen=True) class Checkout: """A job-local srt-slurm checkout with InferenceX recipes staged.""" @@ -49,6 +57,15 @@ def _checked(result: subprocess.CompletedProcess[str], action: str) -> None: raise LaunchError(f"{action} failed (exit {result.returncode})") +def apply_debug_prs(destination: Path, model_prefix: str, framework: str) -> None: + """Apply pinned upstream commits only to the matching job's disposable checkout.""" + for repository, commit in SRT_DEBUG_PRS.get((model_prefix, framework), ()): + # Cherry-pick needs the parent; depth=1 makes the change look like a root commit. + _git("-C", destination, "fetch", "--no-tags", "--depth=2", repository, commit) + _git("-C", destination, "cherry-pick", "--no-commit", commit) + print(f"Applied temporary upstream change {repository}@{commit}", flush=True) + + def checkout_dir(run: SrtRun, *, shared: bool) -> Path: """Where a multi-node checkout lives, named for its run: shared-run-root or the workspace.""" request = run.request @@ -84,6 +101,7 @@ def prepare_checkout(run: SrtRun, destination: Path, *, power: bool) -> Checkout ) for patch in sorted((run.workspace / PATCHES).glob("*.patch")): _git("-C", destination, "apply", patch) + apply_debug_prs(destination, run.request.model_prefix, run.request.framework) head = _git("-C", destination, "rev-parse", "HEAD", capture=True) if head != commit: raise LaunchError(f"srt-slurm checkout is at {head}, expected {commit}") diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index b39508a725..d866f59b6d 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -105,6 +105,12 @@ class SrtLane: mounts=( LaneMount(Match(), "aiperf-cache", "/aiperf_mmap_cache"), LaneMount(Match(frameworks=any_of("tilert")), "it-share-data", "/models"), + # Temporary image/kernel ABI prerequisite, not a shared-MR or engine patch. + LaneMount( + Match(any_of("kimik3"), frameworks=any_of("vllm-disagg")), + "ionic-provider", + read_only=True, + ), ), stage_recipe_images=Match(any_of("kimik3"), frameworks=any_of("vllm-disagg")), eval_unsets=( diff --git a/inferencex-e2e/infx/tests/launch/test_k3_moriio_setup.py b/inferencex-e2e/infx/tests/launch/test_k3_moriio_setup.py new file mode 100644 index 0000000000..1df332b20b --- /dev/null +++ b/inferencex-e2e/infx/tests/launch/test_k3_moriio_setup.py @@ -0,0 +1,91 @@ +"""The temporary worker backport must not execute in a standalone router image.""" + +import hashlib +import os +import subprocess +from pathlib import Path + +import pytest + +SCRIPT = ( + Path(__file__).resolve().parents[3] + / "benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-debug.sh" +) + + +@pytest.mark.parametrize( + ("router", "worker"), [(True, False), (False, True), (True, True), (False, False)] +) +def test_only_standalone_router_skips_engine_setup(tmp_path, router, worker): + binaries = tmp_path / "bin" + binaries.mkdir() + for name, present in (("vllm-router", router), ("vllm", worker)): + if present: + binary = binaries / name + binary.write_text("#!/bin/bash\nexit 0\n") + binary.chmod(0o755) + # A failed package lookup must remain fatal for workers or an unknown image. + python = binaries / "python3" + python.write_text("#!/bin/bash\necho 'package lookup failed' >&2\nexit 7\n") + python.chmod(0o755) + completed = subprocess.run( + ["/bin/bash", str(SCRIPT)], + env={"PATH": str(binaries), "LANG": os.environ.get("LANG", "C")}, + capture_output=True, + text=True, + timeout=10, + ) + if router and not worker: + assert completed.returncode == 0 + assert "no engine backport required" in completed.stdout + assert completed.stderr == "" + else: + assert completed.returncode == 7 + assert "package lookup failed" in completed.stderr + assert "no engine backport required" not in completed.stdout + + +@pytest.mark.parametrize("failure", [None, "checksum", "context"]) +def test_verified_patch_changes_only_vllm_and_rejects_invalid_input(tmp_path, failure): + package_root = tmp_path / "site-packages" + package = package_root / "vllm" + package.mkdir(parents=True) + worker = package / "worker.py" + initial = "VALUE = 9\n" if failure == "context" else "VALUE = 1\n" + worker.write_text(initial) + unrelated = package_root / "other.txt" + unrelated.write_text("untouched\n") + patch = tmp_path / "backport.patch" + patch.write_text( + "diff --git a/vllm/worker.py b/vllm/worker.py\n" + "--- a/vllm/worker.py\n+++ b/vllm/worker.py\n@@ -1 +1 @@\n" + "-VALUE = 1\n+VALUE = 2\n" + "diff --git a/other.txt b/other.txt\n" + "--- a/other.txt\n+++ b/other.txt\n@@ -1 +1 @@\n" + "-untouched\n+changed\n" + ) + checksum = hashlib.sha256(patch.read_bytes()).hexdigest() + if failure == "checksum": + patch.write_text(patch.read_text() + "corrupted\n") + completed = subprocess.run( + [ + "/bin/bash", + "-c", + 'source "$1"; apply_verified_patch "$2" "$3" "$4"', + "test", + str(SCRIPT), + str(package_root), + str(patch), + checksum, + ], + capture_output=True, + text=True, + timeout=10, + ) + assert unrelated.read_text() == "untouched\n" + if failure is None: + assert completed.returncode == 0, completed.stderr + assert worker.read_text() == "VALUE = 2\n" + else: + assert completed.returncode != 0 + assert worker.read_text() == initial diff --git a/inferencex-e2e/infx/tests/launch/test_srt_debug_prs.py b/inferencex-e2e/infx/tests/launch/test_srt_debug_prs.py new file mode 100644 index 0000000000..a7e934c938 --- /dev/null +++ b/inferencex-e2e/infx/tests/launch/test_srt_debug_prs.py @@ -0,0 +1,46 @@ +"""The temporary upstream backport changes only a matching disposable checkout.""" + +import subprocess + +import pytest + +from infx.launch.context import LaunchError +from infx.launch.drivers.srt.checkout import SRT_DEBUG_PRS, apply_debug_prs + + +def test_pinned_debug_backport_applies_real_commit_and_rejects_conflict(tmp_path, monkeypatch): + upstream = tmp_path / "upstream" + upstream.mkdir() + + def git(*args, cwd=upstream): + return subprocess.run(["git", *args], cwd=cwd, check=True, capture_output=True, text=True).stdout.strip() + + git("init", "--quiet") + git("config", "user.name", "Fixture") + git("config", "user.email", "fixture@example.invalid") + source = upstream / "feature.txt" + source.write_text("binding: missing\n") + git("add", "feature.txt") + git("commit", "--quiet", "-m", "base") + base = git("rev-parse", "HEAD") + source.write_text("binding: resolved\n") + git("commit", "--quiet", "-am", "upstream feature") + patch = git("rev-parse", "HEAD") + job = tmp_path / "job" + git("clone", "--quiet", str(upstream), str(job)) + git("checkout", "--quiet", "--detach", base, cwd=job) + monkeypatch.setitem(SRT_DEBUG_PRS, ("fixture", "vllm"), ((str(upstream), patch),)) + + apply_debug_prs(job, "unrelated", "vllm") + assert (job / "feature.txt").read_text() == "binding: missing\n" + apply_debug_prs(job, "fixture", "vllm") + assert (job / "feature.txt").read_text() == "binding: resolved\n" + assert git("rev-parse", "HEAD", cwd=job) == base + + conflicting = tmp_path / "conflicting" + git("clone", "--quiet", str(job), str(conflicting)) + (conflicting / "feature.txt").write_text("binding: incompatible\n") + with pytest.raises(LaunchError, match="cherry-pick"): + apply_debug_prs(conflicting, "fixture", "vllm") + assert (conflicting / "feature.txt").read_text() == "binding: incompatible\n" + assert source.read_text() == "binding: resolved\n" diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 44dfa3983f..596d177c11 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9222,3 +9222,83 @@ - "通过 Python launcher 增加原生 Kimi-K3 PD,使用 FP32 SSM、FP8 KV、GMU0.90 和上游 DSpark 校验,保留 recipe 管理的并发与拓扑配置。" - "使用固定 digest 的官方 ROCm nightly 替换自定义镜像,不携带 shared-MR 或候选 provider bind;新镜像仍待运行验收。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Temporarily apply pinned upstream srt-slurm#508 to the job-local clone and merged vLLM#57700 Python changes to the official-nightly worker containers. Fail startup if the online fetch, checksum or patch application fails; remove this debug layer after advancing the respective upstream pins." + - "临时在任务内应用固定的上游 srt-slurm#508,以及已合入的 vLLM#57700 纯 Python 修改;在线获取、校验或应用失败即终止启动,待分别升级上游 pin 后删除该 debug 层。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Restore the Kimi-K3 vLLM PD lane's single-file Ionic provider bind as a read-only node-local volume after the official nightly rejected the host kernel ABI. Keep the image's libibverbs core, serving settings and router unchanged; no shared-MR patch. Current-image device-open and end-to-end qualification remain required." + - "官方 nightly 的 Ionic provider 拒绝主机内核 ABI,因此为 Kimi-K3 vLLM PD 路径恢复节点本地单文件只读挂载。保留镜像原有 libibverbs 核心库、serving 参数与 router,不加入 shared-MR;仍需验证当前镜像的设备打开及端到端运行。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Temporarily backport only vLLM#58968's independent-heartbeat helper and worker lifecycle at 9cb82832079ede6638b5a5f35ba0d60d203f57a0 to prevent compute-worker GIL stalls from expiring discovery registration. Preserve payloads, intervals, FP32 SSM, GMU0.90, router and data-plane settings; exclude the rest of that PR. Remove this debug subset once the official image contains the upstream fix." + - "临时提取固定版本 vLLM#58968 的独立心跳 helper 与 worker 生命周期,避免计算线程持有 GIL 时 discovery 注册过期。保留 payload、间隔、FP32 SSM、GMU0.90、router 及数据面设置,不携带该 PR 其余修改;官方镜像包含上游修复后删除此 debug 子集。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Temporarily backport only the FULL pre-forward context from vLLM#58968 at 7595667abf90c03b546c92189d05e5d73392910e so the existing MoRIIO READ completion wait runs before FULL graph replay. Preserve the pinned image, heartbeat backport, FP32 SSM, GMU0.90, graph modes, DSpark, router and benchmark acceptance criteria; omit all other changes from that PR." + - "临时仅提取固定版本 vLLM#58968 的 FULL pre-forward 上下文,使现有 MoRIIO READ 完成等待发生在 FULL graph replay 之前。保留镜像、既有心跳补丁、FP32 SSM、GMU0.90、图模式、DSpark、router 和 benchmark 验收标准,不携带该 PR 其余修改。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Run a diagnostic full-feature control with vLLM#58968 at 7595667abf90c03b546c92189d05e5d73392910e and extracted synchronous READ zeroing fix #59164 at d5e6faa954024d3f3090edd1e769b9efe4feb1cb, retaining merged #57700 and the existing independent heartbeat. Preserve the image, FP32 SSM, GMU0.90, graphs, DSpark, router and benchmark criteria. This is a baseline experiment before patch reduction, not production qualification." + - "使用固定版本 vLLM#58968 的完整运行时代码与已拆出的 #59164 同步 READ zeroing 修复建立实验对照,保留已合入的 #57700 和独立心跳。镜像、FP32 SSM、GMU0.90、图模式、DSpark、router 及 benchmark 判据不变;这是精简补丁前的基线实验,不代表生产验证通过。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Move the minimal Kimi-K3 FP32 PD diagnostic candidate from official ROCm nightly 36768d1b to digest-pinned ac68c308, which includes #57700. Retain only #59164's synchronous READ zeroing backport; omit #58968 and independent heartbeat. Preserve serving settings, DSpark synthetic acceptance and benchmark criteria. The previous full-feature control remains in history; this image-changing experiment is not production qualification or a strict single-variable comparison." + - "将 Kimi-K3 FP32 PD 最小依赖诊断候选从官方 ROCm nightly 36768d1b 更新为按 digest 固定且已包含 #57700 的 ac68c308。仅保留 #59164 同步 READ zeroing 回补,省略 #58968 与独立心跳;serving 参数、DSpark synthetic acceptance 和 benchmark 判据不变。历史中保留完整特性对照;本次换镜像实验不代表生产验证通过,也不是严格单变量比较。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Consolidate the debug engine stack to immutable official ROCm nightly ac68c308 plus pinned #59164 and the Kimi-K3 streaming-EOF subset, runtime-equivalent to YukioZzz/vllm@535727bcc. Preserve FP32 SSM, GMU0.90, topology, automatic acceptance and 1% error gates; no heartbeat or full #58968 backport. Supporting matched 1P1D c48 evidence does not qualify the current InferenceX 1P2D candidate." + - "将 debug engine 栈收束为不可变官方 ROCm nightly ac68c308,加固定 #59164 与和 YukioZzz/vllm@535727bcc 运行时等价的 Kimi-K3 streaming-EOF 子集。保留 FP32 SSM、GMU0.90、拓扑、自动 acceptance 和 1% 错误门槛,不加入心跳或全量 #58968。配对 1P1D c48 证据不等于当前 InferenceX 1P2D 候选已通过验证。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Add a diagnostic synthetic-only rollback of vLLM #58784 input gathering on the pinned nightly + #59164 + EOF stack. Normal/block verification retains invalid-slot rejection. Preserve golden acceptance selection, FP32 SSM, GMU0.90, recipes and 1% error gates; the experiment is not a general correctness fix or a claim of performance equivalence." + - "在固定 nightly + #59164 + EOF 栈上新增仅针对 synthetic 的 vLLM #58784 输入 gather 回退实验;普通/block 验证仍拒绝无效槽位。保持 golden acceptance 自动选择、FP32 SSM、GMU0.90、recipe 和 1% 错误门槛;该实验不是通用正确性修复,也不代表已证明性能口径等价。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Remove the EOF backport from the Kimi-K3 FP32 PD candidate. Retain immutable official nightly ac68c308, pinned #59164 and synthetic-only legacy draft gathering; use the upstream parser unchanged. Preserve all serving settings, automatic golden acceptance and request-error gates. Earlier EOF-enabled results do not qualify this candidate." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 From e27698658cd83baab43cf7e1af832d665fecf116 Mon Sep 17 00:00:00 2001 From: Yichao Zhu Date: Fri, 2 Oct 2026 15:21:23 +0800 Subject: [PATCH 4/4] debug(ci): scope K3 PD sweep to one GSM8K evaluation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 本轮保留全部五个吞吐点,仅选择 1P2D c48 完整 GSM8K;不修改 evaluator 参数或验收门槛。此提交需在完整上游评测前删除。 --- inferencex-e2e/docs/k3-pd-native.md | 2 + inferencex-e2e/docs/k3-pd-native_zh.md | 2 + inferencex-e2e/infx/matrix/plan.py | 3 +- .../tests/workflows/test_k3_pd_debug_eval.py | 82 +++++++++++++++++++ .../infx/workflows/k3_pd_debug_eval.py | 48 +++++++++++ 5 files changed, 136 insertions(+), 1 deletion(-) create mode 100644 inferencex-e2e/infx/tests/workflows/test_k3_pd_debug_eval.py create mode 100644 inferencex-e2e/infx/workflows/k3_pd_debug_eval.py diff --git a/inferencex-e2e/docs/k3-pd-native.md b/inferencex-e2e/docs/k3-pd-native.md index a7e953d3d2..2b2a1219ae 100644 --- a/inferencex-e2e/docs/k3-pd-native.md +++ b/inferencex-e2e/docs/k3-pd-native.md @@ -40,6 +40,8 @@ The debug layer also retains the cluster-declared, read-only single-file `ionic- ## Evidence and remaining qualification +The current PR sweep temporarily retains all five one-hour throughput points and only the generated 1P2D c48 full GSM8K evaluation. A separate debug commit filters the PR #3582 plan after native generation; it does not change evaluator inputs, real block rejection, sample limits, score thresholds or benchmark gates. It rejects an unexpected scope rather than silently reducing it. Other PRs and manual workflows are unaffected. Remove this selection commit before claiming the repository's full evaluation coverage; the omitted vendor checks remain unqualified. + EOF is excluded from the current candidate. Earlier EOF regression and throughput results remain historical evidence for a different parser stack, not qualification of this candidate. The matched [1P1D c48 one-hour run](https://github.com/billishyahao/InferenceMINI/actions/runs/36847087330) used the same immutable worker image, #59164 and byte-identical EOF runtime. Warmup: 531 valid / 0 empty; profile: 4040 valid / 4 empty (0.0989%). Submission, result export and native cleanup completed. This is supporting runtime evidence from a separately adapted harness, not a sweep on this PR's current commit or an accuracy certification. diff --git a/inferencex-e2e/docs/k3-pd-native_zh.md b/inferencex-e2e/docs/k3-pd-native_zh.md index 9e277aab44..012d977725 100644 --- a/inferencex-e2e/docs/k3-pd-native_zh.md +++ b/inferencex-e2e/docs/k3-pd-native_zh.md @@ -40,6 +40,8 @@ Debug 层还保留集群声明的单文件只读 `ionic-provider` 挂载,用 ## 证据与剩余验收 +本轮 PR sweep 临时保留全部五个一小时吞吐点,仅选取原生生成的 1P2D c48 完整 GSM8K 精度任务。独立 debug commit 在原生矩阵生成后收窄 PR #3582 的评测范围,不改变 evaluator 输入、真实 block rejection、样本上限、分数门槛或 benchmark 判据;遇到意外范围会直接报错,不静默缩减。其他 PR 和手动 workflow 不受影响。宣称满足仓库完整评测覆盖之前必须删除该选择提交;本轮省略的厂商检查不视为已验收。 + 当前候选不带 EOF。此前 EOF 回归和吞吐结果仅作为另一套 parser 栈的历史证据,不代表当前候选已通过验证。 配对 [1P1D c48 一小时作业](https://github.com/billishyahao/InferenceMINI/actions/runs/36847087330) 使用相同不可变 worker 镜像、#59164 和字节一致的 EOF 运行时代码。Warmup 为 531 valid / 0 empty;profile 为 4040 valid / 4 empty(0.0989%),完成提交、结果导出及原生收尾。这是另一套适配 harness 的运行支持证据,不是本 PR 当前 commit 的 sweep 或精度认证。 diff --git a/inferencex-e2e/infx/matrix/plan.py b/inferencex-e2e/infx/matrix/plan.py index 3d6ee34e97..bdd463deac 100644 --- a/inferencex-e2e/infx/matrix/plan.py +++ b/inferencex-e2e/infx/matrix/plan.py @@ -13,6 +13,7 @@ import yaml from infx.config import MASTER_CONFIGS, RUNNER_CONFIG, git_path_at_ref +from infx.workflows.k3_pd_debug_eval import select_debug_eval from .generate import ( EvalMode, @@ -494,7 +495,7 @@ def generate_current( suffix = "agentic_evals" if result.get("scenario-type") == "agentic-coding" else "evals" final_results[prefix + suffix].append(result) - return ChangelogMatrixEntry.model_validate(final_results) + return ChangelogMatrixEntry.model_validate(select_debug_eval(final_results)) def main() -> None: diff --git a/inferencex-e2e/infx/tests/workflows/test_k3_pd_debug_eval.py b/inferencex-e2e/infx/tests/workflows/test_k3_pd_debug_eval.py new file mode 100644 index 0000000000..868559db09 --- /dev/null +++ b/inferencex-e2e/infx/tests/workflows/test_k3_pd_debug_eval.py @@ -0,0 +1,82 @@ +"""Behavior of the explicitly temporary PR evaluation scope.""" + +import copy +import json +import subprocess +import sys + +import pytest + +from infx.workflows.k3_pd_debug_eval import select_debug_eval + + +@pytest.fixture +def plan(): + def row(decode, conc, framework): + return { + "prefill": {"num-worker": 1}, + "decode": {"num-worker": decode}, + "conc": [conc], + "eval-conc": conc, + "eval-framework": framework, + "eval-suite": "" if framework == "lm-eval" else "vendor-suite", + "threshold": 0.9, + "duration": 3600, + } + + return { + "changelog_metadata": { + "entries": [ + { + "config-keys": ["kimik3-fp4-mi355x-vllm-disagg-agentic"], + "pr-link": "https://github.com/SemiAnalysisAI/InferenceX/pull/3582", + } + ] + }, + "multi_node": { + "agentic": [row(d, c, None) for d, c in ((1, 1), (1, 10), (2, 24), (1, 48), (2, 48))] + }, + "multinode_agentic_evals": [ + row(1, 48, "lm-eval"), + row(2, 48, "lm-eval"), + row(2, 48, "kimi-vendor"), + ], + "evals": [], + "agentic_evals": [], + "multinode_evals": [], + } + + +def test_cli_preserves_benchmarks_and_complete_selected_evaluator(plan): + completed = subprocess.run( + [sys.executable, "-m", "infx.workflows.k3_pd_debug_eval"], + input=json.dumps(plan), + capture_output=True, + text=True, + check=True, + ) + result = json.loads(completed.stdout) + assert result == {**plan, "multinode_agentic_evals": [plan["multinode_agentic_evals"][1]]} + + +def test_unrelated_pr_is_unchanged(): + unrelated = {"unrelated": "input"} + assert select_debug_eval(unrelated) is unrelated + + +@pytest.mark.parametrize("change", ["missing", "duplicate", "throughput", "scope", "family"]) +def test_unexpected_scope_fails_without_modifying_input(plan, change): + if change == "missing": + plan["multinode_agentic_evals"].pop(1) + elif change == "duplicate": + plan["multinode_agentic_evals"].append(plan["multinode_agentic_evals"][1]) + elif change == "throughput": + plan["multi_node"]["agentic"].pop() + elif change == "scope": + plan["changelog_metadata"]["entries"][0]["config-keys"].append("unrelated") + else: + plan["evals"].append({"unrelated": "eval"}) + original = copy.deepcopy(plan) + with pytest.raises(ValueError): + select_debug_eval(plan) + assert plan == original diff --git a/inferencex-e2e/infx/workflows/k3_pd_debug_eval.py b/inferencex-e2e/infx/workflows/k3_pd_debug_eval.py new file mode 100644 index 0000000000..4949490428 --- /dev/null +++ b/inferencex-e2e/infx/workflows/k3_pd_debug_eval.py @@ -0,0 +1,48 @@ +"""Temporary PR #3582 evaluation selection; remove before upstream qualification.""" + +import argparse +import json +import sys + + +def select_debug_eval(plan: dict) -> dict: + """Preserve throughput and retain the generated 1P2D c48 GSM8K row only.""" + entries = plan.get("changelog_metadata", {}).get("entries", []) + if not entries or any( + entry.get("pr-link") != "https://github.com/SemiAnalysisAI/InferenceX/pull/3582" + for entry in entries + ): + return plan + keys = {key for entry in entries for key in entry["config-keys"]} + if keys != {"kimik3-fp4-mi355x-vllm-disagg-agentic"}: + raise ValueError("Temporary K3 selection requires the isolated K3 PD changelog") + benchmarks = plan["multi_node"].get("agentic", []) + points = [(row["decode"]["num-worker"], row["conc"]) for row in benchmarks] + if sorted(points) != [(1, [1]), (1, [10]), (1, [48]), (2, [24]), (2, [48])]: + raise ValueError("Temporary K3 selection requires the unchanged throughput matrix") + selected = [ + row + for row in plan["multinode_agentic_evals"] + if row.get("eval-framework") == "lm-eval" + and row.get("eval-suite", "") == "" + and row["prefill"]["num-worker"] == 1 + and row["decode"]["num-worker"] == 2 + and row["conc"] == [48] + and row["eval-conc"] == 48 + ] + if len(selected) != 1: + raise ValueError("Expected exactly one generated 1P2D c48 GSM8K evaluation") + if any(plan.get(key) for key in ("evals", "agentic_evals", "multinode_evals")): + raise ValueError("Unexpected evaluation family in the isolated K3 PD sweep") + return {**plan, "multinode_agentic_evals": selected} + + +def main() -> None: + """Filter the generated plan without modifying evaluator inputs or scores.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.parse_args() + print(json.dumps(select_debug_eval(json.load(sys.stdin)))) + + +if __name__ == "__main__": + main()