From cb0d0a71fd5ca92cbfb995d122219cb467291653 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Wed, 30 Sep 2026 11:29:58 -0500 Subject: [PATCH 1/6] feat(launch): route llmd-vllm multinode jobs through infx.launch Add an llm-d driver that submits the job through the llm-d wrapper, attaches to the Slurm job, streams its log, checks its final state and stages results, agentic and eval artifacts. Route llmd-vllm multinode requests to it on gb200-nv and resolve DeepSeek-V4-Pro to the node-local numa1 checkpoint. Restore the GB200 DeepSeek-V4-Pro disagg wrapper. job.slurm no longer scancels its own allocation when the coordinator finishes; it stops the srun step and exits 0, so the job ends COMPLETED instead of CANCELLED, which infx.launch reports as a failure. Driver port by Ilya Markov from #2719. --- .../dsv4_fp4_gb200_llmd-vllm-disagg.sh | 12 ++ .../benchmarks/multi_node/llm-d/README.md | 2 +- .../benchmarks/multi_node/llm-d/job.slurm | 42 ++--- .../benchmarks/multi_node/llm-d/server.sh | 4 +- .../infx/launch/drivers/__init__.py | 3 +- inferencex-e2e/infx/launch/drivers/llmd.py | 152 ++++++++++++++++++ .../infx/launch/drivers/srt/models.py | 5 + inferencex-e2e/infx/launch/policy.py | 26 +++ inferencex-e2e/infx/launch/request.py | 8 + .../infx/tests/launch/test_llmd_driver.py | 115 +++++++++++++ .../infx/tests/launch/test_srt_policy.py | 1 + 11 files changed, 347 insertions(+), 23 deletions(-) create mode 100755 inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh create mode 100644 inferencex-e2e/infx/launch/drivers/llmd.py create mode 100644 inferencex-e2e/infx/tests/launch/test_llmd_driver.py diff --git a/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh b/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh new file mode 100755 index 0000000000..d2a49f2dc3 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh @@ -0,0 +1,12 @@ +#!/usr/bin/env bash +# DeepSeek-V4-Pro FP4 GB200 llm-d vLLM P/D disagg. infx.launch (drivers/llmd.py) runs this +# and reads the Slurm job id submit.sh prints on stdout. +set -eo pipefail + +export GPUS_PER_NODE=4 TIME_LIMIT="${TIME_LIMIT:-08:00:00}" CONTAINER_IMAGE="$IMAGE" +export PREFILL_WORKERS="${PREFILL_WORKERS:-${PREFILL_NUM_WORKERS:-1}}" +export DECODE_WORKERS="${DECODE_WORKERS:-${DECODE_NUM_WORKERS:-1}}" + +cd "$(dirname "$0")/llm-d" +exec bash ./submit.sh "$PREFILL_NODES" "$DECODE_NODES" \ + "$ISL" "$OSL" "${CONC_LIST// /x}" inf "$RANDOM_RANGE_RATIO" diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/README.md b/inferencex-e2e/benchmarks/multi_node/llm-d/README.md index 81dbd51995..f8e34492dc 100644 --- a/inferencex-e2e/benchmarks/multi_node/llm-d/README.md +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/README.md @@ -9,7 +9,7 @@ InferenceX itself owns the SLURM job, no vendor multi-node tool involved. |---|---| | `submit.sh` | sbatch wrapper. Validates env, exports tuning vars, returns `JOB_ID`. May read `slurm.time_limit` from the recipe to override `TIME_LIMIT`. | | `job.slurm` | sbatch entrypoint. Allocates `PREFILL_NODES + DECODE_NODES` nodes, derives per-node IPs, runs one Docker container per node via `srun`, threads role assignment env into each. | -| `server.sh` | Per-node entry. Reads `NODE_RANK = SLURM_PROCID`, picks role, starts vLLM (with the wide-EP / DeepEP / NIXL flag set from the llm-d wide-EP-lws guide), starts the pd-sidecar on each leader, and on the decode leader additionally writes `endpoints.yaml`, starts EPP + Envoy, runs `benchmark_serving.py`, and `scancel`s the job. | +| `server.sh` | Per-node entry. Reads `NODE_RANK = SLURM_PROCID`, picks role, starts vLLM (with the wide-EP / DeepEP / NIXL flag set from the llm-d wide-EP-lws guide), starts the pd-sidecar on each leader, and on the decode leader additionally writes `endpoints.yaml`, starts EPP + Envoy, runs `benchmark_serving.py`, and writes a done marker; `job.slurm` then stops the srun step and exits 0 so the job ends `COMPLETED`. | ## Topology diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm b/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm index fe36301b6a..b84931d66e 100644 --- a/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm @@ -63,25 +63,29 @@ export DOCKER_CONT_NAME : "${BENCHMARK_LOGS_DIR:?BENCHMARK_LOGS_DIR not set}" DOCKER_MOUNT_PATH="/workspace" -cleanup() { - echo "[${SLURM_JOB_ID}] cleanup on $(hostname)" - [[ -n "${WATCHER_PID:-}" ]] && kill "$WATCHER_PID" 2>/dev/null || true -} -trap cleanup INT TERM HUP EXIT - -# Coordinator-done watcher. server.sh on the decode coordinator writes -# this marker after the bench finishes; we then scancel the allocation -# from outside the container (the image has no SLURM client tools). -# Without this, workers `wait` on local vLLM forever and the job runs -# to TIME_LIMIT. +# server.sh on the decode coordinator writes this marker once the benchmark and +# eval are done. Workers keep vLLM up until stopped, so the step is stopped then +# and the batch script exits 0: the job ends COMPLETED, the state infx.launch +# accepts as success. Cancelling the allocation instead would end it CANCELLED. BENCH_DONE_MARKER="$BENCHMARK_LOGS_DIR/.bench_done.$SLURM_JOB_ID" rm -f "$BENCH_DONE_MARKER" -( - while [[ ! -f "$BENCH_DONE_MARKER" ]]; do sleep 5; done - echo "[${SLURM_JOB_ID}] coordinator finished; scancel'ing job" - scancel "$SLURM_JOB_ID" 2>/dev/null || true -) & -WATCHER_PID=$! + +run_until_bench_done() { + "$@" & + local step=$! rc=0 + while kill -0 "$step" 2>/dev/null; do + if [[ -f "$BENCH_DONE_MARKER" ]]; then + echo "[${SLURM_JOB_ID}] coordinator finished; stopping the srun step" + kill -TERM "$step" 2>/dev/null || true + break + fi + sleep 5 + done + wait "$step" || rc=$? + [[ -f "$BENCH_DONE_MARKER" ]] && return 0 + echo "[${SLURM_JOB_ID}] srun step exited rc=$rc before the coordinator finished" >&2 + return "$rc" +} # Container engine: 'docker' (default) for clusters where the SLURM # user can talk to /var/run/docker.sock (e.g. h200-dgxc-slurm); 'pyxis' @@ -102,7 +106,7 @@ done if [[ "$LLMD_CONTAINER_ENGINE" == "docker" ]]; then # One docker run per node, one task per node. server.sh dispatches by NODE_RANK. - srun \ + run_until_bench_done srun \ --kill-on-bad-exit=1 \ --signal=TERM@30 \ --unbuffered \ @@ -261,7 +265,7 @@ elif [[ "$LLMD_CONTAINER_ENGINE" == "pyxis" ]]; then # MODEL_DIR / BENCHMARK_LOGS_DIR / NODE_RANK are translated to their # in-container values inside bash -lc (host MODEL_DIR is the source # path of the bind mount, but server.sh expects /models inside). - srun \ + run_until_bench_done srun \ --kill-on-bad-exit=1 \ --signal=TERM@30 \ --unbuffered \ diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh b/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh index a4e8a94047..2f45cdeaad 100755 --- a/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh @@ -569,8 +569,8 @@ PY ) fi - # Signal job.slurm (outside the container, where scancel exists) to release - # the allocation; without it workers wait until TIME_LIMIT. + # Signal job.slurm (outside the container) to stop the srun step and exit 0; + # without it workers wait until TIME_LIMIT. touch "$BENCHMARK_LOGS_DIR/.bench_done.$SLURM_JOB_ID" else # Workers (prefill leader, prefill/decode workers): keep vLLM alive. diff --git a/inferencex-e2e/infx/launch/drivers/__init__.py b/inferencex-e2e/infx/launch/drivers/__init__.py index bab33b1cb6..996bb5738d 100644 --- a/inferencex-e2e/infx/launch/drivers/__init__.py +++ b/inferencex-e2e/infx/launch/drivers/__init__.py @@ -13,7 +13,7 @@ from infx.launch import policy from infx.launch.backends import backend_class from infx.launch.context import Launch, LaunchError -from infx.launch.drivers import legacy, script, srt +from infx.launch.drivers import legacy, llmd, script, srt from infx.launch.policy import LaunchPath, launch_path if TYPE_CHECKING: @@ -38,6 +38,7 @@ class Route: LaunchPath.SCRIPT: Route(None, script.run), LaunchPath.LEGACY_TILERT: Route("slurm", legacy.run_tilert), LaunchPath.LEGACY_AMD_UTILS: Route("slurm", legacy.run_amd_utils), + LaunchPath.LLMD: Route("slurm", llmd.run), } diff --git a/inferencex-e2e/infx/launch/drivers/llmd.py b/inferencex-e2e/infx/launch/drivers/llmd.py new file mode 100644 index 0000000000..809104458a --- /dev/null +++ b/inferencex-e2e/infx/launch/drivers/llmd.py @@ -0,0 +1,152 @@ +"""GB200 llm-d vLLM multinode jobs submitted through benchmarks/multi_node/llm-d/submit.sh.""" + +from __future__ import annotations + +import os +import shutil +import subprocess +import sys +from pathlib import Path + +from infx.launch import artifacts, policy, proc +from infx.launch.backends.base import BackendError +from infx.launch.backends.slurm import cli +from infx.launch.context import Launch, LaunchError +from infx.launch.drivers.srt import models +from infx.launch.drivers.srt.run import slurm_backend +from infx.launch.request import LlmdRequest, RequestError + +CANCEL_TIMEOUT_S = 600.0 + + +def _bench_script(request: LlmdRequest) -> Path: + model_tag = request.exp_name.split("_", 1)[0] + kind = "disagg" if request.disagg else "agg" + script = ( + request.workspace + / f"benchmarks/multi_node/{model_tag}_{request.precision}_gb200_llmd-vllm-{kind}.sh" + ) + if not script.is_file(): + raise LaunchError(f"llm-d wrapper not found: {script}") + return script + + +def _find_eval_dir(logs_dir: Path) -> Path | None: + for root, dirs, _files in os.walk(logs_dir): + if "eval_results" in dirs: + return Path(root) / "eval_results" + return None + + +def _stage_agentic(logs_dir: Path, workspace: Path) -> None: + agentic = logs_dir / "agentic" + if not agentic.is_dir(): + return + staged = workspace / "LOGS" / "agentic" + staged.mkdir(parents=True, exist_ok=True) + for entry in agentic.iterdir(): + destination = staged / entry.name + if entry.is_dir(): + shutil.copytree(entry, destination, dirs_exist_ok=True) + elif entry.is_file(): + shutil.copy2(entry, destination) + + +def run(launch: Launch) -> int: + """Submit the llm-d Slurm job, follow its log, and stage benchmark artifacts.""" + backend = slurm_backend(launch) + request = LlmdRequest.from_env(launch.request.env) + if launch.cluster.id not in policy.LLMD_CLUSTERS: + raise LaunchError(f"llmd-vllm is not configured for cluster {launch.cluster.id!r}") + + checkpoint = models.checkpoint(launch.cluster, request) + if checkpoint is None: + raise LaunchError( + f"cluster {launch.cluster.id!r} stages no checkpoint for MODEL={request.model}" + ) + model_path = models.host_path(launch.cluster, checkpoint) + if not (model_path / "config.json").is_file(): + raise LaunchError(f"model checkpoint is unavailable: {model_path / 'config.json'}") + + squash = backend.prepare_image(request.image) + logs_dir = request.workspace / "benchmark_logs" + logs_dir.mkdir(parents=True, exist_ok=True) + + account = backend.settings.account or cli.default_account() + if not account: + raise RequestError.missing("SLURM_ACCOUNT") + + env = policy.runtime_env( + launch.cluster, + request, + models.job_env(launch.cluster, request, str(model_path)), + { + "SLURM_PARTITION": backend.settings.partition, + "SLURM_ACCOUNT": account, + "MODEL_PATH": str(model_path), + "MODEL_NAME": request.model, + "LLMD_CONTAINER_ENGINE": "pyxis", + "LLMD_SQUASH_FILE": squash.reference, + "BENCHMARK_LOGS_DIR": str(logs_dir), + }, + ) + + script = _bench_script(request) + argv = ["bash", str(script)] + proc.echo(argv, env) + submitted = subprocess.run( + argv, + stdout=subprocess.PIPE, + stderr=sys.stderr, + text=True, + env=env, + cwd=request.workspace, + check=False, + ) + job_id = submitted.stdout.strip() + if submitted.returncode != 0 or not job_id: + print("ERROR: llm-d submit wrapper failed before returning a Slurm job id", file=sys.stderr) + return 1 + if not (job_id.isascii() and job_id.isdigit()): + print( + f"ERROR: llm-d submit wrapper printed {job_id!r} instead of a Slurm job id", + file=sys.stderr, + ) + return 1 + + log_file = logs_dir / f"slurm_job-{job_id}.out" + job = backend.attach(job_id, log=log_file, outputs=logs_dir) + print(f"Submitted llm-d job: {job_id}", flush=True) + + launch.life.callback( + artifacts.bundle_server_logs, logs_dir, request.workspace / "multinode_server_logs.tar.gz" + ) + launch.life.callback(backend.cancel, job, wait_s=CANCEL_TIMEOUT_S) + + try: + backend.stream_logs(job) + except BackendError: + return 1 + + status = backend.state(job) + rc = 0 if status.succeeded else 1 + + for result_file in sorted(logs_dir.glob(f"{request.result_filename}*.json")): + try: + artifacts.copy_to_workspace(result_file, request.workspace / result_file.name) + except artifacts.ArtifactError as error: + print(f"ERROR: {error}", file=sys.stderr) + rc = 1 + + if request.is_agentic and not request.eval_only: + _stage_agentic(logs_dir, request.workspace) + + if request.run_eval: + eval_dir = _find_eval_dir(logs_dir) or logs_dir / "eval_results" + try: + artifacts.copy_eval_artifacts(eval_dir, request.workspace) + except artifacts.ArtifactError as error: + print(f"ERROR: {error}", file=sys.stderr) + rc = 1 + + return rc diff --git a/inferencex-e2e/infx/launch/drivers/srt/models.py b/inferencex-e2e/infx/launch/drivers/srt/models.py index a8f1dad3fc..7faf1000eb 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/models.py +++ b/inferencex-e2e/infx/launch/drivers/srt/models.py @@ -46,6 +46,11 @@ class Override: Override(Match(model_glob="*/DeepSeek-V4-Pro-0813"), entry="DeepSeek-V4-Pro-0813"), ), "gb200-nv": ( + Override( + Match(any_of("dsv4"), any_of("fp4"), any_of("llmd-vllm")), + entry="DeepSeek-V4-Pro@numa1", + served_name="deepseek-ai/DeepSeek-V4-Pro", + ), Override( Match(any_of("dsr1"), any_of("fp4"), any_of("dynamo-sglang")), entry="deepseek-r1-0528-fp4-v2", diff --git a/inferencex-e2e/infx/launch/policy.py b/inferencex-e2e/infx/launch/policy.py index f9d26ab88c..76c135500e 100644 --- a/inferencex-e2e/infx/launch/policy.py +++ b/inferencex-e2e/infx/launch/policy.py @@ -15,6 +15,7 @@ from typing import TYPE_CHECKING from infx.clusters.slurm import SlurmSettings +from infx.launch.context import LaunchError if TYPE_CHECKING: from infx.clusters import Cluster @@ -61,6 +62,7 @@ class LaunchPath(StrEnum): SCRIPT = "script" LEGACY_TILERT = "legacy-tilert" LEGACY_AMD_UTILS = "legacy-amd-utils" + LLMD = "llmd" NATIVE_SRT_LANES: dict[str, tuple[Match, ...]] = { @@ -79,8 +81,18 @@ class LaunchPath(StrEnum): } +LLMD_CLUSTERS: frozenset[str] = frozenset({"gb200-nv"}) + + def launch_path(cluster_id: str, request: LaunchRequest) -> LaunchPath: if request.is_multinode: + if request.framework == "llmd-vllm": + if cluster_id not in LLMD_CLUSTERS: + raise LaunchError( + f"llmd-vllm is not configured for cluster {cluster_id!r}; " + f"supported clusters: {', '.join(sorted(LLMD_CLUSTERS))}" + ) + return LaunchPath.LLMD if any(lane(request) for lane in NATIVE_SRT_LANES.get(cluster_id, ())): return LaunchPath.SRT_NATIVE if cluster_id in LEGACY_TILERT and request.framework == "tilert": @@ -228,6 +240,11 @@ def keys(table: Mapping[str, object]) -> list[str]: for key in keys(table) if key not in clusters ] + problems += [ + f"LLMD_CLUSTERS[{cluster_id!r}]: no such cluster" + for cluster_id in LLMD_CLUSTERS + if only in (None, cluster_id) and cluster_id not in clusters + ] for cluster_id in keys(LEGACY_TILERT): cluster = clusters.get(cluster_id) settings = cluster.scheduler_settings if cluster is not None else None @@ -248,4 +265,13 @@ def keys(table: Mapping[str, object]) -> list[str]: for name in lane.host_setup_env if name not in host_env ] + for cluster_id in LLMD_CLUSTERS: + if only not in (None, cluster_id): + continue + cluster = clusters.get(cluster_id) + settings = cluster.scheduler_settings if cluster is not None else None + if isinstance(settings, SlurmSettings) and settings.squash is None: + problems.append( + f"LLMD_CLUSTERS[{cluster_id!r}]: no slurm.squash for Pyxis image import" + ) return problems diff --git a/inferencex-e2e/infx/launch/request.py b/inferencex-e2e/infx/launch/request.py index fe8863b2ec..db7f35ff8c 100644 --- a/inferencex-e2e/infx/launch/request.py +++ b/inferencex-e2e/infx/launch/request.py @@ -176,3 +176,11 @@ class AmdUtilsRequest(LegacyRequest): model: str = Field(alias="MODEL") user: str | None = Field(None, alias="USER") keep_logs: OneFlag = Field(False, alias="KEEP_LOGS") + + +class LlmdRequest(SrtRequest): + """A GB200 llm-d vLLM multinode job submitted through benchmarks/multi_node/llm-d.""" + + model: str = Field(alias="MODEL") + exp_name: str = Field(alias="EXP_NAME") + disagg: TrueFlag = Field(alias="DISAGG") diff --git a/inferencex-e2e/infx/tests/launch/test_llmd_driver.py b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py new file mode 100644 index 0000000000..77b855324c --- /dev/null +++ b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py @@ -0,0 +1,115 @@ +"""GB200 llm-d vLLM launches through infx.launch instead of runners/launch_gb200-nv.sh.""" + +import json +import subprocess +import tarfile +from pathlib import Path + +import pytest + +from infx.tests.launch.fake_slurm import ( + base_env, + install_fakes, + launch, + make_workspace, + runner_for, + sandbox_runner_config, +) + +LLMD_SUBMIT = """#!/usr/bin/env bash +set -e +env > "$GITHUB_WORKSPACE/submitted.env" +logs="$BENCHMARK_LOGS_DIR" +job="$logs/slurm_job-4299" +mkdir -p "$logs/agentic/conc_128" "$job/eval_results" +echo '{"conc": 128}' > "$logs/point-identity_conc128.json" +echo trace > "$logs/agentic/conc_128/profile.json" +echo '{"score": 1}' > "$job/eval_results/results_gsm8k.json" +echo 'server log' > "$logs/server.log" +echo 'benchmark done' > "$logs/slurm_job-4299.out" +echo 'worker warning' > "$logs/slurm_job-4299.err" +echo 'submitting' >&2 +[[ "${NO_JOB_ID:-}" == 1 ]] && exit 1 +echo 4299 +""" + + +@pytest.fixture +def harness(tmp_path): + """Sandboxed gb200-nv cluster, fake Slurm binaries, and a workspace with the llm-d wrapper.""" + sandbox = tmp_path / "sandbox" + sandbox.mkdir() + config = sandbox_runner_config(sandbox) + workspace = make_workspace(tmp_path / "workspace") + wrapper = workspace / "benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh" + wrapper.parent.mkdir(parents=True, exist_ok=True) + wrapper.write_text(LLMD_SUBMIT) + wrapper.chmod(0o755) + + model_root = sandbox / "mnt/numa1/models/DeepSeek-V4-Pro" + model_root.mkdir(parents=True) + (model_root / "config.json").write_text("{}\n") + + logs = tmp_path / "logs" + env = base_env( + fakes=install_fakes(tmp_path / "bin"), logs=logs, workspace=workspace, sandbox=sandbox + ) + env.update( + RUNNER_NAME=runner_for("gb200-nv"), + IS_MULTINODE="true", + IS_AGENTIC="1", + RUN_EVAL="true", + EVAL_ONLY="false", + FRAMEWORK="llmd-vllm", + MODEL="deepseek-ai/DeepSeek-V4-Pro", + MODEL_PREFIX="dsv4", + PRECISION="fp4", + SPEC_DECODING="mtp", + THINKING_MODE="thinking_on", + EXP_NAME="dsv4_agentic_p1x8", + DISAGG="true", + IMAGE="vllm/vllm-openai:v0.21.0", + RESULT_FILENAME="point-identity", + ENROOT_IMPORT_TIME_LIMIT="10", + ) + return config, workspace, env + + +def run_launch(harness) -> subprocess.CompletedProcess[str]: + config, workspace, env = harness + return launch(env, config, workspace) + + +def test_llmd_driver_submits_the_wrapper_and_stages_artifacts(harness): + result = run_launch(harness) + config, workspace, env = harness + + assert result.returncode == 0, result.stdout + result.stderr + submitted = dict( + line.split("=", 1) for line in (workspace / "submitted.env").read_text().splitlines() if "=" in line + ) + model_path = f"{config.parent}/mnt/numa1/models/DeepSeek-V4-Pro" + assert submitted["MODEL_PATH"] == model_path + assert submitted["MODEL_NAME"] == env["MODEL"] + assert submitted["LLMD_CONTAINER_ENGINE"] == "pyxis" + assert submitted["LLMD_SQUASH_FILE"] + assert submitted["BENCHMARK_LOGS_DIR"] == f"{workspace}/benchmark_logs" + assert submitted["SLURM_PARTITION"] == "batch" + assert submitted["SLURM_ACCOUNT"] == "benchmark" + + assert json.loads((workspace / "point-identity_conc128.json").read_text()) == {"conc": 128} + assert (workspace / "LOGS/agentic/conc_128/profile.json").read_text() == "trace\n" + assert json.loads((workspace / "results_gsm8k.json").read_text()) == {"score": 1} + with tarfile.open(workspace / "multinode_server_logs.tar.gz") as bundle: + assert "./server.log" in bundle.getnames() + assert "submitting" in result.stderr + + +def test_llmd_driver_fails_when_the_wrapper_prints_no_job_id(harness): + _, workspace, env = harness + env["NO_JOB_ID"] = "1" + + result = launch(env, harness[0], workspace) + + assert result.returncode == 1 + assert "failed before returning a Slurm job id" in result.stderr diff --git a/inferencex-e2e/infx/tests/launch/test_srt_policy.py b/inferencex-e2e/infx/tests/launch/test_srt_policy.py index ce4b3d00d5..d8d8200a6c 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_policy.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_policy.py @@ -56,6 +56,7 @@ def cluster(tmp_path, single_node_models: str = "staged") -> Cluster: ("b200-nscale", dict(MULTI, MODEL_PREFIX="glm5.1", PRECISION="fp8", FRAMEWORK="tilert", SPEC_DECODING="mtp", IS_AGENTIC="0"), LaunchPath.LEGACY_TILERT), ("mi355x-amds", dict(IS_MULTINODE="true", FRAMEWORK="atom-disagg"), LaunchPath.LEGACY_AMD_UTILS), ("mi355x-amds", dict(MULTI, FRAMEWORK="sglang-disagg"), LaunchPath.SRT_MULTI), + ("gb200-nv", dict(MULTI, FRAMEWORK="llmd-vllm", MODEL_PREFIX="dsv4", PRECISION="fp4", SPEC_DECODING="mtp"), LaunchPath.LLMD), ("gb200-nv", dict(MULTI, FRAMEWORK="tilert"), LaunchPath.SRT_MULTI), ("b300-dsxe", dict(SINGLE, MODEL_PREFIX="dsv41flash", FRAMEWORK="sglang", IS_AGENTIC="1"), LaunchPath.SRT_BATCH), ("b300-dsxe", dict(SINGLE, MODEL_PREFIX="dsv41flash", FRAMEWORK="sglang", IS_AGENTIC="1", INFX_BATCH_REENTRY="1"), LaunchPath.SRT_SINGLE), From 4d0813ffd687d12ed4db7cf2a746f68d820f7286 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Wed, 30 Sep 2026 11:42:18 -0500 Subject: [PATCH 2/6] chore(launch): trim llm-d comments --- .../benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh | 2 -- inferencex-e2e/benchmarks/multi_node/llm-d/README.md | 2 +- inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm | 5 +---- inferencex-e2e/benchmarks/multi_node/llm-d/server.sh | 3 +-- inferencex-e2e/infx/tests/launch/test_llmd_driver.py | 2 +- 5 files changed, 4 insertions(+), 10 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh b/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh index d2a49f2dc3..537688298d 100755 --- a/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh +++ b/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh @@ -1,6 +1,4 @@ #!/usr/bin/env bash -# DeepSeek-V4-Pro FP4 GB200 llm-d vLLM P/D disagg. infx.launch (drivers/llmd.py) runs this -# and reads the Slurm job id submit.sh prints on stdout. set -eo pipefail export GPUS_PER_NODE=4 TIME_LIMIT="${TIME_LIMIT:-08:00:00}" CONTAINER_IMAGE="$IMAGE" diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/README.md b/inferencex-e2e/benchmarks/multi_node/llm-d/README.md index f8e34492dc..b7394da371 100644 --- a/inferencex-e2e/benchmarks/multi_node/llm-d/README.md +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/README.md @@ -9,7 +9,7 @@ InferenceX itself owns the SLURM job, no vendor multi-node tool involved. |---|---| | `submit.sh` | sbatch wrapper. Validates env, exports tuning vars, returns `JOB_ID`. May read `slurm.time_limit` from the recipe to override `TIME_LIMIT`. | | `job.slurm` | sbatch entrypoint. Allocates `PREFILL_NODES + DECODE_NODES` nodes, derives per-node IPs, runs one Docker container per node via `srun`, threads role assignment env into each. | -| `server.sh` | Per-node entry. Reads `NODE_RANK = SLURM_PROCID`, picks role, starts vLLM (with the wide-EP / DeepEP / NIXL flag set from the llm-d wide-EP-lws guide), starts the pd-sidecar on each leader, and on the decode leader additionally writes `endpoints.yaml`, starts EPP + Envoy, runs `benchmark_serving.py`, and writes a done marker; `job.slurm` then stops the srun step and exits 0 so the job ends `COMPLETED`. | +| `server.sh` | Per-node entry. Reads `NODE_RANK = SLURM_PROCID`, picks role, starts vLLM (with the wide-EP / DeepEP / NIXL flag set from the llm-d wide-EP-lws guide), starts the pd-sidecar on each leader, and on the decode leader additionally writes `endpoints.yaml`, starts EPP + Envoy, runs `benchmark_serving.py`, and signals `job.slurm` to end the job. | ## Topology diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm b/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm index b84931d66e..433dc33ee5 100644 --- a/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/job.slurm @@ -63,10 +63,7 @@ export DOCKER_CONT_NAME : "${BENCHMARK_LOGS_DIR:?BENCHMARK_LOGS_DIR not set}" DOCKER_MOUNT_PATH="/workspace" -# server.sh on the decode coordinator writes this marker once the benchmark and -# eval are done. Workers keep vLLM up until stopped, so the step is stopped then -# and the batch script exits 0: the job ends COMPLETED, the state infx.launch -# accepts as success. Cancelling the allocation instead would end it CANCELLED. +# Stop the step and exit 0 once the coordinator is done, so the job ends COMPLETED, not CANCELLED. BENCH_DONE_MARKER="$BENCHMARK_LOGS_DIR/.bench_done.$SLURM_JOB_ID" rm -f "$BENCH_DONE_MARKER" diff --git a/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh b/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh index 2f45cdeaad..0933a380ee 100755 --- a/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh +++ b/inferencex-e2e/benchmarks/multi_node/llm-d/server.sh @@ -569,8 +569,7 @@ PY ) fi - # Signal job.slurm (outside the container) to stop the srun step and exit 0; - # without it workers wait until TIME_LIMIT. + # job.slurm stops the srun step once this marker exists. touch "$BENCHMARK_LOGS_DIR/.bench_done.$SLURM_JOB_ID" else # Workers (prefill leader, prefill/decode workers): keep vLLM alive. diff --git a/inferencex-e2e/infx/tests/launch/test_llmd_driver.py b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py index 77b855324c..148e7244c1 100644 --- a/inferencex-e2e/infx/tests/launch/test_llmd_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py @@ -1,4 +1,4 @@ -"""GB200 llm-d vLLM launches through infx.launch instead of runners/launch_gb200-nv.sh.""" +"""The llm-d driver: submit through the wrapper, attach, stage artifacts.""" import json import subprocess From fcdd31172030e6d3abba870e83210cb2484904fd Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Wed, 30 Sep 2026 11:46:32 -0500 Subject: [PATCH 3/6] feat(launch): run llm-d submit.sh directly; re-add GB200 llm-d 8k1k test point Drop the per-model llm-d wrapper script. The driver passes topology, sequence lengths and concurrency to submit.sh and sets GPUS_PER_NODE, TIME_LIMIT, CONTAINER_IMAGE and worker counts itself. llmd-vllm supports only P/D disaggregated points. Re-add the dsv4-fp4-gb200-llmd-vllm low-latency 8k1k point (1P DEP8 + 1D TP8, conc 1) to exercise the path end to end. --- .../dsv4_fp4_gb200_llmd-vllm-disagg.sh | 10 ----- inferencex-e2e/configs/nvidia-master.yaml | 37 +++++++++++++++++ inferencex-e2e/infx/launch/drivers/llmd.py | 40 +++++++++++-------- inferencex-e2e/infx/launch/request.py | 8 +++- .../infx/tests/launch/test_llmd_driver.py | 22 +++++++--- inferencex-e2e/perf-changelog.yaml | 6 +++ 6 files changed, 90 insertions(+), 33 deletions(-) delete mode 100755 inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh diff --git a/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh b/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh deleted file mode 100755 index 537688298d..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh +++ /dev/null @@ -1,10 +0,0 @@ -#!/usr/bin/env bash -set -eo pipefail - -export GPUS_PER_NODE=4 TIME_LIMIT="${TIME_LIMIT:-08:00:00}" CONTAINER_IMAGE="$IMAGE" -export PREFILL_WORKERS="${PREFILL_WORKERS:-${PREFILL_NUM_WORKERS:-1}}" -export DECODE_WORKERS="${DECODE_WORKERS:-${DECODE_NUM_WORKERS:-1}}" - -cd "$(dirname "$0")/llm-d" -exec bash ./submit.sh "$PREFILL_NODES" "$DECODE_NODES" \ - "$ISL" "$OSL" "${CONC_LIST// /x}" inf "$RANDOM_RANGE_RATIO" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 07c47e6e40..1062873581 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -4076,6 +4076,43 @@ dsr1-fp4-b200-dynamo-sglang-mtp: ep: 8 dp-attn: true +dsv4-fp4-gb200-llmd-vllm: + image: quay.io/rh-ee-imarkov/llm-d-nokube-vllm:vllm0.26@sha256:a9095d4c835935c4070be2040de0a5ef3b44098f603092ab66d743b0e731b7b4 + model: deepseek-ai/DeepSeek-V4-Pro + model-prefix: dsv4 + runner: gb200 + precision: fp4 + framework: llmd-vllm + router: { name: llm-d-router, version: "0.9.0" } + kv-p2p-transfer: nixl + multinode: true + disagg: true + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + # Low latency: 1 prefill DEP8 + 1 decode TP8. + - spec-decoding: "none" + conc-list: [1] + prefill: + num-worker: 1 + tp: 1 + ep: 8 + dp-attn: true + additional-settings: + - "PREFILL_NODES=2" + - "GPUS_PER_NODE=4" + - "CONFIG_FILE=dsv4-fp4-gb200-low-latency.yaml" + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "DECODE_NODES=2" + - "GPUS_PER_NODE=4" + qwen3.5-fp8-gb200-dynamo-sglang: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 model: Qwen/Qwen3.5-397B-A17B-FP8 diff --git a/inferencex-e2e/infx/launch/drivers/llmd.py b/inferencex-e2e/infx/launch/drivers/llmd.py index 809104458a..60ceb67cd1 100644 --- a/inferencex-e2e/infx/launch/drivers/llmd.py +++ b/inferencex-e2e/infx/launch/drivers/llmd.py @@ -17,18 +17,8 @@ from infx.launch.request import LlmdRequest, RequestError CANCEL_TIMEOUT_S = 600.0 - - -def _bench_script(request: LlmdRequest) -> Path: - model_tag = request.exp_name.split("_", 1)[0] - kind = "disagg" if request.disagg else "agg" - script = ( - request.workspace - / f"benchmarks/multi_node/{model_tag}_{request.precision}_gb200_llmd-vllm-{kind}.sh" - ) - if not script.is_file(): - raise LaunchError(f"llm-d wrapper not found: {script}") - return script +DEFAULT_TIME_LIMIT = "08:00:00" +LLMD_DIR = "benchmarks/multi_node/llm-d" def _find_eval_dir(logs_dir: Path) -> Path | None: @@ -58,6 +48,8 @@ def run(launch: Launch) -> int: request = LlmdRequest.from_env(launch.request.env) if launch.cluster.id not in policy.LLMD_CLUSTERS: raise LaunchError(f"llmd-vllm is not configured for cluster {launch.cluster.id!r}") + if not request.disagg: + raise LaunchError("llmd-vllm supports only P/D disaggregated points") checkpoint = models.checkpoint(launch.cluster, request) if checkpoint is None: @@ -85,14 +77,28 @@ def run(launch: Launch) -> int: "SLURM_ACCOUNT": account, "MODEL_PATH": str(model_path), "MODEL_NAME": request.model, + "CONTAINER_IMAGE": request.image, + "GPUS_PER_NODE": str(launch.cluster.gpus_per_node), + "TIME_LIMIT": request.env.get("TIME_LIMIT") or DEFAULT_TIME_LIMIT, + "PREFILL_WORKERS": str(request.prefill_num_workers), + "DECODE_WORKERS": str(request.decode_num_workers), "LLMD_CONTAINER_ENGINE": "pyxis", "LLMD_SQUASH_FILE": squash.reference, "BENCHMARK_LOGS_DIR": str(logs_dir), }, ) - script = _bench_script(request) - argv = ["bash", str(script)] + argv = [ + "bash", + "submit.sh", + str(request.prefill_nodes), + str(request.decode_nodes), + str(request.isl), + str(request.osl), + "x".join(map(str, request.conc_list)), + "inf", + request.random_range_ratio, + ] proc.echo(argv, env) submitted = subprocess.run( argv, @@ -100,16 +106,16 @@ def run(launch: Launch) -> int: stderr=sys.stderr, text=True, env=env, - cwd=request.workspace, + cwd=request.workspace / LLMD_DIR, check=False, ) job_id = submitted.stdout.strip() if submitted.returncode != 0 or not job_id: - print("ERROR: llm-d submit wrapper failed before returning a Slurm job id", file=sys.stderr) + print("ERROR: llm-d submit.sh failed before returning a Slurm job id", file=sys.stderr) return 1 if not (job_id.isascii() and job_id.isdigit()): print( - f"ERROR: llm-d submit wrapper printed {job_id!r} instead of a Slurm job id", + f"ERROR: llm-d submit.sh printed {job_id!r} instead of a Slurm job id", file=sys.stderr, ) return 1 diff --git a/inferencex-e2e/infx/launch/request.py b/inferencex-e2e/infx/launch/request.py index db7f35ff8c..d9b05804f6 100644 --- a/inferencex-e2e/infx/launch/request.py +++ b/inferencex-e2e/infx/launch/request.py @@ -182,5 +182,11 @@ class LlmdRequest(SrtRequest): """A GB200 llm-d vLLM multinode job submitted through benchmarks/multi_node/llm-d.""" model: str = Field(alias="MODEL") - exp_name: str = Field(alias="EXP_NAME") disagg: TrueFlag = Field(alias="DISAGG") + prefill_nodes: int = Field(alias="PREFILL_NODES") + decode_nodes: int = Field(alias="DECODE_NODES") + prefill_num_workers: int = Field(1, alias="PREFILL_NUM_WORKERS") + decode_num_workers: int = Field(1, alias="DECODE_NUM_WORKERS") + isl: int = Field(alias="ISL") + osl: int = Field(alias="OSL") + random_range_ratio: str = Field(alias="RANDOM_RANGE_RATIO") diff --git a/inferencex-e2e/infx/tests/launch/test_llmd_driver.py b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py index 148e7244c1..fdabded421 100644 --- a/inferencex-e2e/infx/tests/launch/test_llmd_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py @@ -1,4 +1,4 @@ -"""The llm-d driver: submit through the wrapper, attach, stage artifacts.""" +"""The llm-d driver: run submit.sh, attach, stage artifacts.""" import json import subprocess @@ -19,6 +19,7 @@ LLMD_SUBMIT = """#!/usr/bin/env bash set -e env > "$GITHUB_WORKSPACE/submitted.env" +echo "$PWD $*" > "$GITHUB_WORKSPACE/submitted.args" logs="$BENCHMARK_LOGS_DIR" job="$logs/slurm_job-4299" mkdir -p "$logs/agentic/conc_128" "$job/eval_results" @@ -36,12 +37,12 @@ @pytest.fixture def harness(tmp_path): - """Sandboxed gb200-nv cluster, fake Slurm binaries, and a workspace with the llm-d wrapper.""" + """Sandboxed gb200-nv cluster, fake Slurm binaries, and a workspace with a fake submit.sh.""" sandbox = tmp_path / "sandbox" sandbox.mkdir() config = sandbox_runner_config(sandbox) workspace = make_workspace(tmp_path / "workspace") - wrapper = workspace / "benchmarks/multi_node/dsv4_fp4_gb200_llmd-vllm-disagg.sh" + wrapper = workspace / "benchmarks/multi_node/llm-d/submit.sh" wrapper.parent.mkdir(parents=True, exist_ok=True) wrapper.write_text(LLMD_SUBMIT) wrapper.chmod(0o755) @@ -66,7 +67,12 @@ def harness(tmp_path): PRECISION="fp4", SPEC_DECODING="mtp", THINKING_MODE="thinking_on", - EXP_NAME="dsv4_agentic_p1x8", + PREFILL_NODES="2", + DECODE_NODES="2", + ISL="8192", + OSL="1024", + RANDOM_RANGE_RATIO="0.8", + CONC_LIST="256 512", DISAGG="true", IMAGE="vllm/vllm-openai:v0.21.0", RESULT_FILENAME="point-identity", @@ -96,6 +102,12 @@ def test_llmd_driver_submits_the_wrapper_and_stages_artifacts(harness): assert submitted["BENCHMARK_LOGS_DIR"] == f"{workspace}/benchmark_logs" assert submitted["SLURM_PARTITION"] == "batch" assert submitted["SLURM_ACCOUNT"] == "benchmark" + assert submitted["GPUS_PER_NODE"] == "4" + assert submitted["CONTAINER_IMAGE"] == env["IMAGE"] + assert (workspace / "submitted.args").read_text().split() == [ + str(workspace / "benchmarks/multi_node/llm-d"), + *"2 2 8192 1024 256x512 inf 0.8".split(), + ] assert json.loads((workspace / "point-identity_conc128.json").read_text()) == {"conc": 128} assert (workspace / "LOGS/agentic/conc_128/profile.json").read_text() == "trace\n" @@ -112,4 +124,4 @@ def test_llmd_driver_fails_when_the_wrapper_prints_no_job_id(harness): result = launch(env, harness[0], workspace) assert result.returncode == 1 - assert "failed before returning a Slurm job id" in result.stderr + assert "submit.sh failed before returning a Slurm job id" in result.stderr diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index d7fc64f0d6..ef6a776139 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9149,3 +9149,9 @@ - "Use AITER attention and allreduce fusion, FP8 KV (fp8_e4m3), and EAGLE MTP (3 steps, topk 1, 4 draft tokens); do not enable ROCm INT4 quick all-reduce. HiCache uses ratio 1.5, write_through_selective, kernel I/O and page_first layout. The matching cookbook recipe is sgl-project/sglang#41849." - "The embedded MTP head runs at its stored precision: block-FP8 expert weights and checkpoint dtype for mtp.fc and gates. No separate draft, draft dtype override or submission-side quantization is used." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3602 + +- config-keys: + - dsv4-fp4-gb200-llmd-vllm + description: + - "Re-add the GB200 llm-d low-latency 8k1k point (1P DEP8 + 1D TP8, conc 1) to validate llm-d on infx.launch" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3613 From 4b2ffc9c36fa8ad2f480b0008533425251d8747a Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Wed, 30 Sep 2026 11:55:13 -0500 Subject: [PATCH 4/6] refactor(launch): drop the llm-d cluster whitelist Route every multinode llmd-vllm request to the llm-d driver. The driver requires slurm.squash and a staged checkpoint, which is what the whitelist stood in for. --- inferencex-e2e/infx/launch/drivers/llmd.py | 6 +++--- inferencex-e2e/infx/launch/policy.py | 23 ---------------------- inferencex-e2e/infx/launch/request.py | 2 +- 3 files changed, 4 insertions(+), 27 deletions(-) diff --git a/inferencex-e2e/infx/launch/drivers/llmd.py b/inferencex-e2e/infx/launch/drivers/llmd.py index 60ceb67cd1..4ac3756707 100644 --- a/inferencex-e2e/infx/launch/drivers/llmd.py +++ b/inferencex-e2e/infx/launch/drivers/llmd.py @@ -1,4 +1,4 @@ -"""GB200 llm-d vLLM multinode jobs submitted through benchmarks/multi_node/llm-d/submit.sh.""" +"""llm-d vLLM multinode jobs submitted through benchmarks/multi_node/llm-d/submit.sh.""" from __future__ import annotations @@ -46,8 +46,8 @@ def run(launch: Launch) -> int: """Submit the llm-d Slurm job, follow its log, and stage benchmark artifacts.""" backend = slurm_backend(launch) request = LlmdRequest.from_env(launch.request.env) - if launch.cluster.id not in policy.LLMD_CLUSTERS: - raise LaunchError(f"llmd-vllm is not configured for cluster {launch.cluster.id!r}") + if backend.settings.squash is None: + raise LaunchError(f"llmd-vllm: cluster {launch.cluster.id!r} has no slurm.squash") if not request.disagg: raise LaunchError("llmd-vllm supports only P/D disaggregated points") diff --git a/inferencex-e2e/infx/launch/policy.py b/inferencex-e2e/infx/launch/policy.py index 76c135500e..7819422343 100644 --- a/inferencex-e2e/infx/launch/policy.py +++ b/inferencex-e2e/infx/launch/policy.py @@ -15,7 +15,6 @@ from typing import TYPE_CHECKING from infx.clusters.slurm import SlurmSettings -from infx.launch.context import LaunchError if TYPE_CHECKING: from infx.clusters import Cluster @@ -81,17 +80,9 @@ class LaunchPath(StrEnum): } -LLMD_CLUSTERS: frozenset[str] = frozenset({"gb200-nv"}) - - def launch_path(cluster_id: str, request: LaunchRequest) -> LaunchPath: if request.is_multinode: if request.framework == "llmd-vllm": - if cluster_id not in LLMD_CLUSTERS: - raise LaunchError( - f"llmd-vllm is not configured for cluster {cluster_id!r}; " - f"supported clusters: {', '.join(sorted(LLMD_CLUSTERS))}" - ) return LaunchPath.LLMD if any(lane(request) for lane in NATIVE_SRT_LANES.get(cluster_id, ())): return LaunchPath.SRT_NATIVE @@ -240,11 +231,6 @@ def keys(table: Mapping[str, object]) -> list[str]: for key in keys(table) if key not in clusters ] - problems += [ - f"LLMD_CLUSTERS[{cluster_id!r}]: no such cluster" - for cluster_id in LLMD_CLUSTERS - if only in (None, cluster_id) and cluster_id not in clusters - ] for cluster_id in keys(LEGACY_TILERT): cluster = clusters.get(cluster_id) settings = cluster.scheduler_settings if cluster is not None else None @@ -265,13 +251,4 @@ def keys(table: Mapping[str, object]) -> list[str]: for name in lane.host_setup_env if name not in host_env ] - for cluster_id in LLMD_CLUSTERS: - if only not in (None, cluster_id): - continue - cluster = clusters.get(cluster_id) - settings = cluster.scheduler_settings if cluster is not None else None - if isinstance(settings, SlurmSettings) and settings.squash is None: - problems.append( - f"LLMD_CLUSTERS[{cluster_id!r}]: no slurm.squash for Pyxis image import" - ) return problems diff --git a/inferencex-e2e/infx/launch/request.py b/inferencex-e2e/infx/launch/request.py index d9b05804f6..a7acbb3579 100644 --- a/inferencex-e2e/infx/launch/request.py +++ b/inferencex-e2e/infx/launch/request.py @@ -179,7 +179,7 @@ class AmdUtilsRequest(LegacyRequest): class LlmdRequest(SrtRequest): - """A GB200 llm-d vLLM multinode job submitted through benchmarks/multi_node/llm-d.""" + """An llm-d vLLM multinode job submitted through benchmarks/multi_node/llm-d.""" model: str = Field(alias="MODEL") disagg: TrueFlag = Field(alias="DISAGG") From 62cb067f924eb1c3e190bd05ade1fc06af05e269 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Thu, 1 Oct 2026 13:23:04 -0500 Subject: [PATCH 5/6] fix(launch): leave node-local llm-d checkpoints to the job /mnt/numa1 exists only on compute nodes, so the runner host cannot see the checkpoint. job.slurm already checks it on every allocated node. --- inferencex-e2e/infx/launch/drivers/llmd.py | 2 +- inferencex-e2e/infx/tests/launch/test_llmd_driver.py | 9 +++++++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/inferencex-e2e/infx/launch/drivers/llmd.py b/inferencex-e2e/infx/launch/drivers/llmd.py index 4ac3756707..92587dd90b 100644 --- a/inferencex-e2e/infx/launch/drivers/llmd.py +++ b/inferencex-e2e/infx/launch/drivers/llmd.py @@ -57,7 +57,7 @@ def run(launch: Launch) -> int: f"cluster {launch.cluster.id!r} stages no checkpoint for MODEL={request.model}" ) model_path = models.host_path(launch.cluster, checkpoint) - if not (model_path / "config.json").is_file(): + if not checkpoint.node_local and not (model_path / "config.json").is_file(): raise LaunchError(f"model checkpoint is unavailable: {model_path / 'config.json'}") squash = backend.prepare_image(request.image) diff --git a/inferencex-e2e/infx/tests/launch/test_llmd_driver.py b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py index fdabded421..179fa083b4 100644 --- a/inferencex-e2e/infx/tests/launch/test_llmd_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_llmd_driver.py @@ -125,3 +125,12 @@ def test_llmd_driver_fails_when_the_wrapper_prints_no_job_id(harness): assert result.returncode == 1 assert "submit.sh failed before returning a Slurm job id" in result.stderr + + +def test_llmd_driver_leaves_node_local_checkpoints_to_the_job(harness): + config, workspace, env = harness + (config.parent / "mnt/numa1/models/DeepSeek-V4-Pro/config.json").unlink() + + result = launch(env, config, workspace) + + assert result.returncode == 0, result.stdout + result.stderr From 7fd158a17131c73a700c496acbebf98166e5effd Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Fri, 2 Oct 2026 16:57:24 -0500 Subject: [PATCH 6/6] chore: drop the GB200 llm-d 8k1k test point The point validated the llm-d path end to end in e2e run 36906427804; it is not a benchmark to publish. --- inferencex-e2e/configs/nvidia-master.yaml | 37 ----------------------- inferencex-e2e/perf-changelog.yaml | 6 ---- 2 files changed, 43 deletions(-) diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 0cca1c7b47..0e1f65e659 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -4076,43 +4076,6 @@ dsr1-fp4-b200-dynamo-sglang-mtp: ep: 8 dp-attn: true -dsv4-fp4-gb200-llmd-vllm: - image: quay.io/rh-ee-imarkov/llm-d-nokube-vllm:vllm0.26@sha256:a9095d4c835935c4070be2040de0a5ef3b44098f603092ab66d743b0e731b7b4 - model: deepseek-ai/DeepSeek-V4-Pro - model-prefix: dsv4 - runner: gb200 - precision: fp4 - framework: llmd-vllm - router: { name: llm-d-router, version: "0.9.0" } - kv-p2p-transfer: nixl - multinode: true - disagg: true - scenarios: - fixed-seq-len: - - isl: 8192 - osl: 1024 - search-space: - # Low latency: 1 prefill DEP8 + 1 decode TP8. - - spec-decoding: "none" - conc-list: [1] - prefill: - num-worker: 1 - tp: 1 - ep: 8 - dp-attn: true - additional-settings: - - "PREFILL_NODES=2" - - "GPUS_PER_NODE=4" - - "CONFIG_FILE=dsv4-fp4-gb200-low-latency.yaml" - decode: - num-worker: 1 - tp: 8 - ep: 1 - dp-attn: false - additional-settings: - - "DECODE_NODES=2" - - "GPUS_PER_NODE=4" - qwen3.5-fp8-gb200-dynamo-sglang: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 model: Qwen/Qwen3.5-397B-A17B-FP8 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 385c64fc45..b5f4f5572b 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9275,9 +9275,3 @@ - "Update the GB300 DeepSeek-V4.1-Flash vLLM AgentX image from nightly ac68c308 to nightly-dev-arm64-cu130-ac9126e58aa7 and enable FlashInfer autotuning." - "Run TP4 at concurrency 1-16 and replace TP2 with DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) at concurrency 8-192 behind a consistent-hash vLLM Router." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3652 - -- config-keys: - - dsv4-fp4-gb200-llmd-vllm - description: - - "Re-add the GB200 llm-d low-latency 8k1k point (1P DEP8 + 1D TP8, conc 1) to validate llm-d on infx.launch" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3613