From 38cc815ca8513913753f1ea370b4702adcb04e83 Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Tue, 29 Sep 2026 01:26:17 +0000 Subject: [PATCH 1/6] fix: align four single-node srt-slurm recipe images with master configs #3428 ported these AgentX recipes from the legacy scripts as they were before the 09-25 image bumps (#3361, #3362, #3334, #3420), which changed only the master images. Every point of these keys now fails before submission with 'Single-node SRT image: recipe/matrix'. The SGLang v0.5.20 recipes also take the --cuda-graph-max-bs-decode rename that #3362 and #3334 applied to the legacy scripts; v0.5.20 no longer accepts the deprecated --cuda-graph-max-bs alias (sgl-project/sglang#38375). Co-authored-by: Cursor --- .../dsv4/sglang/b200-fp4-mtp/agentic.yaml | 26 +++++++++---------- .../vllm/mi355x-fp4-mtp/agentic.yaml | 2 +- .../glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml | 16 ++++++------ .../vllm/mi300x-fp8-mtp/agentic.yaml | 2 +- 4 files changed, 23 insertions(+), 23 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b200-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b200-fp4-mtp/agentic.yaml index 36bba702cb..65551cf4ae 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b200-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b200-fp4-mtp/agentic.yaml @@ -6,7 +6,7 @@ base: name: dsv4-fp4-b200-sglang-agentic model: path: hf:deepseek-ai/DeepSeek-V4-Pro-0813 - container: lmsysorg/sglang:v0.5.19-cu130 + container: lmsysorg/sglang:v0.5.20-cu130 precision: fp4 resources: gpu_type: b200 @@ -85,7 +85,7 @@ override_tp8_c1: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 2 - cuda-graph-max-bs: 2 + cuda-graph-max-bs-decode: 2 benchmark: env: CONC: '1' @@ -100,7 +100,7 @@ override_tp8_c2: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 4 - cuda-graph-max-bs: 4 + cuda-graph-max-bs-decode: 4 benchmark: env: CONC: '2' @@ -115,7 +115,7 @@ override_tp8_c3: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 6 - cuda-graph-max-bs: 6 + cuda-graph-max-bs-decode: 6 benchmark: env: CONC: '3' @@ -130,7 +130,7 @@ override_tp8_c4: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 8 - cuda-graph-max-bs: 8 + cuda-graph-max-bs-decode: 8 benchmark: env: CONC: '4' @@ -145,7 +145,7 @@ override_tp8_c5: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 10 - cuda-graph-max-bs: 10 + cuda-graph-max-bs-decode: 10 benchmark: env: CONC: '5' @@ -160,7 +160,7 @@ override_tp8_hicache_c8: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 16 - cuda-graph-max-bs: 16 + cuda-graph-max-bs-decode: 16 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -181,7 +181,7 @@ override_tp8_hicache_c10: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 20 - cuda-graph-max-bs: 20 + cuda-graph-max-bs-decode: 20 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -202,7 +202,7 @@ override_tp8_hicache_c16: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 32 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -246,7 +246,7 @@ override_dep8_hicache_c64: swa-full-tokens-ratio: 0.02 chunked-prefill-size: 49152 max-running-requests: 128 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -294,7 +294,7 @@ override_dep8_hicache_c96: swa-full-tokens-ratio: 0.02 chunked-prefill-size: 49152 max-running-requests: 192 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -342,7 +342,7 @@ override_dep8_hicache_c128: swa-full-tokens-ratio: 0.02 chunked-prefill-size: 49152 max-running-requests: 256 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -391,7 +391,7 @@ override_dep8_hicache_c160: swa-full-tokens-ratio: 0.02 chunked-prefill-size: 49152 max-running-requests: 320 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml index 9ab84be6dd..fc67c49e8b 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml @@ -6,7 +6,7 @@ base: name: dsv41flash-fp4-mi355x-vllm-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - container: vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 + container: vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098 precision: fp4 resources: gpu_type: mi355x diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml index f329216a2d..6a94c07703 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml @@ -5,7 +5,7 @@ base: name: glm5.2-fp8-mi325x-sglang-agentic model: path: hf:zai-org/GLM-5.2-FP8 - container: lmsysorg/sglang:v0.5.19-rocm720-mi30x + container: lmsysorg/sglang:v0.5.20-rocm720-mi30x precision: fp8 resources: gpu_type: mi325x @@ -76,7 +76,7 @@ override_tp8_c1: agg: args: max-running-requests: 2 - cuda-graph-max-bs: 2 + cuda-graph-max-bs-decode: 2 benchmark: env: CONC: '1' @@ -86,7 +86,7 @@ override_tp8_c2: agg: args: max-running-requests: 4 - cuda-graph-max-bs: 4 + cuda-graph-max-bs-decode: 4 benchmark: env: CONC: '2' @@ -96,7 +96,7 @@ override_tp8_c3: agg: args: max-running-requests: 6 - cuda-graph-max-bs: 6 + cuda-graph-max-bs-decode: 6 benchmark: env: CONC: '3' @@ -106,7 +106,7 @@ override_tp8_c4: agg: args: max-running-requests: 8 - cuda-graph-max-bs: 8 + cuda-graph-max-bs-decode: 8 benchmark: env: CONC: '4' @@ -116,7 +116,7 @@ override_tp8_c5: agg: args: max-running-requests: 10 - cuda-graph-max-bs: 10 + cuda-graph-max-bs-decode: 10 benchmark: env: CONC: '5' @@ -126,7 +126,7 @@ override_tp8_c6: agg: args: max-running-requests: 12 - cuda-graph-max-bs: 12 + cuda-graph-max-bs-decode: 12 benchmark: env: CONC: '6' @@ -136,7 +136,7 @@ override_tp8_c8: agg: args: max-running-requests: 16 - cuda-graph-max-bs: 16 + cuda-graph-max-bs-decode: 16 benchmark: env: CONC: '8' diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml index 151ce47cc7..7abe435501 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml @@ -5,7 +5,7 @@ base: name: minimaxm3-fp8-mi300x-vllm-agentic model: path: hf:MiniMaxAI/MiniMax-M3-MXFP8 - container: vllm/vllm-openai-rocm:v0.29.0 + container: vllm/vllm-openai-rocm:v0.30.0 precision: fp8 resources: gpu_type: mi300x From 8c6adba73abdb73dca59a1b0677e9f3796c830e1 Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Tue, 29 Sep 2026 01:26:17 +0000 Subject: [PATCH 2/6] test: check srt-slurm recipe images against master configs single_node.py rejects a point whose recipe container differs from the matrix image only once a GPU job starts, and the multi-node path has no such check: srtctl pulls a literal container missing from the alias map. Check both statically, including identity.container.image. Co-authored-by: Cursor --- .../tests/srt_slurm/test_recipe_images.py | 89 +++++++++++++++++++ 1 file changed, 89 insertions(+) create mode 100644 inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py b/inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py new file mode 100644 index 0000000000..a4310d848c --- /dev/null +++ b/inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py @@ -0,0 +1,89 @@ +"""Keep checked-in srt-slurm recipe images identical to their master-config images.""" + +import sys +from collections import defaultdict +from pathlib import Path + +import yaml + +from infx.srt_slurm.synthetic_acceptance import selected_recipes + +ROOT = Path(__file__).resolve().parents[3] +sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) +MULTI_NODE_RECIPES = "benchmarks/multi_node/srt-slurm-recipes/" + + +def recipe_references(node): + """Yield (multinode, reference) for every srt-slurm recipe a master entry names.""" + if isinstance(node, dict): + for key, value in node.items(): + if key == "srt-recipe": + yield False, value + else: + yield from recipe_references(value) + elif isinstance(node, list): + for item in node: + yield from recipe_references(item) + elif isinstance(node, str) and node.startswith("CONFIG_FILE="): + path = node.removeprefix("CONFIG_FILE=") + # Launchers stage multi-node recipes under recipes/ in the srt-slurm checkout. + for staged in ("recipes/", MULTI_NODE_RECIPES): + if path.startswith(staged): + yield True, MULTI_NODE_RECIPES + path.removeprefix(staged) + + +def master_references(): + """Yield (config key, image, multinode, reference) from every master config.""" + for master in sorted((ROOT / "configs").glob("*-master.yaml")): + for key, entry in yaml.safe_load(master.read_text()).items(): + for multinode, reference in sorted(set(recipe_references(entry))): + yield key, entry["image"], multinode, reference + + +def variants(reference): + path, _, selector = reference.partition(":") + recipe = yaml.safe_load((ROOT / path).read_text()) + return path, selected_recipes(recipe, selector or None) + + +def canonical_image(image): + # Enroot spells the nvcr.io registry separator as "#"; launchers key aliases either way. + return image.replace("nvcr.io#", "nvcr.io/", 1) + + +def test_single_node_recipes_use_their_master_image(): + references = [ + (key, image, reference) + for key, image, multinode, reference in master_references() + if not multinode + ] + images = defaultdict(set) + for _, image, reference in references: + images[reference.partition(":")[0]].add(image) + problems = set() + for key, image, reference in references: + path, selected = variants(reference) + containers = {recipe["model"]["container"] for _, recipe in selected} + if image not in containers: + problems.add(f"{key}: no variant of {reference} uses master image {image}") + for container in containers - images[path]: + problems.add(f"{path}: {container} is not the image of any master key using it") + assert not problems, "\n".join(sorted(problems)) + + +def test_multi_node_recipes_use_their_master_image(): + problems = set() + for key, image, multinode, reference in master_references(): + if not multinode: + continue + path, selected = variants(reference) + for name, recipe in selected: + label = f"{key}: {path}" + (f":{name}" if name else "") + container = recipe["model"]["container"] + # srtctl pulls a literal missing from the alias map, so a stale one runs silently. + if ":" in container and canonical_image(container) != canonical_image(image): + problems.add(f"{label} model.container {container} != master image {image}") + identity = ((recipe.get("identity") or {}).get("container") or {}).get("image") + if identity is not None and canonical_image(identity) != canonical_image(image): + problems.add(f"{label} identity.container.image {identity} != master image {image}") + assert not problems, "\n".join(sorted(problems)) From 75c7220f153286de3c52d769a714cd7a3956057f Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Tue, 29 Sep 2026 01:40:52 +0000 Subject: [PATCH 3/6] chore: record srt-slurm recipe image alignment in perf-changelog Co-authored-by: Cursor --- inferencex-e2e/perf-changelog.yaml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 6c28e71ec8..fa6500597d 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9024,3 +9024,19 @@ description: - "Pin B200 DeepSeek-V4.1-Flash vLLM to nightly ddd6fbca with FlashInfer sparse attention, MXFP4 indexer KV, sparse logits, fp8 KV cache, and a 7200 s readiness timeout." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3459 + +- config-keys: + - glm5.2-fp8-mi325x-sglang-agentic-mtp + - minimaxm3-fp8-mi300x-vllm-agentic-mtp + - dsv41flash-fp4-mi355x-vllm-agentic-dspark + - dsv4-fp4-b200-sglang-agentic-hicache-mtp + scenario-type: + - agentic-coding + description: + - "Align the single-node srt-slurm recipe model.container with the unchanged master image for four AgentX keys whose recipes #3428 ported from the legacy scripts as they were before the image bumps in #3361, #3362, #3334 and #3420; every point of these keys would fail before submission with 'Single-node SRT image: recipe/matrix'." + - "Recipe containers: glm5.2-fp8-mi325x-sglang-agentic-mtp lmsysorg/sglang:v0.5.19-rocm720-mi30x -> lmsysorg/sglang:v0.5.20-rocm720-mi30x; minimaxm3-fp8-mi300x-vllm-agentic-mtp vllm/vllm-openai-rocm:v0.29.0 -> vllm/vllm-openai-rocm:v0.30.0; dsv41flash-fp4-mi355x-vllm-agentic-dspark vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 -> vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098; dsv4-fp4-b200-sglang-agentic-hicache-mtp lmsysorg/sglang:v0.5.19-cu130 -> lmsysorg/sglang:v0.5.20-cu130." + - "The two SGLang v0.5.20 recipes also rename cuda-graph-max-bs to cuda-graph-max-bs-decode, as #3362 and #3334 did in the legacy scripts, because v0.5.20 removed the deprecated alias (sgl-project/sglang#38375). All other serving flags and sweep points are unchanged." + - "将四个 AgentX key 的单节点 srt-slurm 配方 model.container 与未改动的主配置镜像对齐。这些配方由 #3428 从旧脚本移植而来,而移植所依据的是 #3361、#3362、#3334 和 #3420 升级镜像之前的版本;此前这些 key 的每个点都会在提交前以 'Single-node SRT image: recipe/matrix' 失败。" + - "配方镜像:glm5.2-fp8-mi325x-sglang-agentic-mtp lmsysorg/sglang:v0.5.19-rocm720-mi30x -> lmsysorg/sglang:v0.5.20-rocm720-mi30x;minimaxm3-fp8-mi300x-vllm-agentic-mtp vllm/vllm-openai-rocm:v0.29.0 -> vllm/vllm-openai-rocm:v0.30.0;dsv41flash-fp4-mi355x-vllm-agentic-dspark vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 -> vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098;dsv4-fp4-b200-sglang-agentic-hicache-mtp lmsysorg/sglang:v0.5.19-cu130 -> lmsysorg/sglang:v0.5.20-cu130。" + - "两个 SGLang v0.5.20 配方同时将 cuda-graph-max-bs 改为 cuda-graph-max-bs-decode,与 #3362 和 #3334 对旧脚本的修改一致,因为 v0.5.20 已移除该弃用别名(sgl-project/sglang#38375)。其余服务参数和 sweep 点均保持不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX From 926da189d483b2ee8e234eca0545e46962a20561 Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Tue, 29 Sep 2026 03:15:28 +0000 Subject: [PATCH 4/6] chore: set perf-changelog PR link to #3567 Co-authored-by: Cursor --- inferencex-e2e/perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index fa6500597d..5170df0626 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9039,4 +9039,4 @@ - "将四个 AgentX key 的单节点 srt-slurm 配方 model.container 与未改动的主配置镜像对齐。这些配方由 #3428 从旧脚本移植而来,而移植所依据的是 #3361、#3362、#3334 和 #3420 升级镜像之前的版本;此前这些 key 的每个点都会在提交前以 'Single-node SRT image: recipe/matrix' 失败。" - "配方镜像:glm5.2-fp8-mi325x-sglang-agentic-mtp lmsysorg/sglang:v0.5.19-rocm720-mi30x -> lmsysorg/sglang:v0.5.20-rocm720-mi30x;minimaxm3-fp8-mi300x-vllm-agentic-mtp vllm/vllm-openai-rocm:v0.29.0 -> vllm/vllm-openai-rocm:v0.30.0;dsv41flash-fp4-mi355x-vllm-agentic-dspark vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 -> vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098;dsv4-fp4-b200-sglang-agentic-hicache-mtp lmsysorg/sglang:v0.5.19-cu130 -> lmsysorg/sglang:v0.5.20-cu130。" - "两个 SGLang v0.5.20 配方同时将 cuda-graph-max-bs 改为 cuda-graph-max-bs-decode,与 #3362 和 #3334 对旧脚本的修改一致,因为 v0.5.20 已移除该弃用别名(sgl-project/sglang#38375)。其余服务参数和 sweep 点均保持不变。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3567 From f8544be3b083794822b1f163ebe9f90d34503ce7 Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Tue, 29 Sep 2026 03:35:32 +0000 Subject: [PATCH 5/6] chore: narrow the recipe image fix to B200 and MI355X Restore the MI325X GLM-5.2 and MI300X MiniMax-M3 recipes and drop the static image test from this PR; they will follow separately. The changelog entry now selects only the B200 and MI355X keys. Co-authored-by: Cursor --- .../glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml | 16 ++-- .../vllm/mi300x-fp8-mtp/agentic.yaml | 2 +- .../tests/srt_slurm/test_recipe_images.py | 89 ------------------- inferencex-e2e/perf-changelog.yaml | 14 ++- 4 files changed, 15 insertions(+), 106 deletions(-) delete mode 100644 inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml index 6a94c07703..f329216a2d 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/glm5.2/sglang/mi325x-fp8-mtp/agentic.yaml @@ -5,7 +5,7 @@ base: name: glm5.2-fp8-mi325x-sglang-agentic model: path: hf:zai-org/GLM-5.2-FP8 - container: lmsysorg/sglang:v0.5.20-rocm720-mi30x + container: lmsysorg/sglang:v0.5.19-rocm720-mi30x precision: fp8 resources: gpu_type: mi325x @@ -76,7 +76,7 @@ override_tp8_c1: agg: args: max-running-requests: 2 - cuda-graph-max-bs-decode: 2 + cuda-graph-max-bs: 2 benchmark: env: CONC: '1' @@ -86,7 +86,7 @@ override_tp8_c2: agg: args: max-running-requests: 4 - cuda-graph-max-bs-decode: 4 + cuda-graph-max-bs: 4 benchmark: env: CONC: '2' @@ -96,7 +96,7 @@ override_tp8_c3: agg: args: max-running-requests: 6 - cuda-graph-max-bs-decode: 6 + cuda-graph-max-bs: 6 benchmark: env: CONC: '3' @@ -106,7 +106,7 @@ override_tp8_c4: agg: args: max-running-requests: 8 - cuda-graph-max-bs-decode: 8 + cuda-graph-max-bs: 8 benchmark: env: CONC: '4' @@ -116,7 +116,7 @@ override_tp8_c5: agg: args: max-running-requests: 10 - cuda-graph-max-bs-decode: 10 + cuda-graph-max-bs: 10 benchmark: env: CONC: '5' @@ -126,7 +126,7 @@ override_tp8_c6: agg: args: max-running-requests: 12 - cuda-graph-max-bs-decode: 12 + cuda-graph-max-bs: 12 benchmark: env: CONC: '6' @@ -136,7 +136,7 @@ override_tp8_c8: agg: args: max-running-requests: 16 - cuda-graph-max-bs-decode: 16 + cuda-graph-max-bs: 16 benchmark: env: CONC: '8' diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml index 7abe435501..151ce47cc7 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml @@ -5,7 +5,7 @@ base: name: minimaxm3-fp8-mi300x-vllm-agentic model: path: hf:MiniMaxAI/MiniMax-M3-MXFP8 - container: vllm/vllm-openai-rocm:v0.30.0 + container: vllm/vllm-openai-rocm:v0.29.0 precision: fp8 resources: gpu_type: mi300x diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py b/inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py deleted file mode 100644 index a4310d848c..0000000000 --- a/inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py +++ /dev/null @@ -1,89 +0,0 @@ -"""Keep checked-in srt-slurm recipe images identical to their master-config images.""" - -import sys -from collections import defaultdict -from pathlib import Path - -import yaml - -from infx.srt_slurm.synthetic_acceptance import selected_recipes - -ROOT = Path(__file__).resolve().parents[3] -sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) -MULTI_NODE_RECIPES = "benchmarks/multi_node/srt-slurm-recipes/" - - -def recipe_references(node): - """Yield (multinode, reference) for every srt-slurm recipe a master entry names.""" - if isinstance(node, dict): - for key, value in node.items(): - if key == "srt-recipe": - yield False, value - else: - yield from recipe_references(value) - elif isinstance(node, list): - for item in node: - yield from recipe_references(item) - elif isinstance(node, str) and node.startswith("CONFIG_FILE="): - path = node.removeprefix("CONFIG_FILE=") - # Launchers stage multi-node recipes under recipes/ in the srt-slurm checkout. - for staged in ("recipes/", MULTI_NODE_RECIPES): - if path.startswith(staged): - yield True, MULTI_NODE_RECIPES + path.removeprefix(staged) - - -def master_references(): - """Yield (config key, image, multinode, reference) from every master config.""" - for master in sorted((ROOT / "configs").glob("*-master.yaml")): - for key, entry in yaml.safe_load(master.read_text()).items(): - for multinode, reference in sorted(set(recipe_references(entry))): - yield key, entry["image"], multinode, reference - - -def variants(reference): - path, _, selector = reference.partition(":") - recipe = yaml.safe_load((ROOT / path).read_text()) - return path, selected_recipes(recipe, selector or None) - - -def canonical_image(image): - # Enroot spells the nvcr.io registry separator as "#"; launchers key aliases either way. - return image.replace("nvcr.io#", "nvcr.io/", 1) - - -def test_single_node_recipes_use_their_master_image(): - references = [ - (key, image, reference) - for key, image, multinode, reference in master_references() - if not multinode - ] - images = defaultdict(set) - for _, image, reference in references: - images[reference.partition(":")[0]].add(image) - problems = set() - for key, image, reference in references: - path, selected = variants(reference) - containers = {recipe["model"]["container"] for _, recipe in selected} - if image not in containers: - problems.add(f"{key}: no variant of {reference} uses master image {image}") - for container in containers - images[path]: - problems.add(f"{path}: {container} is not the image of any master key using it") - assert not problems, "\n".join(sorted(problems)) - - -def test_multi_node_recipes_use_their_master_image(): - problems = set() - for key, image, multinode, reference in master_references(): - if not multinode: - continue - path, selected = variants(reference) - for name, recipe in selected: - label = f"{key}: {path}" + (f":{name}" if name else "") - container = recipe["model"]["container"] - # srtctl pulls a literal missing from the alias map, so a stale one runs silently. - if ":" in container and canonical_image(container) != canonical_image(image): - problems.add(f"{label} model.container {container} != master image {image}") - identity = ((recipe.get("identity") or {}).get("container") or {}).get("image") - if identity is not None and canonical_image(identity) != canonical_image(image): - problems.add(f"{label} identity.container.image {identity} != master image {image}") - assert not problems, "\n".join(sorted(problems)) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 5170df0626..4166a6c733 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9026,17 +9026,15 @@ pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3459 - config-keys: - - glm5.2-fp8-mi325x-sglang-agentic-mtp - - minimaxm3-fp8-mi300x-vllm-agentic-mtp - dsv41flash-fp4-mi355x-vllm-agentic-dspark - dsv4-fp4-b200-sglang-agentic-hicache-mtp scenario-type: - agentic-coding description: - - "Align the single-node srt-slurm recipe model.container with the unchanged master image for four AgentX keys whose recipes #3428 ported from the legacy scripts as they were before the image bumps in #3361, #3362, #3334 and #3420; every point of these keys would fail before submission with 'Single-node SRT image: recipe/matrix'." - - "Recipe containers: glm5.2-fp8-mi325x-sglang-agentic-mtp lmsysorg/sglang:v0.5.19-rocm720-mi30x -> lmsysorg/sglang:v0.5.20-rocm720-mi30x; minimaxm3-fp8-mi300x-vllm-agentic-mtp vllm/vllm-openai-rocm:v0.29.0 -> vllm/vllm-openai-rocm:v0.30.0; dsv41flash-fp4-mi355x-vllm-agentic-dspark vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 -> vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098; dsv4-fp4-b200-sglang-agentic-hicache-mtp lmsysorg/sglang:v0.5.19-cu130 -> lmsysorg/sglang:v0.5.20-cu130." - - "The two SGLang v0.5.20 recipes also rename cuda-graph-max-bs to cuda-graph-max-bs-decode, as #3362 and #3334 did in the legacy scripts, because v0.5.20 removed the deprecated alias (sgl-project/sglang#38375). All other serving flags and sweep points are unchanged." - - "将四个 AgentX key 的单节点 srt-slurm 配方 model.container 与未改动的主配置镜像对齐。这些配方由 #3428 从旧脚本移植而来,而移植所依据的是 #3361、#3362、#3334 和 #3420 升级镜像之前的版本;此前这些 key 的每个点都会在提交前以 'Single-node SRT image: recipe/matrix' 失败。" - - "配方镜像:glm5.2-fp8-mi325x-sglang-agentic-mtp lmsysorg/sglang:v0.5.19-rocm720-mi30x -> lmsysorg/sglang:v0.5.20-rocm720-mi30x;minimaxm3-fp8-mi300x-vllm-agentic-mtp vllm/vllm-openai-rocm:v0.29.0 -> vllm/vllm-openai-rocm:v0.30.0;dsv41flash-fp4-mi355x-vllm-agentic-dspark vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 -> vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098;dsv4-fp4-b200-sglang-agentic-hicache-mtp lmsysorg/sglang:v0.5.19-cu130 -> lmsysorg/sglang:v0.5.20-cu130。" - - "两个 SGLang v0.5.20 配方同时将 cuda-graph-max-bs 改为 cuda-graph-max-bs-decode,与 #3362 和 #3334 对旧脚本的修改一致,因为 v0.5.20 已移除该弃用别名(sgl-project/sglang#38375)。其余服务参数和 sweep 点均保持不变。" + - "Align the single-node srt-slurm recipe model.container with the unchanged master image for two AgentX keys whose recipes #3428 ported from the legacy scripts as they were before the image bumps in #3420 and #3334; every point of these keys would fail before submission with 'Single-node SRT image: recipe/matrix'." + - "Recipe containers: dsv41flash-fp4-mi355x-vllm-agentic-dspark vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 -> vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098; dsv4-fp4-b200-sglang-agentic-hicache-mtp lmsysorg/sglang:v0.5.19-cu130 -> lmsysorg/sglang:v0.5.20-cu130." + - "The B200 SGLang v0.5.20 recipe also renames cuda-graph-max-bs to cuda-graph-max-bs-decode, as #3334 did in the legacy script, because v0.5.20 removed the deprecated alias (sgl-project/sglang#38375). All other serving flags and sweep points are unchanged." + - "将两个 AgentX key 的单节点 srt-slurm 配方 model.container 与未改动的主配置镜像对齐。这些配方由 #3428 从旧脚本移植而来,而移植所依据的是 #3420 和 #3334 升级镜像之前的版本;此前这些 key 的每个点都会在提交前以 'Single-node SRT image: recipe/matrix' 失败。" + - "配方镜像:dsv41flash-fp4-mi355x-vllm-agentic-dspark vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 -> vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098;dsv4-fp4-b200-sglang-agentic-hicache-mtp lmsysorg/sglang:v0.5.19-cu130 -> lmsysorg/sglang:v0.5.20-cu130。" + - "B200 的 SGLang v0.5.20 配方同时将 cuda-graph-max-bs 改为 cuda-graph-max-bs-decode,与 #3334 对旧脚本的修改一致,因为 v0.5.20 已移除该弃用别名(sgl-project/sglang#38375)。其余服务参数和 sweep 点均保持不变。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3567 From 9d011a3660a304579a22dc64e47eb140d2bc9b47 Mon Sep 17 00:00:00 2001 From: Chun Fang Date: Tue, 29 Sep 2026 04:26:53 +0000 Subject: [PATCH 6/6] test: check srt-slurm recipe images against master configs Restore the static check with the MI325X GLM-5.2 and MI300X MiniMax-M3 keys exempt until their recipes are aligned, and run CI Tests when master configs or srt-slurm recipes change so YAML-only PRs are checked before any GPU job. Co-authored-by: Cursor --- .github/workflows/ci.yml | 4 + .../tests/srt_slurm/test_recipe_images.py | 94 +++++++++++++++++++ 2 files changed, 98 insertions(+) create mode 100644 inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 504dafcdf9..83ac57b92c 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -13,6 +13,10 @@ on: - 'inferencex-e2e/infx/ruff.toml' - '**/pytest.ini' - 'inferencex-e2e/utils/srt-slurm' + # Tests check recipe containers against master images. + - 'inferencex-e2e/configs/*-master.yaml' + - 'inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/**' + - 'inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/**' push: branches: [main] paths: *python-paths diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py b/inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py new file mode 100644 index 0000000000..b8bf0d1756 --- /dev/null +++ b/inferencex-e2e/infx/tests/srt_slurm/test_recipe_images.py @@ -0,0 +1,94 @@ +"""Keep checked-in srt-slurm recipe images identical to their master-config images.""" + +import sys +from collections import defaultdict +from pathlib import Path + +import yaml + +from infx.srt_slurm.synthetic_acceptance import selected_recipes + +ROOT = Path(__file__).resolve().parents[3] +sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) +MULTI_NODE_RECIPES = "benchmarks/multi_node/srt-slurm-recipes/" +# Their recipes still name a stale image on main; remove a key once its recipe is aligned. +KNOWN_STALE_KEYS = { + "glm5.2-fp8-mi325x-sglang-agentic-mtp", + "minimaxm3-fp8-mi300x-vllm-agentic-mtp", +} + + +def recipe_references(node): + """Yield (multinode, reference) for every srt-slurm recipe a master entry names.""" + if isinstance(node, dict): + for key, value in node.items(): + if key == "srt-recipe": + yield False, value + else: + yield from recipe_references(value) + elif isinstance(node, list): + for item in node: + yield from recipe_references(item) + elif isinstance(node, str) and node.startswith("CONFIG_FILE="): + path = node.removeprefix("CONFIG_FILE=") + # Launchers stage multi-node recipes under recipes/ in the srt-slurm checkout. + for staged in ("recipes/", MULTI_NODE_RECIPES): + if path.startswith(staged): + yield True, MULTI_NODE_RECIPES + path.removeprefix(staged) + + +def master_references(): + """Yield (config key, image, multinode, reference) from every master config.""" + for master in sorted((ROOT / "configs").glob("*-master.yaml")): + for key, entry in yaml.safe_load(master.read_text()).items(): + for multinode, reference in sorted(set(recipe_references(entry))): + yield key, entry["image"], multinode, reference + + +def variants(reference): + path, _, selector = reference.partition(":") + recipe = yaml.safe_load((ROOT / path).read_text()) + return path, selected_recipes(recipe, selector or None) + + +def canonical_image(image): + # Enroot spells the nvcr.io registry separator as "#"; launchers key aliases either way. + return image.replace("nvcr.io#", "nvcr.io/", 1) + + +def test_single_node_recipes_use_their_master_image(): + references = [ + (key, image, reference) + for key, image, multinode, reference in master_references() + if not multinode and key not in KNOWN_STALE_KEYS + ] + images = defaultdict(set) + for _, image, reference in references: + images[reference.partition(":")[0]].add(image) + problems = set() + for key, image, reference in references: + path, selected = variants(reference) + containers = {recipe["model"]["container"] for _, recipe in selected} + if image not in containers: + problems.add(f"{key}: no variant of {reference} uses master image {image}") + for container in containers - images[path]: + problems.add(f"{path}: {container} is not the image of any master key using it") + assert not problems, "\n".join(sorted(problems)) + + +def test_multi_node_recipes_use_their_master_image(): + problems = set() + for key, image, multinode, reference in master_references(): + if not multinode: + continue + path, selected = variants(reference) + for name, recipe in selected: + label = f"{key}: {path}" + (f":{name}" if name else "") + container = recipe["model"]["container"] + # srtctl pulls a literal missing from the alias map, so a stale one runs silently. + if ":" in container and canonical_image(container) != canonical_image(image): + problems.add(f"{label} model.container {container} != master image {image}") + identity = ((recipe.get("identity") or {}).get("container") or {}).get("image") + if identity is not None and canonical_image(identity) != canonical_image(image): + problems.add(f"{label} identity.container.image {identity} != master image {image}") + assert not problems, "\n".join(sorted(problems))