diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b200-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b200-fp4-mtp/agentic.yaml index 36bba702cb..65551cf4ae 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b200-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/sglang/b200-fp4-mtp/agentic.yaml @@ -6,7 +6,7 @@ base: name: dsv4-fp4-b200-sglang-agentic model: path: hf:deepseek-ai/DeepSeek-V4-Pro-0813 - container: lmsysorg/sglang:v0.5.19-cu130 + container: lmsysorg/sglang:v0.5.20-cu130 precision: fp4 resources: gpu_type: b200 @@ -85,7 +85,7 @@ override_tp8_c1: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 2 - cuda-graph-max-bs: 2 + cuda-graph-max-bs-decode: 2 benchmark: env: CONC: '1' @@ -100,7 +100,7 @@ override_tp8_c2: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 4 - cuda-graph-max-bs: 4 + cuda-graph-max-bs-decode: 4 benchmark: env: CONC: '2' @@ -115,7 +115,7 @@ override_tp8_c3: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 6 - cuda-graph-max-bs: 6 + cuda-graph-max-bs-decode: 6 benchmark: env: CONC: '3' @@ -130,7 +130,7 @@ override_tp8_c4: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 8 - cuda-graph-max-bs: 8 + cuda-graph-max-bs-decode: 8 benchmark: env: CONC: '4' @@ -145,7 +145,7 @@ override_tp8_c5: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 10 - cuda-graph-max-bs: 10 + cuda-graph-max-bs-decode: 10 benchmark: env: CONC: '5' @@ -160,7 +160,7 @@ override_tp8_hicache_c8: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 16 - cuda-graph-max-bs: 16 + cuda-graph-max-bs-decode: 16 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -181,7 +181,7 @@ override_tp8_hicache_c10: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 20 - cuda-graph-max-bs: 20 + cuda-graph-max-bs-decode: 20 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -202,7 +202,7 @@ override_tp8_hicache_c16: swa-full-tokens-ratio: 0.1 chunked-prefill-size: 8192 max-running-requests: 32 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -246,7 +246,7 @@ override_dep8_hicache_c64: swa-full-tokens-ratio: 0.02 chunked-prefill-size: 49152 max-running-requests: 128 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -294,7 +294,7 @@ override_dep8_hicache_c96: swa-full-tokens-ratio: 0.02 chunked-prefill-size: 49152 max-running-requests: 192 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -342,7 +342,7 @@ override_dep8_hicache_c128: swa-full-tokens-ratio: 0.02 chunked-prefill-size: 49152 max-running-requests: 256 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct @@ -391,7 +391,7 @@ override_dep8_hicache_c160: swa-full-tokens-ratio: 0.02 chunked-prefill-size: 49152 max-running-requests: 320 - cuda-graph-max-bs: 32 + cuda-graph-max-bs-decode: 32 enable-hierarchical-cache: true hicache-write-policy: write_through hicache-io-backend: direct diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml index 9ab84be6dd..fc67c49e8b 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml @@ -6,7 +6,7 @@ base: name: dsv41flash-fp4-mi355x-vllm-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - container: vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 + container: vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098 precision: fp4 resources: gpu_type: mi355x diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9b9c48a0ef..9c337b0f72 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9168,3 +9168,17 @@ - "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid." - "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605 + +- config-keys: + - dsv41flash-fp4-mi355x-vllm-agentic-dspark + - dsv4-fp4-b200-sglang-agentic-hicache-mtp + scenario-type: + - agentic-coding + description: + - "Align the single-node srt-slurm recipe model.container with the unchanged master image for two AgentX keys whose recipes #3428 ported from the legacy scripts as they were before the image bumps in #3420 and #3334; every point of these keys would fail before submission with 'Single-node SRT image: recipe/matrix'." + - "Recipe containers: dsv41flash-fp4-mi355x-vllm-agentic-dspark vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 -> vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098; dsv4-fp4-b200-sglang-agentic-hicache-mtp lmsysorg/sglang:v0.5.19-cu130 -> lmsysorg/sglang:v0.5.20-cu130." + - "The B200 SGLang v0.5.20 recipe also renames cuda-graph-max-bs to cuda-graph-max-bs-decode, as #3334 did in the legacy script, because v0.5.20 removed the deprecated alias (sgl-project/sglang#38375). All other serving flags and sweep points are unchanged." + - "将两个 AgentX key 的单节点 srt-slurm 配方 model.container 与未改动的主配置镜像对齐。这些配方由 #3428 从旧脚本移植而来,而移植所依据的是 #3420 和 #3334 升级镜像之前的版本;此前这些 key 的每个点都会在提交前以 'Single-node SRT image: recipe/matrix' 失败。" + - "配方镜像:dsv41flash-fp4-mi355x-vllm-agentic-dspark vllm/vllm-openai-rocm:nightly-rocm100-7f1a5398e9610d96c473931a26c0e12bbe0d0423 -> vllm/vllm-openai-rocm:nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098;dsv4-fp4-b200-sglang-agentic-hicache-mtp lmsysorg/sglang:v0.5.19-cu130 -> lmsysorg/sglang:v0.5.20-cu130。" + - "B200 的 SGLang v0.5.20 配方同时将 cuda-graph-max-bs 改为 cuda-graph-max-bs-decode,与 #3334 对旧脚本的修改一致,因为 v0.5.20 已移除该弃用别名(sgl-project/sglang#38375)。其余服务参数和 sweep 点均保持不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3567