From 760f9ec750c4780694678b47141e7713e4e77107 Mon Sep 17 00:00:00 2001 From: Guanbao Yu Date: Thu, 24 Sep 2026 07:43:31 +0000 Subject: [PATCH 1/6] feat(agentx): bump Kimi-K3 FP4 MI355X ATOM image to 0924 and track recipe MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Track recipes/Agentic-Kimi-K3.md as retuned in ROCm/ATOM#2382: enable FlyDSL FP8 prefill attention on every band, and hold a ready prefill for four decode passes from concurrency 16 up. The published concurrency set and every other launch argument are unchanged. 将 MI355X Kimi-K3 FP4 ATOM AgentX 提交切换到 0924 镜像,并跟随 ROCm/ATOM#2382 重调后的 recipe:全部并发开启 FlyDSL FP8 prefill attention;并发 16 及以上时让就绪的 prefill 等待 4 个 decode 轮次。 已发布的并发点集合与其余启动参数保持不变。 Co-Authored-By: Claude Opus 5 --- .../single_node/agentic/kimik3_fp4_mi355x_atom_mtp.sh | 9 +++++++++ configs/amd-master.yaml | 2 +- perf-changelog.yaml | 11 +++++++++++ 3 files changed, 21 insertions(+), 1 deletion(-) diff --git a/benchmarks/single_node/agentic/kimik3_fp4_mi355x_atom_mtp.sh b/benchmarks/single_node/agentic/kimik3_fp4_mi355x_atom_mtp.sh index 764e40e461..13155407d2 100644 --- a/benchmarks/single_node/agentic/kimik3_fp4_mi355x_atom_mtp.sh +++ b/benchmarks/single_node/agentic/kimik3_fp4_mi355x_atom_mtp.sh @@ -129,6 +129,13 @@ esac export ATOM_ENABLE_REPLAYSSM export AITER_REUSE_IDENTICAL_COMM_GROUPS +# From concurrency 16 up, hold a ready prefill for 4 decode passes instead of +# interleaving it into every step; MAX_QUEUE_MS bounds the wait. 14 runs without it. +if [ "$CONC" -ge 16 ]; then + export ATOM_PREFILL_DECODE_INTERVAL=4 + export ATOM_PREFILL_DELAYER_MAX_QUEUE_MS=5000 +fi + # Full CUDA graphs over [1 .. window * (1 + draft tokens)]: the verify step # submits one row per draft token on top of the accepted token, so capturing # only up to the window would send every speculative decode down the eager path. @@ -199,6 +206,8 @@ export AITER_FLYDSL_STAGE2_FP8=1 # costs more in evictions than its reuse is worth on these traces. export ATOM_STATE_CHECKPOINT_DEMAND=0 export ATOM_GDN_SSM_DTYPE=fp16 +# FlyDSL FP8 prefill attention; ATOM defaults it off. +export ATOM_USE_FLYDSL_FP8_PREFILL_ATTN=1 # https://github.com/SemiAnalysisAI/InferenceX/blob/main/golden_al_distribution/kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method.yaml # 7 draft tokens -> AL 3.84 diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index 421d55c483..0aec4a55d1 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -614,7 +614,7 @@ kimik3-fp4-mi355x-vllm-agentic-mtp: # wider in-flight window needs the deeper pool to keep the paged KV # resident. kimik3-fp4-mi355x-atom-agentic-mtp: - image: rocm/atom-dev:ubuntu24.04_py3.12_pytorch_release_2.10.0_kimi_k3_agentic_0911 + image: rocm/atom-dev:ubuntu24.04_py3.12_pytorch_release_2.10.0_kimi_k3_agentic_0924 model: moonshotai/Kimi-K3 model-prefix: kimik3 runner: cluster:mi355x-amds diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 5655eb4067..d261cb5689 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8815,3 +8815,14 @@ - Retain min(64*concurrency, 1024) SWA prefix tails within the measured GB200 KV budget; select prefill/decode interval 16 only at TP4 C16 after canonical performance and full GSM8K qualification. - "Match the published vLLM grid exactly: TP2/EP1 and TP4/EP1, each C1/2/4/8/16/32/64/128. Use stock DSpark precision; TP2 uses static memory 0.92, chunk 2048, graph/running cap 16, interval 16, expandable allocator and min(128*CONC,1024) SWA tails. Requalify all 16 performance and full GSM8K points. Avoid deadline-only failures with the supported 12-hour partition maximum for C64/C128 performance while preserving complete warmup and 3600-second scoring." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3347 + +- config-keys: + - kimik3-fp4-mi355x-atom-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Move the MI355X Kimi-K3 FP4 ATOM AgentX submission to image rocm/atom-dev:ubuntu24.04_py3.12_pytorch_release_2.10.0_kimi_k3_agentic_0924, tracking recipes/Agentic-Kimi-K3.md as retuned in ROCm/ATOM#2382." + - "Enable FlyDSL FP8 prefill attention (ATOM_USE_FLYDSL_FP8_PREFILL_ATTN=1) on every concurrency; ATOM defaults it off." + - "Hold a ready prefill for four decode passes from concurrency 16 up (ATOM_PREFILL_DECODE_INTERVAL=4, ATOM_PREFILL_DELAYER_MAX_QUEUE_MS=5000). Concurrency 1, 4 and 14 are unchanged." + - "Keep the published concurrency set [1, 4, 14, 16, 48, 56, 72], max-num-seqs, max-num-batched-tokens, gpu-memory-utilization, the CUDA-graph ladder, dcp-size, draft depth, synthetic acceptance, ReplaySSM placement and LMCache sizing unchanged." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX From 7625dc8fc144bd33625ab00461872bda3249cfd2 Mon Sep 17 00:00:00 2001 From: Guanbao Yu Date: Thu, 24 Sep 2026 07:44:28 +0000 Subject: [PATCH 2/6] chore(agentx): point perf-changelog entry at PR 3407 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 perf-changelog 条目的 pr-link 指向 PR 3407。 Co-Authored-By: Claude Opus 5 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index d261cb5689..d0de586d27 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8825,4 +8825,4 @@ - "Enable FlyDSL FP8 prefill attention (ATOM_USE_FLYDSL_FP8_PREFILL_ATTN=1) on every concurrency; ATOM defaults it off." - "Hold a ready prefill for four decode passes from concurrency 16 up (ATOM_PREFILL_DECODE_INTERVAL=4, ATOM_PREFILL_DELAYER_MAX_QUEUE_MS=5000). Concurrency 1, 4 and 14 are unchanged." - "Keep the published concurrency set [1, 4, 14, 16, 48, 56, 72], max-num-seqs, max-num-batched-tokens, gpu-memory-utilization, the CUDA-graph ladder, dcp-size, draft depth, synthetic acceptance, ReplaySSM placement and LMCache sizing unchanged." - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3407 From d9c84f1cf81e15de8cff1fd48e9e53da420fd736 Mon Sep 17 00:00:00 2001 From: seungrokj Date: Fri, 25 Sep 2026 19:29:15 -0700 Subject: [PATCH 3/6] chore(agentx): bump Kimi-K3 FP4 MI355X ATOM image to nightly_202609251613 Co-Authored-By: Claude Opus 4.6 --- configs/amd-master.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index da783085e1..c771f07db5 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -690,7 +690,7 @@ kimik3-fp4-mi355x-vllm-agentic-mtp: # wider in-flight window needs the deeper pool to keep the paged KV # resident. kimik3-fp4-mi355x-atom-agentic-mtp: - image: rocm/atom-dev:ubuntu24.04_py3.12_pytorch_release_2.10.0_kimi_k3_agentic_0924 + image: rocm/atom-dev:nightly_202609251613 model: moonshotai/Kimi-K3 model-prefix: kimik3 runner: cluster:mi355x-amds From d716d4caf5eb1eec737ef27c9cb9a506d4a0a521 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 17:37:46 -0500 Subject: [PATCH 4/6] feat(agentx): restore Kimi-K3 MI355X ATOM LMCache bands on srt-slurm Carry srt-slurm patch 507 so ATOM aggregate workers accept extra-kv-connectors, and restore the DCP8 bands from the legacy config and ROCm/ATOM recipes/Agentic-Kimi-K3.md: concurrency 14 and 16 with DSpark 3 and ReplaySSM, 48, 56 and 72 without a draft, all on the in-process lmcache_offload connector (128 GB/rank, 192 GB/rank at 56 and 72). The PrefillDelayer applies from concurrency 16 up. The single-node adapter now reads ATOM's decode-context-parallel-size for the DCP_SIZE check, as it does for vLLM. --- .../kimik3/atom/mi355x-fp4-mtp/agentic.yaml | 147 ++- inferencex-e2e/configs/amd-master.yaml | 4 + inferencex-e2e/infx/srt_slurm/single_node.py | 4 +- .../tests/srt_slurm/test_srt_single_node.py | 6 + inferencex-e2e/perf-changelog.yaml | 2 + .../507-lmcache-server-atom-sglang.patch | 859 ++++++++++++++++++ .../runners/srt-slurm/patches/README.md | 2 +- 7 files changed, 1015 insertions(+), 9 deletions(-) create mode 100644 inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml index 6afefa9b7d..6f40b5636a 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml @@ -1,8 +1,10 @@ -# Kimi-K3 MXFP4 AgentX on MI355X with ATOM DSpark: the interactive band, TP8 -# with GPU-resident KV and the deepest published draft (seven tokens). The -# 1.56 TB checkpoint only fits at TP8. The DCP8 LMCache bands were removed with -# the legacy script (#3461): srtctl reserves ATOM's kv-transfer-config for -# disaggregated workers. +# Kimi-K3 MXFP4 AgentX on MI355X with ATOM DSpark, tracking ROCm/ATOM +# recipes/Agentic-Kimi-K3.md. TP8 only: the 1.56 TB checkpoint is ~195 GB/GPU. +# interactive (1, 4) DCP1, DSpark 7, GPU-resident KV. +# mid (14, 16) DCP8, DSpark 3, ReplaySSM, LMCache 128 GB/rank. +# throughput (48, 56, 72) DCP8, no draft, LMCache 128 or 192 GB/rank. +# LMCache is ATOM's in-process lmcache_offload connector +# (runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch). base: schema: 2 name: kimik3-fp4-mi355x-atom-agentic @@ -87,7 +89,7 @@ base: AIPERF_APPLY_CHAT_TEMPLATE: 'true' # One variant per point. Full graphs cover every verify batch up to 2x CONC -# times the eight-token verify window. +# times the verify window (1 + draft tokens). override_tp8_c1: roles: agg: @@ -105,3 +107,136 @@ override_tp8_c4: benchmark: env: CONC: '4' + +# Mid band: DCP8 shards the KV read across all eight GPUs, ReplaySSM rebuilds +# KDA state from the checkpoint ring, and the paged KV rides the LMCache CPU +# tier. From concurrency 16 up, a ready prefill waits four decode passes. +override_tp8_dcp8_c14_lmcache: + roles: + agg: + args: + decode-context-parallel-size: 8 + num-speculative-tokens: 3 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112]' + extra-kv-connectors: &lmcache_offload + - kv_connector: lmcache_offload + kv_role: offload + env: + <<: &lmcache_env + PYTHONHASHSEED: '0' + LMCACHE_LOCAL_CPU: 'True' + # The offload hash block is block-size 128 x DCP 8, so the KV and the + # state-checkpoint grids coincide. + LMCACHE_CHUNK_SIZE: '1024' + # K3 is a hybrid; without this the paged KV never reaches the CPU tier. + OFFLOAD_KV_FOR_HYBRID: '1' + # Pin the eight ranks to the two sockets explicitly. + LMCACHE_NUMA_MODE: auto + ATOM_NUMA_BIND: '1' + ATOM_NUMA_NODE: 0,0,0,0,1,1,1,1 + ATOM_AUTO_NUMA_BIND: '0' + OFFLOAD_PROFILE: '1' + # One K3 state entry is 54.78 MiB; the default 8 MiB staging buffer + # cannot hold it. + OFFLOAD_GPU_STAGING_CHUNKS: '32' + LMCACHE_MAX_LOCAL_CPU_SIZE: '128' + ATOM_ENABLE_REPLAYSSM: '1' + benchmark: + env: + CONC: '14' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1028' + +override_tp8_dcp8_c16_lmcache: + roles: + agg: + args: + decode-context-parallel-size: 8 + num-speculative-tokens: 3 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128]' + extra-kv-connectors: *lmcache_offload + env: + <<: *lmcache_env + LMCACHE_MAX_LOCAL_CPU_SIZE: '128' + ATOM_ENABLE_REPLAYSSM: '1' + ATOM_PREFILL_DECODE_INTERVAL: '4' + ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: '5000' + benchmark: + env: + CONC: '16' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1028' + +# Throughput band: past the knee the draft no longer pays for itself, and +# max-num-seqs tracks 2x CONC. 56 and 72 need the 192 GB/rank pool and reuse +# identical AITER communicator groups. +override_tp8_dcp8_c48_lmcache: + roles: + agg: + args: + decode-context-parallel-size: 8 + method: null + draft-model: null + num-speculative-tokens: null + max-num-seqs: 96 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96]' + extra-kv-connectors: *lmcache_offload + env: + <<: *lmcache_env + LMCACHE_MAX_LOCAL_CPU_SIZE: '128' + ATOM_PREFILL_DECODE_INTERVAL: '4' + ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: '5000' + benchmark: + env: + CONC: '48' + SPEC_DECODING: mtp + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1028' + +override_tp8_dcp8_c56_lmcache: + roles: + agg: + args: + decode-context-parallel-size: 8 + method: null + draft-model: null + num-speculative-tokens: null + max-num-seqs: 112 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112]' + extra-kv-connectors: *lmcache_offload + env: + <<: *lmcache_env + LMCACHE_MAX_LOCAL_CPU_SIZE: '192' + AITER_REUSE_IDENTICAL_COMM_GROUPS: '1' + ATOM_PREFILL_DECODE_INTERVAL: '4' + ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: '5000' + benchmark: + env: + CONC: '56' + SPEC_DECODING: mtp + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1538' + +override_tp8_dcp8_c72_lmcache: + roles: + agg: + args: + decode-context-parallel-size: 8 + method: null + draft-model: null + num-speculative-tokens: null + max-num-seqs: 144 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128,129,130,131,132,133,134,135,136,137,138,139,140,141,142,143,144]' + extra-kv-connectors: *lmcache_offload + env: + <<: *lmcache_env + LMCACHE_MAX_LOCAL_CPU_SIZE: '192' + AITER_REUSE_IDENTICAL_COMM_GROUPS: '1' + ATOM_PREFILL_DECODE_INTERVAL: '4' + ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: '5000' + benchmark: + env: + CONC: '72' + SPEC_DECODING: mtp + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1538' diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index c4a44b65c5..43c89a3451 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -686,6 +686,10 @@ kimik3-fp4-mi355x-atom-agentic-mtp: - dram-utilization: 0.343 search-space: - { tp: 8, kv-offloading: none, conc-list: [1, 4], spec-decoding: mtp, srt-recipe: benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml } + - { tp: 8, dcp-size: 8, kv-offloading: dram, kv-offload-backend: { name: lmcache, version: "0.4.5" }, conc-list: [14, 16, 48], spec-decoding: mtp, srt-recipe: benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml } + - dram-utilization: 0.513 + search-space: + - { tp: 8, dcp-size: 8, kv-offloading: dram, kv-offload-backend: { name: lmcache, version: "0.4.5" }, conc-list: [56, 72], spec-decoding: mtp, srt-recipe: benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml } minimaxm3-fp4-mi355x-atom-agentic-mtp: image: rocm/atom-dev:nightly_202609171455 diff --git a/inferencex-e2e/infx/srt_slurm/single_node.py b/inferencex-e2e/infx/srt_slurm/single_node.py index 1e6baece4b..0c758d1f45 100644 --- a/inferencex-e2e/infx/srt_slurm/single_node.py +++ b/inferencex-e2e/infx/srt_slurm/single_node.py @@ -127,8 +127,8 @@ def validate_recipe(recipe: dict[str, Any], environment: Mapping[str, str]) -> N if engine == "atom": # Native ATOM derives -tp from the aggregate worker's GPU allocation. expected["ATOM TP"] = (role["gpus"], int(environment["TP"])) - # vLLM shards decode KV across its tensor-parallel ranks. - dcp = str(args.get("decode-context-parallel-size", 1)) if engine == "vllm" else "1" + # vLLM and ATOM shard decode KV across their tensor-parallel ranks. + dcp = str(args.get("decode-context-parallel-size", 1)) if engine in {"vllm", "atom"} else "1" for name, value in {"PP_SIZE": "1", "DCP_SIZE": dcp, "PCP_SIZE": "1"}.items(): expected[name] = (environment[name], value) for name, (actual, wanted) in expected.items(): diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py b/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py index 09e7b4590a..306aea6b6f 100644 --- a/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py +++ b/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py @@ -232,9 +232,15 @@ def test_atom_binding_uses_allocation_tp_and_native_mtp_arguments(point): ({"EP_SIZE": "1"}, "enable-expert-parallel"), ({"TP": "8", "EP_SIZE": "8"}, "ATOM TP"), ({"DP_ATTENTION": "false"}, "DP_ATTENTION"), + ({"DCP_SIZE": "8"}, "DCP_SIZE"), ]: with pytest.raises(ValueError, match=error): runtime_arguments(f"{path}:base", {**env, **changes}) + recipe["roles"]["agg"]["args"]["decode-context-parallel-size"] = 8 + path.write_text(yaml.safe_dump({"base": recipe})) + runtime_arguments(f"{path}:base", {**env, "DCP_SIZE": "8"}) + with pytest.raises(ValueError, match="DCP_SIZE"): + runtime_arguments(f"{path}:base", env) @pytest.mark.parametrize("record,expected", [ diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 91a13a4370..4a957d6d21 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9011,4 +9011,6 @@ description: - "Move the MI355X Kimi-K3 FP4 ATOM AgentX submission to image rocm/atom-dev:nightly_202609251613, tracking recipes/Agentic-Kimi-K3.md as retuned in ROCm/ATOM#2382." - "Enable FlyDSL FP8 prefill attention (ATOM_USE_FLYDSL_FP8_PREFILL_ATTN=1) on every concurrency; ATOM defaults it off." + - "Hold a ready prefill for four decode passes from concurrency 16 up (ATOM_PREFILL_DECODE_INTERVAL=4, ATOM_PREFILL_DELAYER_MAX_QUEUE_MS=5000). Concurrency 1, 4 and 14 are unchanged." + - "Restore the DCP8 LMCache bands on the native srt-slurm recipe: concurrency 14 and 16 (DSpark 3, ReplaySSM) and 48, 56 and 72 (no draft) run ATOM's in-process lmcache_offload connector through roles.agg.args.extra-kv-connectors (srt-slurm patch 507), with 128 GB/rank up to 48 and 192 GB/rank at 56 and 72. Concurrency 1 and 4 stay GPU-resident." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3407 diff --git a/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch new file mode 100644 index 0000000000..b58b06bbaf --- /dev/null +++ b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch @@ -0,0 +1,859 @@ +diff --git a/docs/config-reference.md b/docs/config-reference.md +index d6997e70..7a7178ae 100644 +--- a/docs/config-reference.md ++++ b/docs/config-reference.md +@@ -56,6 +56,28 @@ srt-slurm owns the model path, HTTP port, tensor parallel size, and KV-transfer + contract, so recipes cannot override those arguments. See the complete + [ATOM/AToMesh recipe](../examples/atom/atomesh-disagg.yaml). + ++To run another KV connector next to the Mooncake one (for example ATOM's in-process ++LMCache CPU offload on prefill), list it under `extra-kv-connectors` in that role's ++args. srtctl keeps generating the Mooncake entry, with its handshake port, and wraps ++both in ATOM's `multi` connector. An aggregate role with one extra connector runs it ++on its own. An `lmcache_mp` connector (ATOM's client for the `lmcache-server` ++service) dials the server on its own node, port 8750, unless its ++`kv_connector_extra_config` sets `lmcache.mp.port` or `lmcache.mp.server_urls`. ++ ++```yaml ++roles: ++ prefill: ++ env: ++ PYTHONHASHSEED: "0" # LMCache prefix hashes must agree across offload workers ++ args: ++ extra-kv-connectors: ++ - kv_connector: lmcache_offload ++ kv_role: offload ++ lmcache.local_cpu: true ++ lmcache.max_local_cpu_size: 180 # GB per worker ++ lmcache.chunk_size: 256 ++``` ++ + ```yaml + schema: 2 # Required: recipe layout version + name: "my-benchmark" # Required: job name +diff --git a/docs/schema-reference.md b/docs/schema-reference.md +index 0c172bdd..9ee6090d 100644 +--- a/docs/schema-reference.md ++++ b/docs/schema-reference.md +@@ -624,7 +624,7 @@ vLLM protocol - implements BackendProtocol. + |---|---|---|---| + | `type` | one of `'vllm'` | `'vllm'` | | + | `set_visible_devices` | bool | `False` | Use an environment mask instead of the engine's --device-ids option. | +-| `connector` | str \| None | `'nixl'` | Default KV connector: "nixl", "lmcache", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). | ++| `connector` | str \| None | `'nixl'` | Default KV connector: "nixl", "lmcache", "lmcache-mp", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). | + | `failover` | [VLLMFailoverConfig](#vllmfailoverconfig) \| None | `None` | Shadow engine recovery: when set, every worker runs shadow_engines standby engines on its GPUs next to an implied `gms` service that owns the weights. Dynamo frontend only. | + | `allow_prefill_decode_colocation` | bool | `False` | Allow prefill and decode workers to share one node when the combined GPU request fits within gpus_per_node. Defaults off to preserve existing P/D node separation. | + | `allow_prefill_decode_colocation_across_nodes` | bool | `False` | Extend P/D colocation to multi-node topologies. When enabled together with allow_prefill_decode_colocation, workers are packed contiguously across the minimum number of nodes instead of reserving separate P/D node pools. Defaults off to preserve the original one-node-only policy. | +diff --git a/docs/services.md b/docs/services.md +index ab010da1..04bb9838 100644 +--- a/docs/services.md ++++ b/docs/services.md +@@ -48,7 +48,7 @@ directory. `examples/features/services.yaml` is a runnable version of this. + ```yaml + services: + - name: my-sidecar # required, unique across the list +- type: generic # generic (default) | etcd | nats | mooncake-master | dcgm-exporter | node-exporter | mooncake-store ++ type: generic # generic (default) | etcd | nats | mooncake-master | dcgm-exporter | node-exporter | mooncake-store | lmcache-server + enabled: true # false drops the service (the way to switch an implied one off) + external: null # typed kinds only: use an instance that already runs at this address + command: # argv, not shell-interpreted; required for generic +@@ -104,10 +104,10 @@ services: + | `placement.pool` | none | Run on the nodes another service owns, one instance per node of that pool. Replaces `node`. See [pools.md](pools.md). | + | `placement.per` | `node` | `worker`: one instance per engine worker on each placed node instead of one per node, attached to that worker (its `CUDA_VISIBLE_DEVICES`, the `{worker_*}` placeholders). See [Placement](#placement). | + | `nodes` | none | Whole nodes this service owns: its pool, added to the allocation after the engine roles' nodes, in declaration order. Any number of services may own nodes, next to engine roles or without them. An owner is placed on its own pool (`placement.node: workers`). Not supported with `resources.het_jobs`. See [pools.md](pools.md). | +-| `start` | type default | `etcd`, `nats`: `infra`. `mooncake-master`, `mooncake-store`: `before_workers`. `generic` and the exporters: `after_frontend`. | ++| `start` | type default | `etcd`, `nats`: `infra`. `mooncake-master`, `mooncake-store`, `lmcache-server`: `before_workers`. `generic` and the exporters: `after_frontend`. | + | `readiness` | type default | One probe per node: `port` / `tcp`, `http`, or `log`, plus `timeout_seconds` and `interval_seconds`. The typed kinds gate on their well-known ports when no probe is written. See [Start Order and Readiness](#start-order-and-readiness). Timing out terminates what this stage started and fails the job. | + | `inherit_discovery_env` | `true` | Inject the same `ETCD_ENDPOINTS` / `NATS_SERVER` the Dynamo frontend gets. | +-| `critical` | type default | `generic`: `false`. `mooncake-store`: `true`. | ++| `critical` | type default | `generic`: `false`. `mooncake-store`, `lmcache-server`: `true`. | + | `terminal` | `false` | This service is the job's run: the job ends when every instance of every terminal service has exited, with the worst exit code as the job's. The recipe has no benchmark step (`benchmark.type` stays `manual`); combining the two is refused. See [pools.md](pools.md#ending-the-job-with-a-pool). | + | `metrics` | kind default | Prometheus endpoints the service serves: one `{port, path, nodes, name}` or a list of them (`path` defaults to `/metrics`, `nodes` to `all`, `name` to the service name and is required when there are several). Tachometer scrapes each on every node the service runs on, pools included, as endpoint `_`, and the rows carry `service=`; `nodes: first` scrapes only the service's first node (a cluster head that serves a collector or a router). The exporter kinds declare theirs; write it for anything else that publishes metrics. | + | `preamble` | none | Shell run after the environment is exported and before `command`. | +@@ -302,6 +302,7 @@ environment its process needs; the launch path is shared by every kind. Register + | `node-exporter` | `/bin/node_exporter` with the cpu, infiniband, and meminfo collectors on 9101 in `quay.io/prometheus/node-exporter` | `after_frontend` | `false` | Implied on worker nodes while tachometer runs. Shell-less. `options`: `port`. | + | `process-exporter` | `configs/process-exporter -config.path /process-exporter.yml -web.listen-address=:9256 -threads=true ...` on the bare node | `after_frontend` | `false` | Implied on every allocated node (`placement.node: all`) while tachometer runs. Host-native from the static binary `make setup` installs; skipped with a warning when it is missing. A declared `container` switches to the image's `/bin/process-exporter` with the group file under `/logs`. `options`: `port`, `binary`. | + | `mooncake-store` | `python -m mooncake.mooncake_store_service` | `before_workers` | `true` | Requires a `mooncake-master` entry. Container falls back to the master's. Injects the master's address. | ++| `lmcache-server` | `lmcache server` bound to the RPC and HTTP ports srtctl owns (8750, 8751) | `before_workers` | `true` | One per worker node (`placement.node: workers`); vLLM ranks reach it over localhost with `connector: lmcache-mp`; SGLang workers started with `enable-lmcache` get `LMCACHE_MP_HOST`/`LMCACHE_MP_PORT` pointing at it unless the recipe sets `lmcache-config-file` or `LMCACHE_MP_HOST`. `args` are appended (`--l1-size-gb`, `--chunk-size`, `--max-workers`, ...). Ready when `GET /healthcheck` answers; LMCache must be installed in the job container. | + | `ray` | `ray start --head ...` on the first instance, `ray start --address=:6379 ...` on the rest, both `--block` | `before_workers` | `true` | One Ray cluster across the service's nodes; see [Ray cluster](#ray-cluster). Placement `workers` (default), `head`, or `all`. `options`: `port` (GCS, 6379), `dashboard_port` (8265), `num_gpus` (the node's count). `args` are appended to every `ray start`. | + | `gms` | one `python3 -m gpu_memory_service --device k` per GPU of the worker, supervised by bash, in the job container | `before_workers` | `true` | Implied by `engine.failover`; see [Shadow Engine Recovery](shadow-engine-recovery.md). `placement.per: worker` (required): one instance per vLLM worker in that worker's device view, sockets and lock file under `/srtctl-/_`. Ready when its log says `GMS ready:`; `readiness.timeout_seconds` (default 120) also bounds the servers' startup. | + +diff --git a/examples/README.md b/examples/README.md +index 63eb9d97..373e5413 100644 +--- a/examples/README.md ++++ b/examples/README.md +@@ -32,6 +32,9 @@ Every example is written in the 2.0 layout: `engine:` names the engine (a string + | `features/infra-services.yaml` | etcd and NATS as declared services on a dedicated node with a NATS payload limit; the implied exporters overridden or switched off | + | `features/dynamo-source.yaml` | `dynamo.source:` building Dynamo from a git tag (or a PR head via `--set dynamo.source.rev=refs/pull//head`), pinned to a commit at submit | + | `features/node-hooks.yaml` | `host_setup:` pre-run and post-run commands on each node's bare host, driven by `configs/node-hooks.sh`: arbitrary shell commands, one per line, from `HOOK_PRE` / `HOOK_POST` block scalars in the recipe environment, with per-command logging and a state snapshot | ++| `features/lmcache-server.yaml` | `services[].type: lmcache-server` plus `connector: lmcache-mp`: an LMCache DRAM tier, one server per worker node started before the workers and gated on its healthcheck; LMCache flags go under `args`. Needs a vLLM image that ships LMCache (the `vllm-lmcache` alias). See [../docs/services.md](../docs/services.md#service-types) | ++| `features/lmcache-server-disagg.yaml` | The same tier on the prefill side only: the server placed on prefill nodes and a hand-written `MultiConnector` (NIXL plus LMCache MP) in the prefill args, which srtctl passes through untouched | ++| `features/lmcache-server-sglang.yaml` | The same server next to aggregated SGLang workers: SGLang's `enable-lmcache` flag, with srtctl pointing each worker at the server on its node (`LMCACHE_MP_HOST`/`LMCACHE_MP_PORT`). Needs an SGLang image that ships LMCache (the `sglang-lmcache` alias) | + | `features/vllm-failover.yaml` | `engine.failover:` shadow engine recovery: a GPU Memory Service sidecar and a parked standby engine per vLLM worker, relaunched in place after a crash. Needs a container that ships `gpu_memory_service` (the `dynamo-vllm` alias, an `nvcr.io/nvidia/ai-dynamo/vllm-runtime` image). See [../docs/shadow-engine-recovery.md](../docs/shadow-engine-recovery.md) | + + ## Cluster aliases +@@ -47,6 +50,8 @@ containers: + vllm: /path/to/vllm.sqsh # vLLM image with the vllm-router executable + trtllm: /path/to/tensorrtllm-runtime.sqsh # Dynamo TRT-LLM runtime image (ships ai-dynamo and trtllm-serve) + dynamo-vllm: /path/to/vllm-runtime.sqsh # Dynamo vLLM runtime image (ships ai-dynamo and gpu_memory_service), for features/vllm-failover.yaml ++ vllm-lmcache: /path/to/vllm-lmcache.sqsh # vLLM image with LMCache installed, for features/lmcache-server.yaml and -disagg.yaml ++ sglang-lmcache: /path/to/sglang-lmcache.sqsh # SGLang image with LMCache installed, for features/lmcache-server-sglang.yaml + ``` + + `resources.gpu_type` and `gpus_per_node` are set to `h100` and `8`; change them to match the partition you submit to. +diff --git a/examples/features/lmcache-server-disagg.yaml b/examples/features/lmcache-server-disagg.yaml +new file mode 100644 +index 00000000..95eb632d +--- /dev/null ++++ b/examples/features/lmcache-server-disagg.yaml +@@ -0,0 +1,88 @@ ++# An LMCache DRAM tier on the prefill side of a disaggregated vLLM job. ++# ++# Prefill keeps NIXL for the P/D transfer and adds LMCache for prefix reuse, so its ++# kv-transfer-config is a hand-written MultiConnector (srtctl passes raw JSON through ++# untouched; the `lmcache-mp` shorthand covers the single-connector case). Decode is ++# NIXL only. The server is placed on prefill nodes only (`placement.node: prefill`). ++# ++# Aliases (`qwen3-0.6b`, `vllm-lmcache`) resolve through srtslurm.yaml; see ++# examples/README.md and examples/features/lmcache-server.yaml. ++ ++schema: 2 ++name: "qwen3-0.6b-vllm-lmcache-disagg" ++ ++slurm: ++ time_limit: "00:30:00" ++ ++model: ++ path: "qwen3-0.6b" ++ container: "vllm-lmcache" ++ precision: "bf16" ++ ++resources: ++ gpu_type: "h100" ++ gpus_per_node: 8 ++dynamo: ++ source: ++ pypi: "1.4.2" ++ ++environment: ++ PYTHONHASHSEED: "0" ++ ++frontend: ++ type: dynamo ++ enable_multiple_frontends: false ++ args: ++ router-mode: "kv" ++ ++engine: ++ type: vllm ++ connector: nixl ++roles: ++ prefill: ++ nodes: 1 ++ workers: 1 ++ gpus: 1 ++ env: ++ DYN_HEALTH_CHECK_ENABLED: "false" ++ PYTHONUNBUFFERED: "1" ++ args: ++ # NIXL to decode plus LMCache MP on this node; a raw string here replaces the connector preset. ++ kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both"},{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.host":"tcp://localhost","lmcache.mp.port":8750}}]}}' ++ served-model-name: "Qwen/Qwen3-0.6B" ++ tensor-parallel-size: 1 ++ gpu-memory-utilization: 0.5 ++ max-model-len: 4096 ++ decode: ++ nodes: colocate ++ workers: 1 ++ gpus: 1 ++ env: ++ DYN_HEALTH_CHECK_ENABLED: "false" ++ PYTHONUNBUFFERED: "1" ++ args: ++ served-model-name: "Qwen/Qwen3-0.6B" ++ tensor-parallel-size: 1 ++ gpu-memory-utilization: 0.5 ++ max-model-len: 4096 ++ ++services: ++ - name: lmcache ++ type: lmcache-server ++ placement: ++ node: prefill # decode nodes get no server; with `colocate` that is still this one node ++ args: ++ - --l1-size-gb ++ - "16" ++ - --chunk-size ++ - "256" ++ - --max-workers ++ - "1" ++ preamble: | ++ ulimit -l unlimited ++ ++benchmark: ++ type: "sa-bench" ++ isl: 128 ++ osl: 128 ++ concurrencies: "4x8" +diff --git a/examples/features/lmcache-server-sglang.yaml b/examples/features/lmcache-server-sglang.yaml +new file mode 100644 +index 00000000..66a4fb65 +--- /dev/null ++++ b/examples/features/lmcache-server-sglang.yaml +@@ -0,0 +1,62 @@ ++# An LMCache DRAM tier next to aggregated SGLang workers. ++# ++# `type: lmcache-server` runs the LMCache multiprocess server on every worker node ++# before the workers start (8750 RPC, 8751 HTTP). SGLang's own `enable-lmcache` flag ++# switches the worker to LMCacheUnifiedRadixCache; srtctl sets LMCACHE_MP_HOST and ++# LMCACHE_MP_PORT so it dials the server on its own node. A recipe that passes ++# `lmcache-config-file` or sets LMCACHE_MP_HOST in the role env keeps its own address. ++# ++# The SGLang image must ship LMCache, so this uses a `sglang-lmcache` alias. Aliases ++# (`qwen3-0.6b`, `sglang-lmcache`) resolve through srtslurm.yaml; see examples/README.md. ++ ++schema: 2 ++name: "qwen3-0.6b-sglang-lmcache-agg" ++ ++slurm: ++ time_limit: "00:30:00" ++ ++model: ++ path: "qwen3-0.6b" ++ container: "sglang-lmcache" ++ precision: "bf16" ++ ++resources: ++ gpu_type: "h100" ++ gpus_per_node: 8 ++ ++environment: ++ PYTHONHASHSEED: "0" # chunk hashes must agree across the server and every rank ++ ++frontend: ++ type: sglang ++ enable_multiple_frontends: false ++ ++engine: sglang ++roles: ++ agg: ++ nodes: 1 ++ workers: 1 ++ gpus: 1 ++ args: ++ enable-lmcache: true ++ mem-fraction-static: 0.5 ++ context-length: 4096 ++ ++services: ++ - name: lmcache ++ type: lmcache-server ++ args: ++ - --l1-size-gb ++ - "16" ++ - --chunk-size ++ - "256" ++ - --eviction-policy ++ - LRU ++ preamble: | ++ ulimit -l unlimited ++ ++benchmark: ++ type: "sa-bench" ++ isl: 128 ++ osl: 128 ++ concurrencies: "4x8" +diff --git a/examples/features/lmcache-server.yaml b/examples/features/lmcache-server.yaml +new file mode 100644 +index 00000000..4ecee520 +--- /dev/null ++++ b/examples/features/lmcache-server.yaml +@@ -0,0 +1,78 @@ ++# An LMCache DRAM tier next to aggregated vLLM workers. ++# ++# `type: lmcache-server` runs the LMCache multiprocess server on every worker node ++# before the workers start, on the ports srtctl owns (8750 RPC, 8751 HTTP), and holds ++# the job until `GET /healthcheck` answers. `connector: lmcache-mp` points each vLLM ++# worker's LMCacheMPConnector at the server on its own node. Everything after the ++# ports is LMCache's own CLI: put `lmcache server --help` flags under `args`. ++# ++# The vLLM image must ship LMCache (the connector is imported inside the worker), so ++# this uses a `vllm-lmcache` alias. Aliases (`qwen3-0.6b`, `vllm-lmcache`) resolve ++# through srtslurm.yaml; see examples/README.md. ++ ++schema: 2 ++name: "qwen3-0.6b-vllm-lmcache-agg" ++ ++slurm: ++ time_limit: "00:30:00" ++ ++model: ++ path: "qwen3-0.6b" ++ container: "vllm-lmcache" ++ precision: "bf16" ++ ++resources: ++ gpu_type: "h100" ++ gpus_per_node: 8 ++dynamo: ++ source: ++ pypi: "1.4.2" ++ ++environment: ++ PYTHONHASHSEED: "0" # chunk hashes must agree across the server and every rank ++ ++frontend: ++ type: dynamo ++ enable_multiple_frontends: false ++ args: ++ router-mode: "kv" ++ ++engine: ++ type: vllm ++ connector: lmcache-mp # every role; or per role under roles..args.connector ++roles: ++ agg: ++ nodes: 1 ++ workers: 2 ++ gpus: 1 ++ env: ++ DYN_HEALTH_CHECK_ENABLED: "false" ++ PYTHONUNBUFFERED: "1" ++ args: ++ served-model-name: "Qwen/Qwen3-0.6B" ++ tensor-parallel-size: 1 ++ gpu-memory-utilization: 0.5 ++ max-model-len: 4096 ++ ++services: ++ - name: lmcache ++ type: lmcache-server ++ # placement.node: workers (default) — one instance per worker node. ++ # start: before_workers, critical: true (defaults): a dead server fails the run. ++ args: # appended to `lmcache server --host ... --port 8750 --http-host ... --http-port 8751` ++ - --l1-size-gb ++ - "16" ++ - --chunk-size ++ - "256" ++ - --max-workers ++ - "2" ++ - --eviction-policy ++ - LRU ++ preamble: | ++ ulimit -l unlimited ++ ++benchmark: ++ type: "sa-bench" ++ isl: 128 ++ osl: 128 ++ concurrencies: "4x8" +diff --git a/src/srtctl/backends/atom.py b/src/srtctl/backends/atom.py +index 2f0fc576..8aeea3a0 100644 +--- a/src/srtctl/backends/atom.py ++++ b/src/srtctl/backends/atom.py +@@ -15,7 +15,7 @@ from typing import TYPE_CHECKING, Any, ClassVar, Literal + from marshmallow import Schema + from marshmallow_dataclass import dataclass + +-from srtctl.ports import DYN_SYSTEM_PORT_BASE ++from srtctl.ports import DYN_SYSTEM_PORT_BASE, LMCACHE_SERVER_PORT + + if TYPE_CHECKING: + from srtctl.backends.base import SrunConfig +@@ -141,19 +141,32 @@ class AtomProtocol: + + return endpoints_to_processes(endpoints, base_sys_port=base_sys_port, port_allocator=port_allocator) + +- def _kv_transfer_config(self, process: Process, worker_ip: str) -> str: +- if process.endpoint_mode not in {"prefill", "decode"}: +- raise ValueError("ATOM KV transfer is only valid for prefill/decode workers") +- if process.nixl_port is None: +- raise ValueError("ATOM P/D worker is missing its Mooncake handshake port") +- payload = { +- "kv_role": "kv_producer" if process.endpoint_mode == "prefill" else "kv_consumer", +- "kv_connector": self.connector, +- "proxy_ip": worker_ip, +- "handshake_port": process.nixl_port, +- } +- if self.mooncake_protocol is not None: +- payload["protocol"] = self.mooncake_protocol ++ def _kv_transfer_config( ++ self, process: Process, worker_ip: str, extra_connectors: list[dict[str, Any]] ++ ) -> str | None: ++ """Mooncake for P/D workers plus the role's extra connectors; several are wrapped in ``multi``.""" ++ connectors = [] ++ for connector in extra_connectors: ++ if connector.get("kv_connector") == "lmcache_mp": ++ # Dial the node-local lmcache-server service unless the recipe set a port. ++ extra = {"lmcache.mp.port": LMCACHE_SERVER_PORT, **connector.get("kv_connector_extra_config", {})} ++ connector = {**connector, "kv_connector_extra_config": extra} ++ connectors.append(connector) ++ if process.endpoint_mode in {"prefill", "decode"}: ++ if process.nixl_port is None: ++ raise ValueError("ATOM P/D worker is missing its Mooncake handshake port") ++ mooncake: dict[str, Any] = { ++ "kv_role": "kv_producer" if process.endpoint_mode == "prefill" else "kv_consumer", ++ "kv_connector": self.connector, ++ "proxy_ip": worker_ip, ++ "handshake_port": process.nixl_port, ++ } ++ if self.mooncake_protocol is not None: ++ mooncake["protocol"] = self.mooncake_protocol ++ connectors.insert(0, mooncake) ++ if not connectors: ++ return None ++ payload = connectors[0] if len(connectors) == 1 else {"kv_connector": "multi", "connectors": connectors} + return json.dumps(payload, separators=(",", ":")) + + def build_worker_command( +@@ -175,6 +188,7 @@ class AtomProtocol: + + worker_ip = get_hostname_ip(process.node, runtime.network_interface) + config = self.get_config_for_mode(process.endpoint_mode) ++ extra_connectors = config.pop("extra-kv-connectors", []) + reserved = {"model", "host", "server-port", "tp", "tensor-parallel-size", "kv-transfer-config"} + overlap = reserved.intersection(_canonical_arg_key(key) for key in config) + if overlap: +@@ -196,8 +210,9 @@ class AtomProtocol: + str(len(process.gpu_indices)), + ] + ) +- if process.endpoint_mode in {"prefill", "decode"}: +- command.extend(["--kv-transfer-config", self._kv_transfer_config(process, worker_ip)]) ++ kv_transfer_config = self._kv_transfer_config(process, worker_ip, extra_connectors) ++ if kv_transfer_config is not None: ++ command.extend(["--kv-transfer-config", kv_transfer_config]) + command.extend(_config_to_cli_args(config)) + return command + +diff --git a/src/srtctl/backends/sglang.py b/src/srtctl/backends/sglang.py +index 585935a2..e7e0d618 100644 +--- a/src/srtctl/backends/sglang.py ++++ b/src/srtctl/backends/sglang.py +@@ -27,6 +27,7 @@ from srtctl.backends.sidecar import build_sidecar_launch_command, get_dynamo_sid + from srtctl.ports import ( + DIST_INIT_PORTS, + DYN_SYSTEM_PORT_BASE, ++ LMCACHE_SERVER_PORT, + MOONCAKE_HTTP_METADATA_PORT, + MOONCAKE_MASTER_PORT, + NCCL_PORTS, +@@ -207,10 +208,17 @@ class SGLangProtocol: + def get_process_environment(self, process: "Process") -> dict[str, str]: + """Get process-specific environment variables. + +- SGLang handles kv-events via CLI args (--kv-events-config), so no +- additional process-specific env vars are needed here. ++ A worker started with ``enable-lmcache`` dials the LMCache MP server on its own ++ node (``services[].type: lmcache-server``) unless the recipe points it elsewhere ++ with ``lmcache-config-file`` or ``LMCACHE_MP_HOST`` in the role env. + """ +- return {} ++ mode = process.endpoint_mode ++ config = {key.replace("_", "-"): value for key, value in self.get_config_for_mode(mode).items()} ++ if not config.get("enable-lmcache") or config.get("lmcache-config-file"): ++ return {} ++ if "LMCACHE_MP_HOST" in self.get_environment_for_mode(mode): ++ return {} ++ return {"LMCACHE_MP_HOST": "127.0.0.1", "LMCACHE_MP_PORT": str(LMCACHE_SERVER_PORT)} + + def get_mooncake_worker_env(self, infra_node_ip: str, local_hostname: str) -> dict[str, str]: + """Get mooncake env vars to inject on a specific worker. +diff --git a/src/srtctl/backends/vllm.py b/src/srtctl/backends/vllm.py +index 52ee6737..0fb3e8cb 100644 +--- a/src/srtctl/backends/vllm.py ++++ b/src/srtctl/backends/vllm.py +@@ -35,6 +35,7 @@ from srtctl.ports import ( + HTTP_PORTS, + KV_EVENTS_PORTS, + KVBM_ZMQ_PORTS, ++ LMCACHE_SERVER_PORT, + MOONCAKE_HTTP_METADATA_PORT, + MOONCAKE_MASTER_PORT, + MORIIO_HANDSHAKE_PORTS, +@@ -351,7 +352,7 @@ class VLLMProtocol: + # Use an environment mask instead of the engine's --device-ids option. + set_visible_devices: bool = False + +- # Default KV connector: "nixl", "lmcache", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. ++ # Default KV connector: "nixl", "lmcache", "lmcache-mp", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. + # Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. + # "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. + # dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). +@@ -1282,9 +1283,10 @@ class VLLMProtocol: + overridden = pop_vllm_orchestration_flags(config) + config.setdefault("served-model-name", served_model_name) + +- # A prefill/decode worker gets its KV connector; an aggregate worker has none. +- config.pop("connector", None) +- if mode in {"prefill", "decode"}: ++ # A prefill/decode worker gets its KV connector. An aggregate worker gets ++ # only the one its role names (e.g. lmcache-mp offload), not the P/D default. ++ role_connector = config.pop("connector", None) ++ if mode in {"prefill", "decode"} or role_connector is not None: + kv_transfer_config = self.kv_transfer_config(mode, process, runtime) + if kv_transfer_config is not None: + config.setdefault("kv-transfer-config", kv_transfer_config) +@@ -1760,6 +1762,8 @@ class KVConnector: + kv_role: str | None = "kv_both" + module_path: str | None = None + discovery: bool = False ++ # Static ``kv_connector_extra_config``; a discovery row's topology-derived extras replace it. ++ extra_config: dict[str, Any] | None = None + + def transfer_config(self, mode: WorkerMode) -> dict[str, Any]: + """The ``--kv-transfer-config`` payload for a worker mode, before any topology-derived extras.""" +@@ -1767,6 +1771,8 @@ class KVConnector: + if self.module_path is not None: + payload["kv_connector_module_path"] = self.module_path + payload["kv_role"] = self.kv_role or ("kv_producer" if mode == "prefill" else "kv_consumer") ++ if self.extra_config: ++ payload["kv_connector_extra_config"] = dict(self.extra_config) + return payload + + +@@ -1774,6 +1780,12 @@ class KVConnector: + _CONNECTOR_MAP: dict[str, KVConnector] = { + "nixl": KVConnector("NixlConnector"), + "lmcache": KVConnector("LMCacheConnectorV1"), ++ # Out-of-process LMCache MP server (services[].type: lmcache-server) on the worker's own node. ++ "lmcache-mp": KVConnector( ++ "LMCacheMPConnector", ++ module_path="lmcache.integration.vllm.lmcache_mp_connector", ++ extra_config={"lmcache.mp.host": "tcp://localhost", "lmcache.mp.port": LMCACHE_SERVER_PORT}, ++ ), + "kvbm": KVConnector("DynamoConnector", module_path="kvbm.vllm_integration.connector"), + # AMD MoRI-IO (ROCm): prefill produces and decode consumes KV; workers register with the vLLM Router. + "moriio": KVConnector("MoRIIOConnector", kv_role=None, discovery=True), +diff --git a/src/srtctl/ports.py b/src/srtctl/ports.py +index c9ea7e0b..3de5f4b0 100644 +--- a/src/srtctl/ports.py ++++ b/src/srtctl/ports.py +@@ -46,6 +46,12 @@ MOONCAKE_HTTP_METADATA_PORT = 8701 + # the master lives entirely inside our consolidated 8700-range. + MOONCAKE_METRICS_PORT = 8702 + ++# LMCache multiprocess server (services[].type: lmcache-server), one per worker node, ++# reached by that node's vLLM ranks over localhost. LMCache's own defaults (5555, 8080) ++# sit inside ranges the NIXL side channel and the frontend can reach. ++LMCACHE_SERVER_PORT = 8750 ++LMCACHE_HTTP_PORT = 8751 ++ + # vLLM backend ports. + VLLM_NIXL_PORT_BASE = 5400 + # vLLM Router discovery endpoint (frontend.type: vllm-router with a discovery +diff --git a/src/srtctl/services/__init__.py b/src/srtctl/services/__init__.py +index 715a2f2e..81e22071 100644 +--- a/src/srtctl/services/__init__.py ++++ b/src/srtctl/services/__init__.py +@@ -4,7 +4,7 @@ + """The top-level ``services:`` block: user-declared long-running processes launched next to the job.""" + + # Import kinds to trigger registration. +-from srtctl.services import exporters, generic, gms, infra, mooncake_master, mooncake_store, ray ++from srtctl.services import exporters, generic, gms, infra, lmcache_server, mooncake_master, mooncake_store, ray + from srtctl.services.config import ( + SERVICE_PLACEMENTS, + SERVICE_STARTS, +@@ -42,6 +42,7 @@ __all__ = [ + "gms", + "infra", + "list_service_types", ++ "lmcache_server", + "mooncake_master", + "mooncake_store", + "ray", +diff --git a/src/srtctl/services/lmcache_server.py b/src/srtctl/services/lmcache_server.py +new file mode 100644 +index 00000000..31a1de93 +--- /dev/null ++++ b/src/srtctl/services/lmcache_server.py +@@ -0,0 +1,57 @@ ++# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. ++# SPDX-License-Identifier: Apache-2.0 ++ ++"""``type: lmcache-server``: the LMCache multiprocess server, one per worker node. ++ ++The vLLM connector (``connector: lmcache-mp``) only talks to the server on its ++own node, so the kind runs an instance on every worker node, before workers, ++and gates on the server's HTTP healthcheck. ``args`` are appended to the fixed ++command, which sets only the ports srtctl owns. LMCache must be installed in ++the job container: the connector is imported inside the vLLM worker. ++""" ++ ++from __future__ import annotations ++ ++from typing import TYPE_CHECKING ++ ++from srtctl.ports import LMCACHE_HTTP_PORT, LMCACHE_SERVER_PORT ++from srtctl.services.config import HttpProbe, ServiceReadinessConfig ++from srtctl.services.registry import ServiceKind, ServiceLaunchContext, register_service ++ ++if TYPE_CHECKING: ++ from srtctl.services.config import ServiceConfig ++ ++ ++@register_service("lmcache-server") ++class LMCacheServerService(ServiceKind): ++ """LMCache MP server on each worker node; ``args`` are appended to the fixed command.""" ++ ++ builds_command = True ++ default_start = "before_workers" ++ default_critical = True ++ default_placement = "workers" ++ # L1 preallocation pins host memory; large pools outlast the base class's 120 s. ++ default_readiness_timeout = 300 ++ ++ def build_command(self, service: ServiceConfig, ctx: ServiceLaunchContext) -> list[str]: ++ if service.command is not None: ++ return list(service.effective_command) ++ return [ ++ "lmcache", ++ "server", ++ "--host", ++ "0.0.0.0", ++ "--port", ++ str(LMCACHE_SERVER_PORT), ++ "--http-host", ++ "0.0.0.0", ++ "--http-port", ++ str(LMCACHE_HTTP_PORT), ++ *service.args, ++ ] ++ ++ def readiness(self, service: ServiceConfig, ctx: ServiceLaunchContext) -> ServiceReadinessConfig | None: ++ return ServiceReadinessConfig( ++ http=HttpProbe(port=LMCACHE_HTTP_PORT, path="/healthcheck"), ++ timeout_seconds=self.default_readiness_timeout, ++ ) +diff --git a/tests/test_atom_atomesh.py b/tests/test_atom_atomesh.py +index 937800b6..7f4d017f 100644 +--- a/tests/test_atom_atomesh.py ++++ b/tests/test_atom_atomesh.py +@@ -200,6 +200,59 @@ def test_atom_pd_worker_emits_mooncake_kv_transfer_config(protocol: str | None, + } + + ++def test_atom_wraps_extra_kv_connectors_with_mooncake_in_multi() -> None: ++ """A role's extra-kv-connectors ride next to srtctl's Mooncake connector and never reach the CLI as a flag.""" ++ offload = {"kv_connector": "lmcache_offload", "kv_role": "offload", "lmcache.max_local_cpu_size": 180} ++ backend = AtomProtocol(atom_config=AtomServerConfig(prefill={"extra-kv-connectors": [offload], "max-model-len": 8})) ++ prefill = Process("node0", frozenset(range(8)), 7500, 6100, "prefill", 0, nixl_port=6301) ++ decode = Process("node1", frozenset(range(8)), 7500, 6100, "decode", 0, nixl_port=6302) ++ ++ prefill_command = _build(backend, prefill) ++ decode_command = _build(backend, decode) ++ ++ assert json.loads(prefill_command[prefill_command.index("--kv-transfer-config") + 1]) == { ++ "kv_connector": "multi", ++ "connectors": [ ++ {"kv_role": "kv_producer", "kv_connector": "mooncake", "proxy_ip": WORKER_IP, "handshake_port": 6301}, ++ offload, ++ ], ++ } ++ assert "--extra-kv-connectors" not in prefill_command ++ assert prefill_command[-2:] == ["--max-model-len", "8"] ++ assert json.loads(decode_command[decode_command.index("--kv-transfer-config") + 1])["kv_connector"] == "mooncake" ++ ++ ++def test_atom_aggregate_worker_runs_a_single_extra_connector_unwrapped() -> None: ++ offload = {"kv_connector": "lmcache_offload", "kv_role": "offload"} ++ backend = AtomProtocol(atom_config=AtomServerConfig(aggregated={"extra-kv-connectors": [offload]})) ++ process = Process("node0", frozenset(range(8)), 7500, 6100, "agg", 0) ++ ++ command = _build(backend, process) ++ ++ assert json.loads(command[command.index("--kv-transfer-config") + 1]) == offload ++ ++ ++@pytest.mark.parametrize( ++ ("extra_config", "expected"), ++ [ ++ ({}, {"lmcache.mp.port": 8750}), ++ ( ++ {"lmcache.mp.port": 5555, "lmcache.mp.tp_rank_collapse": True}, ++ {"lmcache.mp.port": 5555, "lmcache.mp.tp_rank_collapse": True}, ++ ), ++ ], ++) ++def test_atom_lmcache_mp_defaults_to_the_lmcache_server_port(extra_config: dict, expected: dict) -> None: ++ """lmcache_mp dials srtctl's lmcache-server on its own node unless the recipe gave an address.""" ++ connector = {"kv_connector": "lmcache_mp", "kv_role": "offload", "kv_connector_extra_config": extra_config} ++ backend = AtomProtocol(atom_config=AtomServerConfig(aggregated={"extra-kv-connectors": [connector]})) ++ process = Process("node0", frozenset(range(8)), 7500, 6100, "agg", 0) ++ ++ command = _build(backend, process) ++ ++ assert json.loads(command[command.index("--kv-transfer-config") + 1])["kv_connector_extra_config"] == expected ++ ++ + def test_atom_rejects_cross_node_model_parallel_endpoint() -> None: + """ATOM cannot coordinate one logical worker across two Slurm nodes.""" + leader = Process("node0", frozenset(range(4)), 7500, 6100, "prefill", 0, nixl_port=6301) +diff --git a/tests/test_services.py b/tests/test_services.py +index c7db10bb..51a4b2ce 100644 +--- a/tests/test_services.py ++++ b/tests/test_services.py +@@ -19,7 +19,14 @@ from srtctl.core.runtime import Nodes, RuntimeContext + from srtctl.core.schema import SrtConfig + from srtctl.core.topology import Endpoint + from srtctl.ports import MOONCAKE_HTTP_METADATA_PORT, MOONCAKE_MASTER_PORT +-from srtctl.services import ServiceConfig, ServiceSourceConfig, list_service_types ++from srtctl.services import ( ++ HttpProbe, ++ ServiceConfig, ++ ServiceLaunchContext, ++ ServiceSourceConfig, ++ get_service_kind, ++ list_service_types, ++) + from srtctl.services.implicit import discovery_env, effective_services, uses_discovery_plane + + SRUN = "srtctl.cli.mixins.service_stage.start_srun_process" +@@ -91,6 +98,7 @@ def test_registered_kinds() -> None: + "etcd", + "generic", + "gms", ++ "lmcache-server", + "mooncake-master", + "mooncake-store", + "nats", +@@ -166,6 +174,34 @@ services: + ) + + ++def test_lmcache_server_defaults_and_command() -> None: ++ kind = get_service_kind("lmcache-server") ++ service = ServiceConfig(name="lmcache", type="lmcache-server", args=["--l1-size-gb", "180"]) ++ ctx = ServiceLaunchContext.preview() ++ ++ assert (service.effective_start, kind.default_critical, kind.default_placement) == ( ++ "before_workers", ++ True, ++ "workers", ++ ) ++ assert kind.build_command(service, ctx) == [ ++ "lmcache", ++ "server", ++ "--host", ++ "0.0.0.0", ++ "--port", ++ "8750", ++ "--http-host", ++ "0.0.0.0", ++ "--http-port", ++ "8751", ++ "--l1-size-gb", ++ "180", ++ ] ++ probe = kind.readiness(service, ctx) ++ assert probe is not None and probe.http == HttpProbe(port=8751, path="/healthcheck") ++ ++ + def test_mooncake_store_defaults_and_requires_master() -> None: + with pytest.raises(ValidationError, match="requires backend.mooncake_kv_store"): + _load("services:\n - name: store\n type: mooncake-store\n placement:\n node: workers\n") +diff --git a/tests/test_sglang_lmcache.py b/tests/test_sglang_lmcache.py +new file mode 100644 +index 00000000..52f3f8d3 +--- /dev/null ++++ b/tests/test_sglang_lmcache.py +@@ -0,0 +1,39 @@ ++# SPDX-FileCopyrightText: Copyright (c) 2026 SemiAnalysis LLC. All rights reserved. ++# SPDX-License-Identifier: Apache-2.0 ++ ++"""SGLang workers reach the node-local LMCache MP server (services[].type: lmcache-server).""" ++ ++import pytest ++ ++from srtctl.backends.sglang import SGLangProtocol, SGLangServerConfig ++from srtctl.core.topology import Process ++ ++ ++def _process(mode: str = "prefill") -> Process: ++ return Process("node0", frozenset(range(8)), 7500, 6100, mode, 0) ++ ++ ++def test_enable_lmcache_points_the_worker_at_the_node_local_server() -> None: ++ backend = SGLangProtocol(sglang_config=SGLangServerConfig(prefill={"enable-lmcache": True})) ++ ++ assert backend.get_process_environment(_process("prefill")) == { ++ "LMCACHE_MP_HOST": "127.0.0.1", ++ "LMCACHE_MP_PORT": "8750", ++ } ++ assert backend.get_process_environment(_process("decode")) == {} ++ ++ ++@pytest.mark.parametrize( ++ ("prefill", "environment"), ++ [ ++ ({"enable_lmcache": True, "lmcache_config_file": "/configs/lmcache.yaml"}, {}), ++ ({"enable-lmcache": True}, {"LMCACHE_MP_HOST": "10.0.0.5"}), ++ ], ++) ++def test_recipe_owned_lmcache_address_is_left_alone(prefill: dict, environment: dict) -> None: ++ backend = SGLangProtocol( ++ sglang_config=SGLangServerConfig(prefill=prefill), ++ prefill_environment=environment, ++ ) ++ ++ assert backend.get_process_environment(_process()) == {} +diff --git a/tests/test_vllm_connectors.py b/tests/test_vllm_connectors.py +index 691990b3..1dfe69ed 100644 +--- a/tests/test_vllm_connectors.py ++++ b/tests/test_vllm_connectors.py +@@ -23,6 +23,14 @@ def test_table_presets_serialize_exactly_as_before(): + assert VLLMProtocol(connector="LMCache").kv_transfer_config("decode") == json.dumps( + {"kv_connector": "LMCacheConnectorV1", "kv_role": "kv_both"} + ) ++ assert VLLMProtocol(connector="lmcache-mp").kv_transfer_config("prefill") == json.dumps( ++ { ++ "kv_connector": "LMCacheMPConnector", ++ "kv_connector_module_path": "lmcache.integration.vllm.lmcache_mp_connector", ++ "kv_role": "kv_both", ++ "kv_connector_extra_config": {"lmcache.mp.host": "tcp://localhost", "lmcache.mp.port": 8750}, ++ } ++ ) + assert VLLMProtocol(connector="kvbm").kv_transfer_config("decode") == json.dumps( + { + "kv_connector": "DynamoConnector", +@@ -67,3 +75,39 @@ def test_a_mode_dependent_role_follows_the_worker_mode(): + assert row.transfer_config("prefill")["kv_role"] == "kv_producer" + assert row.transfer_config("decode")["kv_role"] == "kv_consumer" + assert KVConnector("SomeConnector").transfer_config("prefill")["kv_role"] == "kv_both" ++ ++ ++@pytest.mark.parametrize( ++ ("aggregated", "expected"), ++ [ ++ ({}, None), ++ ({"connector": "none"}, None), ++ ({"connector": "lmcache-mp"}, _CONNECTOR_MAP["lmcache-mp"].transfer_config("agg")), ++ ], ++) ++def test_direct_aggregate_worker_runs_only_its_role_connector(aggregated, expected): ++ """A direct `vllm serve` aggregate worker skips the P/D default connector but keeps the one its role names.""" ++ from pathlib import Path ++ from unittest.mock import MagicMock ++ ++ from srtctl.core.topology import Process ++ ++ backend = VLLMProtocol(connector="nixl", vllm_config=VLLMServerConfig(aggregated=aggregated)) ++ process = Process( ++ node="node0", ++ gpu_indices=frozenset(range(8)), ++ sys_port=8081, ++ http_port=0, ++ endpoint_mode="agg", ++ endpoint_index=0, ++ node_rank=0, ++ ) ++ runtime = MagicMock(model_path=Path("/model"), is_hf_model=False, frontend_port=9000) ++ ++ cmd = backend.build_worker_command( ++ process=process, endpoint_processes=[process], runtime=runtime, frontend_type="vllm" ++ ) ++ ++ assert "--connector" not in cmd ++ kv_config = json.loads(cmd[cmd.index("--kv-transfer-config") + 1]) if "--kv-transfer-config" in cmd else None ++ assert kv_config == expected diff --git a/inferencex-e2e/runners/srt-slurm/patches/README.md b/inferencex-e2e/runners/srt-slurm/patches/README.md index 5792ad0430..1d65209094 100644 --- a/inferencex-e2e/runners/srt-slurm/patches/README.md +++ b/inferencex-e2e/runners/srt-slurm/patches/README.md @@ -6,4 +6,4 @@ Each patch is a temporary fix for an open upstream PR. When the PR merges and th | Patch | Upstream PR | Fix | |-------|-------------|-----| -| _(none)_ | | No patches are currently carried; the pinned submodule (v2.30.0) includes everything the runners need. | +| `507-lmcache-server-atom-sglang.patch` | [SemiAnalysisAI/srt-slurm#32](https://github.com/SemiAnalysisAI/srt-slurm/pull/32) (includes [NVIDIA/srt-slurm#507](https://github.com/NVIDIA/srt-slurm/pull/507)) | LMCache for vLLM, SGLang and ATOM: the `lmcache-server` service, and ATOM `extra-kv-connectors` wrapped with Mooncake in `multi` | From 738749748727832a9bbd430e41359dc87fb3f0d8 Mon Sep 17 00:00:00 2001 From: seungrokj Date: Tue, 29 Sep 2026 10:12:13 -0700 Subject: [PATCH 5/6] chore(agentx): Kimi-K3 FP4 MI355X ATOM image 0925, FlyDSL prefill, prefill delay, DCP8 LMCache - Move kimik3-fp4-mi355x-atom-agentic-mtp to rocm/atom-dev:nightly_202609251613, tracking recipes/Agentic-Kimi-K3.md as retuned in ROCm/ATOM#2382. - Enable FlyDSL FP8 prefill attention (ATOM_USE_FLYDSL_FP8_PREFILL_ATTN=1) at every concurrency. - From conc 16 up, hold a ready prefill for four decode passes (ATOM_PREFILL_DECODE_INTERVAL=4, ATOM_PREFILL_DELAYER_MAX_QUEUE_MS=5000); conc 1, 4 and 14 unchanged. - Restore DCP8 LMCache bands on the native srt-slurm recipe: conc 14/16 (DSpark 3, ReplaySSM) and 48/56/72 (no draft) use ATOM's in-process lmcache_offload connector via roles.agg.args.extra-kv-connectors (srt-slurm patch 507), 128 GB/rank up to 48 and 192 GB/rank at 56/72. Conc 1 and 4 stay GPU-resident. Co-Authored-By: Claude Opus 5.5 From f9c1ba46d5cd92a8f9a79c216dd67a0e7a4487d7 Mon Sep 17 00:00:00 2001 From: seungrokj Date: Tue, 29 Sep 2026 11:00:17 -0700 Subject: [PATCH 6/6] chore(agentx): Kimi-K3 FP4 MI355X ATOM image 0925, FlyDSL prefill, prefill delay, DCP8 LMCache - Move kimik3-fp4-mi355x-atom-agentic-mtp to rocm/atom-dev:nightly_202609251613, tracking recipes/Agentic-Kimi-K3.md as retuned in ROCm/ATOM#2382. - Enable FlyDSL FP8 prefill attention (ATOM_USE_FLYDSL_FP8_PREFILL_ATTN=1) at every concurrency. - From conc 16 up, hold a ready prefill for four decode passes (ATOM_PREFILL_DECODE_INTERVAL=4, ATOM_PREFILL_DELAYER_MAX_QUEUE_MS=5000); conc 1, 4 and 14 unchanged. - Restore DCP8 LMCache bands on the native srt-slurm recipe: conc 14/16 (DSpark 3, ReplaySSM) and 48/56/72 (no draft) use ATOM's in-process lmcache_offload connector via roles.agg.args.extra-kv-connectors (srt-slurm patch 507), 128 GB/rank up to 48 and 192 GB/rank at 56/72. Conc 1 and 4 stay GPU-resident. - Draft model precision unchanged: online_quant_config still excludes every Inferact/Kimi-K3-DSpark linear (layers.*, context_proj), so its weights and activations stay BF16, and it shares the target's FP8 KV cache (kv_cache_dtype fp8). FlyDSL FP8 prefill attention applies only to the target; the draft's block pass runs as decode attention. Co-Authored-By: Claude Opus 5.5 --- inferencex-e2e/perf-changelog.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 4e4718646b..4cc6ebd113 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9091,4 +9091,5 @@ - "Enable FlyDSL FP8 prefill attention (ATOM_USE_FLYDSL_FP8_PREFILL_ATTN=1) on every concurrency; ATOM defaults it off." - "Hold a ready prefill for four decode passes from concurrency 16 up (ATOM_PREFILL_DECODE_INTERVAL=4, ATOM_PREFILL_DELAYER_MAX_QUEUE_MS=5000). Concurrency 1, 4 and 14 are unchanged." - "Restore the DCP8 LMCache bands on the native srt-slurm recipe: concurrency 14 and 16 (DSpark 3, ReplaySSM) and 48, 56 and 72 (no draft) run ATOM's in-process lmcache_offload connector through roles.agg.args.extra-kv-connectors (srt-slurm patch 507), with 128 GB/rank up to 48 and 192 GB/rank at 56 and 72. Concurrency 1 and 4 stay GPU-resident." + - "No change to the Inferact/Kimi-K3-DSpark draft's precision: online_quant_config still excludes every draft linear (layers.*, context_proj), so its weights and activations stay BF16, and it keeps the target's FP8 KV cache (kv_cache_dtype fp8). FlyDSL FP8 prefill attention applies only to the target, since the draft runs its block pass as decode attention." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3407