From 990d9213b1b94886b2630faddbd90ac5d256d437 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 01:39:12 -0400 Subject: [PATCH] [Klaud Cold] Update kimik3-fp4-mi355x-atom-agentic-mtp ATOM image to kimi_k3_agentic_0924 Co-Authored-By: Claude Opus 5.5 (1M context) --- .../kimik3/atom/mi355x-fp4-mtp/agentic.yaml | 2 +- configs/amd-master.yaml | 2 +- perf-changelog.yaml | 6 ++++++ 3 files changed, 8 insertions(+), 2 deletions(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml b/benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml index ac5021e444..81ba5174fe 100644 --- a/benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/kimik3/atom/mi355x-fp4-mtp/agentic.yaml @@ -7,7 +7,7 @@ base: name: kimik3-fp4-mi355x-atom-agentic model: path: hf:moonshotai/Kimi-K3 - container: rocm/atom-dev:ubuntu24.04_py3.12_pytorch_release_2.10.0_kimi_k3_agentic_0911 + container: rocm/atom-dev:ubuntu24.04_py3.12_pytorch_release_2.10.0_kimi_k3_agentic_0924 precision: fp4 resources: gpu_type: mi355x diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index 6eeba9f6d1..63a875ec8c 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -690,7 +690,7 @@ kimik3-fp4-mi355x-vllm-agentic-mtp: # wider in-flight window needs the deeper pool to keep the paged KV # resident. kimik3-fp4-mi355x-atom-agentic-mtp: - image: rocm/atom-dev:ubuntu24.04_py3.12_pytorch_release_2.10.0_kimi_k3_agentic_0911 + image: rocm/atom-dev:ubuntu24.04_py3.12_pytorch_release_2.10.0_kimi_k3_agentic_0924 model: moonshotai/Kimi-K3 model-prefix: kimik3 runner: cluster:mi355x-amds diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 0f361fd6db..1c99188ec6 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8962,3 +8962,9 @@ - "旧镜像切出于 2026-09-16,新镜像领先 441 个提交,包含当前曲线所不含的三项 Kimi-K3 正确性修复:vllm-project/vllm#51483 不再将无状态的首个 chunk 当作 decode,vllm-project/vllm#57098 修复 Kimi-K3 reasoning parser,vllm-project/vllm#57430 支持 routed expert 量化。新镜像还包含本臂在并发大于 4 时均会使用的 ROCm CPU KV 卸载改动,包括 vllm-project/vllm#57160(ROCm CPU KV 卸载改用私有 pinned 张量)与 vllm-project/vllm#50045(卸载背压检测)。" - "This image bump does not change the DSpark draft model data type. Only the image: line changes and kimik3_fp4_mi355x_mtp.sh is unchanged; the draft loads unmodified from the published Inferact/Kimi-K3-DSpark checkpoint via --speculative-config (model=Inferact/Kimi-K3-DSpark, method=dspark). The only dtype in that speculative-config is kv_cache_dtype=fp8, which sets the draft KV-cache storage precision, not the draft weights. No flag overrides or re-quantizes the draft-model weights, so the draft dtype is preserved from its checkpoint across this re-sweep." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3419 + +- config-keys: + - kimik3-fp4-mi355x-atom-agentic-mtp + description: + - "Update ATOM image from kimi_k3_agentic_0911 to kimi_k3_agentic_0924 (latest Kimi-K3 AgentX build) for both the srt-slurm TP8 GPU-resident arm and the legacy-script DCP8 LMCache arms" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3456