Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -1,13 +1,13 @@
# MiniMax-M3 MXFP4 AgentX on MI355X with ATOM EAGLE3 (GQA draft, three tokens)
# and GPU-resident KV at TP2 and TP4. The LMCache bands were removed with the
# legacy script (#3461): srtctl reserves ATOM's kv-transfer-config for
# disaggregated workers.
# at TP2 and TP4, with GPU-resident KV or an in-process LMCache DRAM tier.
# The DRAM variants add ATOM's lmcache_offload connector through
# extra-kv-connectors (runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch).
base:
schema: 2
name: minimaxm3-fp4-mi355x-atom-agentic
model:
path: hf:amd/MiniMax-M3-MXFP4
container: rocm/atom-dev:nightly_202609171455
container: rocm/atom-dev:nightly_202609281543
precision: fp4
resources:
gpu_type: mi355x
Expand Down Expand Up @@ -47,6 +47,8 @@ base:
AITER_QUICK_REDUCE_QUANTIZATION: INT4
AITER_FLYDSL_STAGE2_FP8: '1'
ATOM_FORCE_ATTN_TRITON: '1'
ATOM_PA_FLYDSL: '1'
ATOM_PA_FLYDSL_PLAN: '1'
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
Expand All @@ -57,7 +59,10 @@ base:
AIPERF_APPLY_CHAT_TEMPLATE: 'true'

# One variant per point. Admission is 2x CONC. TP4 shards the indexer across
# ranks at the measured points (15, 20, 24, 28, 32).
# ranks at the measured points (15, 20, 24, 28, and the DRAM points 40, 48).
# The GPU-resident indexer-CP points turn off ATOM's default MiniMax-M3 mono
# decode (ATOM_MONO_ENABLE=0): with it on they hung in an NCCL broadcast
# mid-replay (run 36493321464). The LMCache points already run without mono.
override_tp4_c1:
roles:
agg:
Expand Down Expand Up @@ -136,6 +141,7 @@ override_tp4_c15:
max-num-seqs: 30
env:
ATOM_M3_INDEXER_CP: '1'
ATOM_MONO_ENABLE: '0'
benchmark:
env:
CONC: '15'
Expand All @@ -148,6 +154,7 @@ override_tp4_c20:
max-num-seqs: 40
env:
ATOM_M3_INDEXER_CP: '1'
ATOM_MONO_ENABLE: '0'
benchmark:
env:
CONC: '20'
Expand All @@ -160,6 +167,7 @@ override_tp4_c24:
max-num-seqs: 48
env:
ATOM_M3_INDEXER_CP: '1'
ATOM_MONO_ENABLE: '0'
benchmark:
env:
CONC: '24'
Expand All @@ -172,38 +180,153 @@ override_tp4_c28:
max-num-seqs: 56
env:
ATOM_M3_INDEXER_CP: '1'
ATOM_MONO_ENABLE: '0'
benchmark:
env:
CONC: '28'

override_tp4_c32:
override_tp2_c1:
roles:
agg:
gpus: 4
gpus: 2
args:
max-num-seqs: 2
benchmark:
env:
CONC: '1'

override_tp2_c2:
roles:
agg:
gpus: 2
args:
max-num-seqs: 64
max-num-seqs: 4
benchmark:
env:
CONC: '2'

# DRAM variants: LMCache's in-process CPU tier with ATOM's segmented LRU and
# all-rank lookup, sized at TOTAL_CPU_DRAM_GB / TP (257 GB) per rank.
# PYTHONHASHSEED keeps every rank's chunk keys identical; without it the
# offload hit rate is zero.
override_tp2_c20_lmcache:
roles:
agg:
gpus: 2
args:
max-num-seqs: 40
extra-kv-connectors:
- kv_connector: lmcache_offload
kv_role: offload
env:
ATOM_M3_INDEXER_CP: '1'
PYTHONHASHSEED: '0'
LMCACHE_LOCAL_CPU: 'True'
LMCACHE_MAX_LOCAL_CPU_SIZE: '257'
LMCACHE_CHUNK_SIZE: '256'
LMCACHE_CACHE_POLICY: ATOM_SLRU
LMCACHE_LOOKUP_SERVER_WORKER_IDS: '0,1'
ATOM_PREFIX_CACHE_POLICY: slru
ATOM_PREFIX_CACHE_PROTECTED_RATIO: '0.5'
benchmark:
env:
CONC: '32'
CONC: '20'
KV_OFFLOADING: 'dram'
TOTAL_CPU_DRAM_GB: '515'

override_tp2_c1:
override_tp2_c25_lmcache:
roles:
agg:
gpus: 2
args:
max-num-seqs: 2
max-num-seqs: 50
extra-kv-connectors:
- kv_connector: lmcache_offload
kv_role: offload
env:
PYTHONHASHSEED: '0'
LMCACHE_LOCAL_CPU: 'True'
LMCACHE_MAX_LOCAL_CPU_SIZE: '257'
LMCACHE_CHUNK_SIZE: '256'
LMCACHE_CACHE_POLICY: ATOM_SLRU
LMCACHE_LOOKUP_SERVER_WORKER_IDS: '0,1'
ATOM_PREFIX_CACHE_POLICY: slru
ATOM_PREFIX_CACHE_PROTECTED_RATIO: '0.5'
benchmark:
env:
CONC: '1'
CONC: '25'
KV_OFFLOADING: 'dram'
TOTAL_CPU_DRAM_GB: '515'

override_tp2_c2:
override_tp2_c30_lmcache:
roles:
agg:
gpus: 2
args:
max-num-seqs: 4
max-num-seqs: 60
extra-kv-connectors:
- kv_connector: lmcache_offload
kv_role: offload
env:
PYTHONHASHSEED: '0'
LMCACHE_LOCAL_CPU: 'True'
LMCACHE_MAX_LOCAL_CPU_SIZE: '257'
LMCACHE_CHUNK_SIZE: '256'
LMCACHE_CACHE_POLICY: ATOM_SLRU
LMCACHE_LOOKUP_SERVER_WORKER_IDS: '0,1'
ATOM_PREFIX_CACHE_POLICY: slru
ATOM_PREFIX_CACHE_PROTECTED_RATIO: '0.5'
benchmark:
env:
CONC: '2'
CONC: '30'
KV_OFFLOADING: 'dram'
TOTAL_CPU_DRAM_GB: '515'

override_tp4_c40_lmcache:
roles:
agg:
gpus: 4
args:
max-num-seqs: 80
extra-kv-connectors:
- kv_connector: lmcache_offload
kv_role: offload
env:
ATOM_M3_INDEXER_CP: '1'
PYTHONHASHSEED: '0'
LMCACHE_LOCAL_CPU: 'True'
LMCACHE_MAX_LOCAL_CPU_SIZE: '257'
LMCACHE_CHUNK_SIZE: '256'
LMCACHE_CACHE_POLICY: ATOM_SLRU
LMCACHE_LOOKUP_SERVER_WORKER_IDS: '0,1,2,3'
ATOM_PREFIX_CACHE_POLICY: slru
ATOM_PREFIX_CACHE_PROTECTED_RATIO: '0.5'
benchmark:
env:
CONC: '40'
KV_OFFLOADING: 'dram'
TOTAL_CPU_DRAM_GB: '1030'

override_tp4_c48_lmcache:
roles:
agg:
gpus: 4
args:
max-num-seqs: 96
extra-kv-connectors:
- kv_connector: lmcache_offload
kv_role: offload
env:
ATOM_M3_INDEXER_CP: '1'
PYTHONHASHSEED: '0'
LMCACHE_LOCAL_CPU: 'True'
LMCACHE_MAX_LOCAL_CPU_SIZE: '257'
LMCACHE_CHUNK_SIZE: '256'
LMCACHE_CACHE_POLICY: ATOM_SLRU
LMCACHE_LOOKUP_SERVER_WORKER_IDS: '0,1,2,3'
ATOM_PREFIX_CACHE_POLICY: slru
ATOM_PREFIX_CACHE_PROTECTED_RATIO: '0.5'
benchmark:
env:
CONC: '48'
KV_OFFLOADING: 'dram'
TOTAL_CPU_DRAM_GB: '1030'
6 changes: 4 additions & 2 deletions inferencex-e2e/configs/amd-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -717,7 +717,7 @@ kimik3-fp4-mi355x-vllm-agentic-mtp:
- { tp: 8, ep: 1, dcp-size: 8, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [44, 48, 70], spec-decoding: mtp, srt-recipe: benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4-mtp/agentic.yaml }

minimaxm3-fp4-mi355x-atom-agentic-mtp:
image: rocm/atom-dev:nightly_202609171455
image: rocm/atom-dev:nightly_202609281543
model: amd/MiniMax-M3-MXFP4
model-prefix: minimaxm3
runner: cluster:mi355x-amds
Expand All @@ -730,8 +730,10 @@ minimaxm3-fp4-mi355x-atom-agentic-mtp:
# = TOTAL_CPU_DRAM_GB / TP = node_DRAM * dram-utilization / 8 (TP-independent).
- dram-utilization: 0.687
search-space:
- { tp: 4, kv-offloading: none, conc-list: [1, 2, 4, 5, 8, 10, 12, 15, 20, 24, 28, 32], spec-decoding: mtp, srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/atom/mi355x-fp4-mtp/agentic.yaml }
- { tp: 4, kv-offloading: none, conc-list: [1, 2, 4, 5, 8, 10, 12, 15, 20, 24, 28], spec-decoding: mtp, srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/atom/mi355x-fp4-mtp/agentic.yaml }
- { tp: 2, kv-offloading: none, conc-list: [1, 2], spec-decoding: mtp, srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/atom/mi355x-fp4-mtp/agentic.yaml }
- { tp: 2, kv-offloading: dram, kv-offload-backend: { name: lmcache, version: "0.5.6.dev98+g05fc77a0" }, conc-list: [20, 25, 30], spec-decoding: mtp, srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/atom/mi355x-fp4-mtp/agentic.yaml }
- { tp: 4, kv-offloading: dram, kv-offload-backend: { name: lmcache, version: "0.5.6.dev98+g05fc77a0" }, conc-list: [40, 48], spec-decoding: mtp, srt-recipe: benchmarks/single_node/srt-slurm-recipes/minimaxm3/atom/mi355x-fp4-mtp/agentic.yaml }

dsr1-fp4-mi355x-sglang-disagg:
image: lmsysorg/sglang-rocm:v0.5.17-rocm720-mi35x-20260809
Expand Down
35 changes: 35 additions & 0 deletions inferencex-e2e/docs/configuration-procedures.md
Original file line number Diff line number Diff line change
Expand Up @@ -287,6 +287,41 @@ The pinned image is the official ATOM nightly
The recipe does not patch AITER source at runtime; TP communication
fusion, DSpark K6 and graph capture use the implementation shipped in the image.

### MiniMax-M3 ATOM FlyDSL paged decode and LMCache DRAM tier

`minimaxm3-fp4-mi355x-atom-agentic-mtp` uses
`rocm/atom-dev:nightly_202609281543` with `ATOM_PA_FLYDSL=1` and
`ATOM_PA_FLYDSL_PLAN=1`, following [ROCm/ATOM#2366](https://github.com/ROCm/ATOM/pull/2366)
and the [upstream recipe](https://github.com/ROCm/ATOM/blob/1423fceb08fbe88b2e35c77b320b3073b310a63f/recipes/MiniMax-M3-Agentic-InferenceX.md).
FlyDSL handles supported paged-decode shapes; its work planner balances dense
decode by actual context length. Unsupported shapes retain the Gluon fallback.
Verify the selected route and capture-time work-plan creation in `server.log`.

The TP2 C20/C25/C30 and TP4 C40/C48 points add LMCache's in-process CPU tier.
The `*_lmcache` variants in
`benchmarks/single_node/srt-slurm-recipes/minimaxm3/atom/mi355x-fp4-mtp/agentic.yaml`
list `{kv_connector: lmcache_offload, kv_role: offload}` under
`extra-kv-connectors`, which srtctl renders as
`--kv-transfer-config '{"kv_connector":"lmcache_offload","kv_role":"offload"}'`
(`runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch`). The
`LMCACHE_*` environment variables configure the tier. Each variant declares
`KV_OFFLOADING: dram` and the matrix `TOTAL_CPU_DRAM_GB`, and sizes
`LMCACHE_MAX_LOCAL_CPU_SIZE` at `TOTAL_CPU_DRAM_GB / TP` (257 GB per rank).
`PYTHONHASHSEED=0` is required: without it the ranks hash prompts to different
keys and the offload hit rate is zero. Verify the composed `kv_transfer_config`
line and non-zero `atom:lmcache_loaded_tokens` in `server.log`.

srtctl masks a partial-node worker to GPUs `0..TP-1`, which sit on NUMA node 0,
and HIP pins each rank's CPU tier on its GPU's node, so the tier and the weight
staging buffers (about 720 GB at TP2 and 1.23 TB at TP4) all come from node 0's
1.5 TB. `runners/srt-slurm/hooks/mi355x-amds/setup.sh` drops the page cache before
a `KV_OFFLOADING=dram` job so these pages are free. Without it, pinning reclaims
page cache, ranks finish minutes apart and ATOM's 300 s startup barrier times out
([run 36454319395](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36454319395)).

TP4 C32 is dropped; the GPU-resident TP4 C1-C28 and TP2 C1-C2 points,
EAGLE3 K3, golden AL 2.78 and indexer CP are unchanged.

### DeepSeek-V4.1-Flash DSpark

The GB200 DSpark recipe uses a minimum CUDA graph capture size of 64 tokens to cover concurrent AgentX subagents. This raises c1/c2/c4 from 8/16/32 to 64; c8 and above retain their existing sizes. The full trace, AL 3.51, and Engram UVA settings are preserved; low-concurrency tail latency improvements require CI confirmation.
Expand Down
32 changes: 32 additions & 0 deletions inferencex-e2e/docs/configuration-procedures_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -264,6 +264,38 @@ schedule 和 ragged verification 保持关闭。
配方不再在运行时修改 AITER 源码;TP 通信融合、DSpark K6 和 graph capture
直接使用镜像内实现。

### MiniMax-M3 ATOM FlyDSL paged decode 与 LMCache DRAM 层

`minimaxm3-fp4-mi355x-atom-agentic-mtp` 按照
[ROCm/ATOM#2366](https://github.com/ROCm/ATOM/pull/2366) 和
[上游配方](https://github.com/ROCm/ATOM/blob/1423fceb08fbe88b2e35c77b320b3073b310a63f/recipes/MiniMax-M3-Agentic-InferenceX.md),
使用 `rocm/atom-dev:nightly_202609281543`,启用 `ATOM_PA_FLYDSL=1` 和
`ATOM_PA_FLYDSL_PLAN=1`。FlyDSL 处理支持的 paged-decode shape,work planner
按实际上下文长度均衡 dense decode 工作量;不支持的 shape 仍回退至 Gluon。
从 `server.log` 核对实际路由,以及 work plan 是否在图捕获时创建。

TP2 C20/C25/C30 与 TP4 C40/C48 点位启用 LMCache 进程内 CPU 层。
`benchmarks/single_node/srt-slurm-recipes/minimaxm3/atom/mi355x-fp4-mtp/agentic.yaml`
中的 `*_lmcache` 变体在 `extra-kv-connectors` 下列出
`{kv_connector: lmcache_offload, kv_role: offload}`,srtctl 将其渲染为
`--kv-transfer-config '{"kv_connector":"lmcache_offload","kv_role":"offload"}'`
(`runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch`),CPU 层由
`LMCACHE_*` 环境变量配置。每个变体声明 `KV_OFFLOADING: dram` 和矩阵的
`TOTAL_CPU_DRAM_GB`,并将 `LMCACHE_MAX_LOCAL_CPU_SIZE` 设为
`TOTAL_CPU_DRAM_GB / TP`(每 rank 257 GB)。`PYTHONHASHSEED=0` 必须设置:否则各
rank 对同一 prompt 计算出不同的 key,卸载命中率为零。在 `server.log` 中核对组装后的
`kv_transfer_config` 日志和非零的 `atom:lmcache_loaded_tokens`。

srtctl 将非整节点 worker 限定在 GPU `0..TP-1`,均位于 NUMA 节点 0;HIP 又把每个 rank
的 CPU 层 pin 在其 GPU 所在的节点上,因此 CPU 层与权重 staging buffer(TP2 约 720 GB,
TP4 约 1.23 TB)全部来自节点 0 的 1.5 TB 内存。`runners/srt-slurm/hooks/mi355x-amds/setup.sh`
会在 `KV_OFFLOADING=dram` 的 job 启动前释放 page cache,保证这些内存页空闲;否则 pin
内存时需要回收 page cache,各 rank 完成时间相差数分钟,ATOM 启动时 300 秒的屏障会超时
([run 36454319395](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36454319395))。

移除 TP4 C32;GPU 常驻 KV 的 TP4 C1-C28、TP2 C1-C2 点位、EAGLE3 K3、golden AL 2.78
和 indexer CP 保持不变。

### DeepSeek-V4.1-Flash DSpark

GB200 的 DSpark 配方将 CUDA graph 最小捕获范围设为 64 tokens,以覆盖 AgentX 子代理并发。这会将 c1/c2/c4 的上限从 8/16/32 提升至 64;c8 及以上保持原有大小。完整轨迹、AL 3.51 和 Engram UVA 配置保持不变;需通过 CI 验证低并发尾延迟改善。
Expand Down
12 changes: 12 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9127,3 +9127,15 @@
- "Add Kimi-K3 MXFP4 vLLM agentic-coding on MI355X (TP8, DSpark): dcp1 c1/c4 GPU-resident and c8-c14 with SimpleCPUOffload DRAM offload, plus a new dcp8 c44/c48/c70 throughput band (mtp synthetic acceptance, no draft); vLLM ROCm image vllm/vllm-openai-rocm:nightly-rocm100-36768d1bfd39094681cdbc8cb37d4b31c0729c89"
- "The Inferact/Kimi-K3-DSpark draft (K3DSparkModel, model_type k3_dspark, 5 layers, hidden 7168, torch_dtype bfloat16, no quantization_config) keeps every layer at its pristine dtype. vLLM builds all its modules with quant_config from get_draft_quant_config() (vllm/models/kimi_k3/nvidia/dspark_mla.py), which returns None because K3DSparkModel is explicitly excluded from the DeepSeek-V4 branch that would set draft.quantization = target.quantization (vllm/config/speculative.py:1431); so context_proj (ReplicatedLinear), context_kv_proj (MergedColumnParallelLinear) and each decoder layer's MLA q/kv projections and dense KimiMLP (gate/up/down) load unquantized in BF16 -- no ptpc_fp8, no mxfp4, no INT4 weights. The draft has no FusedMoE/block_sparse_moe, so VLLM_ROCM_USE_AITER_MOE_SITUV2 (A8W4) never touches it, and its KV cache stays fp8 (kv_cache_dtype). This PR only changes draft depth (num_speculative_tokens 4->7) and enables INT4 custom quick all-reduce (VLLM_ROCM_QUICK_REDUCE_QUANTIZATION=INT4, vllm/distributed/device_communicators/quick_all_reduce.py); because the draft shares the target's TP8 group (vllm/v1/spec_decode/draft_model.py), that INT4 applies to its tensor-parallel reductions too -- a collective-reduction transport precision, not any draft weight, activation, or KV dtype."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3561

- config-keys:
- minimaxm3-fp4-mi355x-atom-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Update the image from rocm/atom-dev:nightly_202609171455 to nightly_202609281543 and enable FlyDSL paged decode (ATOM_PA_FLYDSL=1, ATOM_PA_FLYDSL_PLAN=1)."
- "Restore the LMCache DRAM-offload points: TP2 concurrency 20, 25 and 30, and TP4 concurrency 40 and 48 (in-process lmcache_offload, 257 GB CPU tier per rank)."
- "Set ATOM_MONO_ENABLE=0 on TP4 concurrency 15, 20, 24 and 28; the new image enables mono decode by default, which hangs these indexer-CP points."
- "Drop TP4 concurrency 32."
- "No data-type or precision change to the EAGLE3 draft model (Inferact/MiniMax-M3-EAGLE3-GQA): online_quant_config (ptpc_fp8, which ATOM also applies to the draft), kv_cache_dtype fp8 and index-cache-dtype fp8 are unchanged; the FlyDSL, mono-decode and lmcache_offload toggles affect the decode path and KV offload, not draft precision."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3388
16 changes: 13 additions & 3 deletions inferencex-e2e/runners/srt-slurm/hooks/mi355x-amds/setup.sh
Original file line number Diff line number Diff line change
Expand Up @@ -30,16 +30,26 @@ used=$((total - free))
target=$((used + reserved))

echo "MI355X host memory before preparation:"
grep -E '^(MemAvailable|HugePages_Total|HugePages_Free|HugePages_Rsvd|HugePages_Surp|Hugetlb):' "$meminfo"
grep -E '^(MemFree|MemAvailable|HugePages_Total|HugePages_Free|HugePages_Rsvd|HugePages_Surp|Hugetlb):' "$meminfo"

if (( target < total )); then
printf '%s\n' "$target" | sudo -n tee "$nr_hugepages" >/dev/null
fi

# hipHostMalloc pins host memory on the calling GPU's NUMA node, and srtctl gives a
# partial-node worker GPUs 0..TP-1, all on node 0, so a DRAM KV tier and the weight
# staging buffers must fit in node 0's free pages. Pinning while reclaiming page
# cache stalls ranks minutes apart and past the engine's startup barrier.
if [[ "${KV_OFFLOADING:-}" == dram ]]; then
sync
echo 1 | sudo -n tee /proc/sys/vm/drop_caches >/dev/null
fi

after_total=$(read_hugepage_value HugePages_Total)
after_free=$(read_hugepage_value HugePages_Free)
echo "MI355X host memory after preparation:"
grep -E '^(MemAvailable|HugePages_Total|HugePages_Free|HugePages_Rsvd|HugePages_Surp|Hugetlb):' "$meminfo"
echo "MI355X host memory after preparation (KV_OFFLOADING=${KV_OFFLOADING:-unset}):"
grep -E '^(MemFree|MemAvailable|HugePages_Total|HugePages_Free|HugePages_Rsvd|HugePages_Surp|Hugetlb):' "$meminfo"
grep -h ' MemFree:' /sys/devices/system/node/node*/meminfo

if (( after_total - after_free < used )); then
echo "Host preparation released hugepages that were in use" >&2
Expand Down
Loading