Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -1,11 +1,8 @@
#!/usr/bin/env bash
# Start one LMCache MP server per TP rank in the worker container before vLLM,
# as the legacy MI300X MiniMax-M3 AgentX script did. A variant opts in with
# LMCACHE_SHARDS and LMCACHE_L1_SHARD_GB; its kv-transfer-config lists
# tcp://127.0.0.1:5555 through 5555 + LMCACHE_SHARDS - 1.
# Install the ROCm LMCache wheel for the MI300X MiniMax-M3 AgentX LMCache point.
# It runs twice: as the vLLM worker's setup_script (the connector is imported
# there) and in the lmcache-server service's preamble (its own container).
set -euo pipefail
Comment thread
cquil11 marked this conversation as resolved.
[[ -n "${LMCACHE_SHARDS:-}" ]] || exit 0
: "${LMCACHE_L1_SHARD_GB:?}"
version=0.5.3
pip_install=(python3 -m pip install)
if python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages; then
Expand All @@ -18,29 +15,3 @@ fi
"lmcache==${version}" \
--find-links "https://github.com/LMCache/LMCache/releases/expanded_assets/v${version}-rocm"
python3 -c "import cupy; import lmcache.integration.vllm.lmcache_mp_connector; import opentelemetry.exporter.prometheus"

pids=()
for ((shard = 0; shard < LMCACHE_SHARDS; shard++)); do
# Detached so the servers outlive this preamble and serve the vLLM step.
setsid lmcache server \
--host 127.0.0.1 --port $((5555 + shard)) \
--http-host 127.0.0.1 --http-port $((8080 + shard)) \
--l1-size-gb "$LMCACHE_L1_SHARD_GB" --l1-init-size-gb 10 \
--l1-read-ttl-seconds 7200 --chunk-size 256 --max-workers 2 \
--eviction-policy LRU --supported-transfer-mode lmcache_driven \
> "/logs/lmcache_server_${shard}.log" 2>&1 < /dev/null &
pids+=($!)
done
for ((shard = 0; shard < LMCACHE_SHARDS; shard++)); do
for ((attempt = 0; ; attempt++)); do
python3 -c 'import sys, urllib.request; urllib.request.urlopen(sys.argv[1], timeout=2)' \
"http://127.0.0.1:$((8080 + shard))/healthcheck" 2> /dev/null && break
if ! kill -0 "${pids[$shard]}" 2>/dev/null || (( attempt >= 600 )); then
echo "ERROR: LMCache server $shard did not become ready" >&2
tail -n 50 "/logs/lmcache_server_${shard}.log" >&2 || true
exit 1
fi
sleep 1
done
done
echo "LMCache: ${LMCACHE_SHARDS} servers ready, ${LMCACHE_L1_SHARD_GB} GB L1 each"
Original file line number Diff line number Diff line change
@@ -1,11 +1,11 @@
# MiniMax-M3 MXFP8 AgentX on MI300X with vLLM EAGLE3 (GQA draft). The KV cache
# is GPU-resident, or backed by LMCache MP DRAM servers at the offload point.
# is GPU-resident, or backed by an LMCache MP DRAM server at the offload point.
base:
schema: 2
name: minimaxm3-fp8-mi300x-vllm-agentic
model:
path: hf:MiniMaxAI/MiniMax-M3-MXFP8
container: vllm/vllm-openai-rocm:v0.29.0
container: vllm/vllm-openai-rocm:v0.30.0
precision: fp8
resources:
gpu_type: mi300x
Expand All @@ -24,8 +24,6 @@ base:
health_check:
interval_seconds: 10
max_attempts: 360
# Starts the LMCache servers for the variant that sets LMCACHE_SHARDS.
setup_script: lmcache-mp-rocm.sh
roles:
agg:
nodes: 1
Expand Down Expand Up @@ -72,8 +70,9 @@ base:
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:'
AIPERF_APPLY_CHAT_TEMPLATE: 'true'

# One variant per point. Admission is 2x CONC. The DRAM point splits its
# 1298 GB budget into one LMCache L1 shard per TP rank (1298 / 8 = 162 GB).
# One variant per point. Admission is 2x CONC. The DRAM point runs one
# lmcache-server service on the node with the 8 x 162 GB L1 the per-rank
# servers used to split between them.
override_tp8_c2:
roles:
agg:
Expand Down Expand Up @@ -125,14 +124,40 @@ override_tp8_c10:
KV_OFFLOADING: 'none'

override_tp8_c16_lmcache:
# Installs LMCache in the worker container; the service installs its own.
setup_script: lmcache-mp-rocm.sh
services:
- name: lmcache
type: lmcache-server
preamble: bash /configs/lmcache-mp-rocm.sh
# The default probe with more time: the install counts against it.
readiness:
http:
port: 8751
path: /healthcheck
timeout_seconds: 900
args:
- --l1-size-gb
- '1296'
- --l1-init-size-gb
- '10'
- --l1-read-ttl-seconds
- '7200'
- --chunk-size
- '256'
- --max-workers
- '2'
- --eviction-policy
- LRU
- --supported-transfer-mode
- lmcache_driven
roles:
agg:
args:
max-num-seqs: 32
kv-transfer-config: '{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.server_urls":"tcp://127.0.0.1:5555,tcp://127.0.0.1:5556,tcp://127.0.0.1:5557,tcp://127.0.0.1:5558,tcp://127.0.0.1:5559,tcp://127.0.0.1:5560,tcp://127.0.0.1:5561,tcp://127.0.0.1:5562","lmcache.mp.mq_timeout":6000.0}}'
env:
LMCACHE_SHARDS: '8'
LMCACHE_L1_SHARD_GB: '162'
# The lmcache-mp preset (the node's lmcache-server on port 8750) plus the
# measured message queue timeout.
connector: '{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.host":"tcp://localhost","lmcache.mp.port":8750,"lmcache.mp.mq_timeout":6000.0}}'
benchmark:
env:
CONC: '16'
Expand Down
9 changes: 9 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9053,3 +9053,12 @@
description:
- "Update the GB300 DeepSeek-V4.1-Flash SGLang AgentX curve to lmsysorg/sglang:dev-cu13-nightly-0924 (digest pinned): pure TP4 (EP1) at C1/C2 with Engram in HBM, and TP4/EP4 at C4+ with per-rank host Engram, 16K prefill chunks, no prefill-decode interval, 4096 SWA prefix tails and a 128 decode graph batch; all TP4 points use static ragged verify; TP2 is unchanged."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3421

- config-keys:
- minimaxm3-fp8-mi300x-vllm-agentic-mtp
scenario-type:
- agentic-coding
description:
- "All points now run vllm/vllm-openai-rocm:v0.30.0 (previously v0.29.0)."
- "LMCache DRAM point (TP8, concurrency 16): one LMCache server per node with 1296 GB L1 shared by all 8 TP ranks, replacing 8 per-rank servers of 162 GB L1 each."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3545
Loading
Loading