From ebcd9ba8a1d898b937b162f7b6c8760ca82e5db2 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 12:18:49 -0500 Subject: [PATCH 1/6] feat(agentx): run the MiniMax-M3 MI300X LMCache point on the lmcache-server service --- .../configs/lmcache-mp-rocm.sh | 35 +- .../vllm/mi300x-fp8-mtp/agentic.yaml | 43 +- inferencex-e2e/perf-changelog.yaml | 11 + .../507-lmcache-server-atom-sglang.patch | 805 ++++++++++++++++++ .../runners/srt-slurm/patches/README.md | 1 + 5 files changed, 854 insertions(+), 41 deletions(-) create mode 100644 inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/lmcache-mp-rocm.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/lmcache-mp-rocm.sh index a69ec274b1..a8adf36ad7 100755 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/lmcache-mp-rocm.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/lmcache-mp-rocm.sh @@ -1,11 +1,8 @@ #!/usr/bin/env bash -# Start one LMCache MP server per TP rank in the worker container before vLLM, -# as the legacy MI300X MiniMax-M3 AgentX script did. A variant opts in with -# LMCACHE_SHARDS and LMCACHE_L1_SHARD_GB; its kv-transfer-config lists -# tcp://127.0.0.1:5555 through 5555 + LMCACHE_SHARDS - 1. +# Install the ROCm LMCache wheel for the MI300X MiniMax-M3 AgentX LMCache point. +# It runs twice: as the vLLM worker's setup_script (the connector is imported +# there) and in the lmcache-server service's preamble (its own container). set -euo pipefail -[[ -n "${LMCACHE_SHARDS:-}" ]] || exit 0 -: "${LMCACHE_L1_SHARD_GB:?}" version=0.5.3 pip_install=(python3 -m pip install) if python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages; then @@ -18,29 +15,3 @@ fi "lmcache==${version}" \ --find-links "https://github.com/LMCache/LMCache/releases/expanded_assets/v${version}-rocm" python3 -c "import cupy; import lmcache.integration.vllm.lmcache_mp_connector; import opentelemetry.exporter.prometheus" - -pids=() -for ((shard = 0; shard < LMCACHE_SHARDS; shard++)); do - # Detached so the servers outlive this preamble and serve the vLLM step. - setsid lmcache server \ - --host 127.0.0.1 --port $((5555 + shard)) \ - --http-host 127.0.0.1 --http-port $((8080 + shard)) \ - --l1-size-gb "$LMCACHE_L1_SHARD_GB" --l1-init-size-gb 10 \ - --l1-read-ttl-seconds 7200 --chunk-size 256 --max-workers 2 \ - --eviction-policy LRU --supported-transfer-mode lmcache_driven \ - > "/logs/lmcache_server_${shard}.log" 2>&1 < /dev/null & - pids+=($!) -done -for ((shard = 0; shard < LMCACHE_SHARDS; shard++)); do - for ((attempt = 0; ; attempt++)); do - python3 -c 'import sys, urllib.request; urllib.request.urlopen(sys.argv[1], timeout=2)' \ - "http://127.0.0.1:$((8080 + shard))/healthcheck" 2> /dev/null && break - if ! kill -0 "${pids[$shard]}" 2>/dev/null || (( attempt >= 600 )); then - echo "ERROR: LMCache server $shard did not become ready" >&2 - tail -n 50 "/logs/lmcache_server_${shard}.log" >&2 || true - exit 1 - fi - sleep 1 - done -done -echo "LMCache: ${LMCACHE_SHARDS} servers ready, ${LMCACHE_L1_SHARD_GB} GB L1 each" diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml index 151ce47cc7..67a6bdf3f6 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml @@ -1,5 +1,5 @@ # MiniMax-M3 MXFP8 AgentX on MI300X with vLLM EAGLE3 (GQA draft). The KV cache -# is GPU-resident, or backed by LMCache MP DRAM servers at the offload point. +# is GPU-resident, or backed by an LMCache MP DRAM server at the offload point. base: schema: 2 name: minimaxm3-fp8-mi300x-vllm-agentic @@ -24,8 +24,6 @@ base: health_check: interval_seconds: 10 max_attempts: 360 - # Starts the LMCache servers for the variant that sets LMCACHE_SHARDS. - setup_script: lmcache-mp-rocm.sh roles: agg: nodes: 1 @@ -72,8 +70,9 @@ base: AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' AIPERF_APPLY_CHAT_TEMPLATE: 'true' -# One variant per point. Admission is 2x CONC. The DRAM point splits its -# 1298 GB budget into one LMCache L1 shard per TP rank (1298 / 8 = 162 GB). +# One variant per point. Admission is 2x CONC. The DRAM point runs one +# lmcache-server service on the node with the 8 x 162 GB L1 the per-rank +# servers used to split between them. override_tp8_c2: roles: agg: @@ -125,14 +124,40 @@ override_tp8_c10: KV_OFFLOADING: 'none' override_tp8_c16_lmcache: + # Installs LMCache in the worker container; the service installs its own. + setup_script: lmcache-mp-rocm.sh + services: + - name: lmcache + type: lmcache-server + preamble: bash /configs/lmcache-mp-rocm.sh + # The default probe with more time: the install counts against it. + readiness: + http: + port: 8751 + path: /healthcheck + timeout_seconds: 900 + args: + - --l1-size-gb + - '1296' + - --l1-init-size-gb + - '10' + - --l1-read-ttl-seconds + - '7200' + - --chunk-size + - '256' + - --max-workers + - '2' + - --eviction-policy + - LRU + - --supported-transfer-mode + - lmcache_driven roles: agg: args: max-num-seqs: 32 - kv-transfer-config: '{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.server_urls":"tcp://127.0.0.1:5555,tcp://127.0.0.1:5556,tcp://127.0.0.1:5557,tcp://127.0.0.1:5558,tcp://127.0.0.1:5559,tcp://127.0.0.1:5560,tcp://127.0.0.1:5561,tcp://127.0.0.1:5562","lmcache.mp.mq_timeout":6000.0}}' - env: - LMCACHE_SHARDS: '8' - LMCACHE_L1_SHARD_GB: '162' + # connector: lmcache-mp does not reach a direct aggregate worker, so this + # is its preset (the service's port 8750) plus the old message timeout. + kv-transfer-config: '{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.host":"tcp://localhost","lmcache.mp.port":8750,"lmcache.mp.mq_timeout":6000.0}}' benchmark: env: CONC: '16' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index e280ad359f..3b0a2bada7 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -8995,3 +8995,14 @@ - "Validate MiniMax-M3 AgentX on H200 with vLLM FP8 after moving the end-to-end project into inferencex-e2e/. Preserve the existing image, recipe, and benchmark settings." - "将端到端项目迁入 inferencex-e2e/ 后,验证 H200 上的 vLLM FP8 MiniMax-M3 AgentX 运行路径,沿用现有镜像、配方和基准测试设置。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3525 + +- config-keys: + - minimaxm3-fp8-mi300x-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Run the MI300X LMCache DRAM point (TP8, concurrency 16) on srt-slurm's native lmcache-server service instead of servers the setup script started inside the worker container. The service runs one LMCache MP server per worker node on ports 8750/8751 before vLLM starts and gates on its /healthcheck; the vLLM worker's kv-transfer-config points LMCacheMPConnector at tcp://localhost:8750. Needs the lmcache-server srt-slurm patch (NVIDIA/srt-slurm#507)." + - "Behavior change: the point previously ran 8 LMCache servers of 162 GB L1 each, one per TP rank (lmcache.mp.server_urls). It now runs one server with the same 1296 GB total L1 and the same LMCache flags (chunk size 256, 2 workers, LRU, 7200 s read TTL, lmcache_driven transfer), which all 8 ranks share." + - "lmcache-mp-rocm.sh now only installs the ROCm LMCache 0.5.3 wheel; it runs as the LMCache variant's setup_script in the worker container and in the service's preamble. The GPU-resident points are unchanged." + - "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX diff --git a/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch new file mode 100644 index 0000000000..c15c487cee --- /dev/null +++ b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch @@ -0,0 +1,805 @@ +diff --git a/docs/config-reference.md b/docs/config-reference.md +index d6ddf28..023fb45 100644 +--- a/docs/config-reference.md ++++ b/docs/config-reference.md +@@ -56,6 +56,28 @@ srt-slurm owns the model path, HTTP port, tensor parallel size, and KV-transfer + contract, so recipes cannot override those arguments. See the complete + [ATOM/AToMesh recipe](../examples/atom/atomesh-disagg.yaml). + ++To run another KV connector next to the Mooncake one (for example ATOM's in-process ++LMCache CPU offload on prefill), list it under `extra-kv-connectors` in that role's ++args. srtctl keeps generating the Mooncake entry, with its handshake port, and wraps ++both in ATOM's `multi` connector. An aggregate role with one extra connector runs it ++on its own. An `lmcache_mp` connector (ATOM's client for the `lmcache-server` ++service) dials the server on its own node, port 8750, unless its ++`kv_connector_extra_config` sets `lmcache.mp.port` or `lmcache.mp.server_urls`. ++ ++```yaml ++roles: ++ prefill: ++ env: ++ PYTHONHASHSEED: "0" # LMCache prefix hashes must agree across offload workers ++ args: ++ extra-kv-connectors: ++ - kv_connector: lmcache_offload ++ kv_role: offload ++ lmcache.local_cpu: true ++ lmcache.max_local_cpu_size: 180 # GB per worker ++ lmcache.chunk_size: 256 ++``` ++ + ```yaml + schema: 2 # Required: recipe layout version + name: "my-benchmark" # Required: job name +diff --git a/docs/schema-reference.md b/docs/schema-reference.md +index 016b082..a51b17c 100644 +--- a/docs/schema-reference.md ++++ b/docs/schema-reference.md +@@ -622,7 +622,7 @@ vLLM protocol - implements BackendProtocol. + |---|---|---|---| + | `type` | one of `'vllm'` | `'vllm'` | | + | `set_visible_devices` | bool | `False` | Use an environment mask instead of the engine's --device-ids option. | +-| `connector` | str \| None | `'nixl'` | Default KV connector: "nixl", "lmcache", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). | ++| `connector` | str \| None | `'nixl'` | Default KV connector: "nixl", "lmcache", "lmcache-mp", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). | + | `failover` | [VLLMFailoverConfig](#vllmfailoverconfig) \| None | `None` | Shadow engine recovery: when set, every worker runs shadow_engines standby engines on its GPUs next to an implied `gms` service that owns the weights. Dynamo frontend only. | + | `allow_prefill_decode_colocation` | bool | `False` | Allow prefill and decode workers to share one node when the combined GPU request fits within gpus_per_node. Defaults off to preserve existing P/D node separation. | + | `allow_prefill_decode_colocation_across_nodes` | bool | `False` | Extend P/D colocation to multi-node topologies. When enabled together with allow_prefill_decode_colocation, workers are packed contiguously across the minimum number of nodes instead of reserving separate P/D node pools. Defaults off to preserve the original one-node-only policy. | +diff --git a/docs/services.md b/docs/services.md +index ab010da..04bb983 100644 +--- a/docs/services.md ++++ b/docs/services.md +@@ -48,7 +48,7 @@ directory. `examples/features/services.yaml` is a runnable version of this. + ```yaml + services: + - name: my-sidecar # required, unique across the list +- type: generic # generic (default) | etcd | nats | mooncake-master | dcgm-exporter | node-exporter | mooncake-store ++ type: generic # generic (default) | etcd | nats | mooncake-master | dcgm-exporter | node-exporter | mooncake-store | lmcache-server + enabled: true # false drops the service (the way to switch an implied one off) + external: null # typed kinds only: use an instance that already runs at this address + command: # argv, not shell-interpreted; required for generic +@@ -104,10 +104,10 @@ services: + | `placement.pool` | none | Run on the nodes another service owns, one instance per node of that pool. Replaces `node`. See [pools.md](pools.md). | + | `placement.per` | `node` | `worker`: one instance per engine worker on each placed node instead of one per node, attached to that worker (its `CUDA_VISIBLE_DEVICES`, the `{worker_*}` placeholders). See [Placement](#placement). | + | `nodes` | none | Whole nodes this service owns: its pool, added to the allocation after the engine roles' nodes, in declaration order. Any number of services may own nodes, next to engine roles or without them. An owner is placed on its own pool (`placement.node: workers`). Not supported with `resources.het_jobs`. See [pools.md](pools.md). | +-| `start` | type default | `etcd`, `nats`: `infra`. `mooncake-master`, `mooncake-store`: `before_workers`. `generic` and the exporters: `after_frontend`. | ++| `start` | type default | `etcd`, `nats`: `infra`. `mooncake-master`, `mooncake-store`, `lmcache-server`: `before_workers`. `generic` and the exporters: `after_frontend`. | + | `readiness` | type default | One probe per node: `port` / `tcp`, `http`, or `log`, plus `timeout_seconds` and `interval_seconds`. The typed kinds gate on their well-known ports when no probe is written. See [Start Order and Readiness](#start-order-and-readiness). Timing out terminates what this stage started and fails the job. | + | `inherit_discovery_env` | `true` | Inject the same `ETCD_ENDPOINTS` / `NATS_SERVER` the Dynamo frontend gets. | +-| `critical` | type default | `generic`: `false`. `mooncake-store`: `true`. | ++| `critical` | type default | `generic`: `false`. `mooncake-store`, `lmcache-server`: `true`. | + | `terminal` | `false` | This service is the job's run: the job ends when every instance of every terminal service has exited, with the worst exit code as the job's. The recipe has no benchmark step (`benchmark.type` stays `manual`); combining the two is refused. See [pools.md](pools.md#ending-the-job-with-a-pool). | + | `metrics` | kind default | Prometheus endpoints the service serves: one `{port, path, nodes, name}` or a list of them (`path` defaults to `/metrics`, `nodes` to `all`, `name` to the service name and is required when there are several). Tachometer scrapes each on every node the service runs on, pools included, as endpoint `_`, and the rows carry `service=`; `nodes: first` scrapes only the service's first node (a cluster head that serves a collector or a router). The exporter kinds declare theirs; write it for anything else that publishes metrics. | + | `preamble` | none | Shell run after the environment is exported and before `command`. | +@@ -302,6 +302,7 @@ environment its process needs; the launch path is shared by every kind. Register + | `node-exporter` | `/bin/node_exporter` with the cpu, infiniband, and meminfo collectors on 9101 in `quay.io/prometheus/node-exporter` | `after_frontend` | `false` | Implied on worker nodes while tachometer runs. Shell-less. `options`: `port`. | + | `process-exporter` | `configs/process-exporter -config.path /process-exporter.yml -web.listen-address=:9256 -threads=true ...` on the bare node | `after_frontend` | `false` | Implied on every allocated node (`placement.node: all`) while tachometer runs. Host-native from the static binary `make setup` installs; skipped with a warning when it is missing. A declared `container` switches to the image's `/bin/process-exporter` with the group file under `/logs`. `options`: `port`, `binary`. | + | `mooncake-store` | `python -m mooncake.mooncake_store_service` | `before_workers` | `true` | Requires a `mooncake-master` entry. Container falls back to the master's. Injects the master's address. | ++| `lmcache-server` | `lmcache server` bound to the RPC and HTTP ports srtctl owns (8750, 8751) | `before_workers` | `true` | One per worker node (`placement.node: workers`); vLLM ranks reach it over localhost with `connector: lmcache-mp`; SGLang workers started with `enable-lmcache` get `LMCACHE_MP_HOST`/`LMCACHE_MP_PORT` pointing at it unless the recipe sets `lmcache-config-file` or `LMCACHE_MP_HOST`. `args` are appended (`--l1-size-gb`, `--chunk-size`, `--max-workers`, ...). Ready when `GET /healthcheck` answers; LMCache must be installed in the job container. | + | `ray` | `ray start --head ...` on the first instance, `ray start --address=:6379 ...` on the rest, both `--block` | `before_workers` | `true` | One Ray cluster across the service's nodes; see [Ray cluster](#ray-cluster). Placement `workers` (default), `head`, or `all`. `options`: `port` (GCS, 6379), `dashboard_port` (8265), `num_gpus` (the node's count). `args` are appended to every `ray start`. | + | `gms` | one `python3 -m gpu_memory_service --device k` per GPU of the worker, supervised by bash, in the job container | `before_workers` | `true` | Implied by `engine.failover`; see [Shadow Engine Recovery](shadow-engine-recovery.md). `placement.per: worker` (required): one instance per vLLM worker in that worker's device view, sockets and lock file under `/srtctl-/_`. Ready when its log says `GMS ready:`; `readiness.timeout_seconds` (default 120) also bounds the servers' startup. | + +diff --git a/examples/README.md b/examples/README.md +index 487fa7b..67c152c 100644 +--- a/examples/README.md ++++ b/examples/README.md +@@ -31,6 +31,9 @@ Every example is written in the 2.0 layout: `engine:` names the engine (a string + | `features/mlperf-client.yaml` | `benchmark.type: custom` driving the MLPerf inference-endpoint client in its own image; placeholder paths, a reference rather than a runnable example | + | `features/infra-services.yaml` | etcd and NATS as declared services on a dedicated node with a NATS payload limit; the implied exporters overridden or switched off | + | `features/dynamo-source.yaml` | `dynamo.source:` building Dynamo from a git tag (or a PR head via `--set dynamo.source.rev=refs/pull//head`), pinned to a commit at submit | ++| `features/lmcache-server.yaml` | `services[].type: lmcache-server` plus `connector: lmcache-mp`: an LMCache DRAM tier, one server per worker node started before the workers and gated on its healthcheck; LMCache flags go under `args`. Needs a vLLM image that ships LMCache (the `vllm-lmcache` alias). See [../docs/services.md](../docs/services.md#service-types) | ++| `features/lmcache-server-disagg.yaml` | The same tier on the prefill side only: the server placed on prefill nodes and a hand-written `MultiConnector` (NIXL plus LMCache MP) in the prefill args, which srtctl passes through untouched | ++| `features/lmcache-server-sglang.yaml` | The same server next to aggregated SGLang workers: SGLang's `enable-lmcache` flag, with srtctl pointing each worker at the server on its node (`LMCACHE_MP_HOST`/`LMCACHE_MP_PORT`). Needs an SGLang image that ships LMCache (the `sglang-lmcache` alias) | + | `features/vllm-failover.yaml` | `engine.failover:` shadow engine recovery: a GPU Memory Service sidecar and a parked standby engine per vLLM worker, relaunched in place after a crash. Needs a container that ships `gpu_memory_service` (the `dynamo-vllm` alias, an `nvcr.io/nvidia/ai-dynamo/vllm-runtime` image). See [../docs/shadow-engine-recovery.md](../docs/shadow-engine-recovery.md) | + + ## Cluster aliases +@@ -46,6 +49,8 @@ containers: + vllm: /path/to/vllm.sqsh # vLLM image with the vllm-router executable + trtllm: /path/to/tensorrtllm-runtime.sqsh # Dynamo TRT-LLM runtime image (ships ai-dynamo and trtllm-serve) + dynamo-vllm: /path/to/vllm-runtime.sqsh # Dynamo vLLM runtime image (ships ai-dynamo and gpu_memory_service), for features/vllm-failover.yaml ++ vllm-lmcache: /path/to/vllm-lmcache.sqsh # vLLM image with LMCache installed, for features/lmcache-server.yaml and -disagg.yaml ++ sglang-lmcache: /path/to/sglang-lmcache.sqsh # SGLang image with LMCache installed, for features/lmcache-server-sglang.yaml + ``` + + `resources.gpu_type` and `gpus_per_node` are set to `h100` and `8`; change them to match the partition you submit to. +diff --git a/examples/features/lmcache-server-disagg.yaml b/examples/features/lmcache-server-disagg.yaml +new file mode 100644 +index 0000000..95eb632 +--- /dev/null ++++ b/examples/features/lmcache-server-disagg.yaml +@@ -0,0 +1,88 @@ ++# An LMCache DRAM tier on the prefill side of a disaggregated vLLM job. ++# ++# Prefill keeps NIXL for the P/D transfer and adds LMCache for prefix reuse, so its ++# kv-transfer-config is a hand-written MultiConnector (srtctl passes raw JSON through ++# untouched; the `lmcache-mp` shorthand covers the single-connector case). Decode is ++# NIXL only. The server is placed on prefill nodes only (`placement.node: prefill`). ++# ++# Aliases (`qwen3-0.6b`, `vllm-lmcache`) resolve through srtslurm.yaml; see ++# examples/README.md and examples/features/lmcache-server.yaml. ++ ++schema: 2 ++name: "qwen3-0.6b-vllm-lmcache-disagg" ++ ++slurm: ++ time_limit: "00:30:00" ++ ++model: ++ path: "qwen3-0.6b" ++ container: "vllm-lmcache" ++ precision: "bf16" ++ ++resources: ++ gpu_type: "h100" ++ gpus_per_node: 8 ++dynamo: ++ source: ++ pypi: "1.4.2" ++ ++environment: ++ PYTHONHASHSEED: "0" ++ ++frontend: ++ type: dynamo ++ enable_multiple_frontends: false ++ args: ++ router-mode: "kv" ++ ++engine: ++ type: vllm ++ connector: nixl ++roles: ++ prefill: ++ nodes: 1 ++ workers: 1 ++ gpus: 1 ++ env: ++ DYN_HEALTH_CHECK_ENABLED: "false" ++ PYTHONUNBUFFERED: "1" ++ args: ++ # NIXL to decode plus LMCache MP on this node; a raw string here replaces the connector preset. ++ kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both"},{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.host":"tcp://localhost","lmcache.mp.port":8750}}]}}' ++ served-model-name: "Qwen/Qwen3-0.6B" ++ tensor-parallel-size: 1 ++ gpu-memory-utilization: 0.5 ++ max-model-len: 4096 ++ decode: ++ nodes: colocate ++ workers: 1 ++ gpus: 1 ++ env: ++ DYN_HEALTH_CHECK_ENABLED: "false" ++ PYTHONUNBUFFERED: "1" ++ args: ++ served-model-name: "Qwen/Qwen3-0.6B" ++ tensor-parallel-size: 1 ++ gpu-memory-utilization: 0.5 ++ max-model-len: 4096 ++ ++services: ++ - name: lmcache ++ type: lmcache-server ++ placement: ++ node: prefill # decode nodes get no server; with `colocate` that is still this one node ++ args: ++ - --l1-size-gb ++ - "16" ++ - --chunk-size ++ - "256" ++ - --max-workers ++ - "1" ++ preamble: | ++ ulimit -l unlimited ++ ++benchmark: ++ type: "sa-bench" ++ isl: 128 ++ osl: 128 ++ concurrencies: "4x8" +diff --git a/examples/features/lmcache-server-sglang.yaml b/examples/features/lmcache-server-sglang.yaml +new file mode 100644 +index 0000000..66a4fb6 +--- /dev/null ++++ b/examples/features/lmcache-server-sglang.yaml +@@ -0,0 +1,62 @@ ++# An LMCache DRAM tier next to aggregated SGLang workers. ++# ++# `type: lmcache-server` runs the LMCache multiprocess server on every worker node ++# before the workers start (8750 RPC, 8751 HTTP). SGLang's own `enable-lmcache` flag ++# switches the worker to LMCacheUnifiedRadixCache; srtctl sets LMCACHE_MP_HOST and ++# LMCACHE_MP_PORT so it dials the server on its own node. A recipe that passes ++# `lmcache-config-file` or sets LMCACHE_MP_HOST in the role env keeps its own address. ++# ++# The SGLang image must ship LMCache, so this uses a `sglang-lmcache` alias. Aliases ++# (`qwen3-0.6b`, `sglang-lmcache`) resolve through srtslurm.yaml; see examples/README.md. ++ ++schema: 2 ++name: "qwen3-0.6b-sglang-lmcache-agg" ++ ++slurm: ++ time_limit: "00:30:00" ++ ++model: ++ path: "qwen3-0.6b" ++ container: "sglang-lmcache" ++ precision: "bf16" ++ ++resources: ++ gpu_type: "h100" ++ gpus_per_node: 8 ++ ++environment: ++ PYTHONHASHSEED: "0" # chunk hashes must agree across the server and every rank ++ ++frontend: ++ type: sglang ++ enable_multiple_frontends: false ++ ++engine: sglang ++roles: ++ agg: ++ nodes: 1 ++ workers: 1 ++ gpus: 1 ++ args: ++ enable-lmcache: true ++ mem-fraction-static: 0.5 ++ context-length: 4096 ++ ++services: ++ - name: lmcache ++ type: lmcache-server ++ args: ++ - --l1-size-gb ++ - "16" ++ - --chunk-size ++ - "256" ++ - --eviction-policy ++ - LRU ++ preamble: | ++ ulimit -l unlimited ++ ++benchmark: ++ type: "sa-bench" ++ isl: 128 ++ osl: 128 ++ concurrencies: "4x8" +diff --git a/examples/features/lmcache-server.yaml b/examples/features/lmcache-server.yaml +new file mode 100644 +index 0000000..4ecee52 +--- /dev/null ++++ b/examples/features/lmcache-server.yaml +@@ -0,0 +1,78 @@ ++# An LMCache DRAM tier next to aggregated vLLM workers. ++# ++# `type: lmcache-server` runs the LMCache multiprocess server on every worker node ++# before the workers start, on the ports srtctl owns (8750 RPC, 8751 HTTP), and holds ++# the job until `GET /healthcheck` answers. `connector: lmcache-mp` points each vLLM ++# worker's LMCacheMPConnector at the server on its own node. Everything after the ++# ports is LMCache's own CLI: put `lmcache server --help` flags under `args`. ++# ++# The vLLM image must ship LMCache (the connector is imported inside the worker), so ++# this uses a `vllm-lmcache` alias. Aliases (`qwen3-0.6b`, `vllm-lmcache`) resolve ++# through srtslurm.yaml; see examples/README.md. ++ ++schema: 2 ++name: "qwen3-0.6b-vllm-lmcache-agg" ++ ++slurm: ++ time_limit: "00:30:00" ++ ++model: ++ path: "qwen3-0.6b" ++ container: "vllm-lmcache" ++ precision: "bf16" ++ ++resources: ++ gpu_type: "h100" ++ gpus_per_node: 8 ++dynamo: ++ source: ++ pypi: "1.4.2" ++ ++environment: ++ PYTHONHASHSEED: "0" # chunk hashes must agree across the server and every rank ++ ++frontend: ++ type: dynamo ++ enable_multiple_frontends: false ++ args: ++ router-mode: "kv" ++ ++engine: ++ type: vllm ++ connector: lmcache-mp # every role; or per role under roles..args.connector ++roles: ++ agg: ++ nodes: 1 ++ workers: 2 ++ gpus: 1 ++ env: ++ DYN_HEALTH_CHECK_ENABLED: "false" ++ PYTHONUNBUFFERED: "1" ++ args: ++ served-model-name: "Qwen/Qwen3-0.6B" ++ tensor-parallel-size: 1 ++ gpu-memory-utilization: 0.5 ++ max-model-len: 4096 ++ ++services: ++ - name: lmcache ++ type: lmcache-server ++ # placement.node: workers (default) — one instance per worker node. ++ # start: before_workers, critical: true (defaults): a dead server fails the run. ++ args: # appended to `lmcache server --host ... --port 8750 --http-host ... --http-port 8751` ++ - --l1-size-gb ++ - "16" ++ - --chunk-size ++ - "256" ++ - --max-workers ++ - "2" ++ - --eviction-policy ++ - LRU ++ preamble: | ++ ulimit -l unlimited ++ ++benchmark: ++ type: "sa-bench" ++ isl: 128 ++ osl: 128 ++ concurrencies: "4x8" +diff --git a/src/srtctl/backends/atom.py b/src/srtctl/backends/atom.py +index 07082f0..a1c8dcc 100644 +--- a/src/srtctl/backends/atom.py ++++ b/src/srtctl/backends/atom.py +@@ -15,7 +15,7 @@ from typing import TYPE_CHECKING, Any, ClassVar, Literal + from marshmallow import Schema + from marshmallow_dataclass import dataclass + +-from srtctl.ports import DYN_SYSTEM_PORT_BASE ++from srtctl.ports import DYN_SYSTEM_PORT_BASE, LMCACHE_SERVER_PORT + + if TYPE_CHECKING: + from srtctl.backends.base import SrunConfig +@@ -137,19 +137,32 @@ class AtomProtocol: + + return endpoints_to_processes(endpoints, base_sys_port=base_sys_port, port_allocator=port_allocator) + +- def _kv_transfer_config(self, process: Process, worker_ip: str) -> str: +- if process.endpoint_mode not in {"prefill", "decode"}: +- raise ValueError("ATOM KV transfer is only valid for prefill/decode workers") +- if process.nixl_port is None: +- raise ValueError("ATOM P/D worker is missing its Mooncake handshake port") +- payload = { +- "kv_role": "kv_producer" if process.endpoint_mode == "prefill" else "kv_consumer", +- "kv_connector": self.connector, +- "proxy_ip": worker_ip, +- "handshake_port": process.nixl_port, +- } +- if self.mooncake_protocol is not None: +- payload["protocol"] = self.mooncake_protocol ++ def _kv_transfer_config( ++ self, process: Process, worker_ip: str, extra_connectors: list[dict[str, Any]] ++ ) -> str | None: ++ """Mooncake for P/D workers plus the role's extra connectors; several are wrapped in ``multi``.""" ++ connectors = [] ++ for connector in extra_connectors: ++ if connector.get("kv_connector") == "lmcache_mp": ++ # Dial the node-local lmcache-server service unless the recipe set a port. ++ extra = {"lmcache.mp.port": LMCACHE_SERVER_PORT, **connector.get("kv_connector_extra_config", {})} ++ connector = {**connector, "kv_connector_extra_config": extra} ++ connectors.append(connector) ++ if process.endpoint_mode in {"prefill", "decode"}: ++ if process.nixl_port is None: ++ raise ValueError("ATOM P/D worker is missing its Mooncake handshake port") ++ mooncake: dict[str, Any] = { ++ "kv_role": "kv_producer" if process.endpoint_mode == "prefill" else "kv_consumer", ++ "kv_connector": self.connector, ++ "proxy_ip": worker_ip, ++ "handshake_port": process.nixl_port, ++ } ++ if self.mooncake_protocol is not None: ++ mooncake["protocol"] = self.mooncake_protocol ++ connectors.insert(0, mooncake) ++ if not connectors: ++ return None ++ payload = connectors[0] if len(connectors) == 1 else {"kv_connector": "multi", "connectors": connectors} + return json.dumps(payload, separators=(",", ":")) + + def build_worker_command( +@@ -171,6 +184,7 @@ class AtomProtocol: + + worker_ip = get_hostname_ip(process.node, runtime.network_interface) + config = self.get_config_for_mode(process.endpoint_mode) ++ extra_connectors = config.pop("extra-kv-connectors", []) + reserved = {"model", "host", "server-port", "tp", "tensor-parallel-size", "kv-transfer-config"} + overlap = reserved.intersection(_canonical_arg_key(key) for key in config) + if overlap: +@@ -192,8 +206,9 @@ class AtomProtocol: + str(len(process.gpu_indices)), + ] + ) +- if process.endpoint_mode in {"prefill", "decode"}: +- command.extend(["--kv-transfer-config", self._kv_transfer_config(process, worker_ip)]) ++ kv_transfer_config = self._kv_transfer_config(process, worker_ip, extra_connectors) ++ if kv_transfer_config is not None: ++ command.extend(["--kv-transfer-config", kv_transfer_config]) + command.extend(_config_to_cli_args(config)) + return command + +diff --git a/src/srtctl/backends/sglang.py b/src/srtctl/backends/sglang.py +index 641709f..121f060 100644 +--- a/src/srtctl/backends/sglang.py ++++ b/src/srtctl/backends/sglang.py +@@ -27,6 +27,7 @@ from srtctl.backends.sidecar import build_sidecar_launch_command, get_dynamo_sid + from srtctl.ports import ( + DIST_INIT_PORTS, + DYN_SYSTEM_PORT_BASE, ++ LMCACHE_SERVER_PORT, + MOONCAKE_HTTP_METADATA_PORT, + MOONCAKE_MASTER_PORT, + NCCL_PORTS, +@@ -203,10 +204,17 @@ class SGLangProtocol: + def get_process_environment(self, process: "Process") -> dict[str, str]: + """Get process-specific environment variables. + +- SGLang handles kv-events via CLI args (--kv-events-config), so no +- additional process-specific env vars are needed here. ++ A worker started with ``enable-lmcache`` dials the LMCache MP server on its own ++ node (``services[].type: lmcache-server``) unless the recipe points it elsewhere ++ with ``lmcache-config-file`` or ``LMCACHE_MP_HOST`` in the role env. + """ +- return {} ++ mode = process.endpoint_mode ++ config = {key.replace("_", "-"): value for key, value in self.get_config_for_mode(mode).items()} ++ if not config.get("enable-lmcache") or config.get("lmcache-config-file"): ++ return {} ++ if "LMCACHE_MP_HOST" in self.get_environment_for_mode(mode): ++ return {} ++ return {"LMCACHE_MP_HOST": "127.0.0.1", "LMCACHE_MP_PORT": str(LMCACHE_SERVER_PORT)} + + def get_mooncake_worker_env(self, infra_node_ip: str, local_hostname: str) -> dict[str, str]: + """Get mooncake env vars to inject on a specific worker. +diff --git a/src/srtctl/backends/vllm.py b/src/srtctl/backends/vllm.py +index 048627a..5d47ae1 100644 +--- a/src/srtctl/backends/vllm.py ++++ b/src/srtctl/backends/vllm.py +@@ -35,6 +35,7 @@ from srtctl.ports import ( + HTTP_PORTS, + KV_EVENTS_PORTS, + KVBM_ZMQ_PORTS, ++ LMCACHE_SERVER_PORT, + MOONCAKE_HTTP_METADATA_PORT, + MOONCAKE_MASTER_PORT, + MORIIO_HANDSHAKE_PORTS, +@@ -351,7 +352,7 @@ class VLLMProtocol: + # Use an environment mask instead of the engine's --device-ids option. + set_visible_devices: bool = False + +- # Default KV connector: "nixl", "lmcache", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. ++ # Default KV connector: "nixl", "lmcache", "lmcache-mp", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. + # Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. + # "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. + # dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). +@@ -1756,6 +1757,8 @@ class KVConnector: + kv_role: str | None = "kv_both" + module_path: str | None = None + discovery: bool = False ++ # Static ``kv_connector_extra_config``; a discovery row's topology-derived extras replace it. ++ extra_config: dict[str, Any] | None = None + + def transfer_config(self, mode: WorkerMode) -> dict[str, Any]: + """The ``--kv-transfer-config`` payload for a worker mode, before any topology-derived extras.""" +@@ -1763,6 +1766,8 @@ class KVConnector: + if self.module_path is not None: + payload["kv_connector_module_path"] = self.module_path + payload["kv_role"] = self.kv_role or ("kv_producer" if mode == "prefill" else "kv_consumer") ++ if self.extra_config: ++ payload["kv_connector_extra_config"] = dict(self.extra_config) + return payload + + +@@ -1770,6 +1775,12 @@ class KVConnector: + _CONNECTOR_MAP: dict[str, KVConnector] = { + "nixl": KVConnector("NixlConnector"), + "lmcache": KVConnector("LMCacheConnectorV1"), ++ # Out-of-process LMCache MP server (services[].type: lmcache-server) on the worker's own node. ++ "lmcache-mp": KVConnector( ++ "LMCacheMPConnector", ++ module_path="lmcache.integration.vllm.lmcache_mp_connector", ++ extra_config={"lmcache.mp.host": "tcp://localhost", "lmcache.mp.port": LMCACHE_SERVER_PORT}, ++ ), + "kvbm": KVConnector("DynamoConnector", module_path="kvbm.vllm_integration.connector"), + # AMD MoRI-IO (ROCm): prefill produces and decode consumes KV; workers register with the vLLM Router. + "moriio": KVConnector("MoRIIOConnector", kv_role=None, discovery=True), +diff --git a/src/srtctl/ports.py b/src/srtctl/ports.py +index c9ea7e0..3de5f4b 100644 +--- a/src/srtctl/ports.py ++++ b/src/srtctl/ports.py +@@ -46,6 +46,12 @@ MOONCAKE_HTTP_METADATA_PORT = 8701 + # the master lives entirely inside our consolidated 8700-range. + MOONCAKE_METRICS_PORT = 8702 + ++# LMCache multiprocess server (services[].type: lmcache-server), one per worker node, ++# reached by that node's vLLM ranks over localhost. LMCache's own defaults (5555, 8080) ++# sit inside ranges the NIXL side channel and the frontend can reach. ++LMCACHE_SERVER_PORT = 8750 ++LMCACHE_HTTP_PORT = 8751 ++ + # vLLM backend ports. + VLLM_NIXL_PORT_BASE = 5400 + # vLLM Router discovery endpoint (frontend.type: vllm-router with a discovery +diff --git a/src/srtctl/services/__init__.py b/src/srtctl/services/__init__.py +index 715a2f2..81e2207 100644 +--- a/src/srtctl/services/__init__.py ++++ b/src/srtctl/services/__init__.py +@@ -4,7 +4,7 @@ + """The top-level ``services:`` block: user-declared long-running processes launched next to the job.""" + + # Import kinds to trigger registration. +-from srtctl.services import exporters, generic, gms, infra, mooncake_master, mooncake_store, ray ++from srtctl.services import exporters, generic, gms, infra, lmcache_server, mooncake_master, mooncake_store, ray + from srtctl.services.config import ( + SERVICE_PLACEMENTS, + SERVICE_STARTS, +@@ -42,6 +42,7 @@ __all__ = [ + "gms", + "infra", + "list_service_types", ++ "lmcache_server", + "mooncake_master", + "mooncake_store", + "ray", +diff --git a/src/srtctl/services/lmcache_server.py b/src/srtctl/services/lmcache_server.py +new file mode 100644 +index 0000000..31a1de9 +--- /dev/null ++++ b/src/srtctl/services/lmcache_server.py +@@ -0,0 +1,57 @@ ++# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. ++# SPDX-License-Identifier: Apache-2.0 ++ ++"""``type: lmcache-server``: the LMCache multiprocess server, one per worker node. ++ ++The vLLM connector (``connector: lmcache-mp``) only talks to the server on its ++own node, so the kind runs an instance on every worker node, before workers, ++and gates on the server's HTTP healthcheck. ``args`` are appended to the fixed ++command, which sets only the ports srtctl owns. LMCache must be installed in ++the job container: the connector is imported inside the vLLM worker. ++""" ++ ++from __future__ import annotations ++ ++from typing import TYPE_CHECKING ++ ++from srtctl.ports import LMCACHE_HTTP_PORT, LMCACHE_SERVER_PORT ++from srtctl.services.config import HttpProbe, ServiceReadinessConfig ++from srtctl.services.registry import ServiceKind, ServiceLaunchContext, register_service ++ ++if TYPE_CHECKING: ++ from srtctl.services.config import ServiceConfig ++ ++ ++@register_service("lmcache-server") ++class LMCacheServerService(ServiceKind): ++ """LMCache MP server on each worker node; ``args`` are appended to the fixed command.""" ++ ++ builds_command = True ++ default_start = "before_workers" ++ default_critical = True ++ default_placement = "workers" ++ # L1 preallocation pins host memory; large pools outlast the base class's 120 s. ++ default_readiness_timeout = 300 ++ ++ def build_command(self, service: ServiceConfig, ctx: ServiceLaunchContext) -> list[str]: ++ if service.command is not None: ++ return list(service.effective_command) ++ return [ ++ "lmcache", ++ "server", ++ "--host", ++ "0.0.0.0", ++ "--port", ++ str(LMCACHE_SERVER_PORT), ++ "--http-host", ++ "0.0.0.0", ++ "--http-port", ++ str(LMCACHE_HTTP_PORT), ++ *service.args, ++ ] ++ ++ def readiness(self, service: ServiceConfig, ctx: ServiceLaunchContext) -> ServiceReadinessConfig | None: ++ return ServiceReadinessConfig( ++ http=HttpProbe(port=LMCACHE_HTTP_PORT, path="/healthcheck"), ++ timeout_seconds=self.default_readiness_timeout, ++ ) +diff --git a/tests/test_atom_atomesh.py b/tests/test_atom_atomesh.py +index 937800b..7f4d017 100644 +--- a/tests/test_atom_atomesh.py ++++ b/tests/test_atom_atomesh.py +@@ -200,6 +200,59 @@ def test_atom_pd_worker_emits_mooncake_kv_transfer_config(protocol: str | None, + } + + ++def test_atom_wraps_extra_kv_connectors_with_mooncake_in_multi() -> None: ++ """A role's extra-kv-connectors ride next to srtctl's Mooncake connector and never reach the CLI as a flag.""" ++ offload = {"kv_connector": "lmcache_offload", "kv_role": "offload", "lmcache.max_local_cpu_size": 180} ++ backend = AtomProtocol(atom_config=AtomServerConfig(prefill={"extra-kv-connectors": [offload], "max-model-len": 8})) ++ prefill = Process("node0", frozenset(range(8)), 7500, 6100, "prefill", 0, nixl_port=6301) ++ decode = Process("node1", frozenset(range(8)), 7500, 6100, "decode", 0, nixl_port=6302) ++ ++ prefill_command = _build(backend, prefill) ++ decode_command = _build(backend, decode) ++ ++ assert json.loads(prefill_command[prefill_command.index("--kv-transfer-config") + 1]) == { ++ "kv_connector": "multi", ++ "connectors": [ ++ {"kv_role": "kv_producer", "kv_connector": "mooncake", "proxy_ip": WORKER_IP, "handshake_port": 6301}, ++ offload, ++ ], ++ } ++ assert "--extra-kv-connectors" not in prefill_command ++ assert prefill_command[-2:] == ["--max-model-len", "8"] ++ assert json.loads(decode_command[decode_command.index("--kv-transfer-config") + 1])["kv_connector"] == "mooncake" ++ ++ ++def test_atom_aggregate_worker_runs_a_single_extra_connector_unwrapped() -> None: ++ offload = {"kv_connector": "lmcache_offload", "kv_role": "offload"} ++ backend = AtomProtocol(atom_config=AtomServerConfig(aggregated={"extra-kv-connectors": [offload]})) ++ process = Process("node0", frozenset(range(8)), 7500, 6100, "agg", 0) ++ ++ command = _build(backend, process) ++ ++ assert json.loads(command[command.index("--kv-transfer-config") + 1]) == offload ++ ++ ++@pytest.mark.parametrize( ++ ("extra_config", "expected"), ++ [ ++ ({}, {"lmcache.mp.port": 8750}), ++ ( ++ {"lmcache.mp.port": 5555, "lmcache.mp.tp_rank_collapse": True}, ++ {"lmcache.mp.port": 5555, "lmcache.mp.tp_rank_collapse": True}, ++ ), ++ ], ++) ++def test_atom_lmcache_mp_defaults_to_the_lmcache_server_port(extra_config: dict, expected: dict) -> None: ++ """lmcache_mp dials srtctl's lmcache-server on its own node unless the recipe gave an address.""" ++ connector = {"kv_connector": "lmcache_mp", "kv_role": "offload", "kv_connector_extra_config": extra_config} ++ backend = AtomProtocol(atom_config=AtomServerConfig(aggregated={"extra-kv-connectors": [connector]})) ++ process = Process("node0", frozenset(range(8)), 7500, 6100, "agg", 0) ++ ++ command = _build(backend, process) ++ ++ assert json.loads(command[command.index("--kv-transfer-config") + 1])["kv_connector_extra_config"] == expected ++ ++ + def test_atom_rejects_cross_node_model_parallel_endpoint() -> None: + """ATOM cannot coordinate one logical worker across two Slurm nodes.""" + leader = Process("node0", frozenset(range(4)), 7500, 6100, "prefill", 0, nixl_port=6301) +diff --git a/tests/test_services.py b/tests/test_services.py +index c7db10b..51a4b2c 100644 +--- a/tests/test_services.py ++++ b/tests/test_services.py +@@ -19,7 +19,14 @@ from srtctl.core.runtime import Nodes, RuntimeContext + from srtctl.core.schema import SrtConfig + from srtctl.core.topology import Endpoint + from srtctl.ports import MOONCAKE_HTTP_METADATA_PORT, MOONCAKE_MASTER_PORT +-from srtctl.services import ServiceConfig, ServiceSourceConfig, list_service_types ++from srtctl.services import ( ++ HttpProbe, ++ ServiceConfig, ++ ServiceLaunchContext, ++ ServiceSourceConfig, ++ get_service_kind, ++ list_service_types, ++) + from srtctl.services.implicit import discovery_env, effective_services, uses_discovery_plane + + SRUN = "srtctl.cli.mixins.service_stage.start_srun_process" +@@ -91,6 +98,7 @@ def test_registered_kinds() -> None: + "etcd", + "generic", + "gms", ++ "lmcache-server", + "mooncake-master", + "mooncake-store", + "nats", +@@ -166,6 +174,34 @@ services: + ) + + ++def test_lmcache_server_defaults_and_command() -> None: ++ kind = get_service_kind("lmcache-server") ++ service = ServiceConfig(name="lmcache", type="lmcache-server", args=["--l1-size-gb", "180"]) ++ ctx = ServiceLaunchContext.preview() ++ ++ assert (service.effective_start, kind.default_critical, kind.default_placement) == ( ++ "before_workers", ++ True, ++ "workers", ++ ) ++ assert kind.build_command(service, ctx) == [ ++ "lmcache", ++ "server", ++ "--host", ++ "0.0.0.0", ++ "--port", ++ "8750", ++ "--http-host", ++ "0.0.0.0", ++ "--http-port", ++ "8751", ++ "--l1-size-gb", ++ "180", ++ ] ++ probe = kind.readiness(service, ctx) ++ assert probe is not None and probe.http == HttpProbe(port=8751, path="/healthcheck") ++ ++ + def test_mooncake_store_defaults_and_requires_master() -> None: + with pytest.raises(ValidationError, match="requires backend.mooncake_kv_store"): + _load("services:\n - name: store\n type: mooncake-store\n placement:\n node: workers\n") +diff --git a/tests/test_sglang_lmcache.py b/tests/test_sglang_lmcache.py +new file mode 100644 +index 0000000..52f3f8d +--- /dev/null ++++ b/tests/test_sglang_lmcache.py +@@ -0,0 +1,39 @@ ++# SPDX-FileCopyrightText: Copyright (c) 2026 SemiAnalysis LLC. All rights reserved. ++# SPDX-License-Identifier: Apache-2.0 ++ ++"""SGLang workers reach the node-local LMCache MP server (services[].type: lmcache-server).""" ++ ++import pytest ++ ++from srtctl.backends.sglang import SGLangProtocol, SGLangServerConfig ++from srtctl.core.topology import Process ++ ++ ++def _process(mode: str = "prefill") -> Process: ++ return Process("node0", frozenset(range(8)), 7500, 6100, mode, 0) ++ ++ ++def test_enable_lmcache_points_the_worker_at_the_node_local_server() -> None: ++ backend = SGLangProtocol(sglang_config=SGLangServerConfig(prefill={"enable-lmcache": True})) ++ ++ assert backend.get_process_environment(_process("prefill")) == { ++ "LMCACHE_MP_HOST": "127.0.0.1", ++ "LMCACHE_MP_PORT": "8750", ++ } ++ assert backend.get_process_environment(_process("decode")) == {} ++ ++ ++@pytest.mark.parametrize( ++ ("prefill", "environment"), ++ [ ++ ({"enable_lmcache": True, "lmcache_config_file": "/configs/lmcache.yaml"}, {}), ++ ({"enable-lmcache": True}, {"LMCACHE_MP_HOST": "10.0.0.5"}), ++ ], ++) ++def test_recipe_owned_lmcache_address_is_left_alone(prefill: dict, environment: dict) -> None: ++ backend = SGLangProtocol( ++ sglang_config=SGLangServerConfig(prefill=prefill), ++ prefill_environment=environment, ++ ) ++ ++ assert backend.get_process_environment(_process()) == {} +diff --git a/tests/test_vllm_connectors.py b/tests/test_vllm_connectors.py +index 691990b..4bb0974 100644 +--- a/tests/test_vllm_connectors.py ++++ b/tests/test_vllm_connectors.py +@@ -23,6 +23,14 @@ def test_table_presets_serialize_exactly_as_before(): + assert VLLMProtocol(connector="LMCache").kv_transfer_config("decode") == json.dumps( + {"kv_connector": "LMCacheConnectorV1", "kv_role": "kv_both"} + ) ++ assert VLLMProtocol(connector="lmcache-mp").kv_transfer_config("prefill") == json.dumps( ++ { ++ "kv_connector": "LMCacheMPConnector", ++ "kv_connector_module_path": "lmcache.integration.vllm.lmcache_mp_connector", ++ "kv_role": "kv_both", ++ "kv_connector_extra_config": {"lmcache.mp.host": "tcp://localhost", "lmcache.mp.port": 8750}, ++ } ++ ) + assert VLLMProtocol(connector="kvbm").kv_transfer_config("decode") == json.dumps( + { + "kv_connector": "DynamoConnector", diff --git a/inferencex-e2e/runners/srt-slurm/patches/README.md b/inferencex-e2e/runners/srt-slurm/patches/README.md index cb619d0506..1f675b9d0f 100644 --- a/inferencex-e2e/runners/srt-slurm/patches/README.md +++ b/inferencex-e2e/runners/srt-slurm/patches/README.md @@ -7,3 +7,4 @@ Each patch is a temporary fix for an open upstream PR. When the PR merges and th | Patch | Upstream PR | Fix | |-------|-------------|-----| | `504-post-eval-srun-options.patch` | [NVIDIA/srt-slurm#504](https://github.com/NVIDIA/srt-slurm/pull/504) | Forward recipe `srun_options` (e.g. `container-writable`) to post-eval steps | +| `507-lmcache-server-atom-sglang.patch` | [SemiAnalysisAI/srt-slurm#32](https://github.com/SemiAnalysisAI/srt-slurm/pull/32) (includes [NVIDIA/srt-slurm#507](https://github.com/NVIDIA/srt-slurm/pull/507)) | LMCache for vLLM, SGLang and ATOM: the `lmcache-server` service, and ATOM `extra-kv-connectors` wrapped with Mooncake in `multi` | From 440ae121f894a7b5604aa3f2084dc94d38894b31 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 12:19:50 -0500 Subject: [PATCH 2/6] chore(agentx): set the perf-changelog pr-link for #3545 --- inferencex-e2e/perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 3b0a2bada7..5621fe73d0 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9005,4 +9005,4 @@ - "Behavior change: the point previously ran 8 LMCache servers of 162 GB L1 each, one per TP rank (lmcache.mp.server_urls). It now runs one server with the same 1296 GB total L1 and the same LMCache flags (chunk size 256, 2 workers, LRU, 7200 s read TTL, lmcache_driven transfer), which all 8 ranks share." - "lmcache-mp-rocm.sh now only installs the ROCm LMCache 0.5.3 wheel; it runs as the LMCache variant's setup_script in the worker container and in the service's preamble. The GPU-resident points are unchanged." - "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights." - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3545 From 535eee78a5002ccb4cf1064bea95bb58751d2f31 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 12:22:01 -0500 Subject: [PATCH 3/6] fix(agentx): match the MiniMax-M3 MI300X recipe image to its config amd-master.yaml moved minimaxm3-fp8-mi300x-vllm-agentic-mtp to vllm/vllm-openai-rocm:v0.30.0 in #3361, but the recipe still named v0.29.0, so the single-node adapter rejected every point. --- .../minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml index 67a6bdf3f6..7003774821 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml @@ -5,7 +5,7 @@ base: name: minimaxm3-fp8-mi300x-vllm-agentic model: path: hf:MiniMaxAI/MiniMax-M3-MXFP8 - container: vllm/vllm-openai-rocm:v0.29.0 + container: vllm/vllm-openai-rocm:v0.30.0 precision: fp8 resources: gpu_type: mi300x From cd2dbd52dc02818fa5698f88cff9ca1a93ad7051 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 12:38:36 -0500 Subject: [PATCH 4/6] fix(agentx): name the lmcache-mp connector on the MiniMax-M3 MI300X LMCache point The refreshed lmcache-server patch (SemiAnalysisAI/srt-slurm#32 at 181b2e4) keeps a role's connector on a direct vllm serve aggregate worker, so the variant names connector: lmcache-mp instead of hand-writing its kv-transfer-config. The message queue timeout falls back to LMCache's default (300 s) instead of 6000 s. --- .../vllm/mi300x-fp8-mtp/agentic.yaml | 5 +- .../507-lmcache-server-atom-sglang.patch | 64 +++++++++++++++++-- 2 files changed, 61 insertions(+), 8 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml index 7003774821..3ba22cb3a3 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml @@ -155,9 +155,8 @@ override_tp8_c16_lmcache: agg: args: max-num-seqs: 32 - # connector: lmcache-mp does not reach a direct aggregate worker, so this - # is its preset (the service's port 8750) plus the old message timeout. - kv-transfer-config: '{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.host":"tcp://localhost","lmcache.mp.port":8750,"lmcache.mp.mq_timeout":6000.0}}' + # LMCacheMPConnector to the lmcache-server service on this node (port 8750). + connector: lmcache-mp benchmark: env: CONC: '16' diff --git a/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch index c15c487cee..92ef255197 100644 --- a/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch +++ b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch @@ -460,7 +460,7 @@ index 641709f..121f060 100644 def get_mooncake_worker_env(self, infra_node_ip: str, local_hostname: str) -> dict[str, str]: """Get mooncake env vars to inject on a specific worker. diff --git a/src/srtctl/backends/vllm.py b/src/srtctl/backends/vllm.py -index 048627a..5d47ae1 100644 +index 048627a..79168fd 100644 --- a/src/srtctl/backends/vllm.py +++ b/src/srtctl/backends/vllm.py @@ -35,6 +35,7 @@ from srtctl.ports import ( @@ -480,7 +480,21 @@ index 048627a..5d47ae1 100644 # Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. # "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. # dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). -@@ -1756,6 +1757,8 @@ class KVConnector: +@@ -1278,9 +1279,10 @@ class VLLMProtocol: + overridden = pop_vllm_orchestration_flags(config) + config.setdefault("served-model-name", served_model_name) + +- # A prefill/decode worker gets its KV connector; an aggregate worker has none. +- config.pop("connector", None) +- if mode in {"prefill", "decode"}: ++ # A prefill/decode worker gets its KV connector. An aggregate worker gets ++ # only the one its role names (e.g. lmcache-mp offload), not the P/D default. ++ role_connector = config.pop("connector", None) ++ if mode in {"prefill", "decode"} or role_connector is not None: + kv_transfer_config = self.kv_transfer_config(mode, process, runtime) + if kv_transfer_config is not None: + config.setdefault("kv-transfer-config", kv_transfer_config) +@@ -1756,6 +1758,8 @@ class KVConnector: kv_role: str | None = "kv_both" module_path: str | None = None discovery: bool = False @@ -489,7 +503,7 @@ index 048627a..5d47ae1 100644 def transfer_config(self, mode: WorkerMode) -> dict[str, Any]: """The ``--kv-transfer-config`` payload for a worker mode, before any topology-derived extras.""" -@@ -1763,6 +1766,8 @@ class KVConnector: +@@ -1763,6 +1767,8 @@ class KVConnector: if self.module_path is not None: payload["kv_connector_module_path"] = self.module_path payload["kv_role"] = self.kv_role or ("kv_producer" if mode == "prefill" else "kv_consumer") @@ -498,7 +512,7 @@ index 048627a..5d47ae1 100644 return payload -@@ -1770,6 +1775,12 @@ class KVConnector: +@@ -1770,6 +1776,12 @@ class KVConnector: _CONNECTOR_MAP: dict[str, KVConnector] = { "nixl": KVConnector("NixlConnector"), "lmcache": KVConnector("LMCacheConnectorV1"), @@ -785,7 +799,7 @@ index 0000000..52f3f8d + + assert backend.get_process_environment(_process()) == {} diff --git a/tests/test_vllm_connectors.py b/tests/test_vllm_connectors.py -index 691990b..4bb0974 100644 +index 691990b..1dfe69e 100644 --- a/tests/test_vllm_connectors.py +++ b/tests/test_vllm_connectors.py @@ -23,6 +23,14 @@ def test_table_presets_serialize_exactly_as_before(): @@ -803,3 +817,43 @@ index 691990b..4bb0974 100644 assert VLLMProtocol(connector="kvbm").kv_transfer_config("decode") == json.dumps( { "kv_connector": "DynamoConnector", +@@ -67,3 +75,39 @@ def test_a_mode_dependent_role_follows_the_worker_mode(): + assert row.transfer_config("prefill")["kv_role"] == "kv_producer" + assert row.transfer_config("decode")["kv_role"] == "kv_consumer" + assert KVConnector("SomeConnector").transfer_config("prefill")["kv_role"] == "kv_both" ++ ++ ++@pytest.mark.parametrize( ++ ("aggregated", "expected"), ++ [ ++ ({}, None), ++ ({"connector": "none"}, None), ++ ({"connector": "lmcache-mp"}, _CONNECTOR_MAP["lmcache-mp"].transfer_config("agg")), ++ ], ++) ++def test_direct_aggregate_worker_runs_only_its_role_connector(aggregated, expected): ++ """A direct `vllm serve` aggregate worker skips the P/D default connector but keeps the one its role names.""" ++ from pathlib import Path ++ from unittest.mock import MagicMock ++ ++ from srtctl.core.topology import Process ++ ++ backend = VLLMProtocol(connector="nixl", vllm_config=VLLMServerConfig(aggregated=aggregated)) ++ process = Process( ++ node="node0", ++ gpu_indices=frozenset(range(8)), ++ sys_port=8081, ++ http_port=0, ++ endpoint_mode="agg", ++ endpoint_index=0, ++ node_rank=0, ++ ) ++ runtime = MagicMock(model_path=Path("/model"), is_hf_model=False, frontend_port=9000) ++ ++ cmd = backend.build_worker_command( ++ process=process, endpoint_processes=[process], runtime=runtime, frontend_type="vllm" ++ ) ++ ++ assert "--connector" not in cmd ++ kv_config = json.loads(cmd[cmd.index("--kv-transfer-config") + 1]) if "--kv-transfer-config" in cmd else None ++ assert kv_config == expected From 8444c86258e5ee9cfe380df8ef083f114a110a78 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 12:49:21 -0500 Subject: [PATCH 5/6] fix(agentx): keep the measured LMCache message queue timeout on MI300X The role connector is now the lmcache-mp preset written out plus lmcache.mp.mq_timeout 6000, so the point keeps its measured timeout instead of falling back to LMCache's 300 s default. --- .../minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml index 3ba22cb3a3..ad1cb5bd46 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/minimaxm3/vllm/mi300x-fp8-mtp/agentic.yaml @@ -155,8 +155,9 @@ override_tp8_c16_lmcache: agg: args: max-num-seqs: 32 - # LMCacheMPConnector to the lmcache-server service on this node (port 8750). - connector: lmcache-mp + # The lmcache-mp preset (the node's lmcache-server on port 8750) plus the + # measured message queue timeout. + connector: '{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.host":"tcp://localhost","lmcache.mp.port":8750,"lmcache.mp.mq_timeout":6000.0}}' benchmark: env: CONC: '16' From 835de73e3557fe36a991f5c497533438d45fa322 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Tue, 29 Sep 2026 09:17:52 -0500 Subject: [PATCH 6/6] docs(changelog): keep only the perf-affecting changes in the #3545 entry --- inferencex-e2e/perf-changelog.yaml | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index dcc24a3197..09e353c217 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9009,8 +9009,6 @@ scenario-type: - agentic-coding description: - - "Benchmark change for the LMCache DRAM point (TP8, concurrency 16): it now runs srt-slurm's lmcache-server service, one LMCache server per node with 1296 GB L1 shared by all 8 TP ranks, replacing the 8 per-rank LMCache servers of 162 GB L1 each. The server starts before vLLM on ports 8750/8751 and gates on its /healthcheck; the other LMCache flags are unchanged (chunk size 256, 2 workers, LRU, 7200 s read TTL, lmcache_driven transfer)." - - "The vLLM worker's connector is LMCacheMPConnector at tcp://localhost:8750, with lmcache.mp.mq_timeout kept at 6000 s. Needs the lmcache-server srt-slurm patch (SemiAnalysisAI/srt-slurm#32, including NVIDIA/srt-slurm#507)." - - "The recipe image is vllm/vllm-openai-rocm:v0.30.0, matching the config. The GPU-resident points are unchanged, and all points are re-measured." - - "The EAGLE3 draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint; no flag overrides or re-quantizes the draft weights." + - "All points now run vllm/vllm-openai-rocm:v0.30.0 (previously v0.29.0)." + - "LMCache DRAM point (TP8, concurrency 16): one LMCache server per node with 1296 GB L1 shared by all 8 TP ranks, replacing 8 per-rank servers of 162 GB L1 each." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3545