From 136ee0222bd44aa360d220e93692d4b54d8f989f Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 11:56:17 -0500 Subject: [PATCH 1/6] feat(agentx): port the MI355X DSV4 ATOM disagg LMCache config to srt-slurm Move dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark (#3158) from the legacy amd_utils path to a native srt-slurm recipe with one override variant per point, matching server_atom.sh for every tier. The conc 256 tier adds ATOM's in-process lmcache_offload next to the Mooncake P/D connector through srt-slurm's extra-kv-connectors, which the new 507-lmcache-server-atom-sglang.patch applies to the pinned submodule (SemiAnalysisAI/srt-slurm#32, including NVIDIA/srt-slurm#507). The golden-acceptance injector now treats atom-disagg as ATOM. --- .../agentx/disagg-lmcache-dspark.yaml | 254 ++++++ inferencex-e2e/configs/amd-master.yaml | 59 +- .../infx/srt_slurm/synthetic_acceptance.py | 1 + inferencex-e2e/perf-changelog.yaml | 9 + .../507-lmcache-server-atom-sglang.patch | 805 ++++++++++++++++++ .../runners/srt-slurm/patches/README.md | 1 + 6 files changed, 1109 insertions(+), 20 deletions(-) create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml create mode 100644 inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml new file mode 100644 index 0000000000..f3addef0d4 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml @@ -0,0 +1,254 @@ +# DeepSeek-V4-Pro-0813 AgentX on MI355X: 1P1D ATOM disaggregation over Mooncake +# RDMA behind AToMesh, with DSpark (three draft tokens). Ported from the legacy +# amd_utils path (#3158; ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md). Three tiers: +# TP8 at concurrency 1-16, DP attention at 64-128, and DP attention plus ATOM's +# in-process LMCache CPU offload on prefill at 256. srtctl generates the Mooncake +# P/D connector; the offload tier adds lmcache_offload next to it +# (runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch). +base: + schema: 2 + name: mi355x-dsv4-pro-0813-atom-agentx-lmcache + model: + path: DeepSeek-V4-Pro-0813 + container: rocm/atom-dev:nightly_202609221542 + precision: fp4 + slurm: + time_limit: "08:00:00" + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atomesh + enable_multiple_frontends: false + args: + log-level: info + disable-health-check: true + disable-circuit-breaker: true + prometheus-port: 29100 + engine: atom + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + env: &environment + PYTHONUNBUFFERED: "1" + PYTHONDONTWRITEBYTECODE: "1" + SAFETENSORS_FAST_GPU: "1" + VLLM_LOG_LEVEL: WARNING + ATOM_LOG_LEVEL: WARNING + AITER_LOG_LEVEL: WARNING + LOG_LEVEL: WARNING + LOGLEVEL: WARNING + ATOM_MOE_GU_ITLV: "1" + AITER_BF16_FP8_MOE_BOUND: "0" + # Rail-isolated RDMA: matched rails replace dest-device affinity. + MC_GID_INDEX: "1" + ATOM_MOONCAKE_MATCHED_RAILS: auto + NCCL_IB_DISABLE: "1" + ATOM_DISABLE_MMAP: "true" + ATOM_NUMA_BIND: "1" + ATOM_DP_MASTER_PORT: "29510" + ATOM_DP_BASE_PORT: "29610" + ATOM_PREFIX_CACHE_POLICY: lru + ATOM_PREFIX_CACHE_PROTECTED_RATIO: "0.5" + args: &server + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + trust-remote-code: true + method: dspark + num-speculative-tokens: 3 + kv_cache_dtype: fp8 + index-cache-dtype: fp4 + # ATOM forces block size 256 for V4. + block-size: 256 + gpu-memory-utilization: 0.9 + max-model-len: 1000000 + max-num-batched-tokens: 16384 + attn-prefill-chunk-size: 16384 + state-checkpoint-interval-tokens: 8192 + level: 3 + cudagraph-mode: FULL + enable-prefix-caching: true + decode: + nodes: 1 + workers: 1 + gpus: 8 + env: + <<: *environment + args: + <<: *server + sbatch_directives: + cpus-per-task: "128" + mem: "0" + srun_options: + mem: "0" + container-writable: "" + container-remap-root: "" + health_check: + max_attempts: 720 + interval_seconds: 5 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro-0813 + # The MI355X launcher collects results from the job workspace. + RESULT_DIR: /infmax-workspace/LOGS/agentic + AGENTIC_OUTPUT_DIR: /infmax-workspace + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "atom:" + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache/hub + TOKENIZERS_PARALLELISM: "false" + TRANSFORMERS_VERBOSITY: error + +# One variant per point; each point runs in its own allocation. Admission is +# 2x CONC on both roles. Decode captures every batch up to min(64, 2x CONC) on +# TP, or CONC/4 per rank under DP attention. DP attention adds TBO on prefill. + +override_tp8_c1: + frontend: + args: + prefill-policy: round_robin + decode-policy: round_robin + atom-pd-rank-mapping-policy: none + roles: + prefill: + args: + max-num-seqs: 2 + decode: + args: + max-num-seqs: 2 + cudagraph-capture-sizes: '[1,2]' + benchmark: + env: + CONC: '1' + +override_tp8_c16: + frontend: + args: + prefill-policy: round_robin + decode-policy: round_robin + atom-pd-rank-mapping-policy: none + roles: + prefill: + args: + max-num-seqs: 32 + decode: + args: + max-num-seqs: 32 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32]' + benchmark: + env: + CONC: '16' + +override_dpa8_c64: + frontend: + args: + dp-aware: true + prefill-policy: cache_aware + decode-policy: cache_aware + cache-threshold: 0.8 + balance-abs-threshold: 20 + balance-rel-threshold: 2.0 + eviction-interval: 300 + atom-pd-rank-mapping-policy: none + roles: + prefill: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + GPU_MAX_HW_QUEUES: '5' + args: + max-num-seqs: 128 + enable-dp-attention: true + enable-tbo: true + decode: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + args: + max-num-seqs: 128 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16]' + enable-dp-attention: true + benchmark: + env: + CONC: '64' + +override_dpa8_c128: + frontend: + args: + dp-aware: true + prefill-policy: cache_aware + decode-policy: cache_aware + cache-threshold: 0.8 + balance-abs-threshold: 20 + balance-rel-threshold: 2.0 + eviction-interval: 300 + atom-pd-rank-mapping-policy: none + roles: + prefill: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + GPU_MAX_HW_QUEUES: '5' + args: + max-num-seqs: 256 + enable-dp-attention: true + enable-tbo: true + decode: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + args: + max-num-seqs: 256 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32]' + enable-dp-attention: true + benchmark: + env: + CONC: '128' + +override_dpa8_c256_lmcache: + frontend: + args: + dp-aware: true + prefill-policy: dp_sticky + decode-policy: dp_sticky + atom-pd-rank-mapping-policy: none + roles: + prefill: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + GPU_MAX_HW_QUEUES: '5' + PYTHONHASHSEED: '0' + OFFLOAD_COPY_WORKERS: '1' + OFFLOAD_MIN_LOAD_TOKENS: '8192' + OFFLOAD_SLOT_STAGING_SLOTS: '4' + args: + max-num-seqs: 512 + enable-dp-attention: true + enable-tbo: true + extra-kv-connectors: + - kv_connector: lmcache_offload + kv_role: offload + offload_layout: hybrid + max_pending_saves: 8 + slot_sidecar_staging_slots: 4 + lmcache.local_cpu: true + # total-cpu-dram-gb 1499 (dram-utilization 0.5) over the eight prefill ranks. + lmcache.max_local_cpu_size: 187 + lmcache.local_disk: null + lmcache.max_local_disk_size: 0 + lmcache.remote_url: null + lmcache.chunk_size: 256 + lmcache.cache_policy: LRU + lmcache.lookup_server_worker_ids: [] + lmcache.store_location: LocalCPUBackend + lmcache.retrieve_locations: [LocalCPUBackend] + decode: + env: + ATOM_ENABLE_PREFILL_DELAYER: '0' + args: + max-num-seqs: 512 + cudagraph-capture-sizes: '[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64]' + enable-dp-attention: true + benchmark: + env: + CONC: '256' + # dp_sticky keys sessions on the correlation id. + AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: '1' diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index 74c709949e..60d676ba08 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -1215,13 +1215,13 @@ dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark: disagg: true scenarios: agentic-coding: - # MTP tiers with GPU-resident KV, unchanged from run 35643358891 apart from - # the image bump. Concurrency 1/16 run plain TP8; 64/128 add DP attention - # (c128 measured in run 35810485934, gsm8k eval in run 35851118405). + # srt-slurm recipe benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml + # (ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md). Concurrency 1/16 run plain + # TP8; 64/128 add DP attention with prefill TBO and the cache-aware router. - dram-utilization: 0.80 search-space: - spec-decoding: "draft_model" - conc-list: [ 1, 16 ] + conc-list: [ 1 ] kv-offloading: none prefill: num-worker: 1 @@ -1229,19 +1229,29 @@ dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark: ep: 1 dp-attn: false additional-settings: - - "PREFILL_NODES=1" + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_tp8_c1" decode: num-worker: 1 tp: 8 ep: 1 dp-attn: false + - spec-decoding: "draft_model" + conc-list: [ 16 ] + kv-offloading: none + prefill: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false additional-settings: - - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" - # DP-attention tier (recipe: concurrency 64-128). TP8 + DP attention, - # prefill TBO only, cache-aware router, no CPU offload. + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_tp8_c16" + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false - spec-decoding: "draft_model" - conc-list: [ 64, 128 ] + conc-list: [ 64 ] kv-offloading: none prefill: num-worker: 1 @@ -1249,18 +1259,30 @@ dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark: ep: 1 dp-attn: true additional-settings: - - "PREFILL_NODES=1" + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_dpa8_c64" decode: num-worker: 1 tp: 8 ep: 1 dp-attn: true + - spec-decoding: "draft_model" + conc-list: [ 128 ] + kv-offloading: none + prefill: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: true additional-settings: - - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" - # DP-attention + CPU offload tier (recipe: concurrency 256). Prefill adds - # the lmcache host KV offload tier; decode is plain Mooncake. DSpark draft - # model and dram-utilization 0.50 as measured in run 35825955863. + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_dpa8_c128" + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: true + # DP attention plus ATOM's in-process LMCache CPU offload on prefill + # (lmcache_offload); decode is plain Mooncake. dram-utilization 0.50 as + # measured in run 35825955863. - dram-utilization: 0.50 search-space: - spec-decoding: "draft_model" @@ -1273,15 +1295,12 @@ dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark: ep: 1 dp-attn: true additional-settings: - - "PREFILL_NODES=1" + - "CONFIG_FILE=recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml:override_dpa8_c256_lmcache" decode: num-worker: 1 tp: 8 ep: 1 dp-attn: true - additional-settings: - - "DECODE_NODES=1" - - "DECODE_MTP_SIZE=3" # GLM-5.2 FP4 agentic-coding benchmark on MI355X via SGLang with MTP speculative # decoding. Two arms: diff --git a/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py b/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py index 45a9f2e153..691c86c009 100644 --- a/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py +++ b/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py @@ -28,6 +28,7 @@ "trt": "trtllm", "dynamo-trt": "trtllm", "atom": "atom", + "atom-disagg": "atom", } SGLANG_VARIABLES = ( "SGLANG_SIMULATE_ACC_LEN", diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index e280ad359f..cf79221271 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -8995,3 +8995,12 @@ - "Validate MiniMax-M3 AgentX on H200 with vLLM FP8 after moving the end-to-end project into inferencex-e2e/. Preserve the existing image, recipe, and benchmark settings." - "将端到端项目迁入 inferencex-e2e/ 后,验证 H200 上的 vLLM FP8 MiniMax-M3 AgentX 运行路径,沿用现有镜像、配方和基准测试设置。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3525 + +- config-keys: + - dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark + scenario-type: + - agentic-coding + description: + - "Port the DeepSeek-V4-Pro-0813 MI355X ATOM 1P1D AgentX config from the legacy amd_utils path to a native srt-slurm recipe (benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml), one override variant per point. Server flags, env, AToMesh routing policies and decode CUDA-graph capture sizes match the legacy server_atom.sh for every tier: TP8 at concurrency 1 and 16, DP attention with prefill TBO at 64 and 128, and DP attention with ATOM's in-process LMCache CPU offload (lmcache_offload, 187 GB per prefill rank) at 256. Golden acceptance 3.01 for DSpark with three draft tokens is now injected by the srt-slurm path, the same value the legacy models_atom.yaml hardcoded." + - "srt-slurm had no way to add LMCache next to its generated Mooncake connector, so runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch applies SemiAnalysisAI/srt-slurm#32 (which includes NVIDIA/srt-slurm#507) to the pinned submodule until it lands upstream." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXXX diff --git a/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch new file mode 100644 index 0000000000..c15c487cee --- /dev/null +++ b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch @@ -0,0 +1,805 @@ +diff --git a/docs/config-reference.md b/docs/config-reference.md +index d6ddf28..023fb45 100644 +--- a/docs/config-reference.md ++++ b/docs/config-reference.md +@@ -56,6 +56,28 @@ srt-slurm owns the model path, HTTP port, tensor parallel size, and KV-transfer + contract, so recipes cannot override those arguments. See the complete + [ATOM/AToMesh recipe](../examples/atom/atomesh-disagg.yaml). + ++To run another KV connector next to the Mooncake one (for example ATOM's in-process ++LMCache CPU offload on prefill), list it under `extra-kv-connectors` in that role's ++args. srtctl keeps generating the Mooncake entry, with its handshake port, and wraps ++both in ATOM's `multi` connector. An aggregate role with one extra connector runs it ++on its own. An `lmcache_mp` connector (ATOM's client for the `lmcache-server` ++service) dials the server on its own node, port 8750, unless its ++`kv_connector_extra_config` sets `lmcache.mp.port` or `lmcache.mp.server_urls`. ++ ++```yaml ++roles: ++ prefill: ++ env: ++ PYTHONHASHSEED: "0" # LMCache prefix hashes must agree across offload workers ++ args: ++ extra-kv-connectors: ++ - kv_connector: lmcache_offload ++ kv_role: offload ++ lmcache.local_cpu: true ++ lmcache.max_local_cpu_size: 180 # GB per worker ++ lmcache.chunk_size: 256 ++``` ++ + ```yaml + schema: 2 # Required: recipe layout version + name: "my-benchmark" # Required: job name +diff --git a/docs/schema-reference.md b/docs/schema-reference.md +index 016b082..a51b17c 100644 +--- a/docs/schema-reference.md ++++ b/docs/schema-reference.md +@@ -622,7 +622,7 @@ vLLM protocol - implements BackendProtocol. + |---|---|---|---| + | `type` | one of `'vllm'` | `'vllm'` | | + | `set_visible_devices` | bool | `False` | Use an environment mask instead of the engine's --device-ids option. | +-| `connector` | str \| None | `'nixl'` | Default KV connector: "nixl", "lmcache", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). | ++| `connector` | str \| None | `'nixl'` | Default KV connector: "nixl", "lmcache", "lmcache-mp", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). | + | `failover` | [VLLMFailoverConfig](#vllmfailoverconfig) \| None | `None` | Shadow engine recovery: when set, every worker runs shadow_engines standby engines on its GPUs next to an implied `gms` service that owns the weights. Dynamo frontend only. | + | `allow_prefill_decode_colocation` | bool | `False` | Allow prefill and decode workers to share one node when the combined GPU request fits within gpus_per_node. Defaults off to preserve existing P/D node separation. | + | `allow_prefill_decode_colocation_across_nodes` | bool | `False` | Extend P/D colocation to multi-node topologies. When enabled together with allow_prefill_decode_colocation, workers are packed contiguously across the minimum number of nodes instead of reserving separate P/D node pools. Defaults off to preserve the original one-node-only policy. | +diff --git a/docs/services.md b/docs/services.md +index ab010da..04bb983 100644 +--- a/docs/services.md ++++ b/docs/services.md +@@ -48,7 +48,7 @@ directory. `examples/features/services.yaml` is a runnable version of this. + ```yaml + services: + - name: my-sidecar # required, unique across the list +- type: generic # generic (default) | etcd | nats | mooncake-master | dcgm-exporter | node-exporter | mooncake-store ++ type: generic # generic (default) | etcd | nats | mooncake-master | dcgm-exporter | node-exporter | mooncake-store | lmcache-server + enabled: true # false drops the service (the way to switch an implied one off) + external: null # typed kinds only: use an instance that already runs at this address + command: # argv, not shell-interpreted; required for generic +@@ -104,10 +104,10 @@ services: + | `placement.pool` | none | Run on the nodes another service owns, one instance per node of that pool. Replaces `node`. See [pools.md](pools.md). | + | `placement.per` | `node` | `worker`: one instance per engine worker on each placed node instead of one per node, attached to that worker (its `CUDA_VISIBLE_DEVICES`, the `{worker_*}` placeholders). See [Placement](#placement). | + | `nodes` | none | Whole nodes this service owns: its pool, added to the allocation after the engine roles' nodes, in declaration order. Any number of services may own nodes, next to engine roles or without them. An owner is placed on its own pool (`placement.node: workers`). Not supported with `resources.het_jobs`. See [pools.md](pools.md). | +-| `start` | type default | `etcd`, `nats`: `infra`. `mooncake-master`, `mooncake-store`: `before_workers`. `generic` and the exporters: `after_frontend`. | ++| `start` | type default | `etcd`, `nats`: `infra`. `mooncake-master`, `mooncake-store`, `lmcache-server`: `before_workers`. `generic` and the exporters: `after_frontend`. | + | `readiness` | type default | One probe per node: `port` / `tcp`, `http`, or `log`, plus `timeout_seconds` and `interval_seconds`. The typed kinds gate on their well-known ports when no probe is written. See [Start Order and Readiness](#start-order-and-readiness). Timing out terminates what this stage started and fails the job. | + | `inherit_discovery_env` | `true` | Inject the same `ETCD_ENDPOINTS` / `NATS_SERVER` the Dynamo frontend gets. | +-| `critical` | type default | `generic`: `false`. `mooncake-store`: `true`. | ++| `critical` | type default | `generic`: `false`. `mooncake-store`, `lmcache-server`: `true`. | + | `terminal` | `false` | This service is the job's run: the job ends when every instance of every terminal service has exited, with the worst exit code as the job's. The recipe has no benchmark step (`benchmark.type` stays `manual`); combining the two is refused. See [pools.md](pools.md#ending-the-job-with-a-pool). | + | `metrics` | kind default | Prometheus endpoints the service serves: one `{port, path, nodes, name}` or a list of them (`path` defaults to `/metrics`, `nodes` to `all`, `name` to the service name and is required when there are several). Tachometer scrapes each on every node the service runs on, pools included, as endpoint `_`, and the rows carry `service=`; `nodes: first` scrapes only the service's first node (a cluster head that serves a collector or a router). The exporter kinds declare theirs; write it for anything else that publishes metrics. | + | `preamble` | none | Shell run after the environment is exported and before `command`. | +@@ -302,6 +302,7 @@ environment its process needs; the launch path is shared by every kind. Register + | `node-exporter` | `/bin/node_exporter` with the cpu, infiniband, and meminfo collectors on 9101 in `quay.io/prometheus/node-exporter` | `after_frontend` | `false` | Implied on worker nodes while tachometer runs. Shell-less. `options`: `port`. | + | `process-exporter` | `configs/process-exporter -config.path /process-exporter.yml -web.listen-address=:9256 -threads=true ...` on the bare node | `after_frontend` | `false` | Implied on every allocated node (`placement.node: all`) while tachometer runs. Host-native from the static binary `make setup` installs; skipped with a warning when it is missing. A declared `container` switches to the image's `/bin/process-exporter` with the group file under `/logs`. `options`: `port`, `binary`. | + | `mooncake-store` | `python -m mooncake.mooncake_store_service` | `before_workers` | `true` | Requires a `mooncake-master` entry. Container falls back to the master's. Injects the master's address. | ++| `lmcache-server` | `lmcache server` bound to the RPC and HTTP ports srtctl owns (8750, 8751) | `before_workers` | `true` | One per worker node (`placement.node: workers`); vLLM ranks reach it over localhost with `connector: lmcache-mp`; SGLang workers started with `enable-lmcache` get `LMCACHE_MP_HOST`/`LMCACHE_MP_PORT` pointing at it unless the recipe sets `lmcache-config-file` or `LMCACHE_MP_HOST`. `args` are appended (`--l1-size-gb`, `--chunk-size`, `--max-workers`, ...). Ready when `GET /healthcheck` answers; LMCache must be installed in the job container. | + | `ray` | `ray start --head ...` on the first instance, `ray start --address=:6379 ...` on the rest, both `--block` | `before_workers` | `true` | One Ray cluster across the service's nodes; see [Ray cluster](#ray-cluster). Placement `workers` (default), `head`, or `all`. `options`: `port` (GCS, 6379), `dashboard_port` (8265), `num_gpus` (the node's count). `args` are appended to every `ray start`. | + | `gms` | one `python3 -m gpu_memory_service --device k` per GPU of the worker, supervised by bash, in the job container | `before_workers` | `true` | Implied by `engine.failover`; see [Shadow Engine Recovery](shadow-engine-recovery.md). `placement.per: worker` (required): one instance per vLLM worker in that worker's device view, sockets and lock file under `/srtctl-/_`. Ready when its log says `GMS ready:`; `readiness.timeout_seconds` (default 120) also bounds the servers' startup. | + +diff --git a/examples/README.md b/examples/README.md +index 487fa7b..67c152c 100644 +--- a/examples/README.md ++++ b/examples/README.md +@@ -31,6 +31,9 @@ Every example is written in the 2.0 layout: `engine:` names the engine (a string + | `features/mlperf-client.yaml` | `benchmark.type: custom` driving the MLPerf inference-endpoint client in its own image; placeholder paths, a reference rather than a runnable example | + | `features/infra-services.yaml` | etcd and NATS as declared services on a dedicated node with a NATS payload limit; the implied exporters overridden or switched off | + | `features/dynamo-source.yaml` | `dynamo.source:` building Dynamo from a git tag (or a PR head via `--set dynamo.source.rev=refs/pull//head`), pinned to a commit at submit | ++| `features/lmcache-server.yaml` | `services[].type: lmcache-server` plus `connector: lmcache-mp`: an LMCache DRAM tier, one server per worker node started before the workers and gated on its healthcheck; LMCache flags go under `args`. Needs a vLLM image that ships LMCache (the `vllm-lmcache` alias). See [../docs/services.md](../docs/services.md#service-types) | ++| `features/lmcache-server-disagg.yaml` | The same tier on the prefill side only: the server placed on prefill nodes and a hand-written `MultiConnector` (NIXL plus LMCache MP) in the prefill args, which srtctl passes through untouched | ++| `features/lmcache-server-sglang.yaml` | The same server next to aggregated SGLang workers: SGLang's `enable-lmcache` flag, with srtctl pointing each worker at the server on its node (`LMCACHE_MP_HOST`/`LMCACHE_MP_PORT`). Needs an SGLang image that ships LMCache (the `sglang-lmcache` alias) | + | `features/vllm-failover.yaml` | `engine.failover:` shadow engine recovery: a GPU Memory Service sidecar and a parked standby engine per vLLM worker, relaunched in place after a crash. Needs a container that ships `gpu_memory_service` (the `dynamo-vllm` alias, an `nvcr.io/nvidia/ai-dynamo/vllm-runtime` image). See [../docs/shadow-engine-recovery.md](../docs/shadow-engine-recovery.md) | + + ## Cluster aliases +@@ -46,6 +49,8 @@ containers: + vllm: /path/to/vllm.sqsh # vLLM image with the vllm-router executable + trtllm: /path/to/tensorrtllm-runtime.sqsh # Dynamo TRT-LLM runtime image (ships ai-dynamo and trtllm-serve) + dynamo-vllm: /path/to/vllm-runtime.sqsh # Dynamo vLLM runtime image (ships ai-dynamo and gpu_memory_service), for features/vllm-failover.yaml ++ vllm-lmcache: /path/to/vllm-lmcache.sqsh # vLLM image with LMCache installed, for features/lmcache-server.yaml and -disagg.yaml ++ sglang-lmcache: /path/to/sglang-lmcache.sqsh # SGLang image with LMCache installed, for features/lmcache-server-sglang.yaml + ``` + + `resources.gpu_type` and `gpus_per_node` are set to `h100` and `8`; change them to match the partition you submit to. +diff --git a/examples/features/lmcache-server-disagg.yaml b/examples/features/lmcache-server-disagg.yaml +new file mode 100644 +index 0000000..95eb632 +--- /dev/null ++++ b/examples/features/lmcache-server-disagg.yaml +@@ -0,0 +1,88 @@ ++# An LMCache DRAM tier on the prefill side of a disaggregated vLLM job. ++# ++# Prefill keeps NIXL for the P/D transfer and adds LMCache for prefix reuse, so its ++# kv-transfer-config is a hand-written MultiConnector (srtctl passes raw JSON through ++# untouched; the `lmcache-mp` shorthand covers the single-connector case). Decode is ++# NIXL only. The server is placed on prefill nodes only (`placement.node: prefill`). ++# ++# Aliases (`qwen3-0.6b`, `vllm-lmcache`) resolve through srtslurm.yaml; see ++# examples/README.md and examples/features/lmcache-server.yaml. ++ ++schema: 2 ++name: "qwen3-0.6b-vllm-lmcache-disagg" ++ ++slurm: ++ time_limit: "00:30:00" ++ ++model: ++ path: "qwen3-0.6b" ++ container: "vllm-lmcache" ++ precision: "bf16" ++ ++resources: ++ gpu_type: "h100" ++ gpus_per_node: 8 ++dynamo: ++ source: ++ pypi: "1.4.2" ++ ++environment: ++ PYTHONHASHSEED: "0" ++ ++frontend: ++ type: dynamo ++ enable_multiple_frontends: false ++ args: ++ router-mode: "kv" ++ ++engine: ++ type: vllm ++ connector: nixl ++roles: ++ prefill: ++ nodes: 1 ++ workers: 1 ++ gpus: 1 ++ env: ++ DYN_HEALTH_CHECK_ENABLED: "false" ++ PYTHONUNBUFFERED: "1" ++ args: ++ # NIXL to decode plus LMCache MP on this node; a raw string here replaces the connector preset. ++ kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both"},{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.host":"tcp://localhost","lmcache.mp.port":8750}}]}}' ++ served-model-name: "Qwen/Qwen3-0.6B" ++ tensor-parallel-size: 1 ++ gpu-memory-utilization: 0.5 ++ max-model-len: 4096 ++ decode: ++ nodes: colocate ++ workers: 1 ++ gpus: 1 ++ env: ++ DYN_HEALTH_CHECK_ENABLED: "false" ++ PYTHONUNBUFFERED: "1" ++ args: ++ served-model-name: "Qwen/Qwen3-0.6B" ++ tensor-parallel-size: 1 ++ gpu-memory-utilization: 0.5 ++ max-model-len: 4096 ++ ++services: ++ - name: lmcache ++ type: lmcache-server ++ placement: ++ node: prefill # decode nodes get no server; with `colocate` that is still this one node ++ args: ++ - --l1-size-gb ++ - "16" ++ - --chunk-size ++ - "256" ++ - --max-workers ++ - "1" ++ preamble: | ++ ulimit -l unlimited ++ ++benchmark: ++ type: "sa-bench" ++ isl: 128 ++ osl: 128 ++ concurrencies: "4x8" +diff --git a/examples/features/lmcache-server-sglang.yaml b/examples/features/lmcache-server-sglang.yaml +new file mode 100644 +index 0000000..66a4fb6 +--- /dev/null ++++ b/examples/features/lmcache-server-sglang.yaml +@@ -0,0 +1,62 @@ ++# An LMCache DRAM tier next to aggregated SGLang workers. ++# ++# `type: lmcache-server` runs the LMCache multiprocess server on every worker node ++# before the workers start (8750 RPC, 8751 HTTP). SGLang's own `enable-lmcache` flag ++# switches the worker to LMCacheUnifiedRadixCache; srtctl sets LMCACHE_MP_HOST and ++# LMCACHE_MP_PORT so it dials the server on its own node. A recipe that passes ++# `lmcache-config-file` or sets LMCACHE_MP_HOST in the role env keeps its own address. ++# ++# The SGLang image must ship LMCache, so this uses a `sglang-lmcache` alias. Aliases ++# (`qwen3-0.6b`, `sglang-lmcache`) resolve through srtslurm.yaml; see examples/README.md. ++ ++schema: 2 ++name: "qwen3-0.6b-sglang-lmcache-agg" ++ ++slurm: ++ time_limit: "00:30:00" ++ ++model: ++ path: "qwen3-0.6b" ++ container: "sglang-lmcache" ++ precision: "bf16" ++ ++resources: ++ gpu_type: "h100" ++ gpus_per_node: 8 ++ ++environment: ++ PYTHONHASHSEED: "0" # chunk hashes must agree across the server and every rank ++ ++frontend: ++ type: sglang ++ enable_multiple_frontends: false ++ ++engine: sglang ++roles: ++ agg: ++ nodes: 1 ++ workers: 1 ++ gpus: 1 ++ args: ++ enable-lmcache: true ++ mem-fraction-static: 0.5 ++ context-length: 4096 ++ ++services: ++ - name: lmcache ++ type: lmcache-server ++ args: ++ - --l1-size-gb ++ - "16" ++ - --chunk-size ++ - "256" ++ - --eviction-policy ++ - LRU ++ preamble: | ++ ulimit -l unlimited ++ ++benchmark: ++ type: "sa-bench" ++ isl: 128 ++ osl: 128 ++ concurrencies: "4x8" +diff --git a/examples/features/lmcache-server.yaml b/examples/features/lmcache-server.yaml +new file mode 100644 +index 0000000..4ecee52 +--- /dev/null ++++ b/examples/features/lmcache-server.yaml +@@ -0,0 +1,78 @@ ++# An LMCache DRAM tier next to aggregated vLLM workers. ++# ++# `type: lmcache-server` runs the LMCache multiprocess server on every worker node ++# before the workers start, on the ports srtctl owns (8750 RPC, 8751 HTTP), and holds ++# the job until `GET /healthcheck` answers. `connector: lmcache-mp` points each vLLM ++# worker's LMCacheMPConnector at the server on its own node. Everything after the ++# ports is LMCache's own CLI: put `lmcache server --help` flags under `args`. ++# ++# The vLLM image must ship LMCache (the connector is imported inside the worker), so ++# this uses a `vllm-lmcache` alias. Aliases (`qwen3-0.6b`, `vllm-lmcache`) resolve ++# through srtslurm.yaml; see examples/README.md. ++ ++schema: 2 ++name: "qwen3-0.6b-vllm-lmcache-agg" ++ ++slurm: ++ time_limit: "00:30:00" ++ ++model: ++ path: "qwen3-0.6b" ++ container: "vllm-lmcache" ++ precision: "bf16" ++ ++resources: ++ gpu_type: "h100" ++ gpus_per_node: 8 ++dynamo: ++ source: ++ pypi: "1.4.2" ++ ++environment: ++ PYTHONHASHSEED: "0" # chunk hashes must agree across the server and every rank ++ ++frontend: ++ type: dynamo ++ enable_multiple_frontends: false ++ args: ++ router-mode: "kv" ++ ++engine: ++ type: vllm ++ connector: lmcache-mp # every role; or per role under roles..args.connector ++roles: ++ agg: ++ nodes: 1 ++ workers: 2 ++ gpus: 1 ++ env: ++ DYN_HEALTH_CHECK_ENABLED: "false" ++ PYTHONUNBUFFERED: "1" ++ args: ++ served-model-name: "Qwen/Qwen3-0.6B" ++ tensor-parallel-size: 1 ++ gpu-memory-utilization: 0.5 ++ max-model-len: 4096 ++ ++services: ++ - name: lmcache ++ type: lmcache-server ++ # placement.node: workers (default) — one instance per worker node. ++ # start: before_workers, critical: true (defaults): a dead server fails the run. ++ args: # appended to `lmcache server --host ... --port 8750 --http-host ... --http-port 8751` ++ - --l1-size-gb ++ - "16" ++ - --chunk-size ++ - "256" ++ - --max-workers ++ - "2" ++ - --eviction-policy ++ - LRU ++ preamble: | ++ ulimit -l unlimited ++ ++benchmark: ++ type: "sa-bench" ++ isl: 128 ++ osl: 128 ++ concurrencies: "4x8" +diff --git a/src/srtctl/backends/atom.py b/src/srtctl/backends/atom.py +index 07082f0..a1c8dcc 100644 +--- a/src/srtctl/backends/atom.py ++++ b/src/srtctl/backends/atom.py +@@ -15,7 +15,7 @@ from typing import TYPE_CHECKING, Any, ClassVar, Literal + from marshmallow import Schema + from marshmallow_dataclass import dataclass + +-from srtctl.ports import DYN_SYSTEM_PORT_BASE ++from srtctl.ports import DYN_SYSTEM_PORT_BASE, LMCACHE_SERVER_PORT + + if TYPE_CHECKING: + from srtctl.backends.base import SrunConfig +@@ -137,19 +137,32 @@ class AtomProtocol: + + return endpoints_to_processes(endpoints, base_sys_port=base_sys_port, port_allocator=port_allocator) + +- def _kv_transfer_config(self, process: Process, worker_ip: str) -> str: +- if process.endpoint_mode not in {"prefill", "decode"}: +- raise ValueError("ATOM KV transfer is only valid for prefill/decode workers") +- if process.nixl_port is None: +- raise ValueError("ATOM P/D worker is missing its Mooncake handshake port") +- payload = { +- "kv_role": "kv_producer" if process.endpoint_mode == "prefill" else "kv_consumer", +- "kv_connector": self.connector, +- "proxy_ip": worker_ip, +- "handshake_port": process.nixl_port, +- } +- if self.mooncake_protocol is not None: +- payload["protocol"] = self.mooncake_protocol ++ def _kv_transfer_config( ++ self, process: Process, worker_ip: str, extra_connectors: list[dict[str, Any]] ++ ) -> str | None: ++ """Mooncake for P/D workers plus the role's extra connectors; several are wrapped in ``multi``.""" ++ connectors = [] ++ for connector in extra_connectors: ++ if connector.get("kv_connector") == "lmcache_mp": ++ # Dial the node-local lmcache-server service unless the recipe set a port. ++ extra = {"lmcache.mp.port": LMCACHE_SERVER_PORT, **connector.get("kv_connector_extra_config", {})} ++ connector = {**connector, "kv_connector_extra_config": extra} ++ connectors.append(connector) ++ if process.endpoint_mode in {"prefill", "decode"}: ++ if process.nixl_port is None: ++ raise ValueError("ATOM P/D worker is missing its Mooncake handshake port") ++ mooncake: dict[str, Any] = { ++ "kv_role": "kv_producer" if process.endpoint_mode == "prefill" else "kv_consumer", ++ "kv_connector": self.connector, ++ "proxy_ip": worker_ip, ++ "handshake_port": process.nixl_port, ++ } ++ if self.mooncake_protocol is not None: ++ mooncake["protocol"] = self.mooncake_protocol ++ connectors.insert(0, mooncake) ++ if not connectors: ++ return None ++ payload = connectors[0] if len(connectors) == 1 else {"kv_connector": "multi", "connectors": connectors} + return json.dumps(payload, separators=(",", ":")) + + def build_worker_command( +@@ -171,6 +184,7 @@ class AtomProtocol: + + worker_ip = get_hostname_ip(process.node, runtime.network_interface) + config = self.get_config_for_mode(process.endpoint_mode) ++ extra_connectors = config.pop("extra-kv-connectors", []) + reserved = {"model", "host", "server-port", "tp", "tensor-parallel-size", "kv-transfer-config"} + overlap = reserved.intersection(_canonical_arg_key(key) for key in config) + if overlap: +@@ -192,8 +206,9 @@ class AtomProtocol: + str(len(process.gpu_indices)), + ] + ) +- if process.endpoint_mode in {"prefill", "decode"}: +- command.extend(["--kv-transfer-config", self._kv_transfer_config(process, worker_ip)]) ++ kv_transfer_config = self._kv_transfer_config(process, worker_ip, extra_connectors) ++ if kv_transfer_config is not None: ++ command.extend(["--kv-transfer-config", kv_transfer_config]) + command.extend(_config_to_cli_args(config)) + return command + +diff --git a/src/srtctl/backends/sglang.py b/src/srtctl/backends/sglang.py +index 641709f..121f060 100644 +--- a/src/srtctl/backends/sglang.py ++++ b/src/srtctl/backends/sglang.py +@@ -27,6 +27,7 @@ from srtctl.backends.sidecar import build_sidecar_launch_command, get_dynamo_sid + from srtctl.ports import ( + DIST_INIT_PORTS, + DYN_SYSTEM_PORT_BASE, ++ LMCACHE_SERVER_PORT, + MOONCAKE_HTTP_METADATA_PORT, + MOONCAKE_MASTER_PORT, + NCCL_PORTS, +@@ -203,10 +204,17 @@ class SGLangProtocol: + def get_process_environment(self, process: "Process") -> dict[str, str]: + """Get process-specific environment variables. + +- SGLang handles kv-events via CLI args (--kv-events-config), so no +- additional process-specific env vars are needed here. ++ A worker started with ``enable-lmcache`` dials the LMCache MP server on its own ++ node (``services[].type: lmcache-server``) unless the recipe points it elsewhere ++ with ``lmcache-config-file`` or ``LMCACHE_MP_HOST`` in the role env. + """ +- return {} ++ mode = process.endpoint_mode ++ config = {key.replace("_", "-"): value for key, value in self.get_config_for_mode(mode).items()} ++ if not config.get("enable-lmcache") or config.get("lmcache-config-file"): ++ return {} ++ if "LMCACHE_MP_HOST" in self.get_environment_for_mode(mode): ++ return {} ++ return {"LMCACHE_MP_HOST": "127.0.0.1", "LMCACHE_MP_PORT": str(LMCACHE_SERVER_PORT)} + + def get_mooncake_worker_env(self, infra_node_ip: str, local_hostname: str) -> dict[str, str]: + """Get mooncake env vars to inject on a specific worker. +diff --git a/src/srtctl/backends/vllm.py b/src/srtctl/backends/vllm.py +index 048627a..5d47ae1 100644 +--- a/src/srtctl/backends/vllm.py ++++ b/src/srtctl/backends/vllm.py +@@ -35,6 +35,7 @@ from srtctl.ports import ( + HTTP_PORTS, + KV_EVENTS_PORTS, + KVBM_ZMQ_PORTS, ++ LMCACHE_SERVER_PORT, + MOONCAKE_HTTP_METADATA_PORT, + MOONCAKE_MASTER_PORT, + MORIIO_HANDSHAKE_PORTS, +@@ -351,7 +352,7 @@ class VLLMProtocol: + # Use an environment mask instead of the engine's --device-ids option. + set_visible_devices: bool = False + +- # Default KV connector: "nixl", "lmcache", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. ++ # Default KV connector: "nixl", "lmcache", "lmcache-mp", "kvbm", "moriio", or a raw JSON string for --kv-transfer-config. + # Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. + # "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. + # dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). +@@ -1756,6 +1757,8 @@ class KVConnector: + kv_role: str | None = "kv_both" + module_path: str | None = None + discovery: bool = False ++ # Static ``kv_connector_extra_config``; a discovery row's topology-derived extras replace it. ++ extra_config: dict[str, Any] | None = None + + def transfer_config(self, mode: WorkerMode) -> dict[str, Any]: + """The ``--kv-transfer-config`` payload for a worker mode, before any topology-derived extras.""" +@@ -1763,6 +1766,8 @@ class KVConnector: + if self.module_path is not None: + payload["kv_connector_module_path"] = self.module_path + payload["kv_role"] = self.kv_role or ("kv_producer" if mode == "prefill" else "kv_consumer") ++ if self.extra_config: ++ payload["kv_connector_extra_config"] = dict(self.extra_config) + return payload + + +@@ -1770,6 +1775,12 @@ class KVConnector: + _CONNECTOR_MAP: dict[str, KVConnector] = { + "nixl": KVConnector("NixlConnector"), + "lmcache": KVConnector("LMCacheConnectorV1"), ++ # Out-of-process LMCache MP server (services[].type: lmcache-server) on the worker's own node. ++ "lmcache-mp": KVConnector( ++ "LMCacheMPConnector", ++ module_path="lmcache.integration.vllm.lmcache_mp_connector", ++ extra_config={"lmcache.mp.host": "tcp://localhost", "lmcache.mp.port": LMCACHE_SERVER_PORT}, ++ ), + "kvbm": KVConnector("DynamoConnector", module_path="kvbm.vllm_integration.connector"), + # AMD MoRI-IO (ROCm): prefill produces and decode consumes KV; workers register with the vLLM Router. + "moriio": KVConnector("MoRIIOConnector", kv_role=None, discovery=True), +diff --git a/src/srtctl/ports.py b/src/srtctl/ports.py +index c9ea7e0..3de5f4b 100644 +--- a/src/srtctl/ports.py ++++ b/src/srtctl/ports.py +@@ -46,6 +46,12 @@ MOONCAKE_HTTP_METADATA_PORT = 8701 + # the master lives entirely inside our consolidated 8700-range. + MOONCAKE_METRICS_PORT = 8702 + ++# LMCache multiprocess server (services[].type: lmcache-server), one per worker node, ++# reached by that node's vLLM ranks over localhost. LMCache's own defaults (5555, 8080) ++# sit inside ranges the NIXL side channel and the frontend can reach. ++LMCACHE_SERVER_PORT = 8750 ++LMCACHE_HTTP_PORT = 8751 ++ + # vLLM backend ports. + VLLM_NIXL_PORT_BASE = 5400 + # vLLM Router discovery endpoint (frontend.type: vllm-router with a discovery +diff --git a/src/srtctl/services/__init__.py b/src/srtctl/services/__init__.py +index 715a2f2..81e2207 100644 +--- a/src/srtctl/services/__init__.py ++++ b/src/srtctl/services/__init__.py +@@ -4,7 +4,7 @@ + """The top-level ``services:`` block: user-declared long-running processes launched next to the job.""" + + # Import kinds to trigger registration. +-from srtctl.services import exporters, generic, gms, infra, mooncake_master, mooncake_store, ray ++from srtctl.services import exporters, generic, gms, infra, lmcache_server, mooncake_master, mooncake_store, ray + from srtctl.services.config import ( + SERVICE_PLACEMENTS, + SERVICE_STARTS, +@@ -42,6 +42,7 @@ __all__ = [ + "gms", + "infra", + "list_service_types", ++ "lmcache_server", + "mooncake_master", + "mooncake_store", + "ray", +diff --git a/src/srtctl/services/lmcache_server.py b/src/srtctl/services/lmcache_server.py +new file mode 100644 +index 0000000..31a1de9 +--- /dev/null ++++ b/src/srtctl/services/lmcache_server.py +@@ -0,0 +1,57 @@ ++# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. ++# SPDX-License-Identifier: Apache-2.0 ++ ++"""``type: lmcache-server``: the LMCache multiprocess server, one per worker node. ++ ++The vLLM connector (``connector: lmcache-mp``) only talks to the server on its ++own node, so the kind runs an instance on every worker node, before workers, ++and gates on the server's HTTP healthcheck. ``args`` are appended to the fixed ++command, which sets only the ports srtctl owns. LMCache must be installed in ++the job container: the connector is imported inside the vLLM worker. ++""" ++ ++from __future__ import annotations ++ ++from typing import TYPE_CHECKING ++ ++from srtctl.ports import LMCACHE_HTTP_PORT, LMCACHE_SERVER_PORT ++from srtctl.services.config import HttpProbe, ServiceReadinessConfig ++from srtctl.services.registry import ServiceKind, ServiceLaunchContext, register_service ++ ++if TYPE_CHECKING: ++ from srtctl.services.config import ServiceConfig ++ ++ ++@register_service("lmcache-server") ++class LMCacheServerService(ServiceKind): ++ """LMCache MP server on each worker node; ``args`` are appended to the fixed command.""" ++ ++ builds_command = True ++ default_start = "before_workers" ++ default_critical = True ++ default_placement = "workers" ++ # L1 preallocation pins host memory; large pools outlast the base class's 120 s. ++ default_readiness_timeout = 300 ++ ++ def build_command(self, service: ServiceConfig, ctx: ServiceLaunchContext) -> list[str]: ++ if service.command is not None: ++ return list(service.effective_command) ++ return [ ++ "lmcache", ++ "server", ++ "--host", ++ "0.0.0.0", ++ "--port", ++ str(LMCACHE_SERVER_PORT), ++ "--http-host", ++ "0.0.0.0", ++ "--http-port", ++ str(LMCACHE_HTTP_PORT), ++ *service.args, ++ ] ++ ++ def readiness(self, service: ServiceConfig, ctx: ServiceLaunchContext) -> ServiceReadinessConfig | None: ++ return ServiceReadinessConfig( ++ http=HttpProbe(port=LMCACHE_HTTP_PORT, path="/healthcheck"), ++ timeout_seconds=self.default_readiness_timeout, ++ ) +diff --git a/tests/test_atom_atomesh.py b/tests/test_atom_atomesh.py +index 937800b..7f4d017 100644 +--- a/tests/test_atom_atomesh.py ++++ b/tests/test_atom_atomesh.py +@@ -200,6 +200,59 @@ def test_atom_pd_worker_emits_mooncake_kv_transfer_config(protocol: str | None, + } + + ++def test_atom_wraps_extra_kv_connectors_with_mooncake_in_multi() -> None: ++ """A role's extra-kv-connectors ride next to srtctl's Mooncake connector and never reach the CLI as a flag.""" ++ offload = {"kv_connector": "lmcache_offload", "kv_role": "offload", "lmcache.max_local_cpu_size": 180} ++ backend = AtomProtocol(atom_config=AtomServerConfig(prefill={"extra-kv-connectors": [offload], "max-model-len": 8})) ++ prefill = Process("node0", frozenset(range(8)), 7500, 6100, "prefill", 0, nixl_port=6301) ++ decode = Process("node1", frozenset(range(8)), 7500, 6100, "decode", 0, nixl_port=6302) ++ ++ prefill_command = _build(backend, prefill) ++ decode_command = _build(backend, decode) ++ ++ assert json.loads(prefill_command[prefill_command.index("--kv-transfer-config") + 1]) == { ++ "kv_connector": "multi", ++ "connectors": [ ++ {"kv_role": "kv_producer", "kv_connector": "mooncake", "proxy_ip": WORKER_IP, "handshake_port": 6301}, ++ offload, ++ ], ++ } ++ assert "--extra-kv-connectors" not in prefill_command ++ assert prefill_command[-2:] == ["--max-model-len", "8"] ++ assert json.loads(decode_command[decode_command.index("--kv-transfer-config") + 1])["kv_connector"] == "mooncake" ++ ++ ++def test_atom_aggregate_worker_runs_a_single_extra_connector_unwrapped() -> None: ++ offload = {"kv_connector": "lmcache_offload", "kv_role": "offload"} ++ backend = AtomProtocol(atom_config=AtomServerConfig(aggregated={"extra-kv-connectors": [offload]})) ++ process = Process("node0", frozenset(range(8)), 7500, 6100, "agg", 0) ++ ++ command = _build(backend, process) ++ ++ assert json.loads(command[command.index("--kv-transfer-config") + 1]) == offload ++ ++ ++@pytest.mark.parametrize( ++ ("extra_config", "expected"), ++ [ ++ ({}, {"lmcache.mp.port": 8750}), ++ ( ++ {"lmcache.mp.port": 5555, "lmcache.mp.tp_rank_collapse": True}, ++ {"lmcache.mp.port": 5555, "lmcache.mp.tp_rank_collapse": True}, ++ ), ++ ], ++) ++def test_atom_lmcache_mp_defaults_to_the_lmcache_server_port(extra_config: dict, expected: dict) -> None: ++ """lmcache_mp dials srtctl's lmcache-server on its own node unless the recipe gave an address.""" ++ connector = {"kv_connector": "lmcache_mp", "kv_role": "offload", "kv_connector_extra_config": extra_config} ++ backend = AtomProtocol(atom_config=AtomServerConfig(aggregated={"extra-kv-connectors": [connector]})) ++ process = Process("node0", frozenset(range(8)), 7500, 6100, "agg", 0) ++ ++ command = _build(backend, process) ++ ++ assert json.loads(command[command.index("--kv-transfer-config") + 1])["kv_connector_extra_config"] == expected ++ ++ + def test_atom_rejects_cross_node_model_parallel_endpoint() -> None: + """ATOM cannot coordinate one logical worker across two Slurm nodes.""" + leader = Process("node0", frozenset(range(4)), 7500, 6100, "prefill", 0, nixl_port=6301) +diff --git a/tests/test_services.py b/tests/test_services.py +index c7db10b..51a4b2c 100644 +--- a/tests/test_services.py ++++ b/tests/test_services.py +@@ -19,7 +19,14 @@ from srtctl.core.runtime import Nodes, RuntimeContext + from srtctl.core.schema import SrtConfig + from srtctl.core.topology import Endpoint + from srtctl.ports import MOONCAKE_HTTP_METADATA_PORT, MOONCAKE_MASTER_PORT +-from srtctl.services import ServiceConfig, ServiceSourceConfig, list_service_types ++from srtctl.services import ( ++ HttpProbe, ++ ServiceConfig, ++ ServiceLaunchContext, ++ ServiceSourceConfig, ++ get_service_kind, ++ list_service_types, ++) + from srtctl.services.implicit import discovery_env, effective_services, uses_discovery_plane + + SRUN = "srtctl.cli.mixins.service_stage.start_srun_process" +@@ -91,6 +98,7 @@ def test_registered_kinds() -> None: + "etcd", + "generic", + "gms", ++ "lmcache-server", + "mooncake-master", + "mooncake-store", + "nats", +@@ -166,6 +174,34 @@ services: + ) + + ++def test_lmcache_server_defaults_and_command() -> None: ++ kind = get_service_kind("lmcache-server") ++ service = ServiceConfig(name="lmcache", type="lmcache-server", args=["--l1-size-gb", "180"]) ++ ctx = ServiceLaunchContext.preview() ++ ++ assert (service.effective_start, kind.default_critical, kind.default_placement) == ( ++ "before_workers", ++ True, ++ "workers", ++ ) ++ assert kind.build_command(service, ctx) == [ ++ "lmcache", ++ "server", ++ "--host", ++ "0.0.0.0", ++ "--port", ++ "8750", ++ "--http-host", ++ "0.0.0.0", ++ "--http-port", ++ "8751", ++ "--l1-size-gb", ++ "180", ++ ] ++ probe = kind.readiness(service, ctx) ++ assert probe is not None and probe.http == HttpProbe(port=8751, path="/healthcheck") ++ ++ + def test_mooncake_store_defaults_and_requires_master() -> None: + with pytest.raises(ValidationError, match="requires backend.mooncake_kv_store"): + _load("services:\n - name: store\n type: mooncake-store\n placement:\n node: workers\n") +diff --git a/tests/test_sglang_lmcache.py b/tests/test_sglang_lmcache.py +new file mode 100644 +index 0000000..52f3f8d +--- /dev/null ++++ b/tests/test_sglang_lmcache.py +@@ -0,0 +1,39 @@ ++# SPDX-FileCopyrightText: Copyright (c) 2026 SemiAnalysis LLC. All rights reserved. ++# SPDX-License-Identifier: Apache-2.0 ++ ++"""SGLang workers reach the node-local LMCache MP server (services[].type: lmcache-server).""" ++ ++import pytest ++ ++from srtctl.backends.sglang import SGLangProtocol, SGLangServerConfig ++from srtctl.core.topology import Process ++ ++ ++def _process(mode: str = "prefill") -> Process: ++ return Process("node0", frozenset(range(8)), 7500, 6100, mode, 0) ++ ++ ++def test_enable_lmcache_points_the_worker_at_the_node_local_server() -> None: ++ backend = SGLangProtocol(sglang_config=SGLangServerConfig(prefill={"enable-lmcache": True})) ++ ++ assert backend.get_process_environment(_process("prefill")) == { ++ "LMCACHE_MP_HOST": "127.0.0.1", ++ "LMCACHE_MP_PORT": "8750", ++ } ++ assert backend.get_process_environment(_process("decode")) == {} ++ ++ ++@pytest.mark.parametrize( ++ ("prefill", "environment"), ++ [ ++ ({"enable_lmcache": True, "lmcache_config_file": "/configs/lmcache.yaml"}, {}), ++ ({"enable-lmcache": True}, {"LMCACHE_MP_HOST": "10.0.0.5"}), ++ ], ++) ++def test_recipe_owned_lmcache_address_is_left_alone(prefill: dict, environment: dict) -> None: ++ backend = SGLangProtocol( ++ sglang_config=SGLangServerConfig(prefill=prefill), ++ prefill_environment=environment, ++ ) ++ ++ assert backend.get_process_environment(_process()) == {} +diff --git a/tests/test_vllm_connectors.py b/tests/test_vllm_connectors.py +index 691990b..4bb0974 100644 +--- a/tests/test_vllm_connectors.py ++++ b/tests/test_vllm_connectors.py +@@ -23,6 +23,14 @@ def test_table_presets_serialize_exactly_as_before(): + assert VLLMProtocol(connector="LMCache").kv_transfer_config("decode") == json.dumps( + {"kv_connector": "LMCacheConnectorV1", "kv_role": "kv_both"} + ) ++ assert VLLMProtocol(connector="lmcache-mp").kv_transfer_config("prefill") == json.dumps( ++ { ++ "kv_connector": "LMCacheMPConnector", ++ "kv_connector_module_path": "lmcache.integration.vllm.lmcache_mp_connector", ++ "kv_role": "kv_both", ++ "kv_connector_extra_config": {"lmcache.mp.host": "tcp://localhost", "lmcache.mp.port": 8750}, ++ } ++ ) + assert VLLMProtocol(connector="kvbm").kv_transfer_config("decode") == json.dumps( + { + "kv_connector": "DynamoConnector", diff --git a/inferencex-e2e/runners/srt-slurm/patches/README.md b/inferencex-e2e/runners/srt-slurm/patches/README.md index cb619d0506..1f675b9d0f 100644 --- a/inferencex-e2e/runners/srt-slurm/patches/README.md +++ b/inferencex-e2e/runners/srt-slurm/patches/README.md @@ -7,3 +7,4 @@ Each patch is a temporary fix for an open upstream PR. When the PR merges and th | Patch | Upstream PR | Fix | |-------|-------------|-----| | `504-post-eval-srun-options.patch` | [NVIDIA/srt-slurm#504](https://github.com/NVIDIA/srt-slurm/pull/504) | Forward recipe `srun_options` (e.g. `container-writable`) to post-eval steps | +| `507-lmcache-server-atom-sglang.patch` | [SemiAnalysisAI/srt-slurm#32](https://github.com/SemiAnalysisAI/srt-slurm/pull/32) (includes [NVIDIA/srt-slurm#507](https://github.com/NVIDIA/srt-slurm/pull/507)) | LMCache for vLLM, SGLang and ATOM: the `lmcache-server` service, and ATOM `extra-kv-connectors` wrapped with Mooncake in `multi` | From 44e2695bc3655e0b7baee9d852c9e0e2fa06279d Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 11:57:10 -0500 Subject: [PATCH 2/6] chore(changelog): link #3543 --- inferencex-e2e/perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index cf79221271..1e82f652df 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9003,4 +9003,4 @@ description: - "Port the DeepSeek-V4-Pro-0813 MI355X ATOM 1P1D AgentX config from the legacy amd_utils path to a native srt-slurm recipe (benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml), one override variant per point. Server flags, env, AToMesh routing policies and decode CUDA-graph capture sizes match the legacy server_atom.sh for every tier: TP8 at concurrency 1 and 16, DP attention with prefill TBO at 64 and 128, and DP attention with ATOM's in-process LMCache CPU offload (lmcache_offload, 187 GB per prefill rank) at 256. Golden acceptance 3.01 for DSpark with three draft tokens is now injected by the srt-slurm path, the same value the legacy models_atom.yaml hardcoded." - "srt-slurm had no way to add LMCache next to its generated Mooncake connector, so runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch applies SemiAnalysisAI/srt-slurm#32 (which includes NVIDIA/srt-slurm#507) to the pinned submodule until it lands upstream." - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3543 From 812d5f9614e38575bb01ceee7c5aeb76fbbbbf29 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 12:01:54 -0500 Subject: [PATCH 3/6] chore(amd): remove the legacy ATOM disagg path dsv4-fp4-mi355x-atom-disagg-agentic-lmcache-dspark was the only config on it. Delete its launch script, server_atom.sh, env_atom.sh and models_atom.yaml, and the atom-disagg branches in job.slurm and server.sh. TileRT keeps the rest of amd_utils. --- .../agentic/dsv4_fp4_mi355x_atom-disagg.sh | 130 ---- .../multi_node/amd_utils/env_atom.sh | 40 - .../benchmarks/multi_node/amd_utils/job.slurm | 104 +-- .../multi_node/amd_utils/models_atom.yaml | 83 --- .../benchmarks/multi_node/amd_utils/server.sh | 6 +- .../multi_node/amd_utils/server_atom.sh | 705 ------------------ 6 files changed, 3 insertions(+), 1065 deletions(-) delete mode 100755 inferencex-e2e/benchmarks/multi_node/agentic/dsv4_fp4_mi355x_atom-disagg.sh delete mode 100755 inferencex-e2e/benchmarks/multi_node/amd_utils/env_atom.sh delete mode 100644 inferencex-e2e/benchmarks/multi_node/amd_utils/models_atom.yaml delete mode 100755 inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh diff --git a/inferencex-e2e/benchmarks/multi_node/agentic/dsv4_fp4_mi355x_atom-disagg.sh b/inferencex-e2e/benchmarks/multi_node/agentic/dsv4_fp4_mi355x_atom-disagg.sh deleted file mode 100755 index 931434e1a7..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/agentic/dsv4_fp4_mi355x_atom-disagg.sh +++ /dev/null @@ -1,130 +0,0 @@ -#!/usr/bin/env bash - -# Agentic trace-replay recipe for a disaggregated ATOM server on MI355X -# (DeepSeek-V4-Pro FP4, 1P1D TP8), mooncake RDMA KV transfer + atomesh router. -# -# CI-style sibling of the former SGLang dsv4_fp4_mi355x_sglang-disagg.sh (same -# agentic trace workload, same submit.sh path), but drives the ATOM engine. -# Modeled on ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md: three concurrency -# tiers selected by the search space -- TP (conc 1-32), DP-attention -# (conc 64-128, no offload), and DP-attention + CPU KV offload (conc 256, -# lmcache_offload multi connector). Per-tier server behavior lives in -# amd_utils/server_atom.sh (IS_AGENTIC branch) and models_atom.yaml -# (DeepSeek-V4-Pro-AgentX). - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -source "$SCRIPT_DIR/../../benchmark_lib.sh" - -check_env_vars \ - CONC_LIST \ - ISL \ - OSL \ - IMAGE \ - SPEC_DECODING \ - MODEL_PATH \ - PREFILL_NUM_WORKERS \ - PREFILL_TP \ - PREFILL_EP \ - PREFILL_DP_ATTN \ - DECODE_NUM_WORKERS \ - DECODE_TP \ - DECODE_EP \ - DECODE_DP_ATTN \ - PREFILL_NODES \ - DECODE_NODES \ - RANDOM_RANGE_RATIO \ - DURATION \ - KV_OFFLOADING \ - IS_AGENTIC \ - FRAMEWORK - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -set -x - -# Use upstreamed multi_node scripts (no external clone needed) -cd "$GITHUB_WORKSPACE/benchmarks/multi_node/amd_utils" || exit 1 - -# Set up ATOM launch script-specific environment variables -export TIME_LIMIT="${TIME_LIMIT:-08:00:00}" -export MODEL_PATH=$MODEL_PATH -export MODEL_NAME=$MODEL_NAME -export CONTAINER_IMAGE=$IMAGE - -# ── Identity / result naming ── -export MODEL_PREFIX="${MODEL_PREFIX:-dsv4}" -export PRECISION="${PRECISION:-fp4}" -export RESULT_FILENAME="${RESULT_FILENAME:-${RUNNER_NAME:-dsv4-fp4-agentic}}" - -# ── Agentic benchmark params ── -export DURATION="${DURATION:-1800}" -# DSV4-Pro max model len for agentic traces (matches single-node recipe). -export MAX_MODEL_LEN="${MAX_MODEL_LEN:-1000000}" - -# ── KV cache offloading (ATOM lmcache_offload CPU tier) ── -# KV_OFFLOADING=none | dram (passed from YAML; none for the TP/DP tiers, dram -# for the conc-256 offload tier). KV_OFFLOAD_BACKEND selects the backend when -# offloading is on; the ATOM PD path only implements the lmcache_offload CPU -# tier, so "lmcache" is the only supported value. The multi-connector JSON and -# per-rank sizing are built in server_atom.sh from TOTAL_CPU_DRAM_GB (aggregate -# budget from the matrix, dram-utilization 0.80). -export KV_OFFLOADING="${KV_OFFLOADING:-none}" -if [[ "$KV_OFFLOADING" != "none" ]]; then - export KV_OFFLOAD_BACKEND="${KV_OFFLOAD_BACKEND:-lmcache}" - # Recipe (DeepSeek-V4-Agentic-PD-Max.md, "The offload settings that matter"): - # both default to values this workload cannot live with. Prefill node only. - export OFFLOAD_SLOT_STAGING_SLOTS="${OFFLOAD_SLOT_STAGING_SLOTS:-4}" - export OFFLOAD_COPY_WORKERS="${OFFLOAD_COPY_WORKERS:-1}" - export OFFLOAD_MIN_LOAD_TOKENS="${OFFLOAD_MIN_LOAD_TOKENS:-8192}" -fi - -# ── MTP ── -# EAGLE/MTP synthetic acceptance length on agentic throughput runs (real target -# verification is used only on eval-only runs). 2.49 per the PD-Max recipe. -export DECODE_MTP_SIZE="${DECODE_MTP_SIZE:-0}" -export SPEC_DECODE_AL="${SPEC_DECODE_AL:-2.49}" - -# Derive EP/DP enable flags from the topology inputs. -if [[ "${PREFILL_EP:-1}" -eq 1 ]]; then -export PREFILL_ENABLE_EP=false -else -export PREFILL_ENABLE_EP=true -fi - -if [[ "$PREFILL_DP_ATTN" == "true" ]]; then -export PREFILL_ENABLE_DP=true -else -export PREFILL_ENABLE_DP=false -fi - -if [[ "${DECODE_EP:-1}" -eq 1 ]]; then -export DECODE_ENABLE_EP=false -else -export DECODE_ENABLE_EP=true -fi - -if [[ "$DECODE_DP_ATTN" == "true" ]]; then -export DECODE_ENABLE_DP=true -else -export DECODE_ENABLE_DP=false -fi - -# Launch the job. CONC_LIST is space-delimited in YAML; submit.sh wants 'x'. -JOB_ID=$(bash ./submit.sh $PREFILL_NODES \ - $PREFILL_NUM_WORKERS \ - $DECODE_NODES \ - $DECODE_NUM_WORKERS \ - $ISL $OSL "${CONC_LIST// /x}" inf \ - ${PREFILL_ENABLE_EP} ${PREFILL_ENABLE_DP} \ - ${DECODE_ENABLE_EP} ${DECODE_ENABLE_DP} \ - ${PREFILL_TP} ${DECODE_TP} \ - ${RANDOM_RANGE_RATIO}) - -if [[ $? -ne 0 ]]; then - echo "Failed to submit job" >&2 - exit 1 -fi - -echo "$JOB_ID" diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/env_atom.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/env_atom.sh deleted file mode 100755 index 71cbdf06ff..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/env_atom.sh +++ /dev/null @@ -1,40 +0,0 @@ -#!/bin/bash -# ATOM/mooncake environment, sourced by server_atom.sh in place of env.sh. -# IBDEVICES: RDMA device names (e.g. ionic_0,ionic_1,...), set by the runner or -# auto-detected. - -set -x - -export PYTHONUNBUFFERED=1 -export PYTHONDONTWRITEBYTECODE=1 - - -if [[ -z "$IBDEVICES" ]]; then - DETECTED=$(ibv_devinfo 2>/dev/null | grep "hca_id:" | awk '{print $2}' | paste -sd',') - if [[ -n "$DETECTED" ]]; then - export IBDEVICES="$DETECTED" - echo "[INFO] Auto-detected IBDEVICES=$IBDEVICES via ibv_devinfo on $(hostname -s)" - else - # ATOM passes no IB device to the server (mooncake picks its own RDMA device via - # proxy_ip/handshake_port), so a missing IBDEVICES is non-fatal here. - echo "[WARN] Unable to detect RDMA devices via ibv_devinfo; IBDEVICES unset (non-fatal for ATOM/mooncake)" >&2 - fi -else - echo "[INFO] Using IBDEVICES=$IBDEVICES (set by runner or environment)" -fi -export IBDEVICES - - -export LD_LIBRARY_PATH=/opt/venv/lib/python3.10/site-packages/mooncake:/opt/rocm/lib:${LD_LIBRARY_PATH:-} - -export SAFETENSORS_FAST_GPU=1 - -export VLLM_LOG_LEVEL=WARNING -export ATOM_LOG_LEVEL=WARNING -export AITER_LOG_LEVEL=WARNING -export LOG_LEVEL=WARNING -export LOGLEVEL=WARNING - -set +x - -echo "[INFO] ATOM env: IBDEVICES=$IBDEVICES LD_LIBRARY_PATH includes mooncake" \ No newline at end of file diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm b/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm index cecc989a0f..119d97bd65 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm @@ -34,9 +34,7 @@ echo "" # Use $(pwd) not BASH_SOURCE — sbatch copies the script to /var/spool/slurmd/ # at runtime, but the CWD remains the submit-time directory (amd_utils/). -if [[ "$ENGINE" == "atom-disagg" ]]; then - MODELS_YAML="$(pwd)/models_atom.yaml" -elif [[ "$ENGINE" == "tilert" ]]; then +if [[ "$ENGINE" == "tilert" ]]; then MODELS_YAML="$(pwd)/models_tilert.yaml" else MODELS_YAML="$(pwd)/models.yaml" @@ -545,26 +543,7 @@ DOCKER_ENV_COMMON=( ) # Engine-specific env vars -if [[ "$ENGINE" == "atom-disagg" ]]; then - check_env_vars \ - PREFILL_PORT DECODE_PORT HANDSHAKE_PORT MEM_FRAC_STATIC KV_CACHE_DTYPE \ - BLOCK_SIZE MAX_NUM_SEQS - DOCKER_ENV_ENGINE=( - -e ATOM_WS_PATH=${WS_PATH} - -e PREFILL_PORT=${PREFILL_PORT} - -e DECODE_PORT=${DECODE_PORT} - -e ROUTER_PORT=${ROUTER_PORT} - -e HANDSHAKE_PORT=${HANDSHAKE_PORT} - -e MEM_FRAC_STATIC=${MEM_FRAC_STATIC} - -e KV_CACHE_DTYPE=${KV_CACHE_DTYPE} - -e BLOCK_SIZE=${BLOCK_SIZE} - -e MAX_NUM_SEQS=${MAX_NUM_SEQS} - -e MAX_MODEL_LEN=${MAX_MODEL_LEN:-} - -e MAX_NUM_BATCHED_TOKENS=${MAX_NUM_BATCHED_TOKENS:-} - -e EXTRA_SERVER_ARGS=\${EXTRA_SERVER_ARGS:-} - -e IBDEVICES=${IBDEVICES:-} - ) -elif [[ "$ENGINE" == "tilert" ]]; then +if [[ "$ENGINE" == "tilert" ]]; then DOCKER_ENV_ENGINE=( -e MODEL_PATH=$DOCKER_MODEL_PATH -e PREFILL_IMAGE=${PREFILL_IMAGE} @@ -673,84 +652,6 @@ echo \"Rank \$SLURM_PROCID on \$(hostname)\" eval \"\$DOCKER_CMD_DETECT\" echo \"[docker-detect] rank \$SLURM_PROCID: DOCKER_CMD=\$DOCKER_CMD\" -# Enable out-of-tree RDMA library mounts for atom-disagg (mooncake requires host RDMA stack) -RDMA_MOUNTS=() -if [[ "$ENGINE" == "atom-disagg" ]]; then - -# When the container base OS differs from the host (e.g. Ubuntu 24.04 image -# on a 22.04 host), the container's bundled libibverbs/libionic may be -# ABI-incompatible with the host kernel drivers. Detect the NIC type and -# bind-mount the host's out-of-tree RDMA userspace libraries into the -# container so the RDMA stack always matches the running kernel. -_detect_nic_type() { - if [[ -n \"\${MORI_NIC_TYPE:-}\" ]]; then echo \"\$MORI_NIC_TYPE\"; return; fi - local bnxt=0 mlx5=0 ionic=0 - if [[ -d /sys/class/infiniband ]]; then - for dev in /sys/class/infiniband/*; do - local name; name=\$(basename \"\$dev\") - case \"\$name\" in - bnxt_re*) ((bnxt++)) ;; mlx5*) ((mlx5++)) ;; ionic*) ((ionic++)) ;; - *) - local drv; drv=\$(basename \"\$(readlink -f \"\$dev/device/driver\" 2>/dev/null)\" 2>/dev/null || true) - case \"\$drv\" in bnxt*) ((bnxt++)) ;; mlx5*) ((mlx5++)) ;; ionic*) ((ionic++)) ;; esac ;; - esac - done - fi - if (( bnxt >= mlx5 && bnxt >= ionic && bnxt > 0 )); then echo bnxt - elif (( ionic >= mlx5 && ionic > 0 )); then echo ionic - else echo mlx5; fi -} - -_find_host_ibverbs() { - for c in /usr/lib64/libibverbs.so.1 /lib/x86_64-linux-gnu/libibverbs.so.1 /usr/lib/x86_64-linux-gnu/libibverbs.so.1.14.39.0 /usr/lib/x86_64-linux-gnu/libibverbs.so.1; do - local r; r=\$(readlink -f \"\$c\" 2>/dev/null || true) - [[ \"\$r\" == *libibverbs.so.1.14.57.0 ]] && continue - if [[ -f \"\$r\" ]]; then echo \"\$r\"; return; fi - done -} - -_NIC_TYPE=\$(_detect_nic_type) -echo \"[rdma] NIC type: \${_NIC_TYPE} on \$(hostname)\" - -if [[ \"\$_NIC_TYPE\" == \"ionic\" || \"\$_NIC_TYPE\" == \"bnxt\" ]]; then - _host_ibv=\$(_find_host_ibverbs) - if [[ -n \"\$_host_ibv\" ]]; then - RDMA_MOUNTS+=(-v \"\$_host_ibv:/lib/x86_64-linux-gnu/libibverbs.so.1\") - fi -fi - -if [[ \"\$_NIC_TYPE\" == \"ionic\" ]]; then - for _dir in /usr/local/lib /usr/lib/x86_64-linux-gnu; do - for _lib in \"\$_dir\"/libionic*.so; do - [[ -f \"\$_lib\" ]] || continue - _real=\$(readlink -f \"\$_lib\") - [[ -f \"\$_real\" ]] && RDMA_MOUNTS+=(-v \"\$_real:\$_real\") - RDMA_MOUNTS+=(-v \"\$_lib:/usr/lib/x86_64-linux-gnu/\$(basename \"\$_lib\")\") - done - done - if [[ -d /usr/lib/x86_64-linux-gnu/libibverbs ]]; then - for _lib in /usr/lib/x86_64-linux-gnu/libibverbs/libionic-rdmav*.so; do - [[ -f \"\$_lib\" ]] && RDMA_MOUNTS+=(-v \"\$_lib:\$_lib\") - done - fi - [[ -d /etc/libibverbs.d ]] && RDMA_MOUNTS+=(-v /etc/libibverbs.d:/etc/libibverbs.d:ro) -elif [[ \"\$_NIC_TYPE\" == \"bnxt\" ]]; then - for _lib in /usr/local/lib/libbnxt_re-rdmav*.so; do - [[ -f \"\$_lib\" ]] && RDMA_MOUNTS+=(-v \"\$_lib:/usr/lib/x86_64-linux-gnu/libibverbs/\$(basename \"\$_lib\")\") - done - for _lib in /usr/local/lib/libbnxt_re.so; do - [[ -f \"\$_lib\" ]] && RDMA_MOUNTS+=(-v \"\$_lib:/usr/lib/x86_64-linux-gnu/\$(basename \"\$_lib\")\") - done - [[ -d /etc/libibverbs.d ]] && RDMA_MOUNTS+=(-v /etc/libibverbs.d:/etc/libibverbs.d:ro) -fi - -if [[ \${#RDMA_MOUNTS[@]} -gt 0 ]]; then - echo \"[rdma] bind-mounts: \${RDMA_MOUNTS[*]}\" -else - echo \"[rdma] no out-of-tree RDMA mounts needed\" -fi -fi # end: if ENGINE == atom-disagg - RANK_IMAGE= if [[ \"$ENGINE\" == \"tilert\" && \"\$SLURM_PROCID\" -lt \"$xP\" ]]; then RANK_IMAGE=\"$PREFILL_IMAGE\" @@ -792,7 +693,6 @@ exec \$DOCKER_CMD run \ -v ${HICACHE_MC_CONFIG}:/config/hicache_mc.env:ro \ ${EXTRA_DOCKER_MOUNTS:-} \ ${CLIENT_DOCKER_MOUNTS} \ - \${RDMA_MOUNTS[@]+"\${RDMA_MOUNTS[@]}"} \ ${DOCKER_ENV_COMMON[*]} \ ${DOCKER_ENV_ENGINE[*]} \ ${CLIENT_DOCKER_ENV} \ diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/models_atom.yaml b/inferencex-e2e/benchmarks/multi_node/amd_utils/models_atom.yaml deleted file mode 100644 index 008edc2116..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/models_atom.yaml +++ /dev/null @@ -1,83 +0,0 @@ -# Model-specific ATOM server configurations for disaggregated inference. -# -# Each top-level key is a MODEL_NAME value (must match the directory name under MODEL_DIR). -# -# To add a new model: add a new top-level entry following the same schema. -# No script changes are required. -# -# Schema: -# : -# env: str # Space-separated KEY=VALUE pairs exported unconditionally -# tp_dp_flags: str # Shared TP+DPA flags (fallback when prefill/decode-specific keys are absent) -# prefill_tp_dp_flags: str # TP+DPA flags for prefill only (overrides tp_dp_flags) -# decode_tp_dp_flags: str # TP+DPA flags for decode only (overrides tp_dp_flags) -# tp_dp_env: str # Space-separated KEY=VALUE pairs exported only in TP+DPA mode -# ep_dp_flags: str # Shared EP+DPA flags (fallback when prefill/decode-specific keys are absent) -# prefill_ep_dp_flags: str # EP+DPA flags for prefill only (overrides ep_dp_flags) -# decode_ep_dp_flags: str # EP+DPA flags for decode only (overrides ep_dp_flags) -# ep_dp_env: str # Space-separated KEY=VALUE pairs exported only in EP+DPA mode -# mtp_flags: str # Flags passed to SPEC_ARGS before $DECODE_MTP_SIZE (e.g. "--method mtp --num-speculative-tokens") -# kv_cache_flags: str # Full --kv_cache_dtype flag string (e.g. "--kv_cache_dtype fp8", or "" for none) -# online_quant_config: str # JSON string passed to --online_quant_config (used when DPA is disabled) -# online_quant_dpa_config: str # JSON string passed to --online_quant_config when DPA is enabled (falls back to online_quant_config) -# block_size: str # --block-size value (overrides server_atom.sh default of 16) -# mem_frac_static: str # --gpu-memory-utilization value (overrides default of 0.85) -# max_model_len: str # --max-model-len value (overrides default of unset) -# max_num_seqs: str # --max-num-seqs value (overrides default of 256) -# max_num_batched_tokens: str # --max-num-batched-tokens value (overrides default of unset) -# scheduler_delay_factor: str # --scheduler-delay-factor value (overrides default of unset) -# Agentic-only (applied by server_atom.sh only when IS_AGENTIC=1): -# attn_prefill_chunk_size: str # --attn-prefill-chunk-size value (unset = omit) -# state_checkpoint_interval_tokens: str # --state-checkpoint-interval-tokens value (unset = omit) -# level: str # --level value (unset = omit) -# spec_decode_acceptance_length: str # --spec-decode-acceptance-length (synthetic AL on throughput runs; unset = omit) - -# Agentic (AgentX trace-replay) variant of DeepSeek-V4-Pro on ATOM PD. Resolved -# by job.slurm/server_atom.sh as '-AgentX' when IS_AGENTIC=1. Differs -# from the throughput entry above per ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md: -# prefix caching (server_atom.sh IS_AGENTIC branch), FP8 KV/index cache, TBO on -# prefill only, agentic prefill-chunk / state-checkpoint / level knobs, and MTP -# with synthetic acceptance on throughput runs. -DeepSeek-V4-Pro-AgentX: &DeepSeek-V4-Pro-AgentX - # Rail-isolated RDMA: matched rails replace dest-device affinity. Omit - # ATOM_MOONCAKE_MATCHED_RAILS if every default P/D HCA pair is mutually - # reachable on the fabric. - # ATOM_NUMA_BIND and the ATOM_DP_* ports are unconditional here: the recipe - # exports them in every tier, including TP (concurrency 1-32) where DP - # attention is off, so they cannot live in tp_dp_env/ep_dp_env (DP-only). - env: "ATOM_MOE_GU_ITLV=1 AITER_BF16_FP8_MOE_BOUND=0 MC_GID_INDEX=1 ATOM_MOONCAKE_MATCHED_RAILS=auto NCCL_IB_DISABLE=1 ATOM_DISABLE_MMAP=true ATOM_NUMA_BIND=1 ATOM_DP_MASTER_PORT=29510 ATOM_DP_BASE_PORT=29610 ATOM_PREFIX_CACHE_POLICY=lru ATOM_PREFIX_CACHE_PROTECTED_RATIO=0.5" - kv_cache_flags: "--kv_cache_dtype fp8 --index-cache-dtype fp4" - # DP-attention tiers (conc 64+): TBO is prefill-only per the recipe. - tp_dp_flags: "--enable-dp-attention" - prefill_tp_dp_flags: "--enable-dp-attention --enable-tbo" - decode_tp_dp_flags: "--enable-dp-attention" - ep_dp_flags: "--enable-expert-parallel --enable-dp-attention" - prefill_ep_dp_flags: "--enable-expert-parallel --enable-dp-attention --enable-tbo" - decode_ep_dp_flags: "--enable-expert-parallel --enable-dp-attention" - # DP-only env for the agentic DP-attention tiers (prefill and decode). Under - # --dp-aware the cache-aware router sends explicit P/D rank hints that take - # priority over engine-local session affinity and load balancing, so - # ATOM_DP_SESSION_AFFINITY and ATOM_DP_LB_REQ_EQUIV are not needed on this - # routing path (recipe). Cross-DP prefill coalescing (ATOM_ENABLE_PREFILL_DELAYER=0) - # is off on both roles per the recipe. NUMA binding and the DP ports moved to - # the unconditional env above because the recipe sets them in the TP tier too. - tp_dp_env: "ATOM_ENABLE_PREFILL_DELAYER=0" - ep_dp_env: "ATOM_ENABLE_PREFILL_DELAYER=0" - # Prefill-only DP env: the recipe sets GPU_MAX_HW_QUEUES=5 on the DP-attention - # prefill node and leaves decode unset. server_atom.sh applies this on prefill - # nodes only. - prefill_dp_env: "GPU_MAX_HW_QUEUES=5" - mtp_flags: "--method dspark --num-speculative-tokens" - # config.py forces block-size 256 for V4 (lcm(4,128)); 16 is ignored/misleading. - block_size: "256" - mem_frac_static: "0.9" - max_num_batched_tokens: "16384" - # Agentic-only server knobs consumed by server_atom.sh IS_AGENTIC branch. - attn_prefill_chunk_size: "16384" - state_checkpoint_interval_tokens: "8192" - level: "3" - spec_decode_acceptance_length: "3.01" - -# The PD-Max recipe runs DSpark, which needs the draft head bundled in the -# -0813 config.json. Same alias split as models.yaml on the SGLang side. -DeepSeek-V4-Pro-0813-AgentX: *DeepSeek-V4-Pro-AgentX diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh index f304f53b40..a3830b57aa 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh @@ -4,7 +4,6 @@ source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only # Multi-Engine Disaggregated Server Dispatcher # Dispatches to the engine-specific server launcher based on ENGINE env var. # ENGINE=sglang-disagg (default) -> server_sglang.sh (SGLang + MoRI) -# ENGINE=atom-disagg -> server_atom.sh (ATOM + mooncake) # ENGINE=tilert -> server_tilert.sh (vLLM prefill + TileRT decode) check_env_vars ENGINE WS_PATH @@ -17,10 +16,7 @@ export WS_PATH ENGINE echo "[DISPATCHER] ENGINE=$ENGINE WS_PATH=$WS_PATH" -if [[ "$ENGINE" == "atom-disagg" ]]; then - export ATOM_WS_PATH="$WS_PATH" - source "$WS_PATH/server_atom.sh" -elif [[ "$ENGINE" == "tilert" ]]; then +if [[ "$ENGINE" == "tilert" ]]; then source "$WS_PATH/server_tilert.sh" else source "$WS_PATH/server_sglang.sh" diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh deleted file mode 100755 index caa187b6bc..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh +++ /dev/null @@ -1,705 +0,0 @@ -#!/bin/bash -# ATOM disaggregated launcher: mooncake RDMA KV transfer and atomesh routing. - -source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only -check_env_vars \ - MODEL_NAME ROUTER_PORT PREFILL_PORT DECODE_PORT HANDSHAKE_PORT \ - MEM_FRAC_STATIC BLOCK_SIZE MAX_NUM_SEQS WAIT_SERVER_TIMEOUT - -check_env_vars \ - NODE0_ADDR NODE_RANK xP yD IPADDRS \ - PREFILL_TP_SIZE DECODE_TP_SIZE PREFILL_ENABLE_EP PREFILL_ENABLE_DP DECODE_ENABLE_EP \ - DECODE_ENABLE_DP DECODE_MTP_SIZE BENCH_INPUT_LEN BENCH_OUTPUT_LEN BENCH_RANDOM_RANGE_RATIO \ - BENCH_REQUEST_RATE BENCH_NUM_PROMPTS_MULTIPLIER BENCH_MAX_CONCURRENCY DRY_RUN GPUS_PER_NODE \ - RUN_EVAL EVAL_ONLY EVAL_FRAMEWORK BENCHMARK_LOGS_DIR MODEL_DIR \ - ATOM_WS_PATH - -EXTRA_SERVER_ARGS="${EXTRA_SERVER_ARGS:-}" - -source $ATOM_WS_PATH/setup_deps.sh -source $ATOM_WS_PATH/env_atom.sh - -# lm-eval with high num_concurrent exhausts the default 1024 FD limit. -ulimit -n 65536 2>/dev/null || ulimit -n 8192 2>/dev/null || true -echo "ulimit -n (open files): $(ulimit -n)" - -host_ip=$(ip route get 1.1.1.1 2>/dev/null | awk '/src/ {print $7}') -if [[ -z "$host_ip" ]]; then - host_ip=$(hostname -I 2>/dev/null | awk '{print $1}') -fi -host_name=$(hostname) - -# ATOM/mooncake handshake IP: the recipe exports this per node (prefill/decode -# IP). Default to this node's resolved IP so it matches the mooncake proxy_ip. -export ATOM_HOST_IP="${ATOM_HOST_IP:-$host_ip}" - -set -x -_yaml_tmp=$(mktemp) -python3 << PYEOF > "$_yaml_tmp" -import yaml -# Resolve the recipe entry the same way server_sglang.sh does: agentic runs -# (IS_AGENTIC) use the '-AgentX' entry, non-agentic runs use the bare -# ''. job.slurm passes MODEL_NAME unchanged (base name), so the -AgentX -# derivation has to happen here. Fall back to the base entry when absent. -with open('${ATOM_WS_PATH}/models_atom.yaml') as f: - _all = yaml.safe_load(f) or {} -_name = '${MODEL_NAME}' -_agentic = '${IS_AGENTIC:-0}'.strip().lower() in ('1', 'true') -_key = f'{_name}-AgentX' if _agentic else _name -m = _all.get(_key, _all.get(_name, {})) -import sys -print(f"Selected models_atom.yaml entry: {_key if _key in _all else _name} (IS_AGENTIC={_agentic})", file=sys.stderr) -def sh(v): return v.replace("'", "'\\''") -print(f"MODEL_ENVS='{sh(m.get('env', ''))}'") -_tp_dp = m.get('tp_dp_flags', '') -print(f"PREFILL_MODEL_TP_DP_FLAGS='{sh(m.get('prefill_tp_dp_flags', _tp_dp))}'") -print(f"DECODE_MODEL_TP_DP_FLAGS='{sh(m.get('decode_tp_dp_flags', _tp_dp))}'") -_ep_dp = m.get('ep_dp_flags', '') -print(f"PREFILL_MODEL_EP_DP_FLAGS='{sh(m.get('prefill_ep_dp_flags', _ep_dp))}'") -print(f"DECODE_MODEL_EP_DP_FLAGS='{sh(m.get('decode_ep_dp_flags', _ep_dp))}'") -print(f"MODEL_TP_DP_ENV='{sh(m.get('tp_dp_env', ''))}'") -print(f"MODEL_EP_DP_ENV='{sh(m.get('ep_dp_env', ''))}'") -print(f"MODEL_PREFILL_DP_ENV='{sh(m.get('prefill_dp_env', ''))}'") -print(f"MODEL_MTP_FLAGS='{sh(m.get('mtp_flags', ''))}'") -print(f"MODEL_KV_ARG='{sh(m.get('kv_cache_flags', ''))}'") -print(f"_ONLINE_QUANT_CONFIG='{sh(m.get('online_quant_config', ''))}'") -print(f"_ONLINE_QUANT_DPA_CONFIG='{sh(m.get('online_quant_dpa_config', m.get('online_quant_config', '')))}'") -print(f"_YAML_BLOCK_SIZE='{sh(m.get('block_size', ''))}'") -print(f"_YAML_MEM_FRAC_STATIC='{sh(m.get('mem_frac_static', ''))}'") -print(f"_YAML_MAX_MODEL_LEN='{sh(m.get('max_model_len', ''))}'") -print(f"_YAML_MAX_NUM_SEQS='{sh(m.get('max_num_seqs', ''))}'") -print(f"_YAML_MAX_NUM_BATCHED_TOKENS='{sh(m.get('max_num_batched_tokens', ''))}'") -print(f"_YAML_SCHEDULER_DELAY_FACTOR='{sh(m.get('scheduler_delay_factor', ''))}'") -print(f"_YAML_ATTN_PREFILL_CHUNK_SIZE='{sh(m.get('attn_prefill_chunk_size', ''))}'") -print(f"_YAML_STATE_CKPT_INTERVAL='{sh(m.get('state_checkpoint_interval_tokens', ''))}'") -print(f"_YAML_LEVEL='{sh(m.get('level', ''))}'") -print(f"_YAML_SPEC_DECODE_AL='{sh(m.get('spec_decode_acceptance_length', ''))}'") -PYEOF -# shellcheck source=/dev/null -source "$_yaml_tmp" -rm -f "$_yaml_tmp" -unset _yaml_tmp - -# Model YAML overrides the caller-provided server tuning. -BLOCK_SIZE="${_YAML_BLOCK_SIZE:-${BLOCK_SIZE}}" -MEM_FRAC_STATIC="${_YAML_MEM_FRAC_STATIC:-${MEM_FRAC_STATIC}}" -MAX_MODEL_LEN="${_YAML_MAX_MODEL_LEN:-${MAX_MODEL_LEN:-}}" -MAX_NUM_SEQS="${_YAML_MAX_NUM_SEQS:-${MAX_NUM_SEQS}}" -MAX_NUM_BATCHED_TOKENS="${_YAML_MAX_NUM_BATCHED_TOKENS:-${MAX_NUM_BATCHED_TOKENS:-}}" -SCHEDULER_DELAY_FACTOR="${_YAML_SCHEDULER_DELAY_FACTOR:-${SCHEDULER_DELAY_FACTOR:-}}" -ATTN_PREFILL_CHUNK_SIZE="${_YAML_ATTN_PREFILL_CHUNK_SIZE:-}" -STATE_CKPT_INTERVAL="${_YAML_STATE_CKPT_INTERVAL:-}" -LEVEL="${_YAML_LEVEL:-}" -# Synthetic acceptance length: YAML > launcher env (SPEC_DECODE_AL). -SPEC_DECODE_AL="${_YAML_SPEC_DECODE_AL:-${SPEC_DECODE_AL:-}}" -unset _YAML_BLOCK_SIZE _YAML_MEM_FRAC_STATIC _YAML_MAX_MODEL_LEN _YAML_MAX_NUM_SEQS _YAML_MAX_NUM_BATCHED_TOKENS _YAML_SCHEDULER_DELAY_FACTOR -unset _YAML_ATTN_PREFILL_CHUNK_SIZE _YAML_STATE_CKPT_INTERVAL _YAML_LEVEL _YAML_SPEC_DECODE_AL - -# ============================================================================= -# Agentic (AgentX trace-replay) run configuration -# ============================================================================= -# Agentic runs (IS_AGENTIC) use the '-AgentX' recipe and differ from the -# throughput path: prefix caching on, per-request max-num-seqs = 2*conc, extra -# ATOM server knobs, dp-sticky router, and an optional CPU KV-offload tier on -# prefill. All of this is gated on IS_AGENTIC_RUN so the throughput path is -# unchanged. Reference: ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md. -IS_AGENTIC_RUN=0 -if [[ "${IS_AGENTIC:-0}" == "1" || "${IS_AGENTIC:-}" == "true" ]]; then - IS_AGENTIC_RUN=1 -fi - -# Largest concurrency in this allocation (BENCH_MAX_CONCURRENCY is x-delimited). -_MAX_CONC=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | tail -1) - -# Prefix caching: agentic runs depend on cross-turn prefix reuse; throughput -# runs keep the server's paged-only behavior. -if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - PREFIX_CACHE_ARG="--enable-prefix-caching" - # Recipe max-num-seqs is 2*concurrency (prefill and decode). - MAX_NUM_SEQS=$((2 * _MAX_CONC)) -else - PREFIX_CACHE_ARG="--no-enable_prefix_caching" -fi - -# Agentic-only server knobs (applied when the model provides them). -AGENTIC_SERVER_ARGS="" -if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - [[ -n "$ATTN_PREFILL_CHUNK_SIZE" ]] && AGENTIC_SERVER_ARGS+=" --attn-prefill-chunk-size ${ATTN_PREFILL_CHUNK_SIZE}" - [[ -n "$STATE_CKPT_INTERVAL" ]] && AGENTIC_SERVER_ARGS+=" --state-checkpoint-interval-tokens ${STATE_CKPT_INTERVAL}" - [[ -n "$LEVEL" ]] && AGENTIC_SERVER_ARGS+=" --level ${LEVEL}" - AGENTIC_SERVER_ARGS+=" --cudagraph-mode FULL" -fi - -IFS=',' read -ra IP_ARRAY <<< "$IPADDRS" - -PREFILL_NODES_PER_WORKER=$(((PREFILL_TP_SIZE + GPUS_PER_NODE - 1) / GPUS_PER_NODE)) -DECODE_NODES_PER_WORKER=$(((DECODE_TP_SIZE + GPUS_PER_NODE - 1) / GPUS_PER_NODE)) -NODE_OFFSET=$((PREFILL_NODES_PER_WORKER * xP)) - -PREFILL_ARGS="" -PREFILL_IPS=() -for i in $(seq 0 $((xP - 1))); do - idx=$((i * PREFILL_NODES_PER_WORKER)) - PREFILL_IPS[$i]="${IP_ARRAY[$idx]}" - PREFILL_ARGS="$PREFILL_ARGS --prefill http://${IP_ARRAY[$idx]}:${PREFILL_PORT}" -done - -DECODE_ARGS="" -DECODE_IPS=() -for i in $(seq 0 $((yD - 1))); do - idx=$((i * DECODE_NODES_PER_WORKER + NODE_OFFSET)) - DECODE_IPS[$i]="${IP_ARRAY[$idx]}" - DECODE_ARGS="$DECODE_ARGS --decode http://${IP_ARRAY[$idx]}:${DECODE_PORT}" -done - -PREFILL_PARALLEL_ARGS=(-tp "$PREFILL_TP_SIZE") #TP -ONLINE_QUANT_ARG="" -if [ "$PREFILL_ENABLE_DP" = "true" ]; then - if [ "$PREFILL_ENABLE_EP" = "true" ]; then #EP+DPA - PREFILL_PARALLEL_ARGS=(-tp "$PREFILL_TP_SIZE" ${PREFILL_MODEL_EP_DP_FLAGS}) - for _dp_env_pair in ${MODEL_EP_DP_ENV}; do export "$_dp_env_pair"; done - else #TP+DPA - PREFILL_PARALLEL_ARGS=(-tp "$PREFILL_TP_SIZE" ${PREFILL_MODEL_TP_DP_FLAGS}) - for _dp_env_pair in ${MODEL_TP_DP_ENV}; do export "$_dp_env_pair"; done - fi - if [[ -n "$_ONLINE_QUANT_DPA_CONFIG" ]]; then - ONLINE_QUANT_ARG="--online_quant_config '${_ONLINE_QUANT_DPA_CONFIG}'" - fi -else - if [[ -n "$_ONLINE_QUANT_CONFIG" ]]; then - ONLINE_QUANT_ARG="--online_quant_config '${_ONLINE_QUANT_CONFIG}'" - fi -fi - -DECODE_PARALLEL_ARGS=(-tp "$DECODE_TP_SIZE") #TP -if [ "$DECODE_ENABLE_DP" = "true" ]; then - if [ "$DECODE_ENABLE_EP" = "true" ]; then #EP+DPA - DECODE_PARALLEL_ARGS=(-tp "$DECODE_TP_SIZE" ${DECODE_MODEL_EP_DP_FLAGS}) - for _dp_env_pair in ${MODEL_EP_DP_ENV}; do export "$_dp_env_pair"; done - else #TP+DPA - DECODE_PARALLEL_ARGS=(-tp "$DECODE_TP_SIZE" ${DECODE_MODEL_TP_DP_FLAGS}) - for _dp_env_pair in ${MODEL_TP_DP_ENV}; do export "$_dp_env_pair"; done - fi -fi -# Prefill-only DP env (e.g. GPU_MAX_HW_QUEUES): the shared DP env above is -# exported on every node, so scope prefill-only knobs by role here. NODE_RANK < -# NODE_OFFSET is a prefill node (see the node-role branch below); the recipe -# leaves these unset on decode. -if [ "$PREFILL_ENABLE_DP" = "true" ] && [ "$NODE_RANK" -lt "$NODE_OFFSET" ]; then - for _dp_env_pair in ${MODEL_PREFILL_DP_ENV}; do export "$_dp_env_pair"; done -fi -unset _dp_env_pair -unset _ONLINE_QUANT_CONFIG _ONLINE_QUANT_DPA_CONFIG - -for _env_pair in ${MODEL_ENVS}; do - export "$_env_pair" -done -unset _env_pair - -SPEC_ARGS=() -if [[ -n "$MODEL_MTP_FLAGS" && "${DECODE_MTP_SIZE}" -gt 0 ]]; then - SPEC_ARGS=(${MODEL_MTP_FLAGS} "$DECODE_MTP_SIZE") - # Agentic throughput runs simulate acceptance at the recipe's synthetic AL; - # eval runs (RUN_EVAL / EVAL_ONLY) need real target verification, so skip it. - if [[ "$IS_AGENTIC_RUN" == "1" && -n "$SPEC_DECODE_AL" \ - && "${EVAL_ONLY:-false}" != "true" && "${RUN_EVAL:-false}" != "true" ]]; then - SPEC_ARGS+=(--spec-decode-acceptance-length "$SPEC_DECODE_AL") - fi -fi - -KV_CACHE_ARG="${MODEL_KV_ARG}" - -MODEL_LEN_ARGS="" -if [[ -n "$MAX_MODEL_LEN" ]]; then - MODEL_LEN_ARGS="${MODEL_LEN_ARGS} --max-model-len ${MAX_MODEL_LEN}" -fi -if [[ -n "$MAX_NUM_BATCHED_TOKENS" ]]; then - MODEL_LEN_ARGS="${MODEL_LEN_ARGS} --max-num-batched-tokens ${MAX_NUM_BATCHED_TOKENS}" -fi -if [[ -n "$SCHEDULER_DELAY_FACTOR" ]]; then - MODEL_LEN_ARGS="${MODEL_LEN_ARGS} --scheduler-delay-factor ${SCHEDULER_DELAY_FACTOR}" -fi - -# ============================================================================= -# PD KV-transfer connectors and router policy -# ============================================================================= -# Decode is always a plain mooncake consumer. Prefill is a plain mooncake -# producer, except on the agentic CPU-offload tier (KV_OFFLOADING=dram) where it -# wraps mooncake + lmcache_offload in a "multi" connector (recipe -# DeepSeek-V4-Agentic-PD-Max.md, "DP attention with CPU offload"). host_ip is -# this node's handshake IP (resolved above). -DECODE_KV_TRANSFER="{\"kv_role\":\"kv_consumer\",\"kv_connector\":\"mooncake\",\"proxy_ip\":\"${host_ip}\",\"handshake_port\":${HANDSHAKE_PORT}}" -PREFILL_KV_TRANSFER="{\"kv_role\":\"kv_producer\",\"kv_connector\":\"mooncake\",\"proxy_ip\":\"${host_ip}\",\"handshake_port\":${HANDSHAKE_PORT}}" -if [[ "$IS_AGENTIC_RUN" == "1" && "${KV_OFFLOADING:-none}" == "dram" ]]; then - # lmcache.max_local_cpu_size is per worker; TOTAL_CPU_DRAM_GB is the - # aggregate CPU budget from the matrix (dram-utilization), so divide by - # GPUS_PER_NODE (one offload worker per GPU rank). - _per_worker_cpu_gb=$(( ${TOTAL_CPU_DRAM_GB:-0} / GPUS_PER_NODE )) - if [[ "$_per_worker_cpu_gb" -le 0 ]]; then _per_worker_cpu_gb=128; fi - # Recipe offload env (prefill node only). These are read by the lmcache - # offload runtime, not encoded in the connector JSON, so they must be in the - # server process env -- the launcher exports them outside the SLURM/Docker - # boundary where they are lost, so set them here. PYTHONHASHSEED=0 keeps the - # LMCache prefix hashes consistent across the offload worker processes. - export PYTHONHASHSEED="${PYTHONHASHSEED:-0}" - export OFFLOAD_COPY_WORKERS="${OFFLOAD_COPY_WORKERS:-1}" - export OFFLOAD_MIN_LOAD_TOKENS="${OFFLOAD_MIN_LOAD_TOKENS:-8192}" - export OFFLOAD_SLOT_STAGING_SLOTS="${OFFLOAD_SLOT_STAGING_SLOTS:-4}" - PREFILL_KV_TRANSFER="{\"kv_connector\":\"multi\",\"connectors\":[{\"kv_role\":\"kv_producer\",\"kv_connector\":\"mooncake\",\"proxy_ip\":\"${host_ip}\",\"handshake_port\":${HANDSHAKE_PORT}},{\"kv_connector\":\"lmcache_offload\",\"kv_role\":\"offload\",\"offload_layout\":\"hybrid\",\"max_pending_saves\":8,\"slot_sidecar_staging_slots\":${OFFLOAD_SLOT_STAGING_SLOTS:-4},\"lmcache.local_cpu\":true,\"lmcache.max_local_cpu_size\":${_per_worker_cpu_gb},\"lmcache.local_disk\":null,\"lmcache.max_local_disk_size\":0,\"lmcache.remote_url\":null,\"lmcache.chunk_size\":256,\"lmcache.cache_policy\":\"LRU\",\"lmcache.lookup_server_worker_ids\":[],\"lmcache.store_location\":\"LocalCPUBackend\",\"lmcache.retrieve_locations\":[\"LocalCPUBackend\"]}]}" -fi - -# Router policy: the agentic DP-attention tiers route cache-aware with balance -# thresholds, the agentic TP tier routes round-robin, and both pin PD rank -# mapping to none. Throughput runs keep random. -ROUTER_POLICY_ARGS="--policy random" -if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - if [[ "$PREFILL_ENABLE_DP" == "true" ]]; then - if [[ "$_MAX_CONC" -eq 256 ]]; then - # conc=256: pin each session to a fixed DP rank (dp_sticky) and let - # aiperf derive the session id from the request correlation id so the - # router can keep the sticky mapping. - ROUTER_POLICY_ARGS="--dp-aware --prefill-policy dp_sticky --decode-policy dp_sticky --atom-pd-rank-mapping-policy none" - export AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID=1 - else - ROUTER_POLICY_ARGS="--dp-aware --prefill-policy cache_aware --decode-policy cache_aware --cache-threshold 0.8 --balance-abs-threshold 20 --balance-rel-threshold 2.0 --eviction-interval 300 --atom-pd-rank-mapping-policy none" - fi - else - ROUTER_POLICY_ARGS="--prefill-policy round_robin --decode-policy round_robin --atom-pd-rank-mapping-policy none" - fi -fi - -cat < prefill node 0 + router; 1..NODE_OFFSET-1 -> prefill; -# NODE_OFFSET.. -> decode. -if [ "$NODE_RANK" -eq 0 ]; then - echo "NODE INFO =======================================" - echo "${host_name}:${host_ip} is Prefill Node 0 + Router" - echo "Prefill TP=${PREFILL_TP_SIZE}, Decode TP=${DECODE_TP_SIZE}" - echo "Prefill servers: ${PREFILL_ARGS}" - echo "Decode servers: ${DECODE_ARGS}" - echo "================================================" - - PREFILL_CMD="python3 -m atom.entrypoints.openai_server \ - --model ${MODEL_DIR}/${MODEL_NAME} \ - --host 0.0.0.0 --server-port ${PREFILL_PORT} \ - --trust-remote-code \ - ${PREFILL_PARALLEL_ARGS[*]} \ - ${SPEC_ARGS[*]} \ - ${KV_CACHE_ARG} \ - --block-size ${BLOCK_SIZE} \ - --gpu-memory-utilization ${MEM_FRAC_STATIC} \ - --max-num-seqs ${MAX_NUM_SEQS} \ - ${MODEL_LEN_ARGS} \ - ${AGENTIC_SERVER_ARGS} \ - ${PREFIX_CACHE_ARG} \ - ${ONLINE_QUANT_ARG} \ - --kv-transfer-config '${PREFILL_KV_TRANSFER}' \ - ${EXTRA_SERVER_ARGS}" - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $PREFILL_CMD" - else - set -x - eval "$PREFILL_CMD" \ - 2>&1 | tee /run_logs/slurm_job-${SLURM_JOB_ID}/prefill0_${host_name}.log & - set +x - prefill0_pid=$! - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting for all servers to be up (timeout=${WAIT_SERVER_TIMEOUT}s)..." - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait for prefill/decode /health endpoints" - else - _deadline=$(( $(date +%s) + WAIT_SERVER_TIMEOUT )) - for _ip in "${PREFILL_IPS[@]}"; do - echo "[wait] prefill http://${_ip}:${PREFILL_PORT}/health" - while ! curl -sf --max-time 10 "http://${_ip}:${PREFILL_PORT}/health" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_deadline ]]; then - echo "[wait][FAIL] prefill ${_ip}:${PREFILL_PORT} not ready after ${WAIT_SERVER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] prefill ${_ip}:${PREFILL_PORT} ready" - done - for _ip in "${DECODE_IPS[@]}"; do - echo "[wait] decode http://${_ip}:${DECODE_PORT}/health" - while ! curl -sf --max-time 10 "http://${_ip}:${DECODE_PORT}/health" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_deadline ]]; then - echo "[wait][FAIL] decode ${_ip}:${DECODE_PORT} not ready after ${WAIT_SERVER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] decode ${_ip}:${DECODE_PORT} ready" - done - fi - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "All servers up. Starting atomesh router..." - - ROUTER_CMD="/usr/local/bin/atomesh launch \ - --host 0.0.0.0 --port ${ROUTER_PORT} \ - --pd-disaggregation \ - ${PREFILL_ARGS} \ - ${DECODE_ARGS} \ - ${ROUTER_POLICY_ARGS} \ - --backend atom \ - --log-level info \ - --disable-health-check \ - --disable-circuit-breaker \ - --prometheus-port 29100" - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $ROUTER_CMD" - else - ROUTER_LOG_FILE="/tmp/slurm_job-${SLURM_JOB_ID}_router_${host_name}.log" - set -x - eval "$ROUTER_CMD" 2>&1 | tee "$ROUTER_LOG_FILE" & - set +x - proxy_pid=$! - - check_env_vars WAIT_LOCAL_ROUTER_TIMEOUT - WAIT_ROUTER_TIMEOUT="${WAIT_ROUTER_TIMEOUT:-$WAIT_LOCAL_ROUTER_TIMEOUT}" - echo "[wait] router http://0.0.0.0:${ROUTER_PORT}/v1/models (timeout=${WAIT_ROUTER_TIMEOUT}s)" - _router_deadline=$(( $(date +%s) + WAIT_ROUTER_TIMEOUT )) - while ! curl -sf --max-time 10 "http://0.0.0.0:${ROUTER_PORT}/v1/models" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_router_deadline ]]; then - echo "[wait][FAIL] router ${ROUTER_PORT}/v1/models not ready after ${WAIT_ROUTER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] router /v1/models ready" - - echo "Router is ready for benchmarking" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Ready for benchmarking on ${host_name}:${host_ip}" - - cd $ATOM_WS_PATH - - export IS_MTP="false" - if [[ -n "$MODEL_MTP_FLAGS" && "${DECODE_MTP_SIZE}" -gt 0 ]]; then - export IS_MTP="true" - fi - - # Select the benchmark runner. - # IS_AGENTIC=1/true -> AgentX trace replay (trace_replay.sh), driven by - # aiperf against the atomesh router on ROUTER_PORT. - # IS_AGENTIC unset/0 -> fixed-seq-len throughput benchmark (bench.sh). - if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - # trace_replay.sh targets ROUTER_PORT and derives MODEL from - # $MODEL_DIR/$MODEL_NAME, which matches the atom server's served-model - # name (its --model path). The atomesh router exposes no /flush_cache, - # and the CI matrix runs one concurrency per allocation, so disable the - # SGLang-specific between-conc cache clear. - export ROUTER_PORT - export DURATION="${DURATION:-1800}" - export CLEAR_CACHE_BETWEEN_CONC="${CLEAR_CACHE_BETWEEN_CONC:-0}" - # trace_replay.sh / benchmark_lib.sh locate utils/aiperf under - # INFMAX_CONTAINER_WORKSPACE (the container repo root). - # The SGLang client-image path sets it in its env-file; the - # in-container ATOM path must set it too -> derive it from ATOM_WS_PATH - # (.../benchmarks/multi_node/amd_utils -> repo root, i.e. /workspace). - export INFMAX_CONTAINER_WORKSPACE="${INFMAX_CONTAINER_WORKSPACE:-${ATOM_WS_PATH%/benchmarks/multi_node/amd_utils}}" - # trace_replay.sh signature: model_path model_name concurrency_list log_path - BENCH_CMD="bash $ATOM_WS_PATH/trace_replay.sh \ - $MODEL_DIR $MODEL_NAME \"${BENCH_MAX_CONCURRENCY}\" /run_logs/slurm_job-${SLURM_JOB_ID}" - echo "Benchmark runner: trace_replay.sh (agentic ATOM, router :${ROUTER_PORT}, KV_OFFLOADING=${KV_OFFLOADING:-none})" - else - BENCH_CMD="bash $ATOM_WS_PATH/bench.sh ${xP} ${yD} $((PREFILL_TP_SIZE*xP)) $((DECODE_TP_SIZE*yD)) \ - $MODEL_DIR $MODEL_NAME /run_logs/slurm_job-${SLURM_JOB_ID} ${BENCH_INPUT_LEN} \ - ${BENCH_OUTPUT_LEN} \"${BENCH_MAX_CONCURRENCY}\" ${BENCH_REQUEST_RATE} \ - ${BENCH_RANDOM_RANGE_RATIO} ${BENCH_NUM_PROMPTS_MULTIPLIER}" - echo "Benchmark runner: bench.sh (fixed-seq-len)" - fi - - if [[ "${EVAL_ONLY}" == "true" ]]; then - echo "EVAL_ONLY mode: skipping throughput benchmark" - elif [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $BENCH_CMD" - else - set -x - eval "$BENCH_CMD" - set +x - fi - - if [[ "${RUN_EVAL}" == "true" ]]; then - echo "Running lm-eval evaluation on Node 0..." - - EVAL_HEALTH_OK=false - for _attempt in 1 2 3; do - if curl -sf --max-time 10 "http://0.0.0.0:${ROUTER_PORT}/health" >/dev/null 2>&1; then - EVAL_HEALTH_OK=true - break - fi - echo "Eval health check attempt $_attempt failed, retrying in 10s..." - sleep 10 - done - - if [[ "$EVAL_HEALTH_OK" != "true" ]]; then - echo "WARNING: Router health check failed after 3 attempts. Skipping eval." - else - pushd /workspace - - # job.slurm's -e allowlist forwards ROUTER_PORT but not PORT, and - # run_lm_eval's check_env_vars guard runs before it parses --port. - export PORT="${ROUTER_PORT}" - - source /workspace/benchmarks/benchmark_lib.sh - - if [[ -n "${EVAL_CONC:-}" ]]; then - export EVAL_CONCURRENT_REQUESTS="${EVAL_CONC}" - else - export EVAL_CONCURRENT_REQUESTS=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | tail -1) - fi - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: run_eval --port ${ROUTER_PORT} (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS})" - else - MODEL_NAME="${MODEL_DIR}/${MODEL_NAME}" run_eval --port "${ROUTER_PORT}" - eval_rc=$? - - if [[ $eval_rc -ne 0 ]]; then - echo "ERROR: run_eval exited rc=$eval_rc; preserving failure artifacts" >&2 - EVAL_FAILED=1 - else - export TP="${PREFILL_TP_SIZE}" - export CONC="${EVAL_CONCURRENT_REQUESTS}" - export PREFILL_TP="${PREFILL_TP_SIZE}" - export PREFILL_EP=1 - export PREFILL_NUM_WORKERS="${xP}" - export DECODE_TP="${DECODE_TP_SIZE}" - export DECODE_EP=1 - export DECODE_NUM_WORKERS="${yD}" - export ISL="${BENCH_INPUT_LEN}" - export OSL="${BENCH_OUTPUT_LEN}" - - MODEL_NAME="${MODEL_DIR}/${MODEL_NAME}" append_lm_eval_summary - - fi - - EVAL_COPY_DIR="/run_logs/slurm_job-${SLURM_JOB_ID}/eval_results" - if stage_eval_artifacts \ - "$EVAL_COPY_DIR" /workspace "${EVAL_RESULT_DIR:-}"; then - echo "Eval artifacts staged in $EVAL_COPY_DIR" - else - echo "ERROR: failed to stage eval artifacts in $EVAL_COPY_DIR" >&2 - EVAL_FAILED=1 - fi - fi - - popd - fi - fi - - LOGS_OUTPUT="${BENCHMARK_LOGS_DIR}/logs" - mkdir -p "$LOGS_OUTPUT" - if [[ "$DRY_RUN" -eq 0 ]]; then - cp -r /run_logs/slurm_job-${SLURM_JOB_ID} "$LOGS_OUTPUT/" - echo "Copied results to $LOGS_OUTPUT/slurm_job-${SLURM_JOB_ID}" - fi - - echo "Waiting 60s before killing router and prefill server..." - sleep 60 - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Killing router and prefill server" - if [[ "$DRY_RUN" -eq 0 ]]; then - kill $proxy_pid - kill $prefill0_pid - fi - - if [[ "${EVAL_FAILED:-0}" -eq 1 ]]; then - echo "ERROR: eval failed; exiting node-0 with rc=1" - exit 1 - fi - -elif [ "$NODE_RANK" -gt 0 ] && [ "$NODE_RANK" -lt "$NODE_OFFSET" ]; then - echo "${host_name}:${host_ip} is Prefill Node (rank ${NODE_RANK})" - - prefill_worker_idx=$((NODE_RANK / PREFILL_NODES_PER_WORKER)) - PREFILL_HEADNODE_IP="${PREFILL_IPS[$prefill_worker_idx]}" - - PREFILL_CMD="python3 -m atom.entrypoints.openai_server \ - --model ${MODEL_DIR}/${MODEL_NAME} \ - --host 0.0.0.0 --server-port ${PREFILL_PORT} \ - --trust-remote-code \ - ${PREFILL_PARALLEL_ARGS[*]} \ - ${SPEC_ARGS[*]} \ - ${KV_CACHE_ARG} \ - --block-size ${BLOCK_SIZE} \ - --gpu-memory-utilization ${MEM_FRAC_STATIC} \ - --max-num-seqs ${MAX_NUM_SEQS} \ - ${MODEL_LEN_ARGS} \ - ${AGENTIC_SERVER_ARGS} \ - ${PREFIX_CACHE_ARG} \ - ${ONLINE_QUANT_ARG} \ - --kv-transfer-config '${PREFILL_KV_TRANSFER}' \ - ${EXTRA_SERVER_ARGS}" - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $PREFILL_CMD" - else - set -x - eval "$PREFILL_CMD" \ - 2>&1 | tee /run_logs/slurm_job-${SLURM_JOB_ID}/prefill_${host_name}.log & - set +x - prefill_pid=$! - trap 'echo "Caught signal, killing prefill (pid=$prefill_pid)"; kill $prefill_pid 2>/dev/null; exit 0' SIGTERM SIGINT - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting for router to be up..." - check_env_vars WAIT_REMOTE_ROUTER_TIMEOUT - WAIT_ROUTER_TIMEOUT="${WAIT_ROUTER_TIMEOUT:-$WAIT_REMOTE_ROUTER_TIMEOUT}" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait for router ${NODE0_ADDR}:${ROUTER_PORT}/health" - else - _router_deadline=$(( $(date +%s) + WAIT_ROUTER_TIMEOUT )) - while ! curl -sf --max-time 10 "http://${NODE0_ADDR}:${ROUTER_PORT}/health" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_router_deadline ]]; then - echo "[wait][FAIL] router ${NODE0_ADDR}:${ROUTER_PORT} not ready after ${WAIT_ROUTER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] router ${NODE0_ADDR}:${ROUTER_PORT} ready" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting until router closes..." - trap 'echo "Caught signal, killing prefill (pid=$prefill_pid)"; kill $prefill_pid 2>/dev/null; exit 0' SIGTERM SIGINT - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait until router ${NODE0_ADDR}:${ROUTER_PORT} closes" - else - while curl -sf --max-time 10 "http://${NODE0_ADDR}:${ROUTER_PORT}/health" >/dev/null 2>&1; do - sleep 10 & - wait $! - done - echo "[wait] router ${NODE0_ADDR}:${ROUTER_PORT} closed" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Killing prefill server (rank ${NODE_RANK})" - if [[ "$DRY_RUN" -eq 0 ]]; then kill $prefill_pid 2>/dev/null; fi - -else - RANK=$((NODE_RANK - NODE_OFFSET)) - echo "${host_name}:${host_ip} is Decode Node (rank ${RANK})" - - _MAX_CONC=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | tail -1) - CUDAGRAPH_SIZES='[1,2,4,8,16,24,32,40,48,56,64,72,80,88,96,104,112,120,128,136,144,152,160,168,176,184,192,200,208,216,224,232,240,248,256]' - - if [[ "$IS_AGENTIC_RUN" == "1" ]]; then - # Recipe max-num-seqs is 2*concurrency. - DECODE_MAX_NUM_SEQS=$((2 * _MAX_CONC)) - # Dense capture ladder per the recipe's per-tier decode sizing: - # TP decode (no DP attention): 1..min(64, 2*conc). - # DP-attention decode: per-rank 1..(conc/4), since max-num-seqs=2*conc - # spreads across the 8 DP ranks (2*conc / 8 = conc/4). - # Every batch size up to the cap gets a graph, which measurably helps - # small-batch agentic decode. - if [[ "$DECODE_ENABLE_DP" == "true" ]]; then - _dense_max=$((_MAX_CONC / 4)) - else - _dense_max=$((2 * _MAX_CONC)) - if [[ "$_dense_max" -gt 64 ]]; then _dense_max=64; fi - fi - if [[ "$_dense_max" -lt 1 ]]; then _dense_max=1; fi - CUDAGRAPH_SIZES="[$(seq -s, 1 "$_dense_max")]" - else - DECODE_MAX_NUM_SEQS="${_MAX_CONC}" - fi - - DECODE_CMD="python3 -m atom.entrypoints.openai_server \ - --model ${MODEL_DIR}/${MODEL_NAME} \ - --host 0.0.0.0 --server-port ${DECODE_PORT} \ - --trust-remote-code \ - ${DECODE_PARALLEL_ARGS[*]} \ - ${SPEC_ARGS[*]} \ - ${KV_CACHE_ARG} \ - --block-size ${BLOCK_SIZE} \ - --gpu-memory-utilization ${MEM_FRAC_STATIC} \ - --max-num-seqs ${DECODE_MAX_NUM_SEQS} \ - ${MODEL_LEN_ARGS} \ - ${AGENTIC_SERVER_ARGS} \ - ${PREFIX_CACHE_ARG} \ - ${ONLINE_QUANT_ARG} \ - --kv-transfer-config '${DECODE_KV_TRANSFER}' \ - --cudagraph-capture-sizes "${CUDAGRAPH_SIZES}" \ - ${EXTRA_SERVER_ARGS}" - - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $DECODE_CMD" - else - set -x - eval "$DECODE_CMD" \ - 2>&1 | tee /run_logs/slurm_job-${SLURM_JOB_ID}/decode_${host_name}.log & - set +x - decode_pid=$! - trap 'echo "Caught signal, killing decode (pid=$decode_pid)"; kill $decode_pid 2>/dev/null; exit 0' SIGTERM SIGINT - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting for router to be up..." - check_env_vars WAIT_REMOTE_ROUTER_TIMEOUT - WAIT_ROUTER_TIMEOUT="${WAIT_ROUTER_TIMEOUT:-$WAIT_REMOTE_ROUTER_TIMEOUT}" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait for router ${NODE0_ADDR}:${ROUTER_PORT}/health" - else - _router_deadline=$(( $(date +%s) + WAIT_ROUTER_TIMEOUT )) - while ! curl -sf --max-time 10 "http://${NODE0_ADDR}:${ROUTER_PORT}/health" >/dev/null 2>&1; do - if [[ $(date +%s) -ge $_router_deadline ]]; then - echo "[wait][FAIL] router ${NODE0_ADDR}:${ROUTER_PORT} not ready after ${WAIT_ROUTER_TIMEOUT}s" >&2 - exit 1 - fi - sleep 10 - done - echo "[wait][OK] router ${NODE0_ADDR}:${ROUTER_PORT} ready" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Waiting until router closes..." - trap 'echo "Caught signal, killing decode (pid=$decode_pid)"; kill $decode_pid 2>/dev/null; exit 0' SIGTERM SIGINT - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: wait until router ${NODE0_ADDR}:${ROUTER_PORT} closes" - else - while curl -sf --max-time 10 "http://${NODE0_ADDR}:${ROUTER_PORT}/health" >/dev/null 2>&1; do - sleep 10 & - wait $! - done - echo "[wait] router ${NODE0_ADDR}:${ROUTER_PORT} closed" - fi - - echo "[-------]" NODE $NODE_RANK "[--------]" - echo "Killing decode server (rank ${RANK})" - if [[ "$DRY_RUN" -eq 0 ]]; then kill $decode_pid 2>/dev/null; fi -fi - -echo "Script completed successfully" -exit 0 From bc215786436a5f6958234ccbe062fbaa53475e57 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 12:37:00 -0500 Subject: [PATCH 4/6] fix(srt-slurm): refresh the lmcache-server patch for the direct aggregate connector Regenerated from SemiAnalysisAI/srt-slurm#32 at 181b2e4: a direct vllm serve aggregate worker now keeps the connector its role names (for example lmcache-mp), instead of dropping it. --- .../507-lmcache-server-atom-sglang.patch | 64 +++++++++++++++++-- 1 file changed, 59 insertions(+), 5 deletions(-) diff --git a/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch index c15c487cee..92ef255197 100644 --- a/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch +++ b/inferencex-e2e/runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch @@ -460,7 +460,7 @@ index 641709f..121f060 100644 def get_mooncake_worker_env(self, infra_node_ip: str, local_hostname: str) -> dict[str, str]: """Get mooncake env vars to inject on a specific worker. diff --git a/src/srtctl/backends/vllm.py b/src/srtctl/backends/vllm.py -index 048627a..5d47ae1 100644 +index 048627a..79168fd 100644 --- a/src/srtctl/backends/vllm.py +++ b/src/srtctl/backends/vllm.py @@ -35,6 +35,7 @@ from srtctl.ports import ( @@ -480,7 +480,21 @@ index 048627a..5d47ae1 100644 # Can be overridden per role by setting "connector" in roles..args; connector_for_mode resolves it. # "moriio" (ROCm MoRI-IO) registers workers with the vLLM Router and needs frontend.type: vllm-router. # dynamo 1.0.0+: translated to --kv-transfer-config (--connector was removed). -@@ -1756,6 +1757,8 @@ class KVConnector: +@@ -1278,9 +1279,10 @@ class VLLMProtocol: + overridden = pop_vllm_orchestration_flags(config) + config.setdefault("served-model-name", served_model_name) + +- # A prefill/decode worker gets its KV connector; an aggregate worker has none. +- config.pop("connector", None) +- if mode in {"prefill", "decode"}: ++ # A prefill/decode worker gets its KV connector. An aggregate worker gets ++ # only the one its role names (e.g. lmcache-mp offload), not the P/D default. ++ role_connector = config.pop("connector", None) ++ if mode in {"prefill", "decode"} or role_connector is not None: + kv_transfer_config = self.kv_transfer_config(mode, process, runtime) + if kv_transfer_config is not None: + config.setdefault("kv-transfer-config", kv_transfer_config) +@@ -1756,6 +1758,8 @@ class KVConnector: kv_role: str | None = "kv_both" module_path: str | None = None discovery: bool = False @@ -489,7 +503,7 @@ index 048627a..5d47ae1 100644 def transfer_config(self, mode: WorkerMode) -> dict[str, Any]: """The ``--kv-transfer-config`` payload for a worker mode, before any topology-derived extras.""" -@@ -1763,6 +1766,8 @@ class KVConnector: +@@ -1763,6 +1767,8 @@ class KVConnector: if self.module_path is not None: payload["kv_connector_module_path"] = self.module_path payload["kv_role"] = self.kv_role or ("kv_producer" if mode == "prefill" else "kv_consumer") @@ -498,7 +512,7 @@ index 048627a..5d47ae1 100644 return payload -@@ -1770,6 +1775,12 @@ class KVConnector: +@@ -1770,6 +1776,12 @@ class KVConnector: _CONNECTOR_MAP: dict[str, KVConnector] = { "nixl": KVConnector("NixlConnector"), "lmcache": KVConnector("LMCacheConnectorV1"), @@ -785,7 +799,7 @@ index 0000000..52f3f8d + + assert backend.get_process_environment(_process()) == {} diff --git a/tests/test_vllm_connectors.py b/tests/test_vllm_connectors.py -index 691990b..4bb0974 100644 +index 691990b..1dfe69e 100644 --- a/tests/test_vllm_connectors.py +++ b/tests/test_vllm_connectors.py @@ -23,6 +23,14 @@ def test_table_presets_serialize_exactly_as_before(): @@ -803,3 +817,43 @@ index 691990b..4bb0974 100644 assert VLLMProtocol(connector="kvbm").kv_transfer_config("decode") == json.dumps( { "kv_connector": "DynamoConnector", +@@ -67,3 +75,39 @@ def test_a_mode_dependent_role_follows_the_worker_mode(): + assert row.transfer_config("prefill")["kv_role"] == "kv_producer" + assert row.transfer_config("decode")["kv_role"] == "kv_consumer" + assert KVConnector("SomeConnector").transfer_config("prefill")["kv_role"] == "kv_both" ++ ++ ++@pytest.mark.parametrize( ++ ("aggregated", "expected"), ++ [ ++ ({}, None), ++ ({"connector": "none"}, None), ++ ({"connector": "lmcache-mp"}, _CONNECTOR_MAP["lmcache-mp"].transfer_config("agg")), ++ ], ++) ++def test_direct_aggregate_worker_runs_only_its_role_connector(aggregated, expected): ++ """A direct `vllm serve` aggregate worker skips the P/D default connector but keeps the one its role names.""" ++ from pathlib import Path ++ from unittest.mock import MagicMock ++ ++ from srtctl.core.topology import Process ++ ++ backend = VLLMProtocol(connector="nixl", vllm_config=VLLMServerConfig(aggregated=aggregated)) ++ process = Process( ++ node="node0", ++ gpu_indices=frozenset(range(8)), ++ sys_port=8081, ++ http_port=0, ++ endpoint_mode="agg", ++ endpoint_index=0, ++ node_rank=0, ++ ) ++ runtime = MagicMock(model_path=Path("/model"), is_hf_model=False, frontend_port=9000) ++ ++ cmd = backend.build_worker_command( ++ process=process, endpoint_processes=[process], runtime=runtime, frontend_type="vllm" ++ ) ++ ++ assert "--connector" not in cmd ++ kv_config = json.loads(cmd[cmd.index("--kv-transfer-config") + 1]) if "--kv-transfer-config" in cmd else None ++ assert kv_config == expected From d0a97e44af6f8dab7645c6fae3f30a3bfe1857dd Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Mon, 28 Sep 2026 12:52:27 -0500 Subject: [PATCH 5/6] fix(srt-slurm): start every srt container in the workspace mount Multi-node srt launches inherit PYTHONPYCACHEPREFIX from the job env and start in the image root when the image sets no WORKDIR (the ATOM images). PyTorch 2.10's generated remote-module import then fails in cache_from_source with IndexError. Single-node already set the container workdir; move that override into apply_srt_recipe so every srt launch gets it. --- inferencex-e2e/infx/srt_slurm/single_node.py | 3 --- .../infx/tests/srt_slurm/test_srt_single_node.py | 5 ++--- .../infx/tests/srt_slurm/test_synthetic_acceptance.py | 10 +++++++++- inferencex-e2e/runners/slurm_utils.sh | 5 ++++- 4 files changed, 15 insertions(+), 8 deletions(-) diff --git a/inferencex-e2e/infx/srt_slurm/single_node.py b/inferencex-e2e/infx/srt_slurm/single_node.py index 1e6baece4b..2afccb75d2 100644 --- a/inferencex-e2e/infx/srt_slurm/single_node.py +++ b/inferencex-e2e/infx/srt_slurm/single_node.py @@ -145,9 +145,6 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: # Exclusive nodes include idle GPUs. Restrict each server/client step to # the serving GPU count so client-side power collection sees the same set. overrides = ["--set", f"srun_options.gpus-per-node={json.dumps(environment['GPU_COUNT'])}"] - # Match the legacy container working directory using the existing repo mount. - # PyTorch's generated module imports fail from / with PYTHONPYCACHEPREFIX set. - overrides += ["--set", 'srun_options.container-workdir="/infmax-workspace"'] if environment.get("SRT_SRUN_OPTIONS"): options = json.loads(environment["SRT_SRUN_OPTIONS"]) if not isinstance(options, dict) or any( diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py b/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py index 09e7b4590a..239675c072 100644 --- a/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py +++ b/inferencex-e2e/infx/tests/srt_slurm/test_srt_single_node.py @@ -54,9 +54,7 @@ def test_native_binding_submits_one_point_and_keeps_server_settings(point): overrides = parse_overrides(argv[1::2], []) actual = copy.deepcopy(recipe) apply_overrides_to_recipe(actual, overrides) - assert actual["srun_options"] == { - "gpus-per-node": "4", "container-workdir": "/infmax-workspace", - } + assert actual["srun_options"] == {"gpus-per-node": "4"} assert actual["benchmark"]["env"] == { "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "USE_CHAT_TEMPLATE": "false", @@ -296,6 +294,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, f"#!{sys.executable}\n" "import json, os, pathlib, sys\n" "assert pathlib.Path('bin/uv').is_file(), 'native bootstrap was skipped'\n" + "assert 'srun_options.container-workdir=\"/infmax-workspace\"' in sys.argv\n" "output = pathlib.Path(sys.argv[sys.argv.index('--output') + 1]) / '42'\n" "logs = output / 'logs'\n" "logs.mkdir(parents=True)\n" diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py b/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py index fc181fdfd1..6bc76ff878 100644 --- a/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py +++ b/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py @@ -443,7 +443,15 @@ def test_shell_forwards_options_and_submission_failure(tmp_path: Path) -> None: ) assert result.returncode == 7, result.stderr argv = json.loads(result.stdout) - assert argv[:5] == ["apply", "-f", str(recipe), "--tags", "a b"] + assert argv[:7] == [ + "apply", + "--set", + 'srun_options.container-workdir="/infmax-workspace"', + "-f", + str(recipe), + "--tags", + "a b", + ] result_recipe = apply_native(yaml.safe_load(recipe.read_text()), argv) assert ( json.loads(result_recipe["roles"]["agg"]["args"]["speculative-config"])[ diff --git a/inferencex-e2e/runners/slurm_utils.sh b/inferencex-e2e/runners/slurm_utils.sh index 50370edfeb..193fc54999 100644 --- a/inferencex-e2e/runners/slurm_utils.sh +++ b/inferencex-e2e/runners/slurm_utils.sh @@ -145,9 +145,12 @@ apply_srt_recipe() { local config="$1" framework="$2" shift 2 # Slurm creates a separate compute venv; do not inherit the login venv marker. + # Every container starts in the InferenceX workspace mount, as the legacy + # launchers did: PyTorch's generated module imports fail from / with + # PYTHONPYCACHEPREFIX set. Caller overrides still win. PYTHONPATH="$INFERENCEX_SLURM_UTILS_DIR/..${PYTHONPATH:+:$PYTHONPATH}" \ env -u VIRTUAL_ENV python3 -m infx.srt_slurm.synthetic_acceptance \ - "$config" "$framework" -- "$@" + "$config" "$framework" -- --set 'srun_options.container-workdir="/infmax-workspace"' "$@" } # One native submission per fixed-sequence or AgentX matrix point, shared across Slurm pools. From 38d8f2b615186d0c0b15f8c0c069344b43d79738 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Tue, 29 Sep 2026 14:08:13 -0500 Subject: [PATCH 6/6] docs(agentx): drop the LMCache patch references for srt-slurm v2.36.0 srt-slurm v2.36.0 ships extra-kv-connectors, so the recipe comment and the changelog entry no longer point at the deleted 507 patch. --- .../dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml | 4 ++-- inferencex-e2e/perf-changelog.yaml | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml index f3addef0d4..6c960f5b01 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml @@ -3,8 +3,8 @@ # amd_utils path (#3158; ATOM recipes/DeepSeek-V4-Agentic-PD-Max.md). Three tiers: # TP8 at concurrency 1-16, DP attention at 64-128, and DP attention plus ATOM's # in-process LMCache CPU offload on prefill at 256. srtctl generates the Mooncake -# P/D connector; the offload tier adds lmcache_offload next to it -# (runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch). +# P/D connector; the offload tier adds lmcache_offload next to it through +# extra-kv-connectors. base: schema: 2 name: mi355x-dsv4-pro-0813-atom-agentx-lmcache diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 14a7b11839..ab8fdf597e 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9111,5 +9111,5 @@ - agentic-coding description: - "Port the DeepSeek-V4-Pro-0813 MI355X ATOM 1P1D AgentX config from the legacy amd_utils path to a native srt-slurm recipe (benchmarks/multi_node/srt-slurm-recipes/dsv4/atom/mi355x-fp4/agentx/disagg-lmcache-dspark.yaml), one override variant per point. Server flags, env, AToMesh routing policies and decode CUDA-graph capture sizes match the legacy server_atom.sh for every tier: TP8 at concurrency 1 and 16, DP attention with prefill TBO at 64 and 128, and DP attention with ATOM's in-process LMCache CPU offload (lmcache_offload, 187 GB per prefill rank) at 256. Golden acceptance 3.01 for DSpark with three draft tokens is now injected by the srt-slurm path, the same value the legacy models_atom.yaml hardcoded." - - "srt-slurm had no way to add LMCache next to its generated Mooncake connector, so runners/srt-slurm/patches/507-lmcache-server-atom-sglang.patch applies SemiAnalysisAI/srt-slurm#32 (which includes NVIDIA/srt-slurm#507) to the pinned submodule until it lands upstream." + - "The LMCache tier uses srt-slurm's extra-kv-connectors (NVIDIA/srt-slurm#507, in v2.36.0) to add lmcache_offload next to the generated Mooncake connector." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3543