Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
#!/usr/bin/env bash
set -eo pipefail

apply_verified_patch() {
local package_root="$1" patch_file="$2" checksum="$3"
printf '%s %s\n' "$checksum" "$patch_file" | sha256sum --check --status
git -C "$package_root" apply --check --include='vllm/*' "$patch_file"
git -C "$package_root" apply --include='vllm/*' "$patch_file"
}

main() {
# Native srt-slurm runs this preamble in both worker and router containers.
# Only the standalone router image is exempt; a worker with missing vLLM must fail.
if command -v vllm-router >/dev/null 2>&1 && ! command -v vllm >/dev/null 2>&1; then
echo 'Standalone vLLM Router image: no engine backport required'
exit 0
fi

# Minimal diagnostic candidate: the pinned nightly already includes #57700.
# Apply only #59164's Python runtime diff to the installed wheel; do not
# replace compiled extensions. Remove this script and setup_script once
# the official image contains the fix and passes qualification.
package_root=$(python3 -c 'import importlib.util; from pathlib import Path; spec = importlib.util.find_spec("vllm"); assert spec and spec.submodule_search_locations, "vLLM package not found"; print(Path(next(iter(spec.submodule_search_locations))).parent)')
# Pin the latest reviewed #59164 head; its runtime diff is equivalent to
# the previously validated d5e6faa9 extraction.
zeroing_patch="$(dirname -- "${BASH_SOURCE[0]}")/k3-moriio-sync-zeroing.patch"
apply_verified_patch "$package_root" "$zeroing_patch" 3da3746e85d53e4a3b17062b4113475d31a86cc07418e87ef5b4bb0c906cad20
echo 'Applied vLLM#59164 at c5b1350f1f2bf10a128127a9b85e93d5f9f18e62'
}

if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
main "$@"
fi
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
diff --git a/vllm/distributed/kv_transfer/kv_connector/v1/base.py b/vllm/distributed/kv_transfer/kv_connector/v1/base.py
index 6df9d1cfb..192b4ea91 100644
--- a/vllm/distributed/kv_transfer/kv_connector/v1/base.py
+++ b/vllm/distributed/kv_transfer/kv_connector/v1/base.py
@@ -543,6 +543,16 @@ class KVConnectorBase_V1(ABC):
"""
pass

+ def get_sync_load_block_ids(self, request: "Request") -> list[int]:
+ """Return blocks whose synchronous load replaces worker zeroing.
+
+ Called after update_state_after_alloc. Each returned block must be
+ fully initialized before forward consumes it. A failed load must abort
+ forward instead of recomputing with uninitialized blocks.
+ Defaults to no blocks.
+ """
+ return []
+
@abstractmethod
def build_connector_meta(
self, scheduler_output: SchedulerOutput
diff --git a/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py b/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py
index b482e6a8c..a6d7ffedc 100644
--- a/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py
+++ b/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py
@@ -303,6 +303,24 @@ class MoRIIOConnector(KVConnectorBase_V1, SupportsHMA):
assert self.connector_scheduler is not None
self.connector_scheduler.on_new_request(request)

+ def get_sync_load_block_ids(self, request: "Request") -> list[int]:
+ scheduler = self.connector_scheduler
+ if (
+ self.mode != MoRIIOMode.READ
+ or not self.kv_transfer_config.is_kv_consumer
+ or scheduler is None
+ or not scheduler._has_mamba
+ or self._vllm_config.cache_config.get_resolved_kv_cache_layout().name
+ not in ("LBHNC", "LBNHC")
+ ):
+ return []
+ pending = scheduler._reqs_need_recv.get(request.request_id)
+ if pending is None:
+ return []
+ # Hybrid READ fills these entire attention pages and aborts on failure.
+ # The scheduler excludes these IDs only from newly allocated page zeroing.
+ return pending[1][0]
+
def build_connector_meta(
self,
scheduler_output: SchedulerOutput,
diff --git a/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py b/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py
index 73b6b0ccf..5cb75e17a 100644
--- a/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py
+++ b/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py
@@ -463,6 +463,12 @@ class MultiConnector(KVConnectorBase_V1, SupportsHMA):
for c in self._connectors:
c.on_new_request(request)

+ def get_sync_load_block_ids(self, request: "Request") -> list[int]:
+ chosen = self._requests_to_connector.get(request.request_id)
+ if chosen is None:
+ return []
+ return self._connectors[chosen].get_sync_load_block_ids(request)
+
def build_connector_meta(
self, scheduler_output: SchedulerOutput
) -> MultiKVConnectorMetadata:
diff --git a/vllm/v1/core/sched/scheduler.py b/vllm/v1/core/sched/scheduler.py
index fcb1421b1..a86e0810a 100644
--- a/vllm/v1/core/sched/scheduler.py
+++ b/vllm/v1/core/sched/scheduler.py
@@ -349,7 +349,7 @@ class Scheduler(SchedulerInterface):

self.has_mamba_layers = kv_cache_config.has_mamba_layers
self.needs_kv_cache_zeroing = kv_cache_config.needs_kv_cache_zeroing
- # Blocks that async KV loads will overwrite this step, skipped from
+ # Blocks that KV loads will overwrite this step, skipped from
# zeroing since the zeroing could race the out-of-band write.
self._skip_zero_block_ids: set[int] = set()
self.need_mamba_block_aligned_split = (
@@ -1295,6 +1295,11 @@ class Scheduler(SchedulerInterface):
if num_external_computed_tokens > 0:
# load_kv_async is False here
has_sync_kv_loads = True
+ if self.needs_kv_cache_zeroing:
+ assert self.connector is not None
+ self._skip_zero_block_ids.update(
+ self.connector.get_sync_load_block_ids(request)
+ )
if self.log_stats:
request.record_event(
EngineCoreEventType.SCHEDULED, scheduled_timestamp
Original file line number Diff line number Diff line change
@@ -0,0 +1,224 @@
# Native Kimi-K3 FP32 SSM PD on an immutable official ROCm nightly.
base:
schema: 2
name: kimik3-fp4-mi355x-vllm-disagg-agentic
# Minimal diagnostic backport until #59164 is included and qualified upstream.
setup_script: k3-moriio-debug.sh
model:
path: Kimi-K3
container: vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19
precision: fp4
slurm:
time_limit: '08:00:00'
sbatch_directives:
mem: '0'
srun_options:
container-remap-root: ''
container-writable: ''
mem: '0'
resources:
gpu_type: mi355x
gpus_per_node: 8
frontend:
type: vllm-router
enable_multiple_frontends: false
orchestrator_placement: head
container_image: docker.io/vllm/vllm-router@sha256:1fbf06701abce8c8cd459414e472d63999c05b201baffc62e00727c10583f5ea
args:
policy: consistent_hash
prefill-policy: consistent_hash
decode-policy: consistent_hash
log-level: info
observability:
enabled: false
tachometer:
enabled: false
health_check:
interval_seconds: 10
max_attempts: 90
engine:
type: vllm
connector: moriio
roles:
prefill:
nodes: 1
workers: 1
gpus: 8
args: &common_args
served-model-name: moonshotai/Kimi-K3
trust-remote-code: true
tensor-parallel-size: 8
decode-context-parallel-size: 1
cp-kv-cache-interleave-size: 1
dcp-comm-backend: a2a
moe-backend: auto
load-format: fastsafetensors
gpu-memory-utilization: 0.90
language-model-only: true
enable-auto-tool-choice: true
tool-call-parser: kimi_k3
reasoning-parser: kimi_k3
max-model-len: 1048576
stream-interval: 10
enable-prefix-caching: true
prefix-match-unit: 128
kv-cache-dtype: fp8
mamba-ssm-cache-dtype: float32
block-size: 128
attention-backend: ROCM_AITER_MLA
attention-config: '{"mla_prefill_backend":"ROCM_AITER_FA","use_prefill_query_quantization":true}'
# The native srt driver supplies the measured probabilistic DSpark4 AL for
# throughput and leaves real block rejection for eval. Draft unchanged.
speculative-config: '{"model":"/models/Inferact-Kimi-K3-DSpark","method":"dspark","num_speculative_tokens":4,"attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}'
kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"MoRIIOConnector","kv_role":"kv_producer","kv_load_failure_policy":"fail","kv_connector_extra_config":{"backend":"rdma"}},{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1799000000000,"lazy_offload":false}}]}}'
env: &common_env
VLLM_USE_V1: '1'
VLLM_ROCM_USE_AITER: '1'
VLLM_ROCM_USE_AITER_MOE_SITUV2_A8W4: '1'
AITER_SITUV2_A8W4: '1'
AITER_BF16_FP8_MOE_BOUND: '0'
VLLM_ROCM_AITER_MLA_ASM_PADDING: asm
VLLM_ROCM_AITER_MLA_DCP_VERIFY: asm
SAFETENSORS_FAST_GPU: '1'
VLLM_SSM_CONV_STATE_LAYOUT: DS
VLLM_KV_CACHE_LAYOUT: HND
NCCL_DMABUF_ENABLE: '0'
HSA_ENABLE_IPC_MODE_LEGACY: '1'
HIP_FORCE_DEV_KERNARG: '1'
PYTHONHASHSEED: '42'
PREFIX_CACHING_HASH_ALGO: sha256
VLLM_USE_DIRECT_DCP_A2A: '0'
VLLM_USE_DIRECT_DCP_Q_GATHER: '0'
VLLM_USE_DIRECT_DCP_KV_GATHER: '0'
VLLM_ALLOW_DCP_FULL_CUDAGRAPH: '1'
VLLM_ENGINE_READY_TIMEOUT_S: '7200'
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1200'
PYTHONNOUSERSITE: '1'
VLLM_SERVER_DEV_MODE: '0'
VLLM_DISABLE_REQUEST_ID_RANDOMIZATION: '1'
TORCH_NCCL_BLOCKING_WAIT: '0'
NCCL_BLOCKING_WAIT: '0'
MORI_IO_SQ_BACKOFF_TIMEOUT_US: '50000'
MORI_IO_QP_MAX_SEND_WR: '16384'
MORI_IO_QP_MAX_CQE: '32768'
MORI_IO_QP_MAX_SGE: '2'
MORI_IO_TC_DISABLE: '0'
decode:
nodes: 1
workers: 1
gpus: 8
args:
<<: *common_args
kv-transfer-config: '{"kv_connector":"MoRIIOConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_connector_extra_config":{"backend":"rdma"}}'
env:
<<: *common_env
VLLM_USE_BREAKABLE_CUDAGRAPH: '0'
benchmark:
type: custom
client_placement: head
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
MODEL: moonshotai/Kimi-K3
PORT: '8000'
IS_MULTINODE: 'true'
KV_OFFLOADING: dram
TOTAL_CPU_DRAM_GB: '1799'
AIPERF_FAILED_REQUEST_THRESHOLD: '0.01'
AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.01'
AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache

override_fp32_c1:
roles:
prefill:
args:
max-num-seqs: 2
max-num-batched-tokens: 16384
speculative-config: '{"model":"/models/Inferact-Kimi-K3-DSpark","method":"dspark","num_speculative_tokens":7,"attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}'
kv-transfer-config: '{"kv_connector":"MoRIIOConnector","kv_role":"kv_producer","kv_load_failure_policy":"fail","kv_connector_extra_config":{"backend":"rdma"}}'
compilation-config: '{"mode":3,"cudagraph_mode":"PIECEWISE","max_cudagraph_capture_size":128,"cudagraph_capture_sizes":[1,16,32,64,128],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}'
env:
VLLM_USE_BREAKABLE_CUDAGRAPH: '1'
decode:
args:
max-num-seqs: 2
max-num-batched-tokens: 512
speculative-config: '{"model":"/models/Inferact-Kimi-K3-DSpark","method":"dspark","num_speculative_tokens":7,"attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}'
compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":16,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}'
benchmark:
concurrencies: [1]
env:
KV_OFFLOADING: none
TOTAL_CPU_DRAM_GB: '0'

override_fp32_c10:
roles:
prefill:
args:
max-num-seqs: 20
max-num-batched-tokens: 8192
compilation-config: '{"mode":3,"cudagraph_mode":"PIECEWISE","max_cudagraph_capture_size":128,"cudagraph_capture_sizes":[1,16,32,64,128],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}'
env:
VLLM_USE_BREAKABLE_CUDAGRAPH: '1'
decode:
args:
max-num-seqs: 20
max-num-batched-tokens: 512
compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":80,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}'
benchmark:
concurrencies: [10]

override_fp32_c24:
roles:
prefill:
args:
decode-context-parallel-size: 8
max-num-seqs: 48
max-num-batched-tokens: 8192
compilation-config: '{"mode":3,"cudagraph_mode":"NONE","max_cudagraph_capture_size":0,"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}'
env:
VLLM_USE_BREAKABLE_CUDAGRAPH: '0'
decode:
nodes: 2
workers: 2
args:
decode-context-parallel-size: 8
max-num-seqs: 24
max-num-batched-tokens: 512
compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":120,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,70,80,90,100,110,120],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}'
benchmark:
concurrencies: [24]

override_fp32_c48: &fp32_c48
roles: &fp32_c48_roles
prefill:
args:
decode-context-parallel-size: 8
max-num-seqs: 96
max-num-batched-tokens: 8192
compilation-config: '{"mode":3,"cudagraph_mode":"NONE","max_cudagraph_capture_size":0,"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}'
env:
VLLM_USE_BREAKABLE_CUDAGRAPH: '0'
# Short-tail segmented MLA can require ~4.1 GB of non-reclaimable scratch.
HSA_NO_SCRATCH_RECLAIM: '0'
decode: &fp32_c48_decode
args: &fp32_c48_decode_args
decode-context-parallel-size: 8
max-num-seqs: 96
max-num-batched-tokens: 512
compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":240,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128,130,140,150,160,170,180,190,200,210,220,230,240],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}'
benchmark:
concurrencies: [48]

override_fp32_c48_1p2d:
<<: *fp32_c48
roles:
<<: *fp32_c48_roles
decode:
<<: *fp32_c48_decode
nodes: 2
workers: 2
args:
<<: *fp32_c48_decode_args
max-num-seqs: 48
Loading
Loading