diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-debug.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-debug.sh new file mode 100644 index 0000000000..c594293d8f --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-debug.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash +set -eo pipefail + +apply_verified_patch() { + local package_root="$1" patch_file="$2" checksum="$3" + printf '%s %s\n' "$checksum" "$patch_file" | sha256sum --check --status + git -C "$package_root" apply --check --include='vllm/*' "$patch_file" + git -C "$package_root" apply --include='vllm/*' "$patch_file" +} + +main() { + # Native srt-slurm runs this preamble in both worker and router containers. + # Only the standalone router image is exempt; a worker with missing vLLM must fail. + if command -v vllm-router >/dev/null 2>&1 && ! command -v vllm >/dev/null 2>&1; then + echo 'Standalone vLLM Router image: no engine backport required' + exit 0 + fi + + # The pinned nightly includes #57700. Backport only #59164 and K3 EOF + # finalization to the installed wheel; do not replace compiled extensions. + # Remove each patch when its fix ships in the qualified official image. + package_root=$(python3 -c 'import importlib.util; from pathlib import Path; spec = importlib.util.find_spec("vllm"); assert spec and spec.submodule_search_locations, "vLLM package not found"; print(Path(next(iter(spec.submodule_search_locations))).parent)') + # Pin the latest reviewed #59164 head; its runtime diff is equivalent to + # the previously validated d5e6faa9 extraction. + zeroing_patch="$(dirname -- "${BASH_SOURCE[0]}")/k3-moriio-sync-zeroing.patch" + apply_verified_patch "$package_root" "$zeroing_patch" 3da3746e85d53e4a3b17062b4113475d31a86cc07418e87ef5b4bb0c906cad20 + echo 'Applied vLLM#59164 at c5b1350f1f2bf10a128127a9b85e93d5f9f18e62' + eof_patch="$(dirname -- "${BASH_SOURCE[0]}")/k3-streaming-eof.patch" + apply_verified_patch "$package_root" "$eof_patch" a48173e44cd527deaa19d4b915b8684a1bf0ca38c5270018cff4384f87489f84 + echo 'Applied Kimi-K3 EOF subset; runtime equivalent to YukioZzz/vllm@535727bcc0156070fe0f4a3b5e46f14e9ba841d5' +} + +if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then + main "$@" +fi diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-sync-zeroing.patch b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-sync-zeroing.patch new file mode 100644 index 0000000000..83e2e1d6cc --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-sync-zeroing.patch @@ -0,0 +1,92 @@ +diff --git a/vllm/distributed/kv_transfer/kv_connector/v1/base.py b/vllm/distributed/kv_transfer/kv_connector/v1/base.py +index 6df9d1cfb..192b4ea91 100644 +--- a/vllm/distributed/kv_transfer/kv_connector/v1/base.py ++++ b/vllm/distributed/kv_transfer/kv_connector/v1/base.py +@@ -543,6 +543,16 @@ class KVConnectorBase_V1(ABC): + """ + pass + ++ def get_sync_load_block_ids(self, request: "Request") -> list[int]: ++ """Return blocks whose synchronous load replaces worker zeroing. ++ ++ Called after update_state_after_alloc. Each returned block must be ++ fully initialized before forward consumes it. A failed load must abort ++ forward instead of recomputing with uninitialized blocks. ++ Defaults to no blocks. ++ """ ++ return [] ++ + @abstractmethod + def build_connector_meta( + self, scheduler_output: SchedulerOutput +diff --git a/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py b/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py +index b482e6a8c..a6d7ffedc 100644 +--- a/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py ++++ b/vllm/distributed/kv_transfer/kv_connector/v1/moriio/moriio_connector.py +@@ -303,6 +303,24 @@ class MoRIIOConnector(KVConnectorBase_V1, SupportsHMA): + assert self.connector_scheduler is not None + self.connector_scheduler.on_new_request(request) + ++ def get_sync_load_block_ids(self, request: "Request") -> list[int]: ++ scheduler = self.connector_scheduler ++ if ( ++ self.mode != MoRIIOMode.READ ++ or not self.kv_transfer_config.is_kv_consumer ++ or scheduler is None ++ or not scheduler._has_mamba ++ or self._vllm_config.cache_config.get_resolved_kv_cache_layout().name ++ not in ("LBHNC", "LBNHC") ++ ): ++ return [] ++ pending = scheduler._reqs_need_recv.get(request.request_id) ++ if pending is None: ++ return [] ++ # Hybrid READ fills these entire attention pages and aborts on failure. ++ # The scheduler excludes these IDs only from newly allocated page zeroing. ++ return pending[1][0] ++ + def build_connector_meta( + self, + scheduler_output: SchedulerOutput, +diff --git a/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py b/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py +index 73b6b0ccf..5cb75e17a 100644 +--- a/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py ++++ b/vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py +@@ -463,6 +463,12 @@ class MultiConnector(KVConnectorBase_V1, SupportsHMA): + for c in self._connectors: + c.on_new_request(request) + ++ def get_sync_load_block_ids(self, request: "Request") -> list[int]: ++ chosen = self._requests_to_connector.get(request.request_id) ++ if chosen is None: ++ return [] ++ return self._connectors[chosen].get_sync_load_block_ids(request) ++ + def build_connector_meta( + self, scheduler_output: SchedulerOutput + ) -> MultiKVConnectorMetadata: +diff --git a/vllm/v1/core/sched/scheduler.py b/vllm/v1/core/sched/scheduler.py +index fcb1421b1..a86e0810a 100644 +--- a/vllm/v1/core/sched/scheduler.py ++++ b/vllm/v1/core/sched/scheduler.py +@@ -349,7 +349,7 @@ class Scheduler(SchedulerInterface): + + self.has_mamba_layers = kv_cache_config.has_mamba_layers + self.needs_kv_cache_zeroing = kv_cache_config.needs_kv_cache_zeroing +- # Blocks that async KV loads will overwrite this step, skipped from ++ # Blocks that KV loads will overwrite this step, skipped from + # zeroing since the zeroing could race the out-of-band write. + self._skip_zero_block_ids: set[int] = set() + self.need_mamba_block_aligned_split = ( +@@ -1295,6 +1295,11 @@ class Scheduler(SchedulerInterface): + if num_external_computed_tokens > 0: + # load_kv_async is False here + has_sync_kv_loads = True ++ if self.needs_kv_cache_zeroing: ++ assert self.connector is not None ++ self._skip_zero_block_ids.update( ++ self.connector.get_sync_load_block_ids(request) ++ ) + if self.log_stats: + request.record_event( + EngineCoreEventType.SCHEDULED, scheduled_timestamp diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-streaming-eof.patch b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-streaming-eof.patch new file mode 100644 index 0000000000..58c659bdae --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/k3-streaming-eof.patch @@ -0,0 +1,141 @@ +diff --git a/vllm/parser/kimi_k3.py b/vllm/parser/kimi_k3.py +index 587e1d5ad..e1342aad0 100644 +--- a/vllm/parser/kimi_k3.py ++++ b/vllm/parser/kimi_k3.py +@@ -10,8 +10,9 @@ from vllm.entrypoints.generate.base.protocol import ( + DeltaMessage, + FunctionCall, + ) +-from vllm.parser.abstract_parser import DelegatingParser ++from vllm.parser.abstract_parser import DelegatingParser, StreamState + from vllm.reasoning.kimi_k3_reasoning_parser import KimiK3ReasoningParser ++from vllm.tool_parsers.kimi_k3_tool_parser import KimiK3ToolParser + + if TYPE_CHECKING: + from vllm.entrypoints.openai.chat_completion.protocol import ( +@@ -93,6 +94,33 @@ class KimiK3Parser(DelegatingParser): + delta_message.tool_calls = [] + return delta_message, False + ++ def finalize_generation( ++ self, ++ delta_message: DeltaMessage | None, ++ request: ChatCompletionRequest | ResponsesRequest, ++ state: StreamState, ++ ) -> DeltaMessage | None: ++ delta_message = super().finalize_generation(delta_message, request, state) ++ if not state.reasoning_ended and isinstance( ++ self._reasoning_parser, KimiK3ReasoningParser ++ ): ++ tail = self._reasoning_parser.finish_reasoning_streaming( ++ state.previous_text ++ ) ++ if tail: ++ if delta_message is None: ++ delta_message = DeltaMessage() ++ delta_message.reasoning = (delta_message.reasoning or "") + tail ++ elif state.reasoning_ended and isinstance(self._tool_parser, KimiK3ToolParser): ++ content_tail = self._tool_parser._extract_response_content( ++ state.previous_text, finished=True ++ ) ++ if content_tail: ++ if delta_message is None: ++ delta_message = DeltaMessage() ++ delta_message.content = (delta_message.content or "") + content_tail ++ return delta_message ++ + def parse_delta( + self, + delta_text: str, +@@ -116,14 +144,16 @@ class KimiK3Parser(DelegatingParser): + self._tool_parser is not None + or not isinstance(self._reasoning_parser, KimiK3ReasoningParser) + or not state.reasoning_ended +- or delta_message is None + ): + return delta_message + + stripped = self._reasoning_parser.strip_content_streaming( + previous_text=previous_content, + current_text=state.previous_text, ++ finished=finished, + ) ++ if delta_message is None: ++ return stripped + delta_message.content = stripped.content if stripped is not None else None + if ( + delta_message.role is None +diff --git a/vllm/reasoning/kimi_k3_reasoning_parser.py b/vllm/reasoning/kimi_k3_reasoning_parser.py +index 1a1c62ba6..5ab9a139e 100644 +--- a/vllm/reasoning/kimi_k3_reasoning_parser.py ++++ b/vllm/reasoning/kimi_k3_reasoning_parser.py +@@ -359,7 +359,16 @@ class KimiK3ReasoningParser(ReasoningParser): + break + return text[:-overlap] if overlap else text + +- def _content_ready_to_emit(self, text: str) -> str: ++ def finish_reasoning_streaming(self, text: str) -> str: ++ """Release an unfinished marker as literal text at end of stream.""" ++ if self._think_close_re.search(text): ++ return "" ++ safe = self._reasoning_text_ready_to_emit(text) ++ opened = self._think_open_re.search(text) ++ body = text[opened.end() :] if opened is not None else text ++ return body[len(safe) :] ++ ++ def _content_ready_to_emit(self, text: str, *, finished: bool = False) -> str: + """Return the content prefix that is safe to stream now. + + Mirrors ``_reasoning_text_ready_to_emit`` but for the post-reasoning +@@ -376,6 +385,9 @@ class KimiK3ReasoningParser(ReasoningParser): + text = self._response_close_re.sub("", text) + text = self._message_close_re.sub("", text) + ++ if finished: ++ return text ++ + # Hold back partial markers at the end + overlap = 0 + for marker in ( +@@ -394,6 +406,8 @@ class KimiK3ReasoningParser(ReasoningParser): + self, + previous_text: str, + current_text: str, ++ *, ++ finished: bool = False, + ) -> DeltaMessage | None: + """Strip XTML content wrappers from streaming deltas after reasoning. + +@@ -404,7 +418,7 @@ class KimiK3ReasoningParser(ReasoningParser): + Works from accumulated text (``previous_text`` / ``current_text`` + already contain only post-reasoning content). + """ +- current_safe = self._content_ready_to_emit(current_text) ++ current_safe = self._content_ready_to_emit(current_text, finished=finished) + previous_safe = self._content_ready_to_emit(previous_text) + if current_safe.startswith(previous_safe): + delta = current_safe[len(previous_safe) :] +diff --git a/vllm/tool_parsers/kimi_k3_tool_parser.py b/vllm/tool_parsers/kimi_k3_tool_parser.py +index 76c6b0aa6..8fcc2fdaa 100644 +--- a/vllm/tool_parsers/kimi_k3_tool_parser.py ++++ b/vllm/tool_parsers/kimi_k3_tool_parser.py +@@ -268,7 +268,9 @@ class KimiK3ToolParser(ToolParser): + return m["c"] or None + return self._strip_response_content(before) + +- def _extract_response_content(self, current_text: str) -> str | None: ++ def _extract_response_content( ++ self, current_text: str, *, finished: bool = False ++ ) -> str | None: + # Streaming response text is computed from the accumulated text. This is + # what keeps split markers from leaking: + # <|open|> / response / <|sep|>Hi -> emit only "Hi" after open closes +@@ -289,6 +291,8 @@ class KimiK3ToolParser(ToolParser): + candidates = [i for i in (tools_start, response_end) if i != -1] + if candidates: + sendable_idx = min(candidates) ++ elif finished: ++ sendable_idx = len(current_text) + else: + overlap = max( + _partial_tag_overlap(current_text, self.response_open), diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml new file mode 100644 index 0000000000..6319298c83 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml @@ -0,0 +1,224 @@ +# Native Kimi-K3 FP32 SSM PD on an immutable official ROCm nightly. +base: + schema: 2 + name: kimik3-fp4-mi355x-vllm-disagg-agentic + # Temporary #59164 and K3 EOF backports; remove as qualified nightlies ship them. + setup_script: k3-moriio-debug.sh + model: + path: Kimi-K3 + container: vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19 + precision: fp4 + slurm: + time_limit: '08:00:00' + sbatch_directives: + mem: '0' + srun_options: + container-remap-root: '' + container-writable: '' + mem: '0' + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: vllm-router + enable_multiple_frontends: false + orchestrator_placement: head + container_image: docker.io/vllm/vllm-router@sha256:1fbf06701abce8c8cd459414e472d63999c05b201baffc62e00727c10583f5ea + args: + policy: consistent_hash + prefill-policy: consistent_hash + decode-policy: consistent_hash + log-level: info + observability: + enabled: false + tachometer: + enabled: false + health_check: + interval_seconds: 10 + max_attempts: 90 + engine: + type: vllm + connector: moriio + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 8 + args: &common_args + served-model-name: moonshotai/Kimi-K3 + trust-remote-code: true + tensor-parallel-size: 8 + decode-context-parallel-size: 1 + cp-kv-cache-interleave-size: 1 + dcp-comm-backend: a2a + moe-backend: auto + load-format: fastsafetensors + gpu-memory-utilization: 0.90 + language-model-only: true + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + max-model-len: 1048576 + stream-interval: 10 + enable-prefix-caching: true + prefix-match-unit: 128 + kv-cache-dtype: fp8 + mamba-ssm-cache-dtype: float32 + block-size: 128 + attention-backend: ROCM_AITER_MLA + attention-config: '{"mla_prefill_backend":"ROCM_AITER_FA","use_prefill_query_quantization":true}' + # The native srt driver supplies the measured probabilistic DSpark4 AL for + # throughput and leaves real block rejection for eval. Draft unchanged. + speculative-config: '{"model":"/models/Inferact-Kimi-K3-DSpark","method":"dspark","num_speculative_tokens":4,"attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_connector_extra_config":{"connectors":[{"kv_connector":"MoRIIOConnector","kv_role":"kv_producer","kv_load_failure_policy":"fail","kv_connector_extra_config":{"backend":"rdma"}},{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":1799000000000,"lazy_offload":false}}]}}' + env: &common_env + VLLM_USE_V1: '1' + VLLM_ROCM_USE_AITER: '1' + VLLM_ROCM_USE_AITER_MOE_SITUV2_A8W4: '1' + AITER_SITUV2_A8W4: '1' + AITER_BF16_FP8_MOE_BOUND: '0' + VLLM_ROCM_AITER_MLA_ASM_PADDING: asm + VLLM_ROCM_AITER_MLA_DCP_VERIFY: asm + SAFETENSORS_FAST_GPU: '1' + VLLM_SSM_CONV_STATE_LAYOUT: DS + VLLM_KV_CACHE_LAYOUT: HND + NCCL_DMABUF_ENABLE: '0' + HSA_ENABLE_IPC_MODE_LEGACY: '1' + HIP_FORCE_DEV_KERNARG: '1' + PYTHONHASHSEED: '42' + PREFIX_CACHING_HASH_ALGO: sha256 + VLLM_USE_DIRECT_DCP_A2A: '0' + VLLM_USE_DIRECT_DCP_Q_GATHER: '0' + VLLM_USE_DIRECT_DCP_KV_GATHER: '0' + VLLM_ALLOW_DCP_FULL_CUDAGRAPH: '1' + VLLM_ENGINE_READY_TIMEOUT_S: '7200' + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1200' + PYTHONNOUSERSITE: '1' + VLLM_SERVER_DEV_MODE: '0' + VLLM_DISABLE_REQUEST_ID_RANDOMIZATION: '1' + TORCH_NCCL_BLOCKING_WAIT: '0' + NCCL_BLOCKING_WAIT: '0' + MORI_IO_SQ_BACKOFF_TIMEOUT_US: '50000' + MORI_IO_QP_MAX_SEND_WR: '16384' + MORI_IO_QP_MAX_CQE: '32768' + MORI_IO_QP_MAX_SGE: '2' + MORI_IO_TC_DISABLE: '0' + decode: + nodes: 1 + workers: 1 + gpus: 8 + args: + <<: *common_args + kv-transfer-config: '{"kv_connector":"MoRIIOConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_connector_extra_config":{"backend":"rdma"}}' + env: + <<: *common_env + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + benchmark: + type: custom + client_placement: head + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + MODEL: moonshotai/Kimi-K3 + PORT: '8000' + IS_MULTINODE: 'true' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '1799' + AIPERF_FAILED_REQUEST_THRESHOLD: '0.01' + AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.01' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + +override_fp32_c1: + roles: + prefill: + args: + max-num-seqs: 2 + max-num-batched-tokens: 16384 + speculative-config: '{"model":"/models/Inferact-Kimi-K3-DSpark","method":"dspark","num_speculative_tokens":7,"attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + kv-transfer-config: '{"kv_connector":"MoRIIOConnector","kv_role":"kv_producer","kv_load_failure_policy":"fail","kv_connector_extra_config":{"backend":"rdma"}}' + compilation-config: '{"mode":3,"cudagraph_mode":"PIECEWISE","max_cudagraph_capture_size":128,"cudagraph_capture_sizes":[1,16,32,64,128],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + env: + VLLM_USE_BREAKABLE_CUDAGRAPH: '1' + decode: + args: + max-num-seqs: 2 + max-num-batched-tokens: 512 + speculative-config: '{"model":"/models/Inferact-Kimi-K3-DSpark","method":"dspark","num_speculative_tokens":7,"attention_backend":"ROCM_AITER_MLA","kv_cache_dtype":"fp8","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' + compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":16,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + benchmark: + concurrencies: [1] + env: + KV_OFFLOADING: none + TOTAL_CPU_DRAM_GB: '0' + +override_fp32_c10: + roles: + prefill: + args: + max-num-seqs: 20 + max-num-batched-tokens: 8192 + compilation-config: '{"mode":3,"cudagraph_mode":"PIECEWISE","max_cudagraph_capture_size":128,"cudagraph_capture_sizes":[1,16,32,64,128],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + env: + VLLM_USE_BREAKABLE_CUDAGRAPH: '1' + decode: + args: + max-num-seqs: 20 + max-num-batched-tokens: 512 + compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":80,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + benchmark: + concurrencies: [10] + +override_fp32_c24: + roles: + prefill: + args: + decode-context-parallel-size: 8 + max-num-seqs: 48 + max-num-batched-tokens: 8192 + compilation-config: '{"mode":3,"cudagraph_mode":"NONE","max_cudagraph_capture_size":0,"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + env: + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + decode: + nodes: 2 + workers: 2 + args: + decode-context-parallel-size: 8 + max-num-seqs: 24 + max-num-batched-tokens: 512 + compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":120,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,70,80,90,100,110,120],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + benchmark: + concurrencies: [24] + +override_fp32_c48: &fp32_c48 + roles: &fp32_c48_roles + prefill: + args: + decode-context-parallel-size: 8 + max-num-seqs: 96 + max-num-batched-tokens: 8192 + compilation-config: '{"mode":3,"cudagraph_mode":"NONE","max_cudagraph_capture_size":0,"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + env: + VLLM_USE_BREAKABLE_CUDAGRAPH: '0' + # Short-tail segmented MLA can require ~4.1 GB of non-reclaimable scratch. + HSA_NO_SCRATCH_RECLAIM: '0' + decode: &fp32_c48_decode + args: &fp32_c48_decode_args + decode-context-parallel-size: 8 + max-num-seqs: 96 + max-num-batched-tokens: 512 + compilation-config: '{"mode":3,"cudagraph_mode":"FULL_DECODE_ONLY","max_cudagraph_capture_size":240,"cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128,130,140,150,160,170,180,190,200,210,220,230,240],"custom_ops":["+fused_rms_norm_gated","+quant_fp8","+grouped_topk","+sparse_attn_indexer","none"]}' + benchmark: + concurrencies: [48] + +override_fp32_c48_1p2d: + <<: *fp32_c48 + roles: + <<: *fp32_c48_roles + decode: + <<: *fp32_c48_decode + nodes: 2 + workers: 2 + args: + <<: *fp32_c48_decode_args + max-num-seqs: 48 diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index 5df8735385..00e0309e90 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -1525,3 +1525,90 @@ dsv41flash-fp4-mi355x-sglang-agentic-dspark: - dram-utilization: 0.60 search-space: - { tp: 4, ep: 4, dp-attn: false, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/mi355x-fp4-mtp/agentic.yaml } + +# Native Kimi-K3 PD with FP32 SSM state. +kimik3-fp4-mi355x-vllm-disagg-agentic: + image: vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19 + model: moonshotai/Kimi-K3 + model-prefix: kimik3 + runner: cluster:mi355x-amds + precision: fp4 + framework: vllm-disagg + router: { name: vllm-router, version: nightly-20260913-83944c4 } + kv-p2p-transfer: moriio + multinode: true + disagg: true + scenarios: + agentic-coding: + - dram-utilization: 0.6 + search-space: + - spec-decoding: mtp + conc-list: [1] + kv-offloading: none + prefill: + num-worker: 1 + tp: 8 + dcp-size: 1 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c1 + decode: { num-worker: 1, tp: 8, dcp-size: 1, ep: 1, dp-attn: false } + - spec-decoding: mtp + conc-list: [10] + kv-offloading: dram + kv-offload-backend: { name: vllm-simple } + prefill: + num-worker: 1 + tp: 8 + dcp-size: 1 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c10 + - TOTAL_CPU_DRAM_GB=1799 + decode: { num-worker: 1, tp: 8, dcp-size: 1, ep: 1, dp-attn: false } + - spec-decoding: mtp + conc-list: [24] + kv-offloading: dram + kv-offload-backend: { name: vllm-simple } + prefill: + num-worker: 1 + tp: 8 + dcp-size: 8 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c24 + - TOTAL_CPU_DRAM_GB=1799 + decode: { num-worker: 2, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } + - spec-decoding: mtp + conc-list: [48] + kv-offloading: dram + kv-offload-backend: { name: vllm-simple } + prefill: + num-worker: 1 + tp: 8 + dcp-size: 8 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c48 + # Recipe enables P-only scratch reclaim; D and GMU0.90 are unchanged. + - TOTAL_CPU_DRAM_GB=1799 + decode: { num-worker: 1, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } + - spec-decoding: mtp + conc-list: [48] + kv-offloading: dram + kv-offload-backend: { name: vllm-simple } + prefill: + num-worker: 1 + tp: 8 + dcp-size: 8 + ep: 1 + dp-attn: false + additional-settings: + - CONFIG_FILE=recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml:override_fp32_c48_1p2d + # Inherits the same P-only scratch-reclaim setting as 1P1D c48. + - TOTAL_CPU_DRAM_GB=1799 + decode: { num-worker: 2, tp: 8, dcp-size: 8, ep: 1, dp-attn: false } diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index 8036d3de69..c12f8a8fd8 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -771,6 +771,7 @@ clusters: MORI_RDMA_TC: "104" models: entries: + Kimi-K3: {root: it-share-data, dir: Kimi-K3} DeepSeek-V4-Pro-0813: {root: it-share-data, dir: DeepSeek-V4-Pro-0813} GLM-5.3: {root: it-share-data, dir: GLM-5.3} Qwen3.5-397B-A17B-FP8: {root: it-share-data, dir: Qwen3.5-397B-A17B-FP8} @@ -786,13 +787,21 @@ clusters: shared-hf-hub-cache: {path: /it-share/hf-hub-cache} aiperf-cache: {path: /it-share/aiperf-cache} it-share-data: {path: /it-share/data} + k3-draft: {path: /it-share/data/Inferact-Kimi-K3-DSpark} + # Selected only by the Kimi-K3 vLLM PD lane; never replace libibverbs core. + ionic-provider: {path: /usr/lib/x86_64-linux-gnu/libibverbs/libionic-rdmav34.so, visibility: node-local} squash: dir: /it-share/gharunners2/srt-slurm/containers visibility: shared import: compute - multi-node-import: false + multi-node-import: true + import-step-args: [--cpus-per-task=4] srt-slurm: network-interface: eno0 + env: + NCCL_IB_HCA: rdma0,rdma1,rdma2,rdma3,rdma4,rdma5,rdma6,rdma7 + NCCL_SOCKET_IFNAME: eno0 + GLOO_SOCKET_IFNAME: eno0 single-node-time-limit: 500 gpus-per-node-directive: true segment-directive: false @@ -805,11 +814,14 @@ clusters: nodes: all volume-mounts: shared-hf-hub-cache: /hf_hub_cache/hub + k3-draft: /models/Inferact-Kimi-K3-DSpark mounts: /dev/kfd: /dev/kfd /dev/dri: /dev/dri + /dev/infiniband: /dev/infiniband /it-share/hf_home: /it-share/hf_home extra: visible_devices_env: ROCR_VISIBLE_DEVICES default_gpu_exporter: null nginx_raise_ulimit: false + default_bash_preamble: ulimit -l unlimited diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index d2df756784..548cc36d5c 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -105,6 +105,14 @@ and preserve resources used by other jobs. ## Procedure index +The Kimi-K3 native PD lane resolves the selected recipe before image provisioning, requires its worker image to match the matrix, and stages the recipe's frontend image through the same backend. This keeps router images recipe-owned and rejects mismatched inputs before an import allocation. Its staged target model, draft mount, fabric devices, worker network environment and memlock preamble are declared in the cluster record; image imports, submission, cancellation and artifact collection use the shared Python launcher. + +The Kimi-K3 vLLM debug setup uses digest-pinned official ROCm nightly `ac68c3087215e0a4f3cdfa218508c6aada57235d`, which already includes #57700, with only #59164 READ-zeroing and the K3 streaming-EOF runtime subset. It omits the heartbeat and other #58968 feature groups. The pinned revisions, tested scope and independent removal conditions are maintained in [Native Kimi-K3 PD](k3-pd-native.md). Serving settings, automatic DSpark acceptance and benchmark criteria are unchanged. + +The Kimi-K3 vLLM PD lane also selects the cluster's node-local `ionic-provider` volume as a read-only single-file mount. This is a temporary compatibility prerequisite for the pinned image's Ionic userspace provider and the host kernel ABI, independent of shared-MR. Other models and frameworks do not select it. Read-only lane assets are neither created nor chmodded on the launch host; the container runtime validates them on the allocated compute node. Keep the image's `libibverbs` core and all unrelated providers unchanged. Remove the volume and its lane entry only after the replacement image, without the mount, enumerates and opens/closes every expected RDMA device on the target fleet. Device-open success is not MR, traffic, throughput, or shutdown qualification. + +The shared Slurm image resolver maps explicit Docker Hub hosts (`docker.io`, `index.docker.io`, and `registry-1.docker.io`) to Enroot's `registry-1.docker.io#repository` registry endpoint, for both slash and `#` input forms. Digest pins are preserved even when a tag is also present. Import validation must exercise the resolved reference against the registry with a cold, task-private cache; a cached squash or mocked import does not validate that boundary. + 1. [Prepare a worktree](#prepare-a-worktree) 2. [Add a model + hardware recipe](#add-a-model--hardware-recipe) 3. [Change a master config](#change-a-master-config) diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 0d023deb1c..c7e83a64b1 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -87,6 +87,14 @@ PowerX 严格校验。现有的 Tachometer 1000 ms / 功耗 exporter 100 ms 采 ## 规程索引 +Kimi-K3 原生 PD 路径在准备镜像前解析所选 recipe,要求 worker 镜像与矩阵一致,并通过同一后端准备 recipe 指定的 frontend 镜像。Router 镜像仍由 recipe 管理;输入不一致会在镜像导入任务申请资源前失败。已部署的目标模型、draft 挂载、网络设备、worker 网络环境及 memlock 前置命令由集群记录声明;镜像导入、任务提交、取消及产物收集沿用共享 Python launcher。 + +Kimi-K3 vLLM debug setup 使用按 digest 固定且已包含 #57700 的官方 ROCm nightly `ac68c3087215e0a4f3cdfa218508c6aada57235d`,仅叠加 #59164 READ-zeroing 和 K3 streaming-EOF 运行时子集,不携带心跳或 #58968 其他特性组。固定版本、已验证范围及独立删除条件统一记录于[原生 Kimi-K3 PD](k3-pd-native_zh.md)。Serving 设置、自动选择的 DSpark acceptance 及 benchmark 判据不变。 + +Kimi-K3 vLLM PD 路径还会选择集群声明的节点本地 `ionic-provider` 卷,以只读方式挂载单个文件。这是固定镜像内的 Ionic 用户态 provider 与主机内核 ABI 之间的临时兼容前置条件,与 shared-MR 无关;其他模型和框架不选择此挂载。Launcher 不会在提交主机上创建只读资产或修改其权限,由容器运行时在已分配的计算节点上验证。保留镜像原有的 `libibverbs` 核心库及其他 provider。只有替换镜像在不挂载 provider 的情况下,能在目标集群枚举并成功打开、关闭全部预期 RDMA 设备,才可移除该卷及路径选择项。设备打开成功不能代替 MR、实际传输、吞吐或退出验证。 + +共享 Slurm 镜像解析器将显式 Docker Hub 主机名(`docker.io`、`index.docker.io` 和 `registry-1.docker.io`)统一映射为 Enroot 的 `registry-1.docker.io#repository` 仓库端点,同时支持斜杠和 `#` 两种输入形式。即使输入还带有 tag,也会保留固定 digest。导入验证必须使用任务独享的空缓存,将解析后的引用交给真实仓库;命中已有 squash 缓存或模拟导入不能证明这一衔接有效。 + 1. [准备 worktree](#准备-worktree) 2. [添加模型 + 硬件配方](#添加模型--硬件配方) 3. [修改主配置](#修改主配置) diff --git a/inferencex-e2e/docs/k3-pd-native.md b/inferencex-e2e/docs/k3-pd-native.md new file mode 100644 index 0000000000..7e25d186fc --- /dev/null +++ b/inferencex-e2e/docs/k3-pd-native.md @@ -0,0 +1,45 @@ +# Native Kimi-K3 PD on MI355X + +**English** | [中文](k3-pd-native_zh.md) + +This draft integrates Kimi-K3 prefill/decode disaggregation through InferenceX's shared Python launcher and native srt-slurm lifecycle. There is no alternate launcher or router algorithm. + +## Configuration ownership + +- `configs/amd-master.yaml` selects `kimik3-fp4-mi355x-vllm-disagg-agentic` and supplies matrix identity and result metadata. +- `benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml` owns the topology, worker/router images and serving settings: MXFP4 target weights, FP32 SSM, FP8 KV and GMU 0.90. +- The existing variants remain 1P1D c1/c10/c48 and 1P2D c24/c48. The latency variant uses DSpark K7 without CPU offload; the others use K4 with prefill SimpleCPUOffload. Draft weights and precision are unchanged. Throughput uses automatic golden acceptance selection; eval uses real verification. +- `configs/runners.yaml` declares staged models, draft/fabric mounts, worker network settings, memlock and image-import policy. `infx/launch/` uses the job-local srt-slurm Python to resolve recipe images, rejects worker-image disagreement before import, stages the recipe-owned router image, and retains shared submission, cancellation and result collection. + +High-concurrency prefill retains `HSA_NO_SCRATCH_RECLAIM=0`. Existing graph modes, transport settings and 1% request-error gates are unchanged. No BF16 SSM, workspace-development, shared-MR, QP/credit implementation or client cancellation patch is added. + +## Pinned runtime and removable debug layer + +The worker image is `vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19`. Master and recipe use the same immutable identity. This image already includes vLLM #57700; no backport of that PR remains. + +The temporary engine setup applies only two checked-in Python runtime patches, verifying SHA256 and `git apply --check` before applying each. It neither replaces compiled extensions nor downloads a moving PR head. A standalone router skips engine setup; a worker with missing vLLM or an incompatible patch fails startup. + +| Dependency | Pinned source | Purpose | +| --- | --- | --- | +| [vLLM #59164](https://github.com/vllm-project/vllm/pull/59164) | `c5b1350f1f2bf10a128127a9b85e93d5f9f18e62` | Exclude synchronous READ destinations from newly allocated KV-page zeroing. | +| Kimi-K3 streaming EOF | [standalone commit](https://github.com/YukioZzz/vllm/commit/535727bcc0156070fe0f4a3b5e46f14e9ba841d5) | Preserve buffered text at generation end in the appropriate output channel. Runtime-only extraction from #58968 at `7595667abf90c03b546c92189d05e5d73392910e`; no full-PR dependency. | +| [srt-slurm #508](https://github.com/NVIDIA/srt-slurm/pull/508) | tested revision `51cee8904a0b402a834887a26008adb79b8cd26b` | Bind discovery topology inside connector templates in the matching job's disposable checkout. | + +Patch hashes: READ zeroing `3da3746e85d53e4a3b17062b4113475d31a86cc07418e87ef5b4bb0c906cad20`; EOF `a48173e44cd527deaa19d4b915b8684a1bf0ca38c5270018cff4384f87489f84`. The EOF runtime is identical to the standalone commit; its tests stay in vLLM rather than being duplicated here. The setup does not include heartbeat, FULL-context, draft-fence or transfer-ownership patches from #58968. + +The debug layer also retains the cluster-declared, read-only single-file `ionic-provider` mount. This is image/kernel ABI compatibility, not shared-MR. Keep the image's libibverbs core and unrelated providers unchanged. The srt-slurm dependency URL, submodule pin and AIPerf source are not replaced by this integration. + +## Removal conditions + +1. Once a qualified official worker image contains either engine fix, update master and recipe together and remove that patch/application step. Delete the worker setup script, recipe reference and its tests after both fixes ship. +2. Independently update the shared srt-slurm pin to a revision containing #508, then remove the job-local cherry-pick and its focused tests. A worker-image update does not upgrade srt-slurm. +3. Remove the provider mount only after the replacement image opens all expected RDMA devices without it on the target fleet. Device enumeration alone does not qualify RDMA traffic. +4. Append the performance changelog and complete the applicable smoke, sweep and eval gates before merge. These temporary engine patches remain a draft debug integration, not an exception to upstream-image policy. + +## Evidence and remaining qualification + +The standalone EOF CPU regression has 56 failures before the fix and 114 passing tests after it, without #59164. Tests cover buffered tails, empty terminal chunks, complete markers and reasoning suppression; this is not a claim that every malformed XTML or tool-call boundary is qualified. + +The matched [1P1D c48 one-hour run](https://github.com/billishyahao/InferenceMINI/actions/runs/36847087330) used the same immutable worker image, #59164 and byte-identical EOF runtime. Warmup: 531 valid / 0 empty; profile: 4040 valid / 4 empty (0.0989%). Submission, result export and native cleanup completed. This is supporting runtime evidence from a separately adapted harness, not a sweep on this PR's current commit or an accuracy certification. + +The earlier [InferenceX full-feature control](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36800732704) used a broader stack. The later #59164-only [InferenceX run](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36814088271) was cancelled while queued, without a GPU runner. Neither qualifies the consolidated candidate. The minimal combination still needs current-PR 1P2D, sweep/eval and loaded-service teardown qualification; omission of other independently identified fixes is not proof that their failure modes are impossible. diff --git a/inferencex-e2e/docs/k3-pd-native_zh.md b/inferencex-e2e/docs/k3-pd-native_zh.md new file mode 100644 index 0000000000..abc8c23373 --- /dev/null +++ b/inferencex-e2e/docs/k3-pd-native_zh.md @@ -0,0 +1,45 @@ +# MI355X 上的原生 Kimi-K3 PD + +[English](k3-pd-native.md) | **中文** + +此 draft 通过 InferenceX 共享 Python launcher 和原生 srt-slurm 生命周期接入 Kimi-K3 prefill/decode 分离,不增加另一套 launcher 或 router 算法。 + +## 配置归属 + +- `configs/amd-master.yaml` 选择 `kimik3-fp4-mi355x-vllm-disagg-agentic`,提供矩阵标识和结果元数据。 +- `benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/mi355x-fp4/agentx/disagg-variants.yaml` 管理拓扑、worker/router 镜像及 serving 设置:MXFP4 目标权重、FP32 SSM、FP8 KV 和 GMU 0.90。 +- 保留现有 1P1D c1/c10/c48 与 1P2D c24/c48。低延迟配置使用 DSpark K7 且不做 CPU offload;其他配置使用 K4 和 prefill SimpleCPUOffload。不改变 draft 权重或精度;吞吐自动选择 golden acceptance,eval 使用真实校验。 +- `configs/runners.yaml` 声明已部署模型、draft/fabric 挂载、worker 网络设置、memlock 和镜像导入策略。`infx/launch/` 使用任务私有 srt-slurm Python 解析 recipe 镜像,导入前拒绝 worker 镜像不一致,准备 recipe 指定的 router 镜像,并沿用共享提交、取消和结果收集。 + +高并发 prefill 保留 `HSA_NO_SCRATCH_RECLAIM=0`。原有图模式、传输设置及 1% 请求错误门槛不变。不加入 BF16 SSM、workspace 开发、shared-MR、QP/credit 实现或客户端取消补丁。 + +## 固定运行栈与可删除 debug 层 + +Worker 镜像为 `vllm/vllm-openai-rocm:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d@sha256:e3fdfb382f2b567718ab6de49a14f5d5695dad84efc6dfd9c38f661b1a763e19`,master 和 recipe 使用完全相同的不可变标识。镜像已包含 vLLM #57700,不再携带该 PR 的 backport。 + +临时 engine setup 仅应用两个随源码交付的 Python 运行时补丁,分别校验 SHA256 和 `git apply --check`,不替换编译扩展,也不下载浮动 PR head。独立 router 跳过 engine setup;worker 缺少 vLLM 或补丁不兼容时终止启动。 + +| 依赖 | 固定来源 | 作用 | +| --- | --- | --- | +| [vLLM #59164](https://github.com/vllm-project/vllm/pull/59164) | `c5b1350f1f2bf10a128127a9b85e93d5f9f18e62` | 从新分配 KV 页的清零列表中排除同步 READ 目标。 | +| Kimi-K3 streaming EOF | [独立 commit](https://github.com/YukioZzz/vllm/commit/535727bcc0156070fe0f4a3b5e46f14e9ba841d5) | 生成结束时把暂存文本保留在对应输出通道。仅提取 #58968 在 `7595667abf90c03b546c92189d05e5d73392910e` 的相关运行时代码,不依赖全量 PR。 | +| [srt-slurm #508](https://github.com/NVIDIA/srt-slurm/pull/508) | 已测版本 `51cee8904a0b402a834887a26008adb79b8cd26b` | 在匹配任务的临时 checkout 中补齐 connector 模板内的 discovery 拓扑绑定。 | + +补丁哈希:READ zeroing 为 `3da3746e85d53e4a3b17062b4113475d31a86cc07418e87ef5b4bb0c906cad20`;EOF 为 `a48173e44cd527deaa19d4b915b8684a1bf0ca38c5270018cff4384f87489f84`。EOF 运行时代码与独立 commit 一致;测试保留在 vLLM,不在此重复维护。不加入 #58968 的心跳、FULL-context、draft-fence 或传输所有权补丁。 + +Debug 层还保留集群声明的单文件只读 `ionic-provider` 挂载,用于镜像与内核 ABI 兼容,与 shared-MR 无关。镜像内 libibverbs 核心库和其他 provider 不变。本集成不替换 srt-slurm 仓库 URL、子模块 pin 或 AIPerf 源码。 + +## 删除条件 + +1. 官方 worker 镜像包含任一 engine 修复并通过验收后,同步更新 master/recipe,删除对应补丁及应用步骤。两项均发布后,删除 worker setup、recipe 引用及其测试。 +2. 独立升级共享 srt-slurm pin 到包含 #508 的版本,再删除任务内 cherry-pick 及专项测试。更换 worker 镜像不会升级 srt-slurm。 +3. 只有替换镜像在目标平台不挂载 provider 也能打开全部预期 RDMA 设备,才删除兼容挂载。设备枚举不能代替 RDMA 流量验收。 +4. 追加性能 changelog,并在合入前完成适用 smoke、sweep 和 eval。临时 engine 补丁仍属于 draft debug 集成,不代表获得上游镜像政策豁免。 + +## 证据与剩余验收 + +独立 EOF CPU 回归在修复前有 56 项失败,修复后完整测试文件 114 项通过,不依赖 #59164。测试覆盖暂存尾部、空终止 chunk、完整标记及 reasoning 隐藏;不代表所有畸形 XTML 或工具调用边界均已验收。 + +配对 [1P1D c48 一小时作业](https://github.com/billishyahao/InferenceMINI/actions/runs/36847087330) 使用相同不可变 worker 镜像、#59164 和字节一致的 EOF 运行时代码。Warmup 为 531 valid / 0 empty;profile 为 4040 valid / 4 empty(0.0989%),完成提交、结果导出及原生收尾。这是另一套适配 harness 的运行支持证据,不是本 PR 当前 commit 的 sweep 或精度认证。 + +此前 [InferenceX 完整特性对照](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36800732704) 使用更广的补丁栈;后续仅带 #59164 的 [InferenceX 作业](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36814088271) 在排队且未获得 GPU runner 时取消,均不构成本次收束候选的验收。最小组合仍需当前 PR 的 1P2D、sweep/eval 和已加载服务收尾验证;省略其他已定位修复不等于证明其故障不可能发生。 diff --git a/inferencex-e2e/infx/launch/backends/slurm/squash.py b/inferencex-e2e/infx/launch/backends/slurm/squash.py index a1d0a41172..47a4c1ee73 100644 --- a/inferencex-e2e/infx/launch/backends/slurm/squash.py +++ b/inferencex-e2e/infx/launch/backends/slurm/squash.py @@ -68,7 +68,8 @@ def enroot_uri(image: str) -> str: Enroot 3.x cannot parse ``tag@digest``, so a pinned image becomes ``registry#repository:digest`` (the digest is immutable, so the tag is dropped). - Pyxis-style ``registry#repo`` input is read as ``registry/repo``. + Pyxis-style ``registry#repo`` input is read as ``registry/repo``. Docker Hub's + public aliases resolve to its registry API, not the docker.io website. """ image = image.replace("#", "/", 1) without_digest, digest = image, "" @@ -79,8 +80,10 @@ def enroot_uri(image: str) -> str: registry, repository = first, without_digest.split("/", 1)[1] else: registry, repository = "registry-1.docker.io", without_digest + if registry in {"docker.io", "index.docker.io"}: + registry = "registry-1.docker.io" if not digest: - if registry == "registry-1.docker.io": + if registry == "registry-1.docker.io" and image == repository: return f"docker://{image}" return f"docker://{registry}#{repository}" directory, _, name = repository.rpartition("/") diff --git a/inferencex-e2e/infx/launch/drivers/srt/checkout.py b/inferencex-e2e/infx/launch/drivers/srt/checkout.py index 822ef43e3b..1912774cc0 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/checkout.py +++ b/inferencex-e2e/infx/launch/drivers/srt/checkout.py @@ -24,6 +24,14 @@ UV_INSTALLER = "https://astral.sh/uv/install.sh" +# Temporary debug layer: delete after the srt-slurm pin includes NVIDIA/srt-slurm#508. +SRT_DEBUG_PRS = { + ("kimik3", "vllm-disagg"): ( + ("https://github.com/NVIDIA/srt-slurm.git", "51cee8904a0b402a834887a26008adb79b8cd26b"), + ), +} + + @dataclass(frozen=True) class Checkout: """A job-local srt-slurm checkout with InferenceX recipes staged.""" @@ -49,6 +57,15 @@ def _checked(result: subprocess.CompletedProcess[str], action: str) -> None: raise LaunchError(f"{action} failed (exit {result.returncode})") +def apply_debug_prs(destination: Path, model_prefix: str, framework: str) -> None: + """Apply pinned upstream commits only to the matching job's disposable checkout.""" + for repository, commit in SRT_DEBUG_PRS.get((model_prefix, framework), ()): + # Cherry-pick needs the parent; depth=1 makes the change look like a root commit. + _git("-C", destination, "fetch", "--no-tags", "--depth=2", repository, commit) + _git("-C", destination, "cherry-pick", "--no-commit", commit) + print(f"Applied temporary upstream change {repository}@{commit}", flush=True) + + def checkout_dir(run: SrtRun, *, shared: bool) -> Path: """Where a multi-node checkout lives, named for its run: shared-run-root or the workspace.""" request = run.request @@ -84,6 +101,7 @@ def prepare_checkout(run: SrtRun, destination: Path, *, power: bool) -> Checkout ) for patch in sorted((run.workspace / PATCHES).glob("*.patch")): _git("-C", destination, "apply", patch) + apply_debug_prs(destination, run.request.model_prefix, run.request.framework) head = _git("-C", destination, "rev-parse", "HEAD", capture=True) if head != commit: raise LaunchError(f"srt-slurm checkout is at {head}, expected {commit}") diff --git a/inferencex-e2e/infx/launch/drivers/srt/config.py b/inferencex-e2e/infx/launch/drivers/srt/config.py index 569156c4c2..6388a0d6d1 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/config.py +++ b/inferencex-e2e/infx/launch/drivers/srt/config.py @@ -17,8 +17,8 @@ from infx.clusters.slurm import slurm_settings from infx.launch.context import LaunchError -from infx.launch.drivers.srt.lanes import srt_time_limit -from infx.launch.drivers.srt.recipe import HEALTH_ATTEMPTS +from infx.launch.drivers.srt.lanes import config_file, srt_time_limit +from infx.launch.drivers.srt.recipe import HEALTH_ATTEMPTS, recipe_images if TYPE_CHECKING: from infx.clusters import Cluster @@ -185,13 +185,21 @@ def create_volume_mounts(run: SrtRun) -> None: def lane_mounts(run: SrtRun, lane: SrtLane) -> list[tuple[str, str]]: - """The (host, container) mounts the lane adds for this request; their hosts are created.""" + """Request-selected mounts; only writable cache directories are created on the launcher. + + Read-only assets may be files present only on compute nodes. Leave their paths and + permissions untouched; the container runtime validates them on the allocated host. + """ mounts: list[tuple[str, str]] = [] for mount in lane.mounts: if mount.when(run.request): host = volume_path(run.cluster, mount.volume) - _create_dir(host, world_writable=mount.world_writable) - mounts.append((str(host), mount.target or str(host))) + target = mount.target or str(host) + if mount.read_only: + target += ":ro" + else: + _create_dir(host, world_writable=mount.world_writable) + mounts.append((str(host), target)) if run.request.framework == "tilert": mounts.append((str(run.workspace), "/infmax-workspace")) return mounts @@ -206,6 +214,11 @@ def write_lane_config( ) -> None: """Stage a multi-node job's images, create its mounts, and write its srtslurm.yaml.""" backend, request = run.backend, run.request + images = ( + recipe_images(run, checkout, config_file(request)) + if lane.stage_recipe_images is not None and lane.stage_recipe_images(request) + else [request.image] + ) container = backend.stage_image( request.image, framework=request.framework, model_prefix=request.model_prefix ).reference @@ -215,6 +228,11 @@ def write_lane_config( else None ) containers: dict[str, str] = {} + for image in images[1:]: + staged = backend.stage_image( + image, framework=request.framework, model_prefix=request.model_prefix + ).reference + containers[image] = containers[pyxis_spelling(image)] = staged if request.framework == "tilert": prefill_image = request.env["PREFILL_IMAGE"] containers[prefill_image] = backend.stage_image(prefill_image).reference diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index f3fb7db89a..756bc0d001 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -23,6 +23,7 @@ class LaneMount: volume: str target: str | None = None world_writable: bool = False + read_only: bool = False @dataclass(frozen=True) @@ -41,6 +42,7 @@ class SrtLane: time_limit: str | None = None long_time_limit: str | None = None long_time: Match | None = None + stage_recipe_images: Match | None = None _DYNAMO = any_of("dynamo-sglang", "dynamo-trt", "dynamo-vllm") @@ -106,7 +108,14 @@ class SrtLane: mounts=( LaneMount(Match(), "aiperf-cache", "/aiperf_mmap_cache"), LaneMount(Match(frameworks=any_of("tilert")), "it-share-data", "/models"), + # Temporary image/kernel ABI prerequisite, not a shared-MR or engine patch. + LaneMount( + Match(any_of("kimik3"), frameworks=any_of("vllm-disagg")), + "ionic-provider", + read_only=True, + ), ), + stage_recipe_images=Match(any_of("kimik3"), frameworks=any_of("vllm-disagg")), eval_unsets=( "roles.prefill.args.ep-dispatch-algorithm", "roles.decode.args.ep-dispatch-algorithm", diff --git a/inferencex-e2e/infx/launch/drivers/srt/recipe.py b/inferencex-e2e/infx/launch/drivers/srt/recipe.py index c03eda1734..aad7c5aaa6 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/recipe.py +++ b/inferencex-e2e/infx/launch/drivers/srt/recipe.py @@ -7,6 +7,7 @@ from __future__ import annotations import fnmatch +import json import re from collections.abc import Sequence from pathlib import Path @@ -14,10 +15,13 @@ import yaml +from infx.launch import proc from infx.launch.context import LaunchError if TYPE_CHECKING: + from infx.launch.drivers.srt.checkout import Checkout from infx.launch.drivers.srt.lanes import SrtLane + from infx.launch.drivers.srt.run import SrtRun from infx.launch.request import LaunchRequest RECIPES_MIRROR = Path("benchmarks/multi_node/srt-slurm-recipes") @@ -37,6 +41,32 @@ def recipe_mirror_path(workspace: Path, config_file: str) -> Path: return workspace / RECIPES_MIRROR / recipe_relpath(config_file).removeprefix("recipes/") +def recipe_images(run: SrtRun, checkout: Checkout, config_file: str) -> list[str]: + """Resolve images in the installed srtctl environment, not the launcher's interpreter.""" + path = recipe_mirror_path(run.workspace, config_file) + selector = config_file.partition(":")[2] + config = f"{path}:{selector}" if selector else str(path) + argv = [ + str(checkout.venv / "bin/python"), "-m", "infx.srt_slurm.recipe_images", + config, run.request.image, + ] # fmt: skip + try: + result = proc.run(argv, env=run.env, cwd=checkout.root, capture=True) + if result.returncode: + raise ValueError(f"resolver exited {result.returncode}: {result.stderr.strip()}") + images = json.loads(result.stdout) + if ( + not isinstance(images, list) + or not images + or images[0] != run.request.image + or not all(isinstance(image, str) and image for image in images) + ): + raise ValueError("resolver did not return the expected image list") + return images + except (OSError, ValueError) as error: + raise LaunchError(f"recipe image resolution failed for {config_file}: {error}") from error + + def rename_job(text: str, name: str) -> str: """Set the top-level ``name:``, the job name srtctl submits.""" return re.sub(r"(?m)^name:.*$", lambda _: f'name: "{name}"', text) diff --git a/inferencex-e2e/infx/srt_slurm/recipe_images.py b/inferencex-e2e/infx/srt_slurm/recipe_images.py new file mode 100644 index 0000000000..4dc88a66d5 --- /dev/null +++ b/inferencex-e2e/infx/srt_slurm/recipe_images.py @@ -0,0 +1,60 @@ +"""Resolve recipe images with the job-local srtctl, before importing any containers.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +import yaml + +from infx.srt_slurm.synthetic_acceptance import selected_recipes + + +def resolve_images(config: str, expected_worker: str) -> list[str]: + """Use native override expansion to validate and deduplicate one recipe's images.""" + path, _, selector = config.partition(":") + try: + raw = yaml.safe_load(Path(path).read_text()) + if not isinstance(raw, dict): + raise ValueError("recipe must be a mapping") + selected = selected_recipes(raw, selector or None) + if len(selected) != 1: + raise ValueError("image provisioning requires exactly one selected recipe") + recipe = selected[0][1] + worker = recipe["model"]["container"] + if worker != expected_worker: + raise ValueError("recipe model.container must match the matrix IMAGE") + frontend_config = recipe.get("frontend", {}) + if not isinstance(frontend_config, dict): + raise TypeError("frontend must be a mapping") + frontend = frontend_config.get("container_image") + images = [worker, *([frontend] if frontend is not None else [])] + if any( + not isinstance(image, str) or not image or any(c.isspace() for c in image) + for image in images + ): + raise ValueError("container identities must be non-empty strings without whitespace") + return list(dict.fromkeys(images)) + except (OSError, KeyError, TypeError, ValueError, yaml.YAMLError) as error: + raise ValueError(f"invalid recipe images for {config}: {error}") from error + + +def main() -> int: + """Print a JSON image list; failed resolution must stop the launcher before import.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("config") + parser.add_argument("expected_worker") + args = parser.parse_args() + try: + images = resolve_images(args.config, args.expected_worker) + except ValueError as error: + print(error, file=sys.stderr) + return 1 + print(json.dumps(images)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/inferencex-e2e/infx/tests/launch/fake_slurm.py b/inferencex-e2e/infx/tests/launch/fake_slurm.py index 851a62b06d..48403e2d72 100644 --- a/inferencex-e2e/infx/tests/launch/fake_slurm.py +++ b/inferencex-e2e/infx/tests/launch/fake_slurm.py @@ -28,13 +28,6 @@ *" init "*) mkdir -p "${@: -1}" ;; *" rev-parse HEAD"*) echo "$FAKE_SRT_COMMIT" ;; esac -""", - "uv": r""" -printf '%s\n' "$*" >> "$FAKE_LOG_DIR/uv.log" -if [[ "$1" == venv ]]; then - dest="${@: -1}"; mkdir -p "$dest/bin" - printf '#!/bin/bash\nexec "%s" "$@"\n' "$FAKE_PYTHON" > "$dest/bin/python"; chmod +x "$dest/bin/python" -fi """, "make": r""" printf '%s\n' "$*" >> "$FAKE_LOG_DIR/make.log" @@ -91,6 +84,24 @@ "rsync": r"""printf '%s\n' "$*" >> "$FAKE_LOG_DIR/rsync.log" """, } +_UV = r""" +import os, pathlib, subprocess, sys, sysconfig +argv = sys.argv[1:] +with open(os.path.join(os.environ["FAKE_LOG_DIR"], "uv.log"), "a") as handle: + handle.write(" ".join(argv) + "\n") +if argv[0] == "venv": + subprocess.run([sys.executable, "-m", "venv", "--without-pip", argv[-1]], check=True) +elif argv[:2] == ["pip", "install"]: + python = argv[argv.index("--python") + 1] + site = subprocess.check_output( + [python, "-c", "import sysconfig; print(sysconfig.get_path('purelib'))"], text=True + ).strip() + # Ordinary dependencies are shared, but srtctl is visible only in the job venv. + pathlib.Path(site, "fixture-dependencies.pth").write_text( + sysconfig.get_path("purelib") + "\n" + os.environ["FAKE_SRT_SOURCE"] + "/src\n" + ) +""" + _SRTCTL = r""" import json, os, pathlib, sys import yaml @@ -163,7 +174,7 @@ def install_fakes(directory: Path) -> Path: binary = directory / name binary.write_text(f"#!/bin/bash\n{body.strip()}\n") binary.chmod(0o755) - for name, body in {"srtctl": _SRTCTL, "sbatch": _SBATCH}.items(): + for name, body in {"uv": _UV, "srtctl": _SRTCTL, "sbatch": _SBATCH}.items(): binary = directory / name binary.write_text(f"#!{sys.executable}\n{body.lstrip()}") binary.chmod(0o755) @@ -241,12 +252,12 @@ def base_env(*, fakes: Path, logs: Path, workspace: Path, sandbox: Path) -> dict } env.update( PATH=f"{fakes}{os.pathsep}{Path(sys.executable).parent}{os.pathsep}/usr/bin{os.pathsep}/bin", - PYTHONPATH=f"{ROOT}{os.pathsep}{ROOT / 'utils/srt-slurm/src'}", + PYTHONPATH=str(ROOT), HOME=str(sandbox / "home"), USER="runner", GITHUB_WORKSPACE=str(workspace), FAKE_LOG_DIR=str(logs), - FAKE_PYTHON=sys.executable, + FAKE_SRT_SOURCE=str(ROOT / "utils/srt-slurm"), FAKE_SRT_COMMIT="0123456789abcdef0123456789abcdef01234567", ENROOT_IMPORT_TIME_LIMIT="10", SALLOC_TIME_LIMIT="10", diff --git a/inferencex-e2e/infx/tests/launch/test_k3_moriio_setup.py b/inferencex-e2e/infx/tests/launch/test_k3_moriio_setup.py new file mode 100644 index 0000000000..1df332b20b --- /dev/null +++ b/inferencex-e2e/infx/tests/launch/test_k3_moriio_setup.py @@ -0,0 +1,91 @@ +"""The temporary worker backport must not execute in a standalone router image.""" + +import hashlib +import os +import subprocess +from pathlib import Path + +import pytest + +SCRIPT = ( + Path(__file__).resolve().parents[3] + / "benchmarks/multi_node/srt-slurm-recipes/configs/k3-moriio-debug.sh" +) + + +@pytest.mark.parametrize( + ("router", "worker"), [(True, False), (False, True), (True, True), (False, False)] +) +def test_only_standalone_router_skips_engine_setup(tmp_path, router, worker): + binaries = tmp_path / "bin" + binaries.mkdir() + for name, present in (("vllm-router", router), ("vllm", worker)): + if present: + binary = binaries / name + binary.write_text("#!/bin/bash\nexit 0\n") + binary.chmod(0o755) + # A failed package lookup must remain fatal for workers or an unknown image. + python = binaries / "python3" + python.write_text("#!/bin/bash\necho 'package lookup failed' >&2\nexit 7\n") + python.chmod(0o755) + completed = subprocess.run( + ["/bin/bash", str(SCRIPT)], + env={"PATH": str(binaries), "LANG": os.environ.get("LANG", "C")}, + capture_output=True, + text=True, + timeout=10, + ) + if router and not worker: + assert completed.returncode == 0 + assert "no engine backport required" in completed.stdout + assert completed.stderr == "" + else: + assert completed.returncode == 7 + assert "package lookup failed" in completed.stderr + assert "no engine backport required" not in completed.stdout + + +@pytest.mark.parametrize("failure", [None, "checksum", "context"]) +def test_verified_patch_changes_only_vllm_and_rejects_invalid_input(tmp_path, failure): + package_root = tmp_path / "site-packages" + package = package_root / "vllm" + package.mkdir(parents=True) + worker = package / "worker.py" + initial = "VALUE = 9\n" if failure == "context" else "VALUE = 1\n" + worker.write_text(initial) + unrelated = package_root / "other.txt" + unrelated.write_text("untouched\n") + patch = tmp_path / "backport.patch" + patch.write_text( + "diff --git a/vllm/worker.py b/vllm/worker.py\n" + "--- a/vllm/worker.py\n+++ b/vllm/worker.py\n@@ -1 +1 @@\n" + "-VALUE = 1\n+VALUE = 2\n" + "diff --git a/other.txt b/other.txt\n" + "--- a/other.txt\n+++ b/other.txt\n@@ -1 +1 @@\n" + "-untouched\n+changed\n" + ) + checksum = hashlib.sha256(patch.read_bytes()).hexdigest() + if failure == "checksum": + patch.write_text(patch.read_text() + "corrupted\n") + completed = subprocess.run( + [ + "/bin/bash", + "-c", + 'source "$1"; apply_verified_patch "$2" "$3" "$4"', + "test", + str(SCRIPT), + str(package_root), + str(patch), + checksum, + ], + capture_output=True, + text=True, + timeout=10, + ) + assert unrelated.read_text() == "untouched\n" + if failure is None: + assert completed.returncode == 0, completed.stderr + assert worker.read_text() == "VALUE = 2\n" + else: + assert completed.returncode != 0 + assert worker.read_text() == initial diff --git a/inferencex-e2e/infx/tests/launch/test_squash_images.py b/inferencex-e2e/infx/tests/launch/test_squash_images.py index 1103521352..ddb2b54d0f 100644 --- a/inferencex-e2e/infx/tests/launch/test_squash_images.py +++ b/inferencex-e2e/infx/tests/launch/test_squash_images.py @@ -191,6 +191,29 @@ def test_enroot_uri_normalization(image, uri): assert enroot_uri(image) == uri +@pytest.mark.parametrize("registry", ["docker.io", "index.docker.io", "registry-1.docker.io"]) +@pytest.mark.parametrize("separator", ["/", "#"]) +@pytest.mark.parametrize(("repository", "resolved"), [ + ("team/image:dev", "team/image:dev"), + ("team/image:dev@sha256:" + "e" * 64, "team/image:sha256:" + "e" * 64), + ("nginx@sha256:" + "e" * 64, "library/nginx:sha256:" + "e" * 64), +]) # fmt: skip +def test_docker_hub_aliases_use_the_registry_api(registry, separator, repository, resolved): + assert enroot_uri(f"{registry}{separator}{repository}") == ( + f"docker://registry-1.docker.io#{resolved}" + ) + + +@pytest.mark.parametrize("mode", ["submit-host", "compute"]) +def test_digest_pinned_docker_hub_alias_reaches_the_importer(tools, tmp_path, mode): + _, logs = tools + image = "docker.io/team/router@sha256:" + "e" * 64 + squash = ensure_image(image, policy(tmp_path, mode), job=Job("7") if mode == "compute" else None) + [imported] = calls(logs["enroot"]) + assert imported["argv"][-1] == "docker://registry-1.docker.io#team/router:sha256:" + "e" * 64 + assert Path(squash).read_text() == VALID + + def test_squash_locations_override_the_cache_field_by_field(): squash = SquashCache.model_validate({ "dir": "/cache", "import": "compute", diff --git a/inferencex-e2e/infx/tests/launch/test_srt_config.py b/inferencex-e2e/infx/tests/launch/test_srt_config.py index dffd525478..58a99752cc 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_config.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_config.py @@ -13,16 +13,19 @@ from infx.launch.backends.slurm import SlurmBackend from infx.launch.context import Launch from infx.launch.drivers.srt.checkout import Checkout, compute_workspace -from infx.launch.drivers.srt.config import SrtJob, pyxis_spelling, render, write +from infx.launch.drivers.srt.config import SrtJob, lane_mounts, pyxis_spelling, render, write +from infx.launch.drivers.srt.lanes import LaneMount, SrtLane from infx.launch.drivers.srt.recipe import HEALTH_ATTEMPTS, prepare_recipe from infx.launch.drivers.srt.run import SrtRun from infx.launch.lifecycle import Lifecycle -from infx.launch.policy import LaunchPath +from infx.launch.policy import LaunchPath, Match, any_of from infx.launch.request import SrtRequest ROOT = Path(__file__).resolve().parents[3] sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) from srtctl.core.config import resolve_config_with_defaults # noqa: E402 +from srtctl.core.schema import ClusterConfig # noqa: E402 +from srtctl.core.slurm import get_container_mounts_str # noqa: E402 def cluster(slurm: dict | None = None, srt: dict | None = None, entries: dict | None = None) -> Cluster: @@ -59,11 +62,15 @@ def test_the_profile_renders_its_facts_and_mounts_a_volume_at_a_second_target(): }, "volume-mounts": {"hub": "/hf_hub_cache/hub"}, "mounts": {"/dev/kfd": "/dev/kfd"}, - "extra": {"visible_devices_env": "ROCR_VISIBLE_DEVICES", "default_gpu_exporter": None}, + "extra": {"visible_devices_env": "ROCR_VISIBLE_DEVICES", "default_gpu_exporter": None, + "default_bash_preamble": "ulimit -l unlimited"}, }, entries={"Model-A": {"root": "data", "dir": "Model-A"}}, ) # fmt: skip config = render(record, job(mounts=[("/share/hub", "/mnt/hf_hub_cache/")], single_node=True)) + native = ClusterConfig.Schema().load(config) + assert native.default_bash_preamble == "ulimit -l unlimited" + assert native.network_interface == "eno0" assert config["default_mounts"] == { "/share/hub": "/hf_hub_cache/hub", "/dev/kfd": "/dev/kfd", @@ -88,6 +95,36 @@ def test_a_host_directory_cannot_be_mounted_at_three_targets(): render(record, job(mounts=[("/share/hub", "/a"), ("/share/hub/", "/b")])) +@pytest.mark.parametrize("asset_exists", [False, True]) +def test_read_only_lane_assets_are_forwarded_without_creating_or_chmodding_them( + tmp_path, asset_exists +): + asset = tmp_path / "provider.so" + if asset_exists: + asset.write_bytes(b"host-owned asset") + asset.chmod(0o440) + record = cluster(slurm={"volumes": { + "provider": {"path": str(asset), "visibility": "node-local"}, + }}) + lane = SrtLane(mounts=(LaneMount( + Match(frameworks=any_of("vllm-disagg")), "provider", read_only=True, + ),)) + run = SimpleNamespace(cluster=record, request=SimpleNamespace(framework="vllm-disagg")) + rendered = render(record, job(mounts=lane_mounts(run, lane))) + native = ClusterConfig.Schema().load(rendered) + command = get_container_mounts_str({ + Path(host): Path(target) for host, target in native.default_mounts.items() + }) + assert command == f"{asset}:{asset}:ro" + if asset_exists: + assert asset.read_bytes() == b"host-owned asset" + assert asset.stat().st_mode & 0o777 == 0o440 + else: + assert not asset.exists() + run.request.framework = "sglang" + assert lane_mounts(run, lane) == [] + + def test_node_exclusions_cpus_and_image_aliases_are_rendered(): record = cluster( slurm={"exclude": ["node-1", "node-2"], "cpus-per-task": 192, diff --git a/inferencex-e2e/infx/tests/launch/test_srt_debug_prs.py b/inferencex-e2e/infx/tests/launch/test_srt_debug_prs.py new file mode 100644 index 0000000000..a7e934c938 --- /dev/null +++ b/inferencex-e2e/infx/tests/launch/test_srt_debug_prs.py @@ -0,0 +1,46 @@ +"""The temporary upstream backport changes only a matching disposable checkout.""" + +import subprocess + +import pytest + +from infx.launch.context import LaunchError +from infx.launch.drivers.srt.checkout import SRT_DEBUG_PRS, apply_debug_prs + + +def test_pinned_debug_backport_applies_real_commit_and_rejects_conflict(tmp_path, monkeypatch): + upstream = tmp_path / "upstream" + upstream.mkdir() + + def git(*args, cwd=upstream): + return subprocess.run(["git", *args], cwd=cwd, check=True, capture_output=True, text=True).stdout.strip() + + git("init", "--quiet") + git("config", "user.name", "Fixture") + git("config", "user.email", "fixture@example.invalid") + source = upstream / "feature.txt" + source.write_text("binding: missing\n") + git("add", "feature.txt") + git("commit", "--quiet", "-m", "base") + base = git("rev-parse", "HEAD") + source.write_text("binding: resolved\n") + git("commit", "--quiet", "-am", "upstream feature") + patch = git("rev-parse", "HEAD") + job = tmp_path / "job" + git("clone", "--quiet", str(upstream), str(job)) + git("checkout", "--quiet", "--detach", base, cwd=job) + monkeypatch.setitem(SRT_DEBUG_PRS, ("fixture", "vllm"), ((str(upstream), patch),)) + + apply_debug_prs(job, "unrelated", "vllm") + assert (job / "feature.txt").read_text() == "binding: missing\n" + apply_debug_prs(job, "fixture", "vllm") + assert (job / "feature.txt").read_text() == "binding: resolved\n" + assert git("rev-parse", "HEAD", cwd=job) == base + + conflicting = tmp_path / "conflicting" + git("clone", "--quiet", str(job), str(conflicting)) + (conflicting / "feature.txt").write_text("binding: incompatible\n") + with pytest.raises(LaunchError, match="cherry-pick"): + apply_debug_prs(conflicting, "fixture", "vllm") + assert (conflicting / "feature.txt").read_text() == "binding: incompatible\n" + assert source.read_text() == "binding: resolved\n" diff --git a/inferencex-e2e/infx/tests/launch/test_srt_driver.py b/inferencex-e2e/infx/tests/launch/test_srt_driver.py index bd11f6903e..2e71ce922a 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_driver.py @@ -226,6 +226,56 @@ def launch_here(monkeypatch, env: dict[str, str], config: Path, cwd: Path) -> in return main(["--runner-config", str(config), "run"]) +@pytest.mark.parametrize("worker", ["test:tag", "mismatched:tag"]) +def test_multinode_recipe_images_are_validated_before_import_and_frontend_is_staged( + harness, worker +): + recipe = yaml.safe_load(LANE_RECIPE) + router = "docker.io/example/router@sha256:" + "d" * 64 + recipe["frontend"] = {"container_image": router} + env = lane_env( + harness, + "lab-a", + yaml.safe_dump({"base": recipe, "override_image": {"model": {"container": worker}}}), + RUNNER_NAME="lab-a_00", + FRAMEWORK="vllm-disagg", + MODEL_PREFIX="kimik3", + MODEL="org/Model", + PRECISION="fp4", + CONFIG_FILE="recipes/test/lane.yaml:override_image", + ) + # A fresh launcher must not inherit test_srt_config's in-process srtctl imports. + completed = subprocess.run( + [sys.executable, "-c", """ +from infx.launch.drivers.srt import lanes +from infx.launch.drivers.srt.lanes import SrtLane +from infx.launch.policy import LaunchPath, Match +from infx.launch.__main__ import main +lanes.SRT_LANES[("lab-a", LaunchPath.SRT_MULTI)] = SrtLane(stage_recipe_images=Match()) +raise SystemExit(main()) +""", "--runner-config", str(lab_config(harness.tmp)), "run"], + env=env, cwd=harness.workspace, capture_output=True, text=True, timeout=120, + ) + rc = completed.returncode + imported = [line.split()[-1] for line in lines(harness.logs, "enroot")] + if worker != "test:tag": + assert rc == 1 + assert "must match" in completed.stderr + assert imported == [] + assert srtctl_calls(harness.logs) == [] + return + assert rc == 0, completed.stdout + completed.stderr + assert imported == [ + "docker://test:tag", + "docker://registry-1.docker.io#example/router:sha256:" + "d" * 64, + ] + config = srtslurm(harness.workspace) + assert config["containers"]["test:tag"].endswith("test_tag.sqsh") + assert config["containers"][router].endswith("docker.io_example_router_sha256_" + "d" * 64 + ".sqsh") + assert config["containers"][router] == config["containers"][router.replace("/", "#", 1)] + assert (harness.workspace / "multinode_server_logs.tar.gz").is_file() + + @pytest.mark.parametrize("cluster_id", LABS) def test_multinode_lane_stages_workflow_artifacts(harness, monkeypatch, cluster_id): lab = LABS[cluster_id] diff --git a/inferencex-e2e/infx/tests/launch/test_srt_recipe_edits.py b/inferencex-e2e/infx/tests/launch/test_srt_recipe_edits.py index 8cdf1aba3e..98373139db 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_recipe_edits.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_recipe_edits.py @@ -1,15 +1,19 @@ """Edits the multi-node lanes apply to the staged recipe copy.""" +from pathlib import Path + import pytest import yaml from infx.launch.drivers.srt.recipe import ( + RECIPES_MIRROR, add_dist_timeout, inject_concurrencies, parse_concurrencies, raise_health_attempts, rename_job, ) +from infx.srt_slurm.recipe_images import resolve_images RECIPE = """name: "upstream" roles: @@ -76,3 +80,42 @@ def test_conc_list_must_be_canonical_positive_integers(): for bad in ("", "08", "0", "-4", "4.0", "4 4", "+4"): with pytest.raises(ValueError): parse_concurrencies(bad) + + +def test_recipe_images_resolve_the_selected_override_and_deduplicate(tmp_path, monkeypatch): + monkeypatch.syspath_prepend(str(Path(__file__).resolve().parents[3] / "utils/srt-slurm/src")) + path = tmp_path / RECIPES_MIRROR / "images.yaml" + path.parent.mkdir(parents=True) + path.write_text( + yaml.safe_dump( + { + "base": { + "model": {"container": "worker:base"}, + "frontend": {"container_image": "router:v1"}, + }, + "override_changed": {"model": {"container": "worker:v2"}}, + "override_shared": {"frontend": {"container_image": "worker:base"}}, + } + ) + ) + assert resolve_images(f"{path}:override_changed", "worker:v2") == [ + "worker:v2", + "router:v1", + ] + assert resolve_images(f"{path}:override_shared", "worker:base") == [ + "worker:base" + ] + with pytest.raises(ValueError, match="exactly one"): + resolve_images(str(path), "worker:base") + with pytest.raises(ValueError, match="must match"): + resolve_images(f"{path}:override_changed", "worker:base") + + +@pytest.mark.parametrize("frontend", [{"container_image": "bad image"}, "not-a-mapping"]) +def test_recipe_images_reject_invalid_frontend_identities(tmp_path, frontend, monkeypatch): + monkeypatch.syspath_prepend(str(Path(__file__).resolve().parents[3] / "utils/srt-slurm/src")) + path = tmp_path / RECIPES_MIRROR / "images.yaml" + path.parent.mkdir(parents=True) + path.write_text(yaml.safe_dump({"model": {"container": "worker:v1"}, "frontend": frontend})) + with pytest.raises(ValueError, match="invalid recipe images"): + resolve_images(str(path), "worker:v1") diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index b8dfc3cf5d..e566b23b94 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9180,3 +9180,77 @@ - "Update the SGLang ROCm image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 on this arm only. The other MI355X arms are left on their current tags." - "No serving flag outside the HiCache block changes, and no other config key is touched. Each of the 19 matrix points resolves to exactly one recipe override, with no override left unused." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3611 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Add native Kimi-K3 PD through the Python launcher with FP32 SSM, FP8 KV, GMU0.90 and upstream DSpark verification. Preserve the recipe-owned concurrency and topology variants." + - "Use official ROCm nightly 36768d1bfd39094681cdbc8cb37d4b31c0729c89 at amd64 digest sha256:50b4f2aec13ffcb11ed846fec049d72250a9c2421eb1ad24c95a5210d46cee4c instead of the custom image; omit shared-MR and the candidate provider bind. Runtime qualification of this image is pending." + - "通过 Python launcher 增加原生 Kimi-K3 PD,使用 FP32 SSM、FP8 KV、GMU0.90 和上游 DSpark 校验,保留 recipe 管理的并发与拓扑配置。" + - "使用固定 digest 的官方 ROCm nightly 替换自定义镜像,不携带 shared-MR 或候选 provider bind;新镜像仍待运行验收。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Temporarily apply pinned upstream srt-slurm#508 to the job-local clone and merged vLLM#57700 Python changes to the official-nightly worker containers. Fail startup if the online fetch, checksum or patch application fails; remove this debug layer after advancing the respective upstream pins." + - "临时在任务内应用固定的上游 srt-slurm#508,以及已合入的 vLLM#57700 纯 Python 修改;在线获取、校验或应用失败即终止启动,待分别升级上游 pin 后删除该 debug 层。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Restore the Kimi-K3 vLLM PD lane's single-file Ionic provider bind as a read-only node-local volume after the official nightly rejected the host kernel ABI. Keep the image's libibverbs core, serving settings and router unchanged; no shared-MR patch. Current-image device-open and end-to-end qualification remain required." + - "官方 nightly 的 Ionic provider 拒绝主机内核 ABI,因此为 Kimi-K3 vLLM PD 路径恢复节点本地单文件只读挂载。保留镜像原有 libibverbs 核心库、serving 参数与 router,不加入 shared-MR;仍需验证当前镜像的设备打开及端到端运行。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Temporarily backport only vLLM#58968's independent-heartbeat helper and worker lifecycle at 9cb82832079ede6638b5a5f35ba0d60d203f57a0 to prevent compute-worker GIL stalls from expiring discovery registration. Preserve payloads, intervals, FP32 SSM, GMU0.90, router and data-plane settings; exclude the rest of that PR. Remove this debug subset once the official image contains the upstream fix." + - "临时提取固定版本 vLLM#58968 的独立心跳 helper 与 worker 生命周期,避免计算线程持有 GIL 时 discovery 注册过期。保留 payload、间隔、FP32 SSM、GMU0.90、router 及数据面设置,不携带该 PR 其余修改;官方镜像包含上游修复后删除此 debug 子集。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Temporarily backport only the FULL pre-forward context from vLLM#58968 at 7595667abf90c03b546c92189d05e5d73392910e so the existing MoRIIO READ completion wait runs before FULL graph replay. Preserve the pinned image, heartbeat backport, FP32 SSM, GMU0.90, graph modes, DSpark, router and benchmark acceptance criteria; omit all other changes from that PR." + - "临时仅提取固定版本 vLLM#58968 的 FULL pre-forward 上下文,使现有 MoRIIO READ 完成等待发生在 FULL graph replay 之前。保留镜像、既有心跳补丁、FP32 SSM、GMU0.90、图模式、DSpark、router 和 benchmark 验收标准,不携带该 PR 其余修改。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Run a diagnostic full-feature control with vLLM#58968 at 7595667abf90c03b546c92189d05e5d73392910e and extracted synchronous READ zeroing fix #59164 at d5e6faa954024d3f3090edd1e769b9efe4feb1cb, retaining merged #57700 and the existing independent heartbeat. Preserve the image, FP32 SSM, GMU0.90, graphs, DSpark, router and benchmark criteria. This is a baseline experiment before patch reduction, not production qualification." + - "使用固定版本 vLLM#58968 的完整运行时代码与已拆出的 #59164 同步 READ zeroing 修复建立实验对照,保留已合入的 #57700 和独立心跳。镜像、FP32 SSM、GMU0.90、图模式、DSpark、router 及 benchmark 判据不变;这是精简补丁前的基线实验,不代表生产验证通过。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Move the minimal Kimi-K3 FP32 PD diagnostic candidate from official ROCm nightly 36768d1b to digest-pinned ac68c308, which includes #57700. Retain only #59164's synchronous READ zeroing backport; omit #58968 and independent heartbeat. Preserve serving settings, DSpark synthetic acceptance and benchmark criteria. The previous full-feature control remains in history; this image-changing experiment is not production qualification or a strict single-variable comparison." + - "将 Kimi-K3 FP32 PD 最小依赖诊断候选从官方 ROCm nightly 36768d1b 更新为按 digest 固定且已包含 #57700 的 ac68c308。仅保留 #59164 同步 READ zeroing 回补,省略 #58968 与独立心跳;serving 参数、DSpark synthetic acceptance 和 benchmark 判据不变。历史中保留完整特性对照;本次换镜像实验不代表生产验证通过,也不是严格单变量比较。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582 + +- config-keys: + - kimik3-fp4-mi355x-vllm-disagg-agentic + scenario-type: + - agentic-coding + description: + - "Consolidate the debug engine stack to immutable official ROCm nightly ac68c308 plus pinned #59164 and the Kimi-K3 streaming-EOF subset, runtime-equivalent to YukioZzz/vllm@535727bcc. Preserve FP32 SSM, GMU0.90, topology, automatic acceptance and 1% error gates; no heartbeat or full #58968 backport. Supporting matched 1P1D c48 evidence does not qualify the current InferenceX 1P2D candidate." + - "将 debug engine 栈收束为不可变官方 ROCm nightly ac68c308,加固定 #59164 与和 YukioZzz/vllm@535727bcc 运行时等价的 Kimi-K3 streaming-EOF 子集。保留 FP32 SSM、GMU0.90、拓扑、自动 acceptance 和 1% 错误门槛,不加入心跳或全量 #58968。配对 1P1D c48 证据不等于当前 InferenceX 1P2D 候选已通过验证。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3582