Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,148 @@
schema: 2
name: kimi-k3-vllm-agg-b300-dcp8-dspark7-maxseq2-mooncake-c1-agentic
model:
path: moonshotai/Kimi-K3
container: ghcr.io/semianalysisai/vllm-openai:efa_pr_58768
precision: fp4
environment: {LD_LIBRARY_PATH: "/opt/amazon/efa/lib:/usr/local/nvidia/lib64:/usr/local/cuda/lib64:/usr/local/nvidia/lib"}
resources:
gpu_type: b300
gpus_per_node: 8
services:
- name: etcd
type: etcd
placement:
node: infra
- name: nats
type: nats
placement:
node: infra
options:
max_payload_mb: 32
- name: mooncake-master
type: mooncake-master
args:
- --default_kv_lease_ttl=60000
- --eviction_high_watermark_ratio=0.95
- --eviction_ratio=0.10
options:
store_config:
metadata_server: P2PHANDSHAKE
global_segment_size: 190GB
local_buffer_size: 4GB
protocol: efa
mode: embedded
enable_offload: false
frontend:
type: dynamo
enable_multiple_frontends: false
args:
dyn-chat-processor: vllm
trust-remote-code: true
tool-call-parser: kimi_k3
reasoning-parser: kimi_k3
enable-auto-tool-choice: true
router-mode: least-loaded
router-session-affinity-ttl-secs: 900
env:
DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600'
DYN_TOKENIZER_CACHE_BYTES: '8589934592'
PYTHONPYCACHEPREFIX: /tmp/vllm-pycache-kimi-k3
engine:
type: vllm
connector: null
roles:
agg:
nodes: 1
workers: 1
gpus: 8
env:
VLLM_ALLREDUCE_USE_FLASHINFER: '1'
VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1'
VLLM_USE_DIRECT_DCP_A2A: '1'
VLLM_USE_DIRECT_DCP_Q_GATHER: '1'
VLLM_USE_DIRECT_DCP_KV_GATHER: '1'
ETCD_LEASE_TTL: '120'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
VLLM_RPC_TIMEOUT: '600000'
VLLM_USE_V2_MODEL_RUNNER: '1'
VLLM_USE_RUST_FRONTEND: '1'
VLLM_MOONCAKE_LOAD_RECV_THREADS: '4'
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0'
VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0'
MC_SLICE_SIZE: '1048576'
WITH_NVIDIA_PEERMEM: '0'
VLLM_LOG_STATS_INTERVAL: '1'
PYTHONHASHSEED: '42'
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
TORCH_CUDA_ARCH_LIST: '10.0'
GLIBC_TUNABLES: glibc.malloc.hugetlb=1
MC_STORE_MEMCPY: '1'
MC_WORKERS_PER_CTX: '4'
args:
kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_offload":false}}'
served-model-name: moonshotai/Kimi-K3
load-format: fastsafetensors
kv-cache-dtype: fp8
mamba-ssm-cache-dtype: bfloat16
gpu-memory-utilization: 0.94
tensor-parallel-size: 8
decode-context-parallel-size: 8
dcp-comm-backend: a2a
max-num-seqs: 2
max-num-batched-tokens: 8192
trust-remote-code: true
max-cudagraph-capture-size: 1024
stream-interval: 10
language-model-only: true
moe-backend: auto
enable-cumem-allocator: true
attention-backend: TOKENSPEED_MLA
attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}'
speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":7,"rejection_sample_method":"block","draft_sample_method":"probabilistic"}'
enable-prefix-caching: true
prefix-match-unit: 128
enable-flashinfer-autotune: true
health_check:
max_attempts: 720
interval_seconds: 10
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: '300'
AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: '0.25'
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:'
AIPERF_HTTP_TCP_USER_TIMEOUT: '900000'
RESULT_DIR: /logs/agentic
PORT: '8000'
IS_MULTINODE: 'true'
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0'
AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache
HF_HUB_CACHE: /hf_hub_cache
WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126
client_placement: head
concurrencies:
- 1
dynamo:
install: true
source:
rev: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b
identity:
model:
repo: moonshotai/Kimi-K3
container:
image: ghcr.io/semianalysisai/vllm-openai:efa_pr_58768
frameworks:
dynamo: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b
slurm:
time_limit: 06:00:00
sbatch_directives:
mem: '0'
cpus-per-task: '72'
comment: '''{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will take extra loading time."}}'''
srun_options:
mem: '0'
container-remap-root: ''
Original file line number Diff line number Diff line change
@@ -0,0 +1,236 @@
schema: 2
name: kimi-k3-vllm-disagg-b300-1p3d-dcp8-dcp8-dspark4-mooncake-c32-agentic
model:
path: moonshotai/Kimi-K3
container: ghcr.io/semianalysisai/vllm-openai:efa_pr_58768
precision: fp4
environment: {LD_LIBRARY_PATH: "/opt/amazon/efa/lib:/usr/local/nvidia/lib64:/usr/local/cuda/lib64:/usr/local/nvidia/lib"}
dynamo:
install: true
source:
rev: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b
slurm:
time_limit: 06:00:00
health_check:
max_attempts: 720
interval_seconds: 10
resources:
gpu_type: b300
gpus_per_node: 8
services:
- name: etcd
type: etcd
placement:
node: infra
- name: nats
type: nats
placement:
node: infra
options:
max_payload_mb: 32
- name: mooncake-master
type: mooncake-master
args:
- --default_kv_lease_ttl=60000
- --eviction_high_watermark_ratio=0.95
- --eviction_ratio=0.10
options:
store_config:
metadata_server: P2PHANDSHAKE
global_segment_size: 190GB
local_buffer_size: 4GB
protocol: efa
mode: embedded
enable_offload: false
frontend:
type: dynamo
enable_multiple_frontends: false
args:
dyn-chat-processor: vllm
trust-remote-code: true
tool-call-parser: kimi_k3
reasoning-parser: kimi_k3
enable-auto-tool-choice: true
router-mode: least-loaded
router-session-affinity-ttl-secs: 900
env:
DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600'
DYN_TOKENIZER_CACHE_BYTES: '8589934592'
PYTHONPYCACHEPREFIX: /tmp/vllm-pycache-kimi-k3
engine:
type: vllm
connector: null
dp_launch_mode: per_node
roles:
prefill:
nodes: 1
workers: 1
gpus: 8
env:
VLLM_USE_DIRECT_DCP_A2A: '1'
VLLM_USE_DIRECT_DCP_Q_GATHER: '1'
VLLM_USE_DIRECT_DCP_KV_GATHER: '1'
VLLM_ALLREDUCE_USE_FLASHINFER: '1'
VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0'
DYN_REQUEST_PLANE: tcp
ETCD_LEASE_TTL: '600'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
VLLM_RPC_TIMEOUT: '600000'
TILELANG_CLEANUP_TEMP_FILES: '1'
VLLM_USE_NCCL_SYMM_MEM: '0'
NCCL_CUMEM_ENABLE: '1'
NCCL_MNNVL_ENABLE: '0'
NCCL_NVLS_ENABLE: '1'
VLLM_SERVER_DEV_MODE: '1'
VLLM_USE_V2_MODEL_RUNNER: '1'
VLLM_USE_RUST_FRONTEND: '1'
VLLM_MOONCAKE_LOAD_RECV_THREADS: '4'
MC_SLICE_SIZE: '1048576'
VLLM_CONNECTOR_PREFETCH_DEPTH: '8'
VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65'
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800'
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0'
NCCL_P2P_LEVEL: NVL
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
NCCL_NET_PLUGIN: none
UCX_MEMTYPE_CACHE: n
UCX_MEMTYPE_REG_WHOLE: n
UCX_RCACHE_MAX_UNRELEASED: '1024'
UCX_TCP_AF_PRIO: inet
VLLM_SSM_CONV_STATE_LAYOUT: DS
DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-b300-pd-dspark-mooncake-{job_id}
PYTHONHASHSEED: '42'
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
TORCH_CUDA_ARCH_LIST: '10.0'
MC_STORE_CLIENT_METRIC: '1'
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_TE_METRIC: '0'
GLIBC_TUNABLES: glibc.malloc.hugetlb=1
MC_STORE_MEMCPY: '1'
MC_WORKERS_PER_CTX: '4'
args:
kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false,"backends":["LIBFABRIC"]}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}'
served-model-name: moonshotai/Kimi-K3
prefix-match-unit: 128
load-format: fastsafetensors
kv-cache-dtype: fp8
mamba-ssm-cache-dtype: bfloat16
gpu-memory-utilization: 0.94
tensor-parallel-size: 8
decode-context-parallel-size: 8
cp-kv-cache-interleave-size: 1
dcp-comm-backend: a2a
enable-cumem-allocator: true
trust-remote-code: true
max-cudagraph-capture-size: 512
stream-interval: 10
language-model-only: true
attention-backend: TOKENSPEED_MLA
attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}'
speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"rejection_sample_method":"block","draft_sample_method":"probabilistic"}'
enable-prefix-caching: true
enable-flashinfer-autotune: true
kv_events: true
decode:
nodes: 3
workers: 3
gpus: 8
env:
VLLM_USE_DIRECT_DCP_A2A: '1'
VLLM_USE_DIRECT_DCP_Q_GATHER: '1'
VLLM_USE_DIRECT_DCP_KV_GATHER: '1'
VLLM_ALLREDUCE_USE_FLASHINFER: '1'
VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0'
DYN_REQUEST_PLANE: tcp
ETCD_LEASE_TTL: '600'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
VLLM_RPC_TIMEOUT: '600000'
TILELANG_CLEANUP_TEMP_FILES: '1'
VLLM_USE_NCCL_SYMM_MEM: '0'
NCCL_CUMEM_ENABLE: '1'
NCCL_MNNVL_ENABLE: '0'
NCCL_NVLS_ENABLE: '1'
VLLM_SERVER_DEV_MODE: '1'
VLLM_USE_V2_MODEL_RUNNER: '1'
VLLM_USE_RUST_FRONTEND: '1'
VLLM_MOONCAKE_LOAD_RECV_THREADS: '4'
MC_SLICE_SIZE: '1048576'
VLLM_CONNECTOR_PREFETCH_DEPTH: '8'
VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65'
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800'
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0'
NCCL_P2P_LEVEL: NVL
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
NCCL_NET_PLUGIN: none
UCX_MEMTYPE_CACHE: n
UCX_MEMTYPE_REG_WHOLE: n
UCX_RCACHE_MAX_UNRELEASED: '1024'
UCX_TCP_AF_PRIO: inet
VLLM_SSM_CONV_STATE_LAYOUT: DS
DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-b300-pd-dspark-mooncake-{job_id}
PYTHONHASHSEED: '42'
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
TORCH_CUDA_ARCH_LIST: '10.0'
MC_STORE_CLIENT_METRIC: '1'
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_TE_METRIC: '0'
GLIBC_TUNABLES: glibc.malloc.hugetlb=1
MC_STORE_MEMCPY: '1'
MC_WORKERS_PER_CTX: '4'
args:
kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false,"backends":["LIBFABRIC"]}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}'
served-model-name: moonshotai/Kimi-K3
prefix-match-unit: 128
load-format: fastsafetensors
kv-cache-dtype: fp8
mamba-ssm-cache-dtype: bfloat16
gpu-memory-utilization: 0.94
tensor-parallel-size: 8
decode-context-parallel-size: 8
cp-kv-cache-interleave-size: 1
dcp-comm-backend: a2a
enable-cumem-allocator: true
trust-remote-code: true
max-cudagraph-capture-size: 512
stream-interval: 10
language-model-only: true
attention-backend: TOKENSPEED_MLA
attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}'
speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"rejection_sample_method":"block","draft_sample_method":"probabilistic"}'
enable-prefix-caching: true
enable-flashinfer-autotune: true
sbatch_directives:
cpus-per-task: '72'
mem: '0'
comment: '''{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will take extra loading time."}}'''
srun_options:
mem: '0'
container-remap-root: ''
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
PORT: '8000'
IS_MULTINODE: 'true'
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0'
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true'
AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400'
AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache
HF_HUB_CACHE: /hf_hub_cache
WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126
client_placement: head
concurrencies:
- 32
identity:
model:
repo: moonshotai/Kimi-K3
container:
image: ghcr.io/semianalysisai/vllm-openai:efa_pr_58768
frameworks:
dynamo: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b
Loading
Loading