Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,244 @@
schema: 2
name: kimi-k3-vllm-disagg-b300-1p1d-dcp8-dcp8-dspark4-mooncake-hugepage-c48-agentic
model:
path: moonshotai/Kimi-K3
container: ghcr.io#semianalysisai/vllm-openai:efa_pr_58768
precision: fp4
environment: {LD_LIBRARY_PATH: "/opt/amazon/efa/lib:/usr/local/nvidia/lib64:/usr/local/cuda/lib64:/usr/local/nvidia/lib"}
dynamo:
install: true
source:
rev: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b
slurm:
time_limit: 06:00:00
health_check:
max_attempts: 720
interval_seconds: 10
resources:
gpu_type: b300
gpus_per_node: 8
services:
- name: etcd
type: etcd
placement:
node: infra
- name: nats
type: nats
placement:
node: infra
options:
max_payload_mb: 32
- name: mooncake-master
type: mooncake-master
args:
- --default_kv_lease_ttl=60000
- --eviction_high_watermark_ratio=0.95
- --eviction_ratio=0.10
env:
MC_STORE_USE_HUGEPAGE: '1'
MC_STORE_HUGEPAGE_SIZE: '2097152'
options:
store_config:
metadata_server: P2PHANDSHAKE
global_segment_size: 190GB
local_buffer_size: 4GB
protocol: efa
mode: embedded
enable_offload: false
frontend:
type: dynamo
enable_multiple_frontends: false
args:
dyn-chat-processor: vllm
trust-remote-code: true
tool-call-parser: kimi_k3
reasoning-parser: kimi_k3
enable-auto-tool-choice: true
router-mode: least-loaded
router-session-affinity-ttl-secs: 900
env:
DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600'
DYN_TOKENIZER_CACHE_BYTES: '8589934592'
engine:
type: vllm
connector: null
dp_launch_mode: per_node
roles:
prefill:
nodes: 1
workers: 1
gpus: 8
env:
VLLM_USE_DIRECT_DCP_A2A: '1'
VLLM_USE_DIRECT_DCP_Q_GATHER: '1'
VLLM_USE_DIRECT_DCP_KV_GATHER: '1'
VLLM_ALLREDUCE_USE_FLASHINFER: '1'
VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0'
DYN_REQUEST_PLANE: tcp
ETCD_LEASE_TTL: '600'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
VLLM_RPC_TIMEOUT: '600000'
TILELANG_CLEANUP_TEMP_FILES: '1'
VLLM_USE_NCCL_SYMM_MEM: '0'
NCCL_CUMEM_ENABLE: '1'
NCCL_MNNVL_ENABLE: '0'
NCCL_NVLS_ENABLE: '1'
VLLM_SERVER_DEV_MODE: '1'
VLLM_USE_V2_MODEL_RUNNER: '1'
VLLM_USE_RUST_FRONTEND: '1'
VLLM_MOONCAKE_LOAD_RECV_THREADS: '4'
MC_SLICE_SIZE: '1048576'
VLLM_CONNECTOR_PREFETCH_DEPTH: '8'
VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65'
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800'
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0'
NCCL_P2P_LEVEL: NVL
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
NCCL_NET_PLUGIN: none
UCX_MEMTYPE_CACHE: n
UCX_MEMTYPE_REG_WHOLE: n
UCX_RCACHE_MAX_UNRELEASED: '1024'
UCX_TCP_AF_PRIO: inet
VLLM_SSM_CONV_STATE_LAYOUT: DS
DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-b300-pd-dspark-mooncake-{job_id}
PYTHONHASHSEED: '42'
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
TORCH_CUDA_ARCH_LIST: '10.0'
MC_STORE_CLIENT_METRIC: '1'
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_TE_METRIC: '0'
GLIBC_TUNABLES: glibc.malloc.hugetlb=1
MC_STORE_MEMCPY: '1'
MC_WORKERS_PER_CTX: '4'
MC_STORE_USE_HUGEPAGE: '1'
MC_STORE_HUGEPAGE_SIZE: '2097152'
args:
enable-logging-iteration-details: true
kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false,"backends":["LIBFABRIC"]}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}'
served-model-name: moonshotai/Kimi-K3
prefix-match-unit: 128
load-format: fastsafetensors
kv-cache-dtype: fp8
mamba-ssm-cache-dtype: bfloat16
gpu-memory-utilization: 0.94
tensor-parallel-size: 8
decode-context-parallel-size: 8
cp-kv-cache-interleave-size: 1
dcp-comm-backend: a2a
enable-cumem-allocator: true
trust-remote-code: true
max-cudagraph-capture-size: 512
stream-interval: 10
language-model-only: true
attention-backend: TOKENSPEED_MLA
attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}'
speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"rejection_sample_method":"block","draft_sample_method":"probabilistic"}'
enable-prefix-caching: true
enable-flashinfer-autotune: true
kv_events: true
decode:
nodes: 1
workers: 1
gpus: 8
env:
VLLM_USE_DIRECT_DCP_A2A: '1'
VLLM_USE_DIRECT_DCP_Q_GATHER: '1'
VLLM_USE_DIRECT_DCP_KV_GATHER: '1'
VLLM_ALLREDUCE_USE_FLASHINFER: '1'
VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0'
DYN_REQUEST_PLANE: tcp
ETCD_LEASE_TTL: '600'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
VLLM_RPC_TIMEOUT: '600000'
TILELANG_CLEANUP_TEMP_FILES: '1'
VLLM_USE_NCCL_SYMM_MEM: '0'
NCCL_CUMEM_ENABLE: '1'
NCCL_MNNVL_ENABLE: '0'
NCCL_NVLS_ENABLE: '1'
VLLM_SERVER_DEV_MODE: '1'
VLLM_USE_V2_MODEL_RUNNER: '1'
VLLM_USE_RUST_FRONTEND: '1'
VLLM_MOONCAKE_LOAD_RECV_THREADS: '4'
MC_SLICE_SIZE: '1048576'
VLLM_CONNECTOR_PREFETCH_DEPTH: '8'
VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65'
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800'
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0'
NCCL_P2P_LEVEL: NVL
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
NCCL_NET_PLUGIN: none
UCX_MEMTYPE_CACHE: n
UCX_MEMTYPE_REG_WHOLE: n
UCX_RCACHE_MAX_UNRELEASED: '1024'
UCX_TCP_AF_PRIO: inet
VLLM_SSM_CONV_STATE_LAYOUT: DS
DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-b300-pd-dspark-mooncake-{job_id}
PYTHONHASHSEED: '42'
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
TORCH_CUDA_ARCH_LIST: '10.0'
MC_STORE_CLIENT_METRIC: '1'
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_TE_METRIC: '0'
GLIBC_TUNABLES: glibc.malloc.hugetlb=1
MC_STORE_MEMCPY: '1'
MC_WORKERS_PER_CTX: '4'
MC_STORE_USE_HUGEPAGE: '1'
MC_STORE_HUGEPAGE_SIZE: '2097152'
args:
enable-logging-iteration-details: true
kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false,"backends":["LIBFABRIC"]}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}'
served-model-name: moonshotai/Kimi-K3
prefix-match-unit: 128
load-format: fastsafetensors
kv-cache-dtype: fp8
mamba-ssm-cache-dtype: bfloat16
gpu-memory-utilization: 0.94
tensor-parallel-size: 8
decode-context-parallel-size: 8
cp-kv-cache-interleave-size: 1
dcp-comm-backend: a2a
enable-cumem-allocator: true
trust-remote-code: true
max-cudagraph-capture-size: 512
stream-interval: 10
language-model-only: true
attention-backend: TOKENSPEED_MLA
attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}'
speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"rejection_sample_method":"block","draft_sample_method":"probabilistic"}'
enable-prefix-caching: true
enable-flashinfer-autotune: true
sbatch_directives:
cpus-per-task: '72'
mem: '0'
comment: '''{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will take extra loading time."}}'''
srun_options:
mem: '0'
container-remap-root: ''
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
PORT: '8000'
IS_MULTINODE: 'true'
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0'
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true'
AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400'
AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache
HF_HUB_CACHE: /hf_hub_cache
WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126
client_placement: head
concurrencies:
- 48
identity:
model:
repo: moonshotai/Kimi-K3
container:
image: ghcr.io#semianalysisai/vllm-openai:efa_pr_58768
frameworks:
dynamo: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b
31 changes: 31 additions & 0 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -1528,6 +1528,37 @@ kimik3-fp4-b300-vllm-agentic-dspark:
# disjoint across arms so exp-names stay unique.
- { tp: 8, ep: 1, dcp-size: 8, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: mooncake, version: "0.3.11.post1" }, conc-list: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 70], srt-recipe: benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml }

kimik3-fp4-b300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg-1p1d-hugepage:
# #3523's 1P1D EFA arm with the worker-embedded Mooncake store segment
# backed by 2 MB hugepages (MC_STORE_USE_HUGEPAGE / MC_STORE_HUGEPAGE_SIZE).
image: ghcr.io#semianalysisai/vllm-openai:efa_pr_58768
model: moonshotai/Kimi-K3
model-prefix: kimik3
runner: cluster:b300-dsxe
precision: fp4
framework: dynamo-vllm
router: { name: dynamo-router, version: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" }

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 (optional) nvidia-master.yaml:1540 records router version "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" for the new kimik3-fp4-b300-...-hugepage config key, but the recipe it points at pins a different dynamo build. disagg-1p1d-dcp8-dcp8-dspark4-mooncake-hugepage-c48.yaml:11 and :244 set dynamo.source.rev / identity.frameworks.dynamo to "cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b" instead. Every sibling kimik3 vllm agentx recipe (gb300/gb200 disagg and agg variants) uses "ba83080ec..." in both the recipe and the matching master-config router.version, so this is the only entry where the two diverge. Anyone using the master config to attribute a sweep's results to a dynamo build gets pointed at the wrong commit. …

Why this was flagged

…Fix: set router.version here to "cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b" to match the recipe's actual dynamo.source.rev (or revert the recipe's rev if the bump was unintentional).

The new master-config entry kimik3-fp4-b300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg-1p1d-hugepage at nvidia-master.yaml:1531-1560 declares router: { version: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" } at line 1540, reached via the CONFIG_FILE additional-setting at line 1559 pointing to the new recipe. That recipe (disagg-1p1d-dcp8-dcp8-dspark4-mooncake-hugepage-c48.yaml:8-11,244) actually pins dynamo.source.rev to "cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b", a different commit not used by any other kimik3 recipe in the repo. All comparable sibling entries (e.g. kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg at nvidia-master.yaml:7832-7860, paired with gb300-fp4/agentx/disagg-1p1d-dcp8-dcp8-dspark4-mooncake.yaml:14-21) keep router.version and dynamo.source.rev identical, showing the master config's router.version field is meant to mirror the recipe's actual dynamo build.…

Verification: nit. Factually accurate discrepancy: nvidia-master.yaml:1540 records router: { name: dynamo-router, version: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" } for the new key, while the recipe it points at (disagg-1p1d-dcp8-dcp8-dspark4-mooncake-hugepage-c48.yaml) pins the dynamo build to a different commit in both places: line 11 rev: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b and line 244…

kv-p2p-transfer: nixl
multinode: true
disagg: true
scenarios:
agentic-coding:
- dram-utilization: 0.75
search-space:
- spec-decoding: mtp
kv-offloading: dram
kv-offload-backend: { name: mooncake, version: "0.3.13.post1" }
conc-list: [48]
prefill:
num-worker: 1
tp: 8
dcp-size: 8
ep: 1
dp-attn: false
additional-settings:
- "CONFIG_FILE=recipes/kimik3/vllm/b300-fp4/agentx/disagg-1p1d-dcp8-dcp8-dspark4-mooncake-hugepage-c48.yaml"
decode: { num-worker: 1, tp: 8, dcp-size: 8, ep: 1, dp-attn: false }

dsr1-fp8-b200-trt:
image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14
model: deepseek-ai/DeepSeek-R1-0528
Expand Down
9 changes: 9 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9012,3 +9012,12 @@
- "Update the GLM-5.2 MXFP4 MI355X SGLang AgentX image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260923 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260924 (digest sha256:baadaba198e23c46c1dd651b5edefae8000bd5430943bd10b246b563905b3ecf); keep all serving flags, HiCache settings, and sweep points unchanged."
- "将 GLM-5.2 MXFP4 MI355X SGLang AgentX 镜像从 lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260923 更新到 lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260924(digest sha256:baadaba198e23c46c1dd651b5edefae8000bd5430943bd10b246b563905b3ecf);其余服务参数、HiCache 设置与 sweep 点保持不变。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3446

- config-keys:
- kimik3-fp4-b300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg-1p1d-hugepage
scenario-type:
- agentic-coding
description:
- "Add a hugepage variant of the Kimi-K3 B300 Mooncake EFA AgentX disaggregated 1P1D DCP8/DCP8 DSpark4 c48 configuration from #3523 on ghcr.io/semianalysisai/vllm-openai:efa_pr_58768 (mooncake-transfer-engine-efa-cuda13 0.3.13.post1). The mooncake-master service and the prefill and decode workers set MC_STORE_USE_HUGEPAGE=1 and MC_STORE_HUGEPAGE_SIZE=2097152 so the embedded Mooncake store segment (190GB global segment, 4GB local buffer) is allocated from 2 MB hugepages; all other recipe settings match #3523."
- "新增 #3523 中 Kimi-K3 B300 Mooncake EFA AgentX 分离式 1P1D DCP8/DCP8 DSpark4 c48 配置的大页变体,镜像为 ghcr.io/semianalysisai/vllm-openai:efa_pr_58768(mooncake-transfer-engine-efa-cuda13 0.3.13.post1)。mooncake-master 服务以及 prefill 与 decode worker 均设置 MC_STORE_USE_HUGEPAGE=1 和 MC_STORE_HUGEPAGE_SIZE=2097152,使内嵌 Mooncake store 段(190GB 全局段、4GB 本地缓冲)使用 2 MB 大页分配;其余配方设置与 #3523 相同。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3557
14 changes: 12 additions & 2 deletions inferencex-e2e/runners/launch_b300-dsxe.sh
Original file line number Diff line number Diff line change
Expand Up @@ -112,8 +112,18 @@ import_squash_image() {
if unsquashfs -l \"$sqsh\" > /dev/null 2>&1; then
exit 0
fi
rm -f \"$sqsh\"
enroot import -o \"$sqsh\" \"docker://$image_ref\"
# Large registry layers (e.g. GHCR) can drop mid-transfer with
# \"curl: (56) Connection reset by peer\". Retry with fewer parallel
# connections; enroot's layer cache keeps finished layers across attempts.
for attempt in 1 2 3 4; do
rm -f \"$sqsh\"
if ENROOT_MAX_CONNECTIONS=\"\${ENROOT_MAX_CONNECTIONS:-4}\" enroot import -o \"$sqsh\" \"docker://$image_ref\"; then
break
fi
[ \"\$attempt\" -eq 4 ] && exit 1
echo \"enroot import attempt \$attempt failed for $image_ref; retrying in \$((attempt * 30))s\" >&2
sleep \$((attempt * 30))
done
unsquashfs -l \"$sqsh\" > /dev/null
" || { echo "Error: enroot import failed for $image_ref -> $sqsh" >&2; exit 1; }

Expand Down
Loading