Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,237 @@
schema: 2
name: kimi-k3-vllm-disagg-b300-1p1d-dcp8-dcp8-dspark4-mooncake-c48-agentic
model:
path: moonshotai/Kimi-K3
container: ghcr.io#semianalysisai/vllm-openai:efa_pr_58768
precision: fp4
environment: {LD_LIBRARY_PATH: "/opt/amazon/efa/lib:/usr/local/nvidia/lib64:/usr/local/cuda/lib64:/usr/local/nvidia/lib"}
dynamo:
install: true
source:
rev: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b
slurm:
time_limit: 06:00:00
health_check:
max_attempts: 720
interval_seconds: 10
resources:
gpu_type: b300
gpus_per_node: 8
services:
- name: etcd
type: etcd
placement:
node: infra
- name: nats
type: nats
placement:
node: infra
options:
max_payload_mb: 32
- name: mooncake-master
type: mooncake-master
args:
- --default_kv_lease_ttl=60000
- --eviction_high_watermark_ratio=0.95
- --eviction_ratio=0.10
options:
store_config:
metadata_server: P2PHANDSHAKE
global_segment_size: 190GB
local_buffer_size: 4GB
protocol: efa
mode: embedded
enable_offload: false
frontend:
type: dynamo
enable_multiple_frontends: false
args:
dyn-chat-processor: vllm
trust-remote-code: true
tool-call-parser: kimi_k3
reasoning-parser: kimi_k3
enable-auto-tool-choice: true
router-mode: least-loaded
router-session-affinity-ttl-secs: 900
env:
DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: '3600'
DYN_TOKENIZER_CACHE_BYTES: '8589934592'
engine:
type: vllm
connector: null
dp_launch_mode: per_node
roles:
prefill:
nodes: 1
workers: 1
gpus: 8
env:
VLLM_USE_DIRECT_DCP_A2A: '1'
VLLM_USE_DIRECT_DCP_Q_GATHER: '1'
VLLM_USE_DIRECT_DCP_KV_GATHER: '1'
VLLM_ALLREDUCE_USE_FLASHINFER: '1'
VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0'
DYN_REQUEST_PLANE: tcp
ETCD_LEASE_TTL: '600'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
VLLM_RPC_TIMEOUT: '600000'
TILELANG_CLEANUP_TEMP_FILES: '1'
VLLM_USE_NCCL_SYMM_MEM: '0'
NCCL_CUMEM_ENABLE: '1'
NCCL_MNNVL_ENABLE: '0'
NCCL_NVLS_ENABLE: '1'
VLLM_SERVER_DEV_MODE: '1'
VLLM_USE_V2_MODEL_RUNNER: '1'
VLLM_USE_RUST_FRONTEND: '1'
VLLM_MOONCAKE_LOAD_RECV_THREADS: '4'
MC_SLICE_SIZE: '1048576'
VLLM_CONNECTOR_PREFETCH_DEPTH: '8'
VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65'
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800'
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0'
NCCL_P2P_LEVEL: NVL
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
NCCL_NET_PLUGIN: none
UCX_MEMTYPE_CACHE: n
UCX_MEMTYPE_REG_WHOLE: n
UCX_RCACHE_MAX_UNRELEASED: '1024'
UCX_TCP_AF_PRIO: inet
VLLM_SSM_CONV_STATE_LAYOUT: DS
DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-b300-pd-dspark-mooncake-{job_id}
PYTHONHASHSEED: '42'
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
TORCH_CUDA_ARCH_LIST: '10.0'
MC_STORE_CLIENT_METRIC: '1'
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_TE_METRIC: '0'
GLIBC_TUNABLES: glibc.malloc.hugetlb=1
MC_STORE_MEMCPY: '1'
MC_WORKERS_PER_CTX: '4'
args:
enable-logging-iteration-details: true
kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false,"backends":["LIBFABRIC"]}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}'
served-model-name: moonshotai/Kimi-K3
prefix-match-unit: 128
load-format: fastsafetensors
kv-cache-dtype: fp8
mamba-ssm-cache-dtype: bfloat16
gpu-memory-utilization: 0.94
tensor-parallel-size: 8
decode-context-parallel-size: 8
cp-kv-cache-interleave-size: 1
dcp-comm-backend: a2a
enable-cumem-allocator: true
trust-remote-code: true
max-cudagraph-capture-size: 512
stream-interval: 10
language-model-only: true
attention-backend: TOKENSPEED_MLA
attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}'
speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"rejection_sample_method":"block","draft_sample_method":"probabilistic"}'
enable-prefix-caching: true
enable-flashinfer-autotune: true
kv_events: true
decode:
nodes: 1
workers: 1
gpus: 8
env:
VLLM_USE_DIRECT_DCP_A2A: '1'
VLLM_USE_DIRECT_DCP_Q_GATHER: '1'
VLLM_USE_DIRECT_DCP_KV_GATHER: '1'
VLLM_ALLREDUCE_USE_FLASHINFER: '1'
VLLM_KIMI_K3_SHARD_SP_SHARED_EXPERT: '0'
DYN_REQUEST_PLANE: tcp
ETCD_LEASE_TTL: '600'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
VLLM_RPC_TIMEOUT: '600000'
TILELANG_CLEANUP_TEMP_FILES: '1'
VLLM_USE_NCCL_SYMM_MEM: '0'
NCCL_CUMEM_ENABLE: '1'
NCCL_MNNVL_ENABLE: '0'
NCCL_NVLS_ENABLE: '1'
VLLM_SERVER_DEV_MODE: '1'
VLLM_USE_V2_MODEL_RUNNER: '1'
VLLM_USE_RUST_FRONTEND: '1'
VLLM_MOONCAKE_LOAD_RECV_THREADS: '4'
MC_SLICE_SIZE: '1048576'
VLLM_CONNECTOR_PREFETCH_DEPTH: '8'
VLLM_CONNECTOR_PREFETCH_KV_CAP: '0.65'
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800'
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0'
NCCL_P2P_LEVEL: NVL
MC_ENABLE_DEST_DEVICE_AFFINITY: '1'
WITH_NVIDIA_PEERMEM: '0'
NCCL_NET_PLUGIN: none
UCX_MEMTYPE_CACHE: n
UCX_MEMTYPE_REG_WHOLE: n
UCX_RCACHE_MAX_UNRELEASED: '1024'
UCX_TCP_AF_PRIO: inet
VLLM_SSM_CONV_STATE_LAYOUT: DS
DG_JIT_CACHE_DIR: /tmp/dg-cache-kimi-k3-b300-pd-dspark-mooncake-{job_id}
PYTHONHASHSEED: '42'
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
TORCH_CUDA_ARCH_LIST: '10.0'
MC_STORE_CLIENT_METRIC: '1'
MC_STORE_CLIENT_METRIC_INTERVAL: '5'
MC_TE_METRIC: '0'
GLIBC_TUNABLES: glibc.malloc.hugetlb=1
MC_STORE_MEMCPY: '1'
MC_WORKERS_PER_CTX: '4'
args:
enable-logging-iteration-details: true
kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false,"backends":["LIBFABRIC"]}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}'
served-model-name: moonshotai/Kimi-K3
prefix-match-unit: 128
load-format: fastsafetensors
kv-cache-dtype: fp8
mamba-ssm-cache-dtype: bfloat16
gpu-memory-utilization: 0.94
tensor-parallel-size: 8
decode-context-parallel-size: 8
cp-kv-cache-interleave-size: 1
dcp-comm-backend: a2a
enable-cumem-allocator: true
trust-remote-code: true
max-cudagraph-capture-size: 512
stream-interval: 10
language-model-only: true
attention-backend: TOKENSPEED_MLA
attention-config: '{"mla_prefill_backend": "TRTLLM_RAGGED", "use_prefill_query_quantization": true}'
speculative-config: '{"model":"Inferact/Kimi-K3-DSpark","attention_backend":"TOKENSPEED_MLA","method":"dspark","num_speculative_tokens":4,"rejection_sample_method":"block","draft_sample_method":"probabilistic"}'
enable-prefix-caching: true
enable-flashinfer-autotune: true
sbatch_directives:
cpus-per-task: '72'
mem: '0'
comment: '''{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"60","reason":"model_loading","description":"Very large model will take extra loading time."}}'''
srun_options:
mem: '0'
container-remap-root: ''
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
PORT: '8000'
IS_MULTINODE: 'true'
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0'
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true'
AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '14400'
AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache
HF_HUB_CACHE: /hf_hub_cache
WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126
client_placement: head
concurrencies:
- 48
identity:
model:
repo: moonshotai/Kimi-K3
container:
image: ghcr.io#semianalysisai/vllm-openai:efa_pr_58768
frameworks:
dynamo: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b
29 changes: 29 additions & 0 deletions configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -1528,6 +1528,35 @@ kimik3-fp4-b300-vllm-agentic-dspark:
# disjoint across arms so exp-names stay unique.
- { tp: 8, ep: 1, dcp-size: 8, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: mooncake, version: "0.3.11.post1" }, conc-list: [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 70], srt-recipe: benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml }

kimik3-fp4-b300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg-1p1d:
image: ghcr.io#semianalysisai/vllm-openai:efa_pr_58768
model: moonshotai/Kimi-K3
model-prefix: kimik3
runner: cluster:b300-dsxe
precision: fp4
framework: dynamo-vllm
router: { name: dynamo-router, version: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" }

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 (optional) The new kimik3-fp4-b300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg-1p1d entry sets router.version to ba83080ecd31c1ce918559e576d3c5bc9e092ff1, but the recipe it points at (disagg-1p1d-dcp8-dcp8-dspark4-mooncake-c48.yaml:11,237) builds dynamo from a different rev, cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b. Every sibling entry (e.g. nvidia-master.yaml:7841, kimik3-fp4-gb300-...-disagg) keeps router.version equal to its recipe's dynamo rev; this one doesn't. Per docs/results-and-ingestion.md, router ({name, version}) is stored as benchmark result metadata, so results from this scenario get tagged and grouped under the wrong dynamo-router build, silently mixing them into Pareto comparisons meant for a different dynamo revision. …

Why this was flagged

…Fix: set router.version to cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b to match the recipe it references.

configs/nvidia-master.yaml:1538 sets router.version to ba83080ecd31c1ce918559e576d3c5bc9e092ff1 for the new kimik3-fp4-b300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg-1p1d key. The recipe it references, benchmarks/multi_node/srt-slurm-recipes/kimik3/vllm/b300-fp4/agentx/disagg-1p1d-dcp8-dcp8-dspark4-mooncake-c48.yaml:11 and :237, sets the dynamo rev actually built into the image to cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b, a different commit. infx/matrix/validation.py's ComponentMetadata schema for router has no check comparing it to the recipe's dynamo rev, so nothing catches the mismatch. docs/results-and-ingestion.md documents router as {name, version} metadata stored with benchmark results, so results are recorded under the wrong dynamo-router version. Sibling entries like nvidia-master.yaml:7841 keep these values equal, showing this is the convention the new entry breaks.

Verification: Severity: normal — this new entry records wrong dynamo/router provenance. configs/nvidia-master.yaml:1538 sets router: { name: dynamo-router, version: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" }, but the recipe it references builds a different dynamo commit: disagg-1p1d-dcp8-dcp8-dspark4-mooncake-c48.yaml:11 dynamo.source.rev: cfada2fd9d17bfa6bb68dbee9d2f455e12577b8b (with `install:…

kv-p2p-transfer: nixl
multinode: true
disagg: true
scenarios:
agentic-coding:
- dram-utilization: 0.75

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 (optional) Anyone reading this scenario's recorded allocated_cpu_dram_gb metric gets a value ~12x too large, corrupting DRAM-offload capacity/efficiency comparisons for this benchmark. dram-utilization: 0.75 here computes total-cpu-dram-gb ≈ 2249 GB (full-node fraction for the 8-GPU prefill worker), which becomes TOTAL_CPU_DRAM_GB and is stored verbatim as allocated_cpu_dram_gb, but the recipe's actual centralized Mooncake pool is only global_segment_size: 190GB (disagg-1p1d-dcp8-dcp8-dspark4-mooncake-c48.yaml:40). Every sibling entry sizes dram-utilization to match its recipe's segment size (e.g. gb300's 0.1775 -> 160GB at nvidia-master.yaml:7849); this one looks copy-pasted from the single-node b300 kimik3 recipe's 0.75/2249GB pairing. …

Why this was flagged

…Fix: pick dram-utilization so computed total-cpu-dram-gb matches the recipe's real DRAM budget (~0.063 here for 190GB), per the convention every other multinode entry follows.

configs/nvidia-master.yaml:1544 sets dram-utilization: 0.75 under kimik3-fp4-b300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg-1p1d. infx/matrix/generate.py's agentic_dram_offload_gb (lines 429-450) uses the prefill worker's full-node GPU fraction (tp=8 on an 8-GPU node -> fraction 1) to turn that into total_cpu_dram_gb ≈ 2249, coincidentally equal to the unrelated single-node recipe's hardcoded TOTAL_CPU_DRAM_GB=2249 (benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml:109). benchmark-multinode-tmpl.yml:178 wires total-cpu-dram-gb into TOTAL_CPU_DRAM_GB; benchmark_lib.sh:212/406 only checks it's a positive integer, so the mismatch is never caught. infx/results/agentic/init.py:191 records it as allocated_cpu_dram_gb, while the recipe's real capacity is global_segment_size: 190GB (disagg-1p1d-dcp8-dcp8-dspark4-mooncake-c48.yaml:40) - about 12x smaller than what gets recorded.

Verification: normal (metric/data-correctness defect in newly added scenario; no serving impact). configs/nvidia-master.yaml:1544 sets dram-utilization: 0.75 for kimik3-fp4-b300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg-1p1d. infx/matrix/generate.py agentic_dram_offload_gb (lines 446-470) computes min(runner DRAM, MAX_AGENTIC_AVAILABLE_CPU_DRAM_MIB=2,861,022 MiB ≈ 2999 GB) × 0.75 × (prefill gpu_count…

search-space:
- spec-decoding: mtp
kv-offloading: dram
kv-offload-backend: { name: mooncake, version: "0.3.13.post1" }
conc-list: [48]
prefill:
num-worker: 1
tp: 8
dcp-size: 8
ep: 1
dp-attn: false
additional-settings:
- "CONFIG_FILE=recipes/kimik3/vllm/b300-fp4/agentx/disagg-1p1d-dcp8-dcp8-dspark4-mooncake-c48.yaml"
decode: { num-worker: 1, tp: 8, dcp-size: 8, ep: 1, dp-attn: false }

dsr1-fp8-b200-trt:
image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14
model: deepseek-ai/DeepSeek-R1-0528
Expand Down
9 changes: 9 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8986,3 +8986,12 @@
- "Hopper 使用 FA3,Blackwell FA4 fp8 descale 问题不适用,EAGLE3 draft 保持 attention_backend FLASH_ATTN。"
- "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3503

- config-keys:
- kimik3-fp4-b300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg-1p1d
scenario-type:
- agentic-coding
description:
- "Add the Kimi-K3 B300 Mooncake EFA AgentX disaggregated 1P1D DCP8/DCP8 DSpark4 configuration at c48 on ghcr.io/semianalysisai/vllm-openai:efa_pr_58768 (vllm-project/vllm#58768 vllm-openai-efa target, with AWS EFA libfabric 2.6.0amzn1.0 and mooncake-transfer-engine-efa-cuda13 0.3.13.post1 built in). Taken from #3405's 1P1D arm; drops its kimik3-b300-efa-setup.sh runtime installer."
- "新增基于 ghcr.io/semianalysisai/vllm-openai:efa_pr_58768(vllm-project/vllm#58768 的 vllm-openai-efa 目标,内置 AWS EFA libfabric 2.6.0amzn1.0 与 mooncake-transfer-engine-efa-cuda13 0.3.13.post1)的 Kimi-K3 B300 Mooncake EFA AgentX 分离式 1P1D DCP8/DCP8 DSpark4 配置(c48)。取自 #3405 的 1P1D 配置;删除其 kimik3-b300-efa-setup.sh 运行时安装脚本。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3523
14 changes: 12 additions & 2 deletions runners/launch_b300-dsxe.sh
Original file line number Diff line number Diff line change
Expand Up @@ -112,8 +112,18 @@ import_squash_image() {
if unsquashfs -l \"$sqsh\" > /dev/null 2>&1; then
exit 0
fi
rm -f \"$sqsh\"
enroot import -o \"$sqsh\" \"docker://$image_ref\"
# Large registry layers (e.g. GHCR) can drop mid-transfer with
# \"curl: (56) Connection reset by peer\". Retry with fewer parallel
# connections; enroot's layer cache keeps finished layers across attempts.
for attempt in 1 2 3 4; do
rm -f \"$sqsh\"
if ENROOT_MAX_CONNECTIONS=\"\${ENROOT_MAX_CONNECTIONS:-4}\" enroot import -o \"$sqsh\" \"docker://$image_ref\"; then
break
fi
[ \"\$attempt\" -eq 4 ] && exit 1
echo \"enroot import attempt \$attempt failed for $image_ref; retrying in \$((attempt * 30))s\" >&2
sleep \$((attempt * 30))
done
unsquashfs -l \"$sqsh\" > /dev/null
" || { echo "Error: enroot import failed for $image_ref -> $sqsh" >&2; exit 1; }

Expand Down
Loading