Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
22 changes: 14 additions & 8 deletions benchmarks/benchmark_lib.sh
Original file line number Diff line number Diff line change
Expand Up @@ -192,6 +192,8 @@ require_agentic_kv_offload_none() {

require_agentic_kv_offload_backend() {
local expected_backend="$1"
# Existing recipes support DRAM; callers opt in explicitly to additional tiers.
local supported_modes="dram ${2:-}"
if [[ -z "${KV_OFFLOADING+x}" || -z "$KV_OFFLOADING" ]]; then
echo "Error: KV_OFFLOADING must be set for agentic benchmarks" >&2
exit 1
Expand All @@ -204,19 +206,23 @@ require_agentic_kv_offload_backend() {
fi
return 1
;;
dram)
dram|nvme|dram+nvme)
if [[ " $supported_modes " != *" $KV_OFFLOADING "* ]]; then
echo "Error: this recipe does not support $KV_OFFLOADING with $expected_backend" >&2
exit 1
fi
if [[ "${KV_OFFLOAD_BACKEND:-}" != "$expected_backend" ]]; then
echo "Error: expected KV_OFFLOAD_BACKEND=$expected_backend when KV_OFFLOADING=dram, got '${KV_OFFLOAD_BACKEND:-}'" >&2
echo "Error: expected KV_OFFLOAD_BACKEND=$expected_backend when KV_OFFLOADING=$KV_OFFLOADING, got '${KV_OFFLOAD_BACKEND:-}'" >&2
exit 1
fi
if [[ ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then
if [[ "$KV_OFFLOADING" != nvme && ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then
echo "Error: DRAM KV offloading requires a positive TOTAL_CPU_DRAM_GB capacity" >&2
exit 1
fi
return 0
;;
*)
echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected one of: none, dram)" >&2
echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected none, dram, nvme, or dram+nvme)" >&2
exit 1
;;
esac
Expand Down Expand Up @@ -398,18 +404,18 @@ if [[ "$_benchmark_caller" == */agentic/* ||
exit 1
fi
;;
dram)
dram|nvme|dram+nvme)
if [[ -z "${KV_OFFLOAD_BACKEND:-}" || "${KV_OFFLOAD_BACKEND:-}" == "none" ]]; then
echo "Error: KV_OFFLOAD_BACKEND is required when KV_OFFLOADING=dram" >&2
echo "Error: KV_OFFLOAD_BACKEND is required when KV_OFFLOADING=$KV_OFFLOADING" >&2
exit 1
fi
if [[ ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then
if [[ "$KV_OFFLOADING" != nvme && ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then
echo "Error: DRAM KV offloading requires a positive configured TOTAL_CPU_DRAM_GB capacity" >&2
exit 1
fi
;;
*)
echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected one of: none, dram)" >&2
echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected none, dram, nvme, or dram+nvme)" >&2
exit 1
;;
esac
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,230 @@
# PR56318 cache-source validation: preserve run34406466616 resources.
# One DEP4 prefill worker/node and one DEP16 decode worker across four nodes.
schema: 2
name: "dsv4-vllm-cache-sources-gb300-dep4-dep16-c256"

# GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one
# DEP16 decode worker at concurrency 256. Decode consumes P/D KV through NIXL
# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead.

model:
path: "deepseek-v4-pro"
container: "cquil11/vllm-cache-sources@sha256:6ce87fcb9ef8b929f7a8270bc6fbba3c219c3440cd77d3bf1ae74db5a1cb5715"
precision: "fp4"

identity:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro"
container:
image: "cquil11/vllm-cache-sources@sha256:6ce87fcb9ef8b929f7a8270bc6fbba3c219c3440cd77d3bf1ae74db5a1cb5715"
frameworks:
dynamo: "1.4.0"

dynamo:
install: true

source:
wheel: "1.4.0"
environment:
# Mooncake prefix-block hashes must match across processes and nodes.
PYTHONHASHSEED: "0"

setup_script: vllm-container-deps.sh

slurm:
time_limit: "8:00:00"

health_check:
max_attempts: 2160
interval_seconds: 10

resources:
gpu_type: "gb300"
gpus_per_node: 4
het_jobs: false
spread_workers: false
services:
- name: etcd
type: etcd
placement:
node: infra
- name: nats
type: nats
placement:
node: infra
options:
max_payload_mb: 32
- name: mooncake-master
type: mooncake-master
options:
store_config:
metadata_server: "P2PHANDSHAKE"
global_segment_size: "180GB"
local_buffer_size: "4GB"
protocol: "rdma"
device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
mode: "embedded"
enable_offload: false
frontend:
type: dynamo
enable_multiple_frontends: false
args:
router-mode: "random"
router-session-affinity-ttl-secs: 900
env:
DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600"
DYN_TCP_CHANNEL_BUFFER: "128"
DYN_TCP_REQUEST_TIMEOUT: "60"

engine:
type: vllm
connector:
dp_launch_mode: per_node
roles:
prefill:
nodes: 1
workers: 1
gpus: 4
env:
HF_HUB_CACHE: "/hf_hub_cache"
HUGGINGFACE_HUB_CACHE: "/hf_hub_cache"
TRANSFORMERS_CACHE: "/hf_hub_cache"
VLLM_ENGINE_READY_TIMEOUT_S: "3600"
VLLM_RPC_TIMEOUT: "600000"
VLLM_LOG_STATS_INTERVAL: "1"
VLLM_V2_WARMUP_MAX_NUM_SEQS: "20"
VLLM_SERVER_DEV_MODE: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_USE_RUST_FRONTEND: "0"
VLLM_MOONCAKE_LOAD_RECV_THREADS: "20"
VLLM_MOONCAKE_STORE_SEND_THREADS: "8"
VLLM_ALLREDUCE_USE_SYMM_MEM: "0"
UCX_MEMTYPE_CACHE: "n"
UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1"
UCX_TLS: "rc,cuda_copy"
NCCL_CUMEM_ENABLE: "1"
NCCL_MNNVL_ENABLE: "1"
NCCL_NVLS_ENABLE: "1"
VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1"
NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0"
VLLM_USE_BREAKABLE_CUDAGRAPH: "1"
VLLM_CONNECTOR_PREFETCH_DEPTH: "8"
VLLM_DSV4_MEGA_FP8_COMBINE: "1"
DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep8-prefill-c256-{job_id}"
MC_ENABLE_DEST_DEVICE_AFFINITY: "1"
MC_STORE_CLIENT_METRIC: "1"
MC_STORE_CLIENT_METRIC_INTERVAL: "5"
MC_TE_METRIC: "0"
args:
kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}'
served-model-name: "deepseek-ai/DeepSeek-V4-Pro"
safetensors-load-strategy: "prefetch"
kv-cache-dtype: "fp8"
tensor-parallel-size: 1
pipeline-parallel-size: 1
data-parallel-size: 4
data-parallel-rpc-port: 13345
enable-cumem-allocator: true
enable-expert-parallel: true
enable-ep-weight-filter: true
max-model-len: 1048576
max-num-seqs: 256
max-num-batched-tokens: 8192
long-prefill-token-threshold: 1024
trust-remote-code: true
no-enable-flashinfer-autotune: true
block-size: 256
gpu-memory-utilization: 0.92
no-disable-hybrid-kv-cache-manager: true
tokenizer-mode: "deepseek_v4"
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}'
speculative-config: '{"method":"mtp","num_speculative_tokens":3}'
moe-backend: "deep_gemm_mega_moe"
numa-bind: true
numa-bind-nodes: [0, 0, 1, 1]
decode:
nodes: 4
workers: 1
gpus: 16

env:
HF_HUB_CACHE: "/hf_hub_cache"
HUGGINGFACE_HUB_CACHE: "/hf_hub_cache"
TRANSFORMERS_CACHE: "/hf_hub_cache"
VLLM_ENGINE_READY_TIMEOUT_S: "3600"
VLLM_RPC_TIMEOUT: "600000"
VLLM_LOG_STATS_INTERVAL: "1"
VLLM_V2_WARMUP_MAX_NUM_SEQS: "20"
VLLM_SERVER_DEV_MODE: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_USE_RUST_FRONTEND: "0"
VLLM_MOONCAKE_LOAD_RECV_THREADS: "4"
VLLM_ALLREDUCE_USE_SYMM_MEM: "0"
UCX_MEMTYPE_CACHE: "n"
UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1"
UCX_TLS: "rc,cuda_copy"
NCCL_CUMEM_ENABLE: "1"
NCCL_MNNVL_ENABLE: "1"
NCCL_NVLS_ENABLE: "1"
VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1"
NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0"
VLLM_DSV4_MEGA_FP8_COMBINE: "1"
DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep32-decode-c256-{job_id}"
MC_ENABLE_DEST_DEVICE_AFFINITY: "1"
MC_STORE_CLIENT_METRIC: "1"
MC_STORE_CLIENT_METRIC_INTERVAL: "5"
MC_TE_METRIC: "0"

args:
kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}'
served-model-name: "deepseek-ai/DeepSeek-V4-Pro"
safetensors-load-strategy: "prefetch"
kv-cache-dtype: "fp8"
tensor-parallel-size: 1
pipeline-parallel-size: 1
data-parallel-size: 16
data-parallel-rpc-port: 13345
enable-cumem-allocator: true
enable-expert-parallel: true
enable-ep-weight-filter: true
max-model-len: 1048576
max-num-seqs: 8
max-num-batched-tokens: 32
trust-remote-code: true
no-enable-flashinfer-autotune: true
block-size: 256
compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}'
max-cudagraph-capture-size: 32
gpu-memory-utilization: 0.90
no-disable-hybrid-kv-cache-manager: true
tokenizer-mode: "deepseek_v4"
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}'
speculative-config: '{"method":"mtp","num_speculative_tokens":3}'
moe-backend: "deep_gemm_mega_moe"
numa-bind: true
numa-bind-nodes: [0, 0, 1, 1]
sbatch_directives:
cpus-per-task: "72"
mem: "0"

srun_options:
container-remap-root: ""

benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:prompt_tokens_cached_by_source"
INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace"
RESULT_DIR: "/logs/agentic"
PORT: "8000"
IS_MULTINODE: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true"
# Avoid concurrent readers observing a mismatched mmap data/index pair.
AIPERF_DATASET_MMAP_CACHE_ENABLED: "false"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126"
Loading
Loading