diff --git a/inferencex-e2e/benchmarks/benchmark_lib.sh b/inferencex-e2e/benchmarks/benchmark_lib.sh index 39a7e22df3..f16c92f86b 100644 --- a/inferencex-e2e/benchmarks/benchmark_lib.sh +++ b/inferencex-e2e/benchmarks/benchmark_lib.sh @@ -199,6 +199,8 @@ require_agentic_kv_offload_none() { require_agentic_kv_offload_backend() { local expected_backend="$1" + # Existing recipes support DRAM; callers opt in explicitly to additional tiers. + local supported_modes="dram ${2:-}" if [[ -z "${KV_OFFLOADING+x}" || -z "$KV_OFFLOADING" ]]; then echo "Error: KV_OFFLOADING must be set for agentic benchmarks" >&2 exit 1 @@ -211,19 +213,23 @@ require_agentic_kv_offload_backend() { fi return 1 ;; - dram) + dram|nvme|dram+nvme) + if [[ " $supported_modes " != *" $KV_OFFLOADING "* ]]; then + echo "Error: this recipe does not support $KV_OFFLOADING with $expected_backend" >&2 + exit 1 + fi if [[ "${KV_OFFLOAD_BACKEND:-}" != "$expected_backend" ]]; then - echo "Error: expected KV_OFFLOAD_BACKEND=$expected_backend when KV_OFFLOADING=dram, got '${KV_OFFLOAD_BACKEND:-}'" >&2 + echo "Error: expected KV_OFFLOAD_BACKEND=$expected_backend when KV_OFFLOADING=$KV_OFFLOADING, got '${KV_OFFLOAD_BACKEND:-}'" >&2 exit 1 fi - if [[ ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then + if [[ "$KV_OFFLOADING" != nvme && ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then echo "Error: DRAM KV offloading requires a positive TOTAL_CPU_DRAM_GB capacity" >&2 exit 1 fi return 0 ;; *) - echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected one of: none, dram)" >&2 + echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected none, dram, nvme, or dram+nvme)" >&2 exit 1 ;; esac @@ -405,18 +411,18 @@ if [[ "$_benchmark_caller" == */agentic/* || exit 1 fi ;; - dram) + dram|nvme|dram+nvme) if [[ -z "${KV_OFFLOAD_BACKEND:-}" || "${KV_OFFLOAD_BACKEND:-}" == "none" ]]; then - echo "Error: KV_OFFLOAD_BACKEND is required when KV_OFFLOADING=dram" >&2 + echo "Error: KV_OFFLOAD_BACKEND is required when KV_OFFLOADING=$KV_OFFLOADING" >&2 exit 1 fi - if [[ ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then + if [[ "$KV_OFFLOADING" != nvme && ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then echo "Error: DRAM KV offloading requires a positive configured TOTAL_CPU_DRAM_GB capacity" >&2 exit 1 fi ;; *) - echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected one of: none, dram)" >&2 + echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected none, dram, nvme, or dram+nvme)" >&2 exit 1 ;; esac diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml new file mode 100644 index 0000000000..41aa39a9a9 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml @@ -0,0 +1,230 @@ +# PR56318 cache-source validation: preserve run34406466616 resources. +# One DEP4 prefill worker/node and one DEP16 decode worker across four nodes. +schema: 2 +name: "dsv4-vllm-cache-sources-gb300-dep4-dep16-c256" + +# GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one +# DEP16 decode worker at concurrency 256. Decode consumes P/D KV through NIXL +# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. + +model: + path: "deepseek-v4-pro" + container: "cquil11/vllm-cache-sources@sha256:188445e9835a0ed9cd022f0e70c61dd3fd9d24792008841d4747da1797e7fe4b" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro" + container: + image: "cquil11/vllm-cache-sources@sha256:188445e9835a0ed9cd022f0e70c61dd3fd9d24792008841d4747da1797e7fe4b" + frameworks: + dynamo: "1.4.0" + +dynamo: + install: true + + source: + wheel: "1.4.0" +environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: "0" + +setup_script: vllm-container-deps.sh + +slurm: + time_limit: "8:00:00" + +health_check: + max_attempts: 2160 + interval_seconds: 10 + +resources: + gpu_type: "gb300" + gpus_per_node: 4 + het_jobs: false + spread_workers: false +services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: "P2PHANDSHAKE" + global_segment_size: "180GB" + local_buffer_size: "4GB" + protocol: "rdma" + device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" + mode: "embedded" + enable_offload: false +frontend: + type: dynamo + enable_multiple_frontends: false + args: + router-mode: "random" + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" + DYN_TCP_CHANNEL_BUFFER: "128" + DYN_TCP_REQUEST_TIMEOUT: "60" + +engine: + type: vllm + connector: + dp_launch_mode: per_node +roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HUB_CACHE: "/hf_hub_cache" + HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" + TRANSFORMERS_CACHE: "/hf_hub_cache" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_RPC_TIMEOUT: "600000" + VLLM_LOG_STATS_INTERVAL: "1" + VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" + VLLM_SERVER_DEV_MODE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_USE_RUST_FRONTEND: "0" + VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" + VLLM_MOONCAKE_STORE_SEND_THREADS: "8" + VLLM_ALLREDUCE_USE_SYMM_MEM: "0" + UCX_MEMTYPE_CACHE: "n" + UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" + UCX_TLS: "rc,cuda_copy" + NCCL_CUMEM_ENABLE: "1" + NCCL_MNNVL_ENABLE: "1" + NCCL_NVLS_ENABLE: "1" + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" + NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" + VLLM_USE_BREAKABLE_CUDAGRAPH: "1" + VLLM_CONNECTOR_PREFETCH_DEPTH: "8" + VLLM_DSV4_MEGA_FP8_COMBINE: "1" + DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep8-prefill-c256-{job_id}" + MC_ENABLE_DEST_DEVICE_AFFINITY: "1" + MC_STORE_CLIENT_METRIC: "1" + MC_STORE_CLIENT_METRIC_INTERVAL: "5" + MC_TE_METRIC: "0" + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: "deepseek-ai/DeepSeek-V4-Pro" + safetensors-load-strategy: "prefetch" + kv-cache-dtype: "fp8" + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 256 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.92 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: "deepseek_v4" + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: "deep_gemm_mega_moe" + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 4 + workers: 1 + gpus: 16 + + env: + HF_HUB_CACHE: "/hf_hub_cache" + HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" + TRANSFORMERS_CACHE: "/hf_hub_cache" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_RPC_TIMEOUT: "600000" + VLLM_LOG_STATS_INTERVAL: "1" + VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" + VLLM_SERVER_DEV_MODE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_USE_RUST_FRONTEND: "0" + VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" + VLLM_ALLREDUCE_USE_SYMM_MEM: "0" + UCX_MEMTYPE_CACHE: "n" + UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" + UCX_TLS: "rc,cuda_copy" + NCCL_CUMEM_ENABLE: "1" + NCCL_MNNVL_ENABLE: "1" + NCCL_NVLS_ENABLE: "1" + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" + NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" + VLLM_DSV4_MEGA_FP8_COMBINE: "1" + DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep32-decode-c256-{job_id}" + MC_ENABLE_DEST_DEVICE_AFFINITY: "1" + MC_STORE_CLIENT_METRIC: "1" + MC_STORE_CLIENT_METRIC_INTERVAL: "5" + MC_TE_METRIC: "0" + + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: "deepseek-ai/DeepSeek-V4-Pro" + safetensors-load-strategy: "prefetch" + kv-cache-dtype: "fp8" + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 16 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 8 + max-num-batched-tokens: 32 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 32 + gpu-memory-utilization: 0.90 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: "deepseek_v4" + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: "deep_gemm_mega_moe" + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] +sbatch_directives: + cpus-per-task: "72" + mem: "0" + +srun_options: + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:prompt_tokens_cached_by_source" + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + IS_MULTINODE: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" + WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml new file mode 100644 index 0000000000..701a48e8d2 --- /dev/null +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml @@ -0,0 +1,127 @@ +# PR56318 cache-source validation: DeepSeek-V4-Pro TP8/MTP3 on B200 with +# three KV offload modes (native DRAM, Simple NVMe, native DRAM+NVMe). +# One variant per search-space row. +base: + schema: 2 + name: dsv4-fp4-b200-vllm-cache-sources + model: + path: hf:deepseek-ai/DeepSeek-V4-Pro + container: cquil11/vllm-cache-sources@sha256:873b8bcb4cc734cfc3e3b7f9d2c1495be329439a944f76a7b356448be6a61641 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + health_check: + interval_seconds: 10 + max_attempts: 720 + setup_script: pr56318-overlay-check.sh + roles: + agg: + nodes: 1 + workers: 1 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro + trust-remote-code: true + kv-cache-dtype: fp8 + block-size: 256 + max-model-len: 1048576 + gpu-memory-utilization: 0.85 + numa-bind: true + enable-cumem-allocator: true + no-enable-flashinfer-autotune: true + tokenizer-mode: deepseek_v4 + tool-call-parser: deepseek_v4 + enable-auto-tool-choice: true + reasoning-parser: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + no-disable-hybrid-kv-cache-manager: true + disable-uvicorn-access-log: true + tensor-parallel-size: 8 + data-parallel-size: 1 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '0' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + VLLM_RPC_TIMEOUT: '600000' + PYTHONHASHSEED: '42' + TORCH_CUDA_ARCH_LIST: '10.0' + PYTHONNOUSERSITE: '1' + VLLM_FLOAT32_MATMUL_PRECISION: high + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:prompt_tokens_cached_by_source' + +# ---------- Variant 1: TP8 c8 native DRAM ---------- +# max-num-seqs = 2 * CONC = 16; capture sizes = {4*n for n=1..16} ∪ {100..500} +override_tp8_c8_dram_native: + roles: + agg: + gpus: 8 + args: + max-num-seqs: 16 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,100,200,300,400,500]}' + kv-transfer-config: '{"kv_connector":"OffloadingConnector","kv_role":"kv_both","kv_connector_extra_config":{"spec_name":"TieringOffloadingSpec","cpu_bytes_to_use":128000000000,"secondary_tiers":[]}}' + benchmark: + env: + CONC: '8' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '128' + +# ---------- Variant 2: TP8 c14 Simple NVMe ---------- +# max-num-seqs = 2 * CONC = 28; disk_capacity_bytes = 1e12 / 8 +override_tp8_c14_nvme_simple: + container_mounts: + "/scratch/inferencex-kv-{job_id}": "/kv-offload" + host_setup: + commands: ["mkdir -m 700 /scratch/inferencex-kv-$SLURM_JOB_ID"] + teardown: ["rm -rf /scratch/inferencex-kv-$SLURM_JOB_ID"] + nodes: workers + roles: + agg: + gpus: 8 + args: + max-num-seqs: 28 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,200,300,400,500]}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"kv_offload_backend":"disk","disk_path":"/kv-offload/cache.bin","disk_capacity_bytes":125000000000,"disk_buffer_slots":4,"lazy_offload":false}}' + benchmark: + env: + CONC: '14' + KV_OFFLOADING: nvme + +# ---------- Variant 3: TP8 c14 native DRAM + NVMe ---------- +# max-num-seqs = 2 * CONC = 28; cpu_bytes = 128e9, fs tier at /kv-offload +override_tp8_c14_dramnvme_native: + container_mounts: + "/scratch/inferencex-kv-{job_id}": "/kv-offload" + host_setup: + commands: ["mkdir -m 700 /scratch/inferencex-kv-$SLURM_JOB_ID"] + teardown: ["rm -rf /scratch/inferencex-kv-$SLURM_JOB_ID"] + nodes: workers + roles: + agg: + gpus: 8 + args: + max-num-seqs: 28 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,200,300,400,500]}' + kv-transfer-config: '{"kv_connector":"OffloadingConnector","kv_role":"kv_both","kv_connector_extra_config":{"spec_name":"TieringOffloadingSpec","cpu_bytes_to_use":128000000000,"secondary_tiers":[{"type":"fs","root_dir":"/kv-offload","locality":"LOCAL"}]}}' + benchmark: + env: + CONC: '14' + KV_OFFLOADING: dram+nvme + TOTAL_CPU_DRAM_GB: '128' diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 63dcf8fff8..5b8d2710e3 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8383,6 +8383,59 @@ dsv41flash-fp4-b200-sglang-agentic-dspark: - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml } - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml } +# PR56318: rerun the four successful hybrid cache-source validation points. +dsv4-fp4-b200-vllm-agentic-cache-sources-mtp: + image: cquil11/vllm-cache-sources@sha256:873b8bcb4cc734cfc3e3b7f9d2c1495be329439a944f76a7b356448be6a61641 + model: deepseek-ai/DeepSeek-V4-Pro + model-prefix: dsv4 + runner: cluster:b200-nscale + precision: fp4 + framework: vllm + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.0592 + search-space: + - { tp: 8, kv-offloading: dram, kv-offload-backend: { name: vllm-native }, spec-decoding: mtp, conc-list: [8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml } + - search-space: + - { tp: 8, kv-offloading: nvme, kv-offload-backend: { name: vllm-simple }, spec-decoding: mtp, conc-list: [14], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml } + - dram-utilization: 0.0592 + search-space: + - { tp: 8, kv-offloading: [dram, nvme], kv-offload-backend: { name: vllm-native }, spec-decoding: mtp, conc-list: [14], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml } + +dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg: + image: cquil11/vllm-cache-sources@sha256:188445e9835a0ed9cd022f0e70c61dd3fd9d24792008841d4747da1797e7fe4b + model: deepseek-ai/DeepSeek-V4-Pro + model-prefix: dsv4 + runner: cluster:gb300-nv + precision: fp4 + framework: dynamo-vllm + kv-p2p-transfer: nixl + multinode: true + disagg: true + scenarios: + agentic-coding: + - dram-utilization: 0.857143 + search-space: + - spec-decoding: mtp + conc-list: [256] + kv-offloading: dram + kv-offload-backend: { name: mooncake, version: "0.3.13.post1" } + router: { name: dynamo-router, version: "1.4.0" } + prefill: + num-worker: 1 + tp: 4 + ep: 4 + dp-attn: true + additional-settings: + - "SLURM_PARTITION=batch_1" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml" + decode: + num-worker: 1 + tp: 16 + ep: 16 + dp-attn: true + dsv41flash-fp4-gb300-vllm-agentic-dspark: image: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e model: deepseek-ai/DeepSeek-V4.1-Flash diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index d2df756784..b3ff95e7c2 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -23,6 +23,12 @@ Use this page for benchmark configuration, recipe, image, and runner changes. It Delete retired entries from the active master configs; they are not archived. Git history and `perf-changelog.yaml` keep the historical settings. For partial retirements, remove only the retired scenarios. Delete unused recipes and model-specific setup as well. Preserve shared dependencies needed by retained SPEED-Bench collectors, including their scheduling scores. See the [deprecation rules](../../AGENTS.md#deprecating-benchmark-configs). +## Cache-source validation + +Single-node AgentX configs accept `kv-offloading: nvme` or `[dram, nvme]` in addition to `none` and `dram`. Tiered entries still require `dram-utilization`, which budgets host memory only. The dedicated DeepSeek-V4-Pro B200 validation recipe supports Simple NVMe and native DRAM/NVMe. Its launcher mounts a job-owned `/scratch/inferencex-kv-` directory and removes it before releasing the allocation. Other recipes must explicitly opt in to NVMe modes. + +The dedicated GB300 cache-source recipe applies a job-local srt-slurm patch to scrape every physical DP worker, including nonleader nodes. It verifies the image's PR-overlay checksums after dependency installation. This patch does not run for other recipes. + ## Dependency submodules Git records the exact dependency commits. [`.gitmodules`](../../.gitmodules) defines the repositories: AIPerf at `utils/aiperf`, NVIDIA srt-slurm at `utils/srt-slurm`. All srt-slurm jobs, including TileRT, use the pinned upstream submodule. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 0d023deb1c..09b59fac1c 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -23,6 +23,12 @@ 退役的配置项直接从启用的主配置中删除,不再归档;历史设置由 Git 历史和 `perf-changelog.yaml` 保留。仅弃用部分场景时,只删除已退役的场景。同时删除不再使用的配方和模型专用初始化逻辑。保留 SPEED-Bench 采集器仍需使用的共享依赖,包括调度评分。参见[弃用规则](../../AGENTS.md#deprecating-benchmark-configs)。 +## 缓存来源验证 + +单节点 AgentX 配置除了 `none` 和 `dram`,还接受 `kv-offloading: nvme` 或 `[dram, nvme]`。分层配置仍须提供 `dram-utilization`,且该值仅用于分配主机内存。专用 DeepSeek-V4-Pro B200 验证配方支持 Simple NVMe 和原生 DRAM/NVMe。launcher 挂载作业专属的 `/scratch/inferencex-kv-` 目录,并在释放资源前将其删除。其他配方必须显式启用 NVMe 模式。 + +专用 GB300 缓存来源配方在作业本地的 srt-slurm 副本中应用补丁,采集每个物理 DP worker 的指标,包括非 leader 节点。依赖安装后会验证镜像中 PR 覆盖文件的校验和。其他配方不会应用此补丁。 + ## 依赖子模块 Git 记录依赖的精确提交版本。[`.gitmodules`](../../.gitmodules) 定义各仓库:AIPerf 位于 `utils/aiperf`,NVIDIA srt-slurm 位于 `utils/srt-slurm`。包括 TileRT 在内的所有 srt-slurm 作业均使用固定的上游子模块。 diff --git a/inferencex-e2e/infx/launch/drivers/srt/checkout.py b/inferencex-e2e/infx/launch/drivers/srt/checkout.py index 822ef43e3b..42c030fb39 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/checkout.py +++ b/inferencex-e2e/infx/launch/drivers/srt/checkout.py @@ -84,6 +84,12 @@ def prepare_checkout(run: SrtRun, destination: Path, *, power: bool) -> Checkout ) for patch in sorted((run.workspace / PATCHES).glob("*.patch")): _git("-C", destination, "apply", patch) + if run.request.config_file == ( + "recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml" + ): + _git( + "-C", destination, "apply", run.workspace / "runners/srt-slurm/validation/pr56318.patch" + ) head = _git("-C", destination, "rev-parse", "HEAD", capture=True) if head != commit: raise LaunchError(f"srt-slurm checkout is at {head}, expected {commit}") diff --git a/inferencex-e2e/infx/launch/drivers/srt/models.py b/inferencex-e2e/infx/launch/drivers/srt/models.py index bb7f701534..63ff070d23 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/models.py +++ b/inferencex-e2e/infx/launch/drivers/srt/models.py @@ -39,6 +39,16 @@ class Override: OVERRIDES: dict[str, tuple[Override, ...]] = { "b200-nscale": ( + Override( + Match( + any_of("dsv4"), + any_of("fp4"), + any_of("vllm"), + multinode=False, + model_glob="deepseek-ai/DeepSeek-V4-Pro", + ), + entry="DeepSeek-V4-Pro-NVFP4", + ), Override( Match(any_of("glm5.1"), frameworks=any_of("tilert")), entry="GLM-5.1-FP8@shared", diff --git a/inferencex-e2e/infx/matrix/generate.py b/inferencex-e2e/infx/matrix/generate.py index 4b64f011df..ffe7158cc1 100644 --- a/inferencex-e2e/infx/matrix/generate.py +++ b/inferencex-e2e/infx/matrix/generate.py @@ -444,7 +444,7 @@ def agentic_dram_offload_gb( budgeted separately if it ever gains its own pool). """ kv_offloading = benchmark.get(Fields.KV_OFFLOADING.value, "none") - if kv_offloading != "dram": + if kv_offloading not in ("dram", ["dram", "nvme"]): return 0 available_mib = min( @@ -475,13 +475,14 @@ def agentic_dram_offload_gb( def agentic_kv_offload_suffix( - kv_offloading: str, + kv_offloading: str | list[str], kv_offload_backend: dict | None, ) -> str: """Return a compact exp-name suffix for agentic KV offload settings.""" if kv_offloading == "none": return "kvnone" - return f"kv{kv_offloading}-{kv_offload_backend['name']}" + mode = "+".join(kv_offloading) if isinstance(kv_offloading, list) else kv_offloading + return f"kv{mode}-{kv_offload_backend['name']}" def multinode_agentic_exp_name( @@ -991,7 +992,9 @@ def _agentic_entries( ) entry.update( { - Fields.KV_OFFLOADING.value: kv_offloading, + Fields.KV_OFFLOADING.value: ( + "+".join(kv_offloading) if isinstance(kv_offloading, list) else kv_offloading + ), Fields.TOTAL_CPU_DRAM_GB.value: total_cpu_dram_gb, Fields.DURATION.value: DEFAULT_AGENTIC_DURATION_SECONDS, Fields.EXP_NAME.value: exp_name, diff --git a/inferencex-e2e/infx/matrix/validation.py b/inferencex-e2e/infx/matrix/validation.py index f44d383ca4..0b5d5b2909 100644 --- a/inferencex-e2e/infx/matrix/validation.py +++ b/inferencex-e2e/infx/matrix/validation.py @@ -15,6 +15,7 @@ from infx.clusters import CLUSTER_LABEL_PREFIX, RunnerInventory DEFAULT_AGENTIC_DURATION_SECONDS = 3600 +type KVOffloadingConfig = Literal["none", "dram", "nvme"] | list[Literal["dram", "nvme"]] """ The below class defines the field names expected to be present in the JSON entries @@ -325,7 +326,9 @@ class SingleNodeAgenticMatrixEntry(BaseModel): default="none", alias=Fields.SPEC_DECODING.value ) conc: int - kv_offloading: Literal["none", "dram"] = Field(alias=Fields.KV_OFFLOADING.value) + kv_offloading: Literal["none", "dram", "nvme", "dram+nvme"] = Field( + alias=Fields.KV_OFFLOADING.value + ) kv_offload_backend: KVOffloadBackendMetadata | None = Field( default=None, alias=Fields.KV_OFFLOAD_BACKEND.value ) @@ -516,7 +519,10 @@ def _validate_kv_offload_fields(self: Any) -> Any: f"{Fields.KV_OFFLOAD_BACKEND.value} requires {Fields.KV_OFFLOADING.value}" ) return self - if self.kv_offloading == "none": + if isinstance(self.kv_offloading, list): + if self.kv_offloading != ["dram", "nvme"]: + raise ValueError("The only supported tier list is ['dram', 'nvme']") + elif self.kv_offloading == "none": if backend is not None: raise ValueError( f"{Fields.KV_OFFLOAD_BACKEND.value} can only be set when " @@ -642,9 +648,7 @@ class AgenticCodingSearchSpaceEntry(BaseModel): prefill: WorkerConfig | None = None decode: WorkerConfig | None = None num_nodes: int | None = Field(default=None, alias=Fields.NUM_NODES.value, gt=0, strict=True) - kv_offloading: Literal["none", "dram"] | None = Field( - default=None, alias=Fields.KV_OFFLOADING.value - ) + kv_offloading: KVOffloadingConfig | None = Field(default=None, alias=Fields.KV_OFFLOADING.value) kv_offload_backend: KVOffloadBackendMetadata | None = Field( default=None, alias=Fields.KV_OFFLOAD_BACKEND.value ) @@ -690,6 +694,8 @@ def validate_topology_fields(self) -> Self: ) _validate_tp_context_topology(self) if has_aggregate_worker or has_complete_multinode: + if self.kv_offloading in ("nvme", ["dram", "nvme"]): + raise ValueError("NVMe offloading currently requires a single-node entry") explicitly_single_node_fields = { "pp", "dcp_size", @@ -725,7 +731,7 @@ class AgenticCodingConfig(BaseModel): @model_validator(mode="after") def validate_dram_offload_capacity(self) -> Self: for entry in self.search_space: - if entry.kv_offloading != "dram": + if entry.kv_offloading not in ("dram", ["dram", "nvme"]): continue if self.dram_utilization is None: raise ValueError( diff --git a/inferencex-e2e/infx/tests/launch/test_srt_policy.py b/inferencex-e2e/infx/tests/launch/test_srt_policy.py index cbaf6d97da..4893549428 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_policy.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_policy.py @@ -106,6 +106,44 @@ def test_a_checkpoint_that_must_be_readable_fails_before_submission(tmp_path, mo assert host_path(c, checkpoint(c, request(MODEL="org/S"))) == tmp_path / "shared/s" +@pytest.mark.parametrize( + ("model", "framework", "multinode", "directory"), + [ + ("DeepSeek-V4-Pro", "vllm", "false", "converted"), + ("DeepSeek-V4-Pro-0813", "vllm", "false", "august"), + ("DeepSeek-V4-Pro", "vllm", "true", "original"), + ("DeepSeek-V4-Pro", "sglang", "false", "original"), + ], +) +def test_b200_original_pro_single_node_vllm_keeps_converted_checkpoint( + tmp_path, + model, + framework, + multinode, + directory, +): + c = cluster(tmp_path) + c.bind_id("b200-nscale") + entry = c.models.entries["M"] + entries = { + name: entry.model_copy(update={"dir": path}) + for name, path in { + "DeepSeek-V4-Pro": "original", + "DeepSeek-V4-Pro-NVFP4": "converted", + "DeepSeek-V4-Pro-0813": "august", + }.items() + } + c = c.model_copy(update={"models": c.models.model_copy(update={"entries": entries})}) + point = request( + MODEL=f"deepseek-ai/{model}", + MODEL_PREFIX="dsv4", + PRECISION="fp4", + FRAMEWORK=framework, + IS_MULTINODE=multinode, + ) + assert single_node_model_path(c, point) == str(tmp_path / "shared" / directory) + + def test_a_points_own_model_path_is_what_its_job_serves(tmp_path, monkeypatch): c = cluster(tmp_path) host = request(MODEL="org/M", MODEL_PATH="/host/m") diff --git a/inferencex-e2e/infx/tests/matrix/test_generate_sweep_configs.py b/inferencex-e2e/infx/tests/matrix/test_generate_sweep_configs.py index 02b8a1198b..d33e7386bb 100644 --- a/inferencex-e2e/infx/tests/matrix/test_generate_sweep_configs.py +++ b/inferencex-e2e/infx/tests/matrix/test_generate_sweep_configs.py @@ -2374,6 +2374,34 @@ def agentic_config(request, sample_single_node_config): class TestAgenticGeneration: + @pytest.mark.parametrize(("mode", "runtime", "budget"), [ + ("dram", "dram", 1199), + ("nvme", "nvme", 0), + (["dram", "nvme"], "dram+nvme", 1199), + ]) + def test_offload_modes_preserve_budget_and_artifact_identity( + self, sample_single_node_config, sample_runner_config, + generate_agentic_sweep, mode, runtime, budget, + ): + config = copy.deepcopy(sample_single_node_config) + entry = next(iter(config.values())) + entry.update(runner="cluster:b300-nv", multinode=False) + entry["scenarios"] = {"agentic-coding": [{ + "dram-utilization": 0.80, + "search-space": [{ + "tp": 4, "kv-offloading": mode, + "kv-offload-backend": {"name": "vllm-native"}, "conc-list": [8], + }], + }]} + original = copy.deepcopy(config) + rows = generate_agentic_sweep(config, sample_runner_config) + assert rows + for row in rows: + assert row["kv-offloading"] == runtime + assert row["total-cpu-dram-gb"] == budget + assert f"kv{runtime}-vllm-native" in row["exp-name"] + assert config == original + def test_point_order_and_input_preservation( self, agentic_config, sample_runner_config, generate_agentic_sweep, ): diff --git a/inferencex-e2e/infx/tests/matrix/test_validation.py b/inferencex-e2e/infx/tests/matrix/test_validation.py index 9d52b65446..7f0341d736 100644 --- a/inferencex-e2e/infx/tests/matrix/test_validation.py +++ b/inferencex-e2e/infx/tests/matrix/test_validation.py @@ -259,6 +259,21 @@ def test_disagg_requires_multinode(self, valid_single_node_matrix_entry): class TestAgenticMatrixEntries: + @pytest.mark.parametrize("mode", [[], ["dram"], ["nvme", "dram"], ["dram", "dram"]]) + def test_rejects_unsupported_tier_lists(self, mode): + with pytest.raises(ValidationError, match="only supported tier list"): + AgenticCodingSearchSpaceEntry(**{ + "tp": 8, "kv-offloading": mode, + "kv-offload-backend": {"name": "vllm-native"}, "conc-list": [8], + }) + + def test_tiered_offload_requires_dram_budget(self): + with pytest.raises(ValidationError, match="dram-utilization"): + AgenticCodingConfig(**{"search-space": [{ + "tp": 8, "kv-offloading": ["dram", "nvme"], + "kv-offload-backend": {"name": "vllm-native"}, "conc-list": [8], + }]}) + def test_arbitrary_backend_is_valid_for_single_node_agentic_entry(self): entry = SingleNodeAgenticMatrixEntry(**{ "image": "cquil/vllm-openai:v0.21.0-8813c92", diff --git a/inferencex-e2e/infx/tests/test_agentic_offload_modes.py b/inferencex-e2e/infx/tests/test_agentic_offload_modes.py new file mode 100644 index 0000000000..63c5ac0338 --- /dev/null +++ b/inferencex-e2e/infx/tests/test_agentic_offload_modes.py @@ -0,0 +1,44 @@ +"""Exercise the shared offload gate with explicit recipe capabilities.""" + +import os +import subprocess +from pathlib import Path + +import pytest + +LIBRARY = Path(__file__).resolve().parents[2] / "benchmarks" / "benchmark_lib.sh" + + +@pytest.mark.parametrize( + ("mode", "capacity", "extra_modes", "expected"), + [ + ("dram", "128", "", 0), + ("nvme", "0", "", 1), + ("nvme", "0", "nvme", 0), + ("dram+nvme", "128", "dram+nvme", 0), + ("dram+nvme", "0", "dram+nvme", 1), + ], +) +def test_recipe_offload_capabilities(mode, capacity, extra_modes, expected): + result = subprocess.run( + [ + "bash", + "-c", + 'source "$1"; ' + 'require_agentic_kv_offload_backend vllm-native "$2"', + "bash", + str(LIBRARY), + extra_modes, + ], + env={ + **os.environ, + "IS_AGENTIC": "0", + "KV_OFFLOADING": mode, + "KV_OFFLOAD_BACKEND": "vllm-native", + "TOTAL_CPU_DRAM_GB": capacity, + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == expected, result.stderr diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index b8dfc3cf5d..31e35e46e7 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9180,3 +9180,36 @@ - "Update the SGLang ROCm image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 on this arm only. The other MI355X arms are left on their current tags." - "No serving flag outside the HiCache block changes, and no other config key is touched. Each of the 19 matrix points resolves to exactly one recipe override, with no override left unused." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3611 + +- config-keys: + - dsv4-fp4-b200-vllm-agentic-cache-sources-mtp + - dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg + scenario-type: + - agentic-coding + no-evals: true + description: + - "Rerun vLLM PR56318 at bd57138f9b98eb64c44ea6a0f10d08c8f639820e with the four proven DeepSeek-V4-Pro hybrid cache-source points: B200 TP8 MTP3 native DRAM c8, Simple NVMe c14, and native DRAM+NVMe c14; GB300 MTP3 NIXL plus Mooncake c256." + - "Preserve B200 GPU utilization 0.85, native DRAM utilization 0.0592, the dedicated 1 TB Simple NVMe budget, 3600-second AgentX profiles, and disabled evals. GB300 is one DEP4 prefill worker on one node plus one DEP16 decode worker across four nodes: five nodes and twenty GPUs total." + - "Keep existing benchmark curves unchanged. Adapt the isolated old-Pro validation recipes to current InferenceX paths and current vLLM flag names, use the Python frontend, and collect all raw backend metrics with the cached-prompt source counter required." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3491 + +- config-keys: + - dsv4-fp4-b200-vllm-agentic-cache-sources-mtp + - dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg + scenario-type: + - agentic-coding + no-evals: true + description: + - "Rerun the four DeepSeek-V4-Pro cache-source points with vLLM PR56318 commit 6ce00ee2ebcf29076d671786c36bf8f8ca71bd0b: B200 native DRAM c8, Simple NVMe c14, native DRAM+NVMe c14, and GB300 NIXL plus Mooncake Store c256." + - "Keep the prior runtimes and serving settings, 3600-second AIPerf AgentX profiles, Python frontend, and Prometheus source collection. Update the attribution overlay for bounded attention ranges and sweep-line accounting." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3491 + +- config-keys: + - dsv4-fp4-b200-vllm-agentic-cache-sources-mtp + - dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg + scenario-type: + - agentic-coding + no-evals: true + description: + - "Restore the original B200 single-node vLLM DeepSeek-V4-Pro-NVFP4 checkpoint mapping after the launcher migration. Repeat the four cache-source validation points with unchanged images and serving settings." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3491 diff --git a/inferencex-e2e/runners/srt-slurm/patches/pr56318-overlay-validation.patch b/inferencex-e2e/runners/srt-slurm/patches/pr56318-overlay-validation.patch new file mode 100644 index 0000000000..6b6bc5bfbe --- /dev/null +++ b/inferencex-e2e/runners/srt-slurm/patches/pr56318-overlay-validation.patch @@ -0,0 +1,14 @@ +diff --git a/configs/patches/pr56318-overlay-check.sh b/configs/patches/pr56318-overlay-check.sh +new file mode 100644 +index 000000000..000000001 +--- /dev/null ++++ b/configs/patches/pr56318-overlay-check.sh +@@ -0,0 +1,8 @@ ++#!/bin/bash ++# Verify PR56318 validation overlay checksums before engine start. ++# A no-op when run on images that do not carry the overlay. ++set -eo pipefail ++OVERLAY=/opt/pr56318-validation/validation-overlay.sha256 ++if [ -f "$OVERLAY" ]; then ++ sha256sum -c "$OVERLAY" ++fi diff --git a/inferencex-e2e/runners/srt-slurm/validation/pr56318.patch b/inferencex-e2e/runners/srt-slurm/validation/pr56318.patch new file mode 100644 index 0000000000..ab3ba570b6 --- /dev/null +++ b/inferencex-e2e/runners/srt-slurm/validation/pr56318.patch @@ -0,0 +1,24 @@ +diff --git a/src/srtctl/cli/mixins/benchmark_stage.py b/src/srtctl/cli/mixins/benchmark_stage.py +--- a/src/srtctl/cli/mixins/benchmark_stage.py ++++ b/src/srtctl/cli/mixins/benchmark_stage.py +@@ -874,7 +874,7 @@ class BenchmarkStageMixin: + env.update(self._get_aiperf_server_metrics_env()) + elif is_custom: + assert logical_endpoints is not None +- env.update(self._get_aiperf_server_metrics_env(logical_endpoints, logical_workers_only=True)) ++ env.update(self._get_aiperf_server_metrics_env()) + if isinstance(runner, AIPerfBenchmarkRunner) and self.config.benchmark.aiperf_package: + env["AIPERF_PACKAGE"] = self.config.benchmark.aiperf_package + +diff --git a/src/srtctl/cli/mixins/worker_stage.py b/src/srtctl/cli/mixins/worker_stage.py +--- a/src/srtctl/cli/mixins/worker_stage.py ++++ b/src/srtctl/cli/mixins/worker_stage.py +@@ -132,6 +132,8 @@ class WorkerStageMixin: + # Skip if dynamo.install is False (container already has dynamo installed) + if installs_dynamo(self.config): + parts.append(self.config.dynamo.get_install_commands()) ++ # Fail before model load if dependency installation replaced the PR overlay. ++ parts.append("sha256sum -c /opt/pr56318-validation/validation-overlay.sha256") + + if not parts: + return None