From 468566a1353d8b14f3dd9d6601a0d53d4af7bf71 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Sat, 26 Sep 2026 15:22:42 -0500 Subject: [PATCH 1/7] test: rerun hybrid cache-source validation for vLLM PR56318 --- benchmarks/benchmark_lib.sh | 22 +- .../cache-sources-dep4-dep16-c256-mtp.yaml | 230 ++++++++++++++++++ .../dsv4_fp4_b200_vllm_cache_sources_mtp.sh | 108 ++++++++ configs/nvidia-master.yaml | 53 ++++ docs/configuration-procedures.md | 6 + docs/configuration-procedures_zh.md | 6 + infx/matrix/generate.py | 11 +- infx/matrix/validation.py | 18 +- .../matrix/test_generate_sweep_configs.py | 28 +++ infx/tests/matrix/test_validation.py | 15 ++ perf-changelog.yaml | 12 + runners/launch_b200-nscale-slurm.sh | 36 ++- runners/launch_gb300-nv.sh | 5 + runners/srt-slurm/validation/pr56318.patch | 24 ++ utils/test_agentic_offload_modes.py | 44 ++++ 15 files changed, 598 insertions(+), 20 deletions(-) create mode 100644 benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml create mode 100644 benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh create mode 100644 runners/srt-slurm/validation/pr56318.patch create mode 100644 utils/test_agentic_offload_modes.py diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index b29ad3686e..fd64b96f68 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -192,6 +192,8 @@ require_agentic_kv_offload_none() { require_agentic_kv_offload_backend() { local expected_backend="$1" + # Existing recipes support DRAM; callers opt in explicitly to additional tiers. + local supported_modes="dram ${2:-}" if [[ -z "${KV_OFFLOADING+x}" || -z "$KV_OFFLOADING" ]]; then echo "Error: KV_OFFLOADING must be set for agentic benchmarks" >&2 exit 1 @@ -204,19 +206,23 @@ require_agentic_kv_offload_backend() { fi return 1 ;; - dram) + dram|nvme|dram+nvme) + if [[ " $supported_modes " != *" $KV_OFFLOADING "* ]]; then + echo "Error: this recipe does not support $KV_OFFLOADING with $expected_backend" >&2 + exit 1 + fi if [[ "${KV_OFFLOAD_BACKEND:-}" != "$expected_backend" ]]; then - echo "Error: expected KV_OFFLOAD_BACKEND=$expected_backend when KV_OFFLOADING=dram, got '${KV_OFFLOAD_BACKEND:-}'" >&2 + echo "Error: expected KV_OFFLOAD_BACKEND=$expected_backend when KV_OFFLOADING=$KV_OFFLOADING, got '${KV_OFFLOAD_BACKEND:-}'" >&2 exit 1 fi - if [[ ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then + if [[ "$KV_OFFLOADING" != nvme && ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then echo "Error: DRAM KV offloading requires a positive TOTAL_CPU_DRAM_GB capacity" >&2 exit 1 fi return 0 ;; *) - echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected one of: none, dram)" >&2 + echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected none, dram, nvme, or dram+nvme)" >&2 exit 1 ;; esac @@ -398,18 +404,18 @@ if [[ "$_benchmark_caller" == */agentic/* || exit 1 fi ;; - dram) + dram|nvme|dram+nvme) if [[ -z "${KV_OFFLOAD_BACKEND:-}" || "${KV_OFFLOAD_BACKEND:-}" == "none" ]]; then - echo "Error: KV_OFFLOAD_BACKEND is required when KV_OFFLOADING=dram" >&2 + echo "Error: KV_OFFLOAD_BACKEND is required when KV_OFFLOADING=$KV_OFFLOADING" >&2 exit 1 fi - if [[ ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then + if [[ "$KV_OFFLOADING" != nvme && ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then echo "Error: DRAM KV offloading requires a positive configured TOTAL_CPU_DRAM_GB capacity" >&2 exit 1 fi ;; *) - echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected one of: none, dram)" >&2 + echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected none, dram, nvme, or dram+nvme)" >&2 exit 1 ;; esac diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml new file mode 100644 index 0000000000..853ca3f404 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml @@ -0,0 +1,230 @@ +# PR56318 cache-source validation: preserve run34406466616 resources. +# One DEP4 prefill worker/node and one DEP16 decode worker across four nodes. +schema: 2 +name: "dsv4-vllm-cache-sources-gb300-dep4-dep16-c256" + +# GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one +# DEP16 decode worker at concurrency 256. Decode consumes P/D KV through NIXL +# and MooncakeStore but skips Mooncake prefix lookup to avoid CPU overhead. + +model: + path: "deepseek-v4-pro" + container: "cquil11/vllm-cache-sources@sha256:6ce87fcb9ef8b929f7a8270bc6fbba3c219c3440cd77d3bf1ae74db5a1cb5715" + precision: "fp4" + +identity: + model: + repo: "deepseek-ai/DeepSeek-V4-Pro" + container: + image: "cquil11/vllm-cache-sources@sha256:6ce87fcb9ef8b929f7a8270bc6fbba3c219c3440cd77d3bf1ae74db5a1cb5715" + frameworks: + dynamo: "1.4.0" + +dynamo: + install: true + + source: + wheel: "1.4.0" +environment: + # Mooncake prefix-block hashes must match across processes and nodes. + PYTHONHASHSEED: "0" + +setup_script: vllm-container-deps.sh + +slurm: + time_limit: "8:00:00" + +health_check: + max_attempts: 2160 + interval_seconds: 10 + +resources: + gpu_type: "gb300" + gpus_per_node: 4 + het_jobs: false + spread_workers: false +services: + - name: etcd + type: etcd + placement: + node: infra + - name: nats + type: nats + placement: + node: infra + options: + max_payload_mb: 32 + - name: mooncake-master + type: mooncake-master + options: + store_config: + metadata_server: "P2PHANDSHAKE" + global_segment_size: "180GB" + local_buffer_size: "4GB" + protocol: "rdma" + device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" + mode: "embedded" + enable_offload: false +frontend: + type: dynamo + enable_multiple_frontends: false + args: + router-mode: "random" + router-session-affinity-ttl-secs: 900 + env: + DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600" + DYN_TCP_CHANNEL_BUFFER: "128" + DYN_TCP_REQUEST_TIMEOUT: "60" + +engine: + type: vllm + connector: + dp_launch_mode: per_node +roles: + prefill: + nodes: 1 + workers: 1 + gpus: 4 + env: + HF_HUB_CACHE: "/hf_hub_cache" + HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" + TRANSFORMERS_CACHE: "/hf_hub_cache" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_RPC_TIMEOUT: "600000" + VLLM_LOG_STATS_INTERVAL: "1" + VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" + VLLM_SERVER_DEV_MODE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_USE_RUST_FRONTEND: "0" + VLLM_MOONCAKE_LOAD_RECV_THREADS: "20" + VLLM_MOONCAKE_STORE_SEND_THREADS: "8" + VLLM_ALLREDUCE_USE_SYMM_MEM: "0" + UCX_MEMTYPE_CACHE: "n" + UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" + UCX_TLS: "rc,cuda_copy" + NCCL_CUMEM_ENABLE: "1" + NCCL_MNNVL_ENABLE: "1" + NCCL_NVLS_ENABLE: "1" + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" + NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" + VLLM_USE_BREAKABLE_CUDAGRAPH: "1" + VLLM_CONNECTOR_PREFETCH_DEPTH: "8" + VLLM_DSV4_MEGA_FP8_COMBINE: "1" + DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep8-prefill-c256-{job_id}" + MC_ENABLE_DEST_DEVICE_AFFINITY: "1" + MC_STORE_CLIENT_METRIC: "1" + MC_STORE_CLIENT_METRIC_INTERVAL: "5" + MC_TE_METRIC: "0" + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_both","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: "deepseek-ai/DeepSeek-V4-Pro" + safetensors-load-strategy: "prefetch" + kv-cache-dtype: "fp8" + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 4 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 256 + max-num-batched-tokens: 8192 + long-prefill-token-threshold: 1024 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + gpu-memory-utilization: 0.92 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: "deepseek_v4" + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: "deep_gemm_mega_moe" + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] + decode: + nodes: 4 + workers: 1 + gpus: 16 + + env: + HF_HUB_CACHE: "/hf_hub_cache" + HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" + TRANSFORMERS_CACHE: "/hf_hub_cache" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_RPC_TIMEOUT: "600000" + VLLM_LOG_STATS_INTERVAL: "1" + VLLM_V2_WARMUP_MAX_NUM_SEQS: "20" + VLLM_SERVER_DEV_MODE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_USE_RUST_FRONTEND: "0" + VLLM_MOONCAKE_LOAD_RECV_THREADS: "4" + VLLM_ALLREDUCE_USE_SYMM_MEM: "0" + UCX_MEMTYPE_CACHE: "n" + UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1" + UCX_TLS: "rc,cuda_copy" + NCCL_CUMEM_ENABLE: "1" + NCCL_MNNVL_ENABLE: "1" + NCCL_NVLS_ENABLE: "1" + VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1" + NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3" + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "0" + VLLM_DSV4_MEGA_FP8_COMBINE: "1" + DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep32-decode-c256-{job_id}" + MC_ENABLE_DEST_DEVICE_AFFINITY: "1" + MC_STORE_CLIENT_METRIC: "1" + MC_STORE_CLIENT_METRIC_INTERVAL: "5" + MC_TE_METRIC: "0" + + args: + kv-transfer-config: '{"kv_connector":"MultiConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"connectors":[{"kv_connector":"NixlConnector","kv_role":"kv_consumer","kv_load_failure_policy":"fail","kv_buffer_device":"cuda","kv_connector_extra_config":{"enforce_handshake_compat":false,"enable_cross_layers_blocks":false}},{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_consumer","kv_connector_extra_config":{"load_async":true,"lookup_async":false,"enable_lookup":false,"enable_cross_layers_blocks":false,"enable_offload":false}}]}}' + served-model-name: "deepseek-ai/DeepSeek-V4-Pro" + safetensors-load-strategy: "prefetch" + kv-cache-dtype: "fp8" + tensor-parallel-size: 1 + pipeline-parallel-size: 1 + data-parallel-size: 16 + data-parallel-rpc-port: 13345 + enable-cumem-allocator: true + enable-expert-parallel: true + enable-ep-weight-filter: true + max-model-len: 1048576 + max-num-seqs: 8 + max-num-batched-tokens: 32 + trust-remote-code: true + no-enable-flashinfer-autotune: true + block-size: 256 + compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}' + max-cudagraph-capture-size: 32 + gpu-memory-utilization: 0.90 + no-disable-hybrid-kv-cache-manager: true + tokenizer-mode: "deepseek_v4" + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + moe-backend: "deep_gemm_mega_moe" + numa-bind: true + numa-bind-nodes: [0, 0, 1, 1] +sbatch_directives: + cpus-per-task: "72" + mem: "0" + +srun_options: + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:prompt_tokens_cached_by_source" + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + IS_MULTINODE: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + # Avoid concurrent readers observing a mismatched mmap data/index pair. + AIPERF_DATASET_MMAP_CACHE_ENABLED: "false" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" + WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh b/benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh new file mode 100644 index 0000000000..0c6a61e9c3 --- /dev/null +++ b/benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh @@ -0,0 +1,108 @@ +#!/usr/bin/env bash +set -eo pipefail +set -x + +# The original DeepSeek-V4-Pro TP8/MTP3 cache-source validation recipe. +# Preserve run 34420875287's cache pressure independently of the 0813 DSpark curve. +source "$(dirname "$0")/../../benchmark_lib.sh" +check_env_vars MODEL MODEL_PATH TP CONC KV_OFFLOADING KV_OFFLOAD_BACKEND +check_env_vars TOTAL_CPU_DRAM_GB RESULT_DIR DURATION PORT GPU_MEMORY_UTILIZATION +check_env_vars EP_SIZE DP_ATTENTION DCP_SIZE PCP_SIZE EVAL_ONLY THINKING_MODE + +if [[ "$TP" != 8 || "$EP_SIZE" != 1 || "$DP_ATTENTION" != false || "$DCP_SIZE" != 1 || "$PCP_SIZE" != 1 ]]; then + echo "Cache-source validation requires TP8 without EP, DP attention, or context parallelism" >&2 + exit 1 +fi +export GPU_COUNT=8 +[[ -d "$MODEL_PATH" ]] || { echo "Missing pre-staged model: $MODEL_PATH" >&2; exit 1; } +nvidia-smi +resolve_trace_source +install_agentic_deps + +export AIPERF_SERVER_METRICS_URLS="http://localhost:$PORT/metrics" +export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="vllm:prompt_tokens_cached_by_source" +export VLLM_ENGINE_READY_TIMEOUT_S=3600 +export VLLM_PREFIX_CACHE_RETENTION_INTERVAL=32768 +export VLLM_USE_V2_MODEL_RUNNER=1 +export VLLM_USE_RUST_FRONTEND=0 +export VLLM_DSV4_MEGA_FP8_COMBINE=1 +export VLLM_RPC_TIMEOUT=600000 +export PYTHONHASHSEED=42 +export TORCH_CUDA_ARCH_LIST=10.0 +export PYTHONNOUSERSITE=1 +export VLLM_FLOAT32_MATMUL_PRECISION=high + +mkdir -p "$RESULT_DIR" +SERVER_LOG="$RESULT_DIR/server.log" +case "$KV_OFFLOAD_BACKEND" in + vllm-native) + require_agentic_kv_offload_backend vllm-native "dram dram+nvme" + SECONDARY_TIERS='[]' + if [[ "$KV_OFFLOADING" == dram+nvme ]]; then + check_env_vars NVME_OFFLOAD_DIR + SECONDARY_TIERS="[{\"type\":\"fs\",\"root_dir\":\"$NVME_OFFLOAD_DIR\",\"locality\":\"LOCAL\"}]" + fi + OFFLOAD_CONFIG="{\"kv_connector\":\"OffloadingConnector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{\"spec_name\":\"TieringOffloadingSpec\",\"cpu_bytes_to_use\":$((TOTAL_CPU_DRAM_GB * 1000000000)),\"secondary_tiers\":$SECONDARY_TIERS}}" + ;; + vllm-simple) + require_agentic_kv_offload_backend vllm-simple nvme + check_env_vars NVME_OFFLOAD_DIR + OFFLOAD_CONFIG="{\"kv_connector\":\"SimpleCPUOffloadConnector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{\"kv_offload_backend\":\"disk\",\"disk_path\":\"$NVME_OFFLOAD_DIR/cache.bin\",\"disk_capacity_bytes\":$((1000000000000 / GPU_COUNT)),\"disk_buffer_slots\":4,\"lazy_offload\":false}}" + ;; + *) echo "Unsupported validation offload backend: $KV_OFFLOAD_BACKEND" >&2; exit 1 ;; +esac + +# Use the same committed golden MTP curve as the shared SRT selector. +SPEC_CONFIG=$(python3 - <<'PY' +import json +import os +from infx.srt_slurm.synthetic_acceptance import GOLDEN_DIR, golden_length + +spec = {"method": "mtp", "num_speculative_tokens": 3} +if os.environ["EVAL_ONLY"] != "true": + spec["rejection_sample_method"] = "synthetic" + spec["synthetic_acceptance_length"] = golden_length( + "dsv4", spec, os.environ["THINKING_MODE"], GOLDEN_DIR + ) +print(json.dumps(spec)) +PY +) +MAX_NUM_SEQS=$((2 * CONC)) +CAPTURE_SIZE_LIST=() +for ((num_seqs = 1; num_seqs <= MAX_NUM_SEQS; num_seqs++)); do + CAPTURE_SIZE_LIST+=("$((num_seqs * 4))") +done +CAPTURE_SIZE_LIST+=(100 200 300 400 500) +CUDA_GRAPH_CAPTURE_SIZES=$(printf '%s\n' "${CAPTURE_SIZE_LIST[@]}" | sort -n -u | paste -sd, -) +COMPILATION_CONFIG="{\"cudagraph_mode\":\"FULL_AND_PIECEWISE\",\"cudagraph_capture_sizes\":[${CUDA_GRAPH_CAPTURE_SIZES}]}" + +{ set +x; } 2>/dev/null +VLLM_CMD=( + vllm serve "$MODEL_PATH" --served-model-name "$MODEL" + --host 0.0.0.0 --port "$PORT" --trust-remote-code + --kv-cache-dtype fp8 --block-size 256 --max-model-len 1048576 + --gpu-memory-utilization "$GPU_MEMORY_UTILIZATION" + --numa-bind --enable-cumem-allocator --no-enable-flashinfer-autotune + --tokenizer-mode deepseek_v4 --tool-call-parser deepseek_v4 + --enable-auto-tool-choice --reasoning-parser deepseek_v4 + --attention-config '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}' + --speculative-config "$SPEC_CONFIG" --no-disable-hybrid-kv-cache-manager + --disable-uvicorn-access-log --compilation-config "$COMPILATION_CONFIG" + --max-num-seqs "$MAX_NUM_SEQS" --tensor-parallel-size "$TP" --data-parallel-size 1 + --kv-transfer-config "$OFFLOAD_CONFIG" +) +printf '%q ' "${VLLM_CMD[@]}" | tee "$RESULT_DIR/vllm_command.txt" +printf '\n' | tee -a "$RESULT_DIR/vllm_command.txt" +sha256sum -c /opt/pr56318-validation/validation-overlay.sha256 | tee "$RESULT_DIR/overlay-verification.log" +"${VLLM_CMD[@]}" > "$SERVER_LOG" 2>&1 & +SERVER_PID=$! +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +if [[ "$EVAL_ONLY" == true ]]; then + run_eval --port "$PORT" +else + build_replay_cmd "$RESULT_DIR" + curl --fail --silent --show-error "$AIPERF_SERVER_METRICS_URLS" > "$RESULT_DIR/metrics-before.prom" + run_agentic_replay_and_write_outputs "$RESULT_DIR" + curl --fail --silent --show-error "$AIPERF_SERVER_METRICS_URLS" > "$RESULT_DIR/metrics-after.prom" +fi diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index f442fed05b..30d7264832 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8385,6 +8385,59 @@ dsv41flash-fp4-b200-sglang-agentic-dspark: - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml } - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml } +# PR56318: rerun the four successful hybrid cache-source validation points. +dsv4-fp4-b200-vllm-agentic-cache-sources-mtp: + image: cquil11/vllm-cache-sources@sha256:6f9e476f6a1e1c25ac9bc1a863329538db465812fb4ac18c6553e988eae09a56 + model: deepseek-ai/DeepSeek-V4-Pro + model-prefix: dsv4 + runner: cluster:b200-nscale + precision: fp4 + framework: vllm + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.0592 + search-space: + - { tp: 8, kv-offloading: dram, kv-offload-backend: { name: vllm-native }, spec-decoding: mtp, conc-list: [8] } + - search-space: + - { tp: 8, kv-offloading: nvme, kv-offload-backend: { name: vllm-simple }, spec-decoding: mtp, conc-list: [14] } + - dram-utilization: 0.0592 + search-space: + - { tp: 8, kv-offloading: [dram, nvme], kv-offload-backend: { name: vllm-native }, spec-decoding: mtp, conc-list: [14] } + +dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg: + image: cquil11/vllm-cache-sources@sha256:6ce87fcb9ef8b929f7a8270bc6fbba3c219c3440cd77d3bf1ae74db5a1cb5715 + model: deepseek-ai/DeepSeek-V4-Pro + model-prefix: dsv4 + runner: cluster:gb300-nv + precision: fp4 + framework: dynamo-vllm + kv-p2p-transfer: nixl + multinode: true + disagg: true + scenarios: + agentic-coding: + - dram-utilization: 0.857143 + search-space: + - spec-decoding: mtp + conc-list: [256] + kv-offloading: dram + kv-offload-backend: { name: mooncake, version: "0.3.13.post1" } + router: { name: dynamo-router, version: "1.4.0" } + prefill: + num-worker: 1 + tp: 4 + ep: 4 + dp-attn: true + additional-settings: + - "SLURM_PARTITION=batch_1" + - "CONFIG_FILE=recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml" + decode: + num-worker: 1 + tp: 16 + ep: 16 + dp-attn: true + dsv41flash-fp4-gb300-vllm-agentic-dspark: image: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e model: deepseek-ai/DeepSeek-V4.1-Flash diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index f57f08c75b..9eb8d1d0ac 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -23,6 +23,12 @@ Use this page for benchmark configuration, recipe, image, and runner changes. It Delete retired entries from the active master configs; they are not archived. Git history and `perf-changelog.yaml` keep the historical settings. For partial retirements, remove only the retired scenarios. Delete retired AMD server-registry entries and model-specific setup from `benchmarks/multi_node/amd_utils/` as well. Preserve shared dependencies needed by retained SPEED-Bench collectors, including their scheduling scores. See the [deprecation rules](../AGENTS.md#deprecating-benchmark-configs). +## Cache-source validation + +Single-node AgentX configs accept `kv-offloading: nvme` or `[dram, nvme]` in addition to `none` and `dram`. Tiered entries still require `dram-utilization`, which budgets host memory only. The dedicated DeepSeek-V4-Pro B200 validation recipe supports Simple NVMe and native DRAM/NVMe. Its launcher mounts a job-owned `/scratch/inferencex-kv-` directory and removes it before releasing the allocation. Other recipes must explicitly opt in to NVMe modes. + +The dedicated GB300 cache-source recipe applies a job-local srt-slurm patch to scrape every physical DP worker, including nonleader nodes. It verifies the image's PR-overlay checksums after dependency installation. This patch does not run for other recipes. + ## Dependency submodules Git records the exact dependency commits. [`.gitmodules`](../.gitmodules) defines the repositories: AIPerf at `utils/aiperf`, NVIDIA srt-slurm at `utils/srt-slurm`. TileRT is a documented manual fork checkout in `setup_srt_slurm()`, not a separate submodule. diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 968553d8a3..1ae3b4b181 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -23,6 +23,12 @@ 退役的配置项直接从启用的主配置中删除,不再归档;历史设置由 Git 历史和 `perf-changelog.yaml` 保留。仅弃用部分场景时,只删除已退役的场景。退役的 AMD 服务注册项和模型专用初始化逻辑也应从 `benchmarks/multi_node/amd_utils/` 中删除。保留 SPEED-Bench 采集器仍需使用的共享依赖,包括调度评分。参见[弃用规则](../AGENTS.md#deprecating-benchmark-configs)。 +## 缓存来源验证 + +单节点 AgentX 配置除了 `none` 和 `dram`,还接受 `kv-offloading: nvme` 或 `[dram, nvme]`。分层配置仍须提供 `dram-utilization`,且该值仅用于分配主机内存。专用 DeepSeek-V4-Pro B200 验证配方支持 Simple NVMe 和原生 DRAM/NVMe。launcher 挂载作业专属的 `/scratch/inferencex-kv-` 目录,并在释放资源前将其删除。其他配方必须显式启用 NVMe 模式。 + +专用 GB300 缓存来源配方在作业本地的 srt-slurm 副本中应用补丁,采集每个物理 DP worker 的指标,包括非 leader 节点。依赖安装后会验证镜像中 PR 覆盖文件的校验和。其他配方不会应用此补丁。 + ## 依赖子模块 Git 记录依赖的精确提交版本。[`.gitmodules`](../.gitmodules) 定义各仓库:AIPerf 位于 `utils/aiperf`,NVIDIA srt-slurm 位于 `utils/srt-slurm`。TileRT 由 `setup_srt_slurm()` 手动检出已记录的分支仓库,不是独立子模块。 diff --git a/infx/matrix/generate.py b/infx/matrix/generate.py index 28d0f2a1f3..0349995731 100644 --- a/infx/matrix/generate.py +++ b/infx/matrix/generate.py @@ -441,7 +441,7 @@ def agentic_dram_offload_gb( budgeted separately if it ever gains its own pool). """ kv_offloading = benchmark.get(Fields.KV_OFFLOADING.value, "none") - if kv_offloading != "dram": + if kv_offloading not in ("dram", ["dram", "nvme"]): return 0 available_mib = min( @@ -472,13 +472,14 @@ def agentic_dram_offload_gb( def agentic_kv_offload_suffix( - kv_offloading: str, + kv_offloading: str | list[str], kv_offload_backend: dict | None, ) -> str: """Return a compact exp-name suffix for agentic KV offload settings.""" if kv_offloading == "none": return "kvnone" - return f"kv{kv_offloading}-{kv_offload_backend['name']}" + mode = "+".join(kv_offloading) if isinstance(kv_offloading, list) else kv_offloading + return f"kv{mode}-{kv_offload_backend['name']}" def multinode_agentic_exp_name( @@ -998,7 +999,9 @@ def _agentic_entries( ) entry.update( { - Fields.KV_OFFLOADING.value: kv_offloading, + Fields.KV_OFFLOADING.value: ( + "+".join(kv_offloading) if isinstance(kv_offloading, list) else kv_offloading + ), Fields.TOTAL_CPU_DRAM_GB.value: total_cpu_dram_gb, Fields.DURATION.value: DEFAULT_AGENTIC_DURATION_SECONDS, Fields.EXP_NAME.value: exp_name, diff --git a/infx/matrix/validation.py b/infx/matrix/validation.py index bf1af5f272..3d432e516a 100644 --- a/infx/matrix/validation.py +++ b/infx/matrix/validation.py @@ -14,6 +14,7 @@ CLUSTER_LABEL_PREFIX = "cluster:" DEFAULT_AGENTIC_DURATION_SECONDS = 3600 +type KVOffloadingConfig = Literal["none", "dram", "nvme"] | list[Literal["dram", "nvme"]] """ The below class defines the field names expected to be present in the JSON entries @@ -325,7 +326,9 @@ class SingleNodeAgenticMatrixEntry(BaseModel): default="none", alias=Fields.SPEC_DECODING.value ) conc: int - kv_offloading: Literal["none", "dram"] = Field(alias=Fields.KV_OFFLOADING.value) + kv_offloading: Literal["none", "dram", "nvme", "dram+nvme"] = Field( + alias=Fields.KV_OFFLOADING.value + ) kv_offload_backend: KVOffloadBackendMetadata | None = Field( default=None, alias=Fields.KV_OFFLOAD_BACKEND.value ) @@ -516,7 +519,10 @@ def _validate_kv_offload_fields(self: Any) -> Any: f"{Fields.KV_OFFLOAD_BACKEND.value} requires {Fields.KV_OFFLOADING.value}" ) return self - if self.kv_offloading == "none": + if isinstance(self.kv_offloading, list): + if self.kv_offloading != ["dram", "nvme"]: + raise ValueError("The only supported tier list is ['dram', 'nvme']") + elif self.kv_offloading == "none": if backend is not None: raise ValueError( f"{Fields.KV_OFFLOAD_BACKEND.value} can only be set when " @@ -642,9 +648,7 @@ class AgenticCodingSearchSpaceEntry(BaseModel): prefill: WorkerConfig | None = None decode: WorkerConfig | None = None num_nodes: int | None = Field(default=None, alias=Fields.NUM_NODES.value, gt=0, strict=True) - kv_offloading: Literal["none", "dram"] | None = Field( - default=None, alias=Fields.KV_OFFLOADING.value - ) + kv_offloading: KVOffloadingConfig | None = Field(default=None, alias=Fields.KV_OFFLOADING.value) kv_offload_backend: KVOffloadBackendMetadata | None = Field( default=None, alias=Fields.KV_OFFLOAD_BACKEND.value ) @@ -690,6 +694,8 @@ def validate_topology_fields(self) -> Self: ) _validate_tp_context_topology(self) if has_aggregate_worker or has_complete_multinode: + if self.kv_offloading in ("nvme", ["dram", "nvme"]): + raise ValueError("NVMe offloading currently requires a single-node entry") explicitly_single_node_fields = { "pp", "dcp_size", @@ -725,7 +731,7 @@ class AgenticCodingConfig(BaseModel): @model_validator(mode="after") def validate_dram_offload_capacity(self) -> Self: for entry in self.search_space: - if entry.kv_offloading != "dram": + if entry.kv_offloading not in ("dram", ["dram", "nvme"]): continue if self.dram_utilization is None: raise ValueError( diff --git a/infx/tests/matrix/test_generate_sweep_configs.py b/infx/tests/matrix/test_generate_sweep_configs.py index e119694d80..f15feb2fe6 100644 --- a/infx/tests/matrix/test_generate_sweep_configs.py +++ b/infx/tests/matrix/test_generate_sweep_configs.py @@ -2495,6 +2495,34 @@ def agentic_config(request, sample_single_node_config): class TestAgenticGeneration: + @pytest.mark.parametrize(("mode", "runtime", "budget"), [ + ("dram", "dram", 1199), + ("nvme", "nvme", 0), + (["dram", "nvme"], "dram+nvme", 1199), + ]) + def test_offload_modes_preserve_budget_and_artifact_identity( + self, sample_single_node_config, sample_runner_config, + generate_agentic_sweep, mode, runtime, budget, + ): + config = copy.deepcopy(sample_single_node_config) + entry = next(iter(config.values())) + entry.update(runner="cluster:b300-nv", multinode=False) + entry["scenarios"] = {"agentic-coding": [{ + "dram-utilization": 0.80, + "search-space": [{ + "tp": 4, "kv-offloading": mode, + "kv-offload-backend": {"name": "vllm-native"}, "conc-list": [8], + }], + }]} + original = copy.deepcopy(config) + rows = generate_agentic_sweep(config, sample_runner_config) + assert rows + for row in rows: + assert row["kv-offloading"] == runtime + assert row["total-cpu-dram-gb"] == budget + assert f"kv{runtime}-vllm-native" in row["exp-name"] + assert config == original + def test_point_order_and_input_preservation( self, agentic_config, sample_runner_config, generate_agentic_sweep, ): diff --git a/infx/tests/matrix/test_validation.py b/infx/tests/matrix/test_validation.py index 62db6815e0..768296f8d6 100644 --- a/infx/tests/matrix/test_validation.py +++ b/infx/tests/matrix/test_validation.py @@ -301,6 +301,21 @@ def test_disagg_requires_multinode(self, valid_single_node_matrix_entry): class TestAgenticMatrixEntries: """Tests for agentic coding validation models.""" + @pytest.mark.parametrize("mode", [[], ["dram"], ["nvme", "dram"], ["dram", "dram"]]) + def test_rejects_unsupported_tier_lists(self, mode): + with pytest.raises(ValidationError, match="only supported tier list"): + AgenticCodingSearchSpaceEntry(**{ + "tp": 8, "kv-offloading": mode, + "kv-offload-backend": {"name": "vllm-native"}, "conc-list": [8], + }) + + def test_tiered_offload_requires_dram_budget(self): + with pytest.raises(ValidationError, match="dram-utilization"): + AgenticCodingConfig(**{"search-space": [{ + "tp": 8, "kv-offloading": ["dram", "nvme"], + "kv-offload-backend": {"name": "vllm-native"}, "conc-list": [8], + }]}) + def test_arbitrary_backend_is_valid_for_single_node_agentic_entry(self): entry = SingleNodeAgenticMatrixEntry(**{ "image": "cquil/vllm-openai:v0.21.0-8813c92", diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 0f361fd6db..04bd39c76e 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8962,3 +8962,15 @@ - "旧镜像切出于 2026-09-16,新镜像领先 441 个提交,包含当前曲线所不含的三项 Kimi-K3 正确性修复:vllm-project/vllm#51483 不再将无状态的首个 chunk 当作 decode,vllm-project/vllm#57098 修复 Kimi-K3 reasoning parser,vllm-project/vllm#57430 支持 routed expert 量化。新镜像还包含本臂在并发大于 4 时均会使用的 ROCm CPU KV 卸载改动,包括 vllm-project/vllm#57160(ROCm CPU KV 卸载改用私有 pinned 张量)与 vllm-project/vllm#50045(卸载背压检测)。" - "This image bump does not change the DSpark draft model data type. Only the image: line changes and kimik3_fp4_mi355x_mtp.sh is unchanged; the draft loads unmodified from the published Inferact/Kimi-K3-DSpark checkpoint via --speculative-config (model=Inferact/Kimi-K3-DSpark, method=dspark). The only dtype in that speculative-config is kv_cache_dtype=fp8, which sets the draft KV-cache storage precision, not the draft weights. No flag overrides or re-quantizes the draft-model weights, so the draft dtype is preserved from its checkpoint across this re-sweep." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3419 + +- config-keys: + - dsv4-fp4-b200-vllm-agentic-cache-sources-mtp + - dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg + scenario-type: + - agentic-coding + no-evals: true + description: + - "Rerun vLLM PR56318 at bd57138f9b98eb64c44ea6a0f10d08c8f639820e with the four proven DeepSeek-V4-Pro hybrid cache-source points: B200 TP8 MTP3 native DRAM c8, Simple NVMe c14, and native DRAM+NVMe c14; GB300 MTP3 NIXL plus Mooncake c256." + - "Preserve B200 GPU utilization 0.85, native DRAM utilization 0.0592, the dedicated 1 TB Simple NVMe budget, 3600-second AgentX profiles, and disabled evals. GB300 is one DEP4 prefill worker on one node plus one DEP16 decode worker across four nodes: five nodes and twenty GPUs total." + - "Keep existing benchmark curves unchanged. Adapt the isolated old-Pro validation recipes to current InferenceX paths and current vLLM flag names, use the Python frontend, and collect all raw backend metrics with the cached-prompt source counter required." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX diff --git a/runners/launch_b200-nscale-slurm.sh b/runners/launch_b200-nscale-slurm.sh index 1c6c7afc04..291538f32b 100755 --- a/runners/launch_b200-nscale-slurm.sh +++ b/runners/launch_b200-nscale-slurm.sh @@ -786,6 +786,10 @@ run_agentic() { if [[ ! -f "$BENCH_SCRIPT" ]]; then BENCH_SCRIPT="${BENCH_BASE}${FRAMEWORK_SUFFIX}${SPEC_SUFFIX}.sh" fi + if [[ "$MODEL" == "deepseek-ai/DeepSeek-V4-Pro" && "$FRAMEWORK" == vllm && "$SPEC_DECODING" == mtp && "$IMAGE" == cquil11/vllm-cache-sources@sha256:6f9e476f6a1e1c25ac9bc1a863329538db465812fb4ac18c6553e988eae09a56 ]]; then + BENCH_SCRIPT="benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh" + export GPU_MEMORY_UTILIZATION=0.85 + fi LOCK_FILE="${SQUASH_FILE}.lock" # TODO(Cam): lmsysorg/sglang:deepseek-v4-blackwell installs sglang editable at @@ -793,7 +797,7 @@ run_agentic() { # the default $GITHUB_WORKSPACE:/workspace/ bind-mount masks the install and # breaks `import sglang`. Mount this one image at /ix instead; drop the # conditional once the image stops installing editable under /workspace. - if [[ "$IMAGE" == *deepseek-v4-blackwell* ]]; then + if [[ "$IMAGE" == *deepseek-v4-blackwell* || "$BENCH_SCRIPT" == *cache_sources_mtp.sh ]]; then CONTAINER_MOUNT_DIR=/ix else CONTAINER_MOUNT_DIR=/workspace @@ -810,9 +814,37 @@ run_agentic() { else CONTAINER_MOUNTS="$GITHUB_WORKSPACE:$CONTAINER_MOUNT_DIR,$MODEL_PATH:$MODEL_PATH,$AIPERF_MMAP_CACHE_HOST_PATH:/aiperf_mmap_cache" fi + if [[ "$BENCH_SCRIPT" == *cache_sources_mtp.sh ]]; then + export INFMAX_CONTAINER_WORKSPACE="$CONTAINER_MOUNT_DIR" + export RESULT_DIR="$CONTAINER_MOUNT_DIR/results" + fi - salloc --partition=$SLURM_PARTITION --account=$SLURM_ACCOUNT --gres=gpu:$GPU_COUNT --exclusive --mem=0 --time="$SALLOC_TIME_LIMIT" --no-shell --job-name="$RUNNER_NAME" + salloc --partition=$SLURM_PARTITION --account=$SLURM_ACCOUNT --gres=gpu:$GPU_COUNT --exclusive --mem=0 --time="$SALLOC_TIME_LIMIT" --no-shell --job-name="$RUNNER_NAME" || return 1 JOB_ID=$(squeue --name="$RUNNER_NAME" -u "$USER" -h -o %A | head -n1) + [[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo "Could not resolve Slurm allocation" >&2; return 1; } + + if [[ "$BENCH_SCRIPT" == *cache_sources_mtp.sh ]]; then + NVME_HOST_DIR="" + cleanup_offload_cache() { + local rc=$? + if [[ -n "$NVME_HOST_DIR" ]]; then + timeout --kill-after=10s 300s srun --jobid="$JOB_ID" rm -rf -- "$NVME_HOST_DIR" || rc=1 + fi + scancel "$JOB_ID" || true + exit "$rc" + } + trap cleanup_offload_cache EXIT + trap 'exit 130' INT + trap 'exit 143' TERM + fi + + if [[ "$KV_OFFLOADING" == *nvme* ]]; then + # This directory belongs only to the validated numeric allocation above. + export NVME_HOST_DIR="/scratch/inferencex-kv-$JOB_ID" + srun --jobid="$JOB_ID" mkdir -m 700 "$NVME_HOST_DIR" || return 1 + export NVME_OFFLOAD_DIR=/kv-offload + CONTAINER_MOUNTS+=",$NVME_HOST_DIR:$NVME_OFFLOAD_DIR" + fi # Bench scripts skip `hf download` when MODEL is a local path. export MODEL="$MODEL_PATH" diff --git a/runners/launch_gb300-nv.sh b/runners/launch_gb300-nv.sh index a09a238e40..dd8a4b9c72 100644 --- a/runners/launch_gb300-nv.sh +++ b/runners/launch_gb300-nv.sh @@ -180,6 +180,11 @@ rm -rf "$SRT_REPO_DIR" setup_srt_slurm "$SRT_REPO_DIR" "$FRAMEWORK" "$USES_DCGM_POWER" || exit 1 +if [[ "$CONFIG_FILE" == recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml ]]; then + # This validation needs all five DP metrics endpoints and verifies the PR overlay after setup. + git -C "$SRT_REPO_DIR" apply "$GITHUB_WORKSPACE/runners/srt-slurm/validation/pr56318.patch" || exit 1 +fi + if [[ "$FRAMEWORK" == "dynamo-trt" && "$MODEL_PREFIX" == "dsv4" ]]; then SRT_SLURM_MODEL_PREFIX="deepseek-ai/DeepSeek-V4-Pro" fi diff --git a/runners/srt-slurm/validation/pr56318.patch b/runners/srt-slurm/validation/pr56318.patch new file mode 100644 index 0000000000..e8d1918038 --- /dev/null +++ b/runners/srt-slurm/validation/pr56318.patch @@ -0,0 +1,24 @@ +diff --git a/src/srtctl/cli/mixins/benchmark_stage.py b/src/srtctl/cli/mixins/benchmark_stage.py +--- a/src/srtctl/cli/mixins/benchmark_stage.py ++++ b/src/srtctl/cli/mixins/benchmark_stage.py +@@ -862,7 +862,7 @@ class BenchmarkStageMixin: + env.update(self._get_aiperf_server_metrics_env()) + elif is_custom: + assert logical_endpoints is not None +- env.update(self._get_aiperf_server_metrics_env(logical_endpoints, logical_workers_only=True)) ++ env.update(self._get_aiperf_server_metrics_env()) + if isinstance(runner, AIPerfBenchmarkRunner) and self.config.benchmark.aiperf_package: + env["AIPERF_PACKAGE"] = self.config.benchmark.aiperf_package + +diff --git a/src/srtctl/cli/mixins/worker_stage.py b/src/srtctl/cli/mixins/worker_stage.py +--- a/src/srtctl/cli/mixins/worker_stage.py ++++ b/src/srtctl/cli/mixins/worker_stage.py +@@ -132,6 +132,8 @@ class WorkerStageMixin: + # Skip if dynamo.install is False (container already has dynamo installed) + if installs_dynamo(self.config): + parts.append(self.config.dynamo.get_install_commands()) ++ # Fail before model load if dependency installation replaced the PR overlay. ++ parts.append("sha256sum -c /opt/pr56318-validation/validation-overlay.sha256") + + if not parts: + return None diff --git a/utils/test_agentic_offload_modes.py b/utils/test_agentic_offload_modes.py new file mode 100644 index 0000000000..12db31ac45 --- /dev/null +++ b/utils/test_agentic_offload_modes.py @@ -0,0 +1,44 @@ +"""Exercise the shared offload gate with explicit recipe capabilities.""" + +import os +import subprocess +from pathlib import Path + +import pytest + +LIBRARY = Path(__file__).resolve().parents[1] / "benchmarks" / "benchmark_lib.sh" + + +@pytest.mark.parametrize( + ("mode", "capacity", "extra_modes", "expected"), + [ + ("dram", "128", "", 0), + ("nvme", "0", "", 1), + ("nvme", "0", "nvme", 0), + ("dram+nvme", "128", "dram+nvme", 0), + ("dram+nvme", "0", "dram+nvme", 1), + ], +) +def test_recipe_offload_capabilities(mode, capacity, extra_modes, expected): + result = subprocess.run( + [ + "bash", + "-c", + 'source "$1"; ' + 'require_agentic_kv_offload_backend vllm-native "$2"', + "bash", + str(LIBRARY), + extra_modes, + ], + env={ + **os.environ, + "IS_AGENTIC": "0", + "KV_OFFLOADING": mode, + "KV_OFFLOAD_BACKEND": "vllm-native", + "TOTAL_CPU_DRAM_GB": capacity, + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == expected, result.stderr From 0a1bf863a33066291176269c2cf397d16548fc02 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Sat, 26 Sep 2026 15:22:59 -0500 Subject: [PATCH 2/7] test: link cache-source validation changelog to PR3491 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 04bd39c76e..a88376029a 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8973,4 +8973,4 @@ - "Rerun vLLM PR56318 at bd57138f9b98eb64c44ea6a0f10d08c8f639820e with the four proven DeepSeek-V4-Pro hybrid cache-source points: B200 TP8 MTP3 native DRAM c8, Simple NVMe c14, and native DRAM+NVMe c14; GB300 MTP3 NIXL plus Mooncake c256." - "Preserve B200 GPU utilization 0.85, native DRAM utilization 0.0592, the dedicated 1 TB Simple NVMe budget, 3600-second AgentX profiles, and disabled evals. GB300 is one DEP4 prefill worker on one node plus one DEP16 decode worker across four nodes: five nodes and twenty GPUs total." - "Keep existing benchmark curves unchanged. Adapt the isolated old-Pro validation recipes to current InferenceX paths and current vLLM flag names, use the Python frontend, and collect all raw backend metrics with the cached-prompt source counter required." - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3491 From 1d542c78a4836cd1b62c1fd816ce05a1cfd7024f Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Sat, 26 Sep 2026 15:25:15 -0500 Subject: [PATCH 3/7] test: follow relocated InferenceX test suite --- {utils => infx/tests}/test_agentic_offload_modes.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) rename {utils => infx/tests}/test_agentic_offload_modes.py (94%) diff --git a/utils/test_agentic_offload_modes.py b/infx/tests/test_agentic_offload_modes.py similarity index 94% rename from utils/test_agentic_offload_modes.py rename to infx/tests/test_agentic_offload_modes.py index 12db31ac45..63c5ac0338 100644 --- a/utils/test_agentic_offload_modes.py +++ b/infx/tests/test_agentic_offload_modes.py @@ -6,7 +6,7 @@ import pytest -LIBRARY = Path(__file__).resolve().parents[1] / "benchmarks" / "benchmark_lib.sh" +LIBRARY = Path(__file__).resolve().parents[2] / "benchmarks" / "benchmark_lib.sh" @pytest.mark.parametrize( From 56136a21fd6e4e64adee01d1e68deb498d2e7ddd Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 16:52:30 -0400 Subject: [PATCH 4/7] refactor: migrate single-node cache-sources from bash to srt-slurm recipe Replace the hand-rolled bash server script (dsv4_fp4_b200_vllm_cache_sources_mtp.sh) with a native srt-slurm recipe at dsv4/vllm/b200-fp4-mtp/cache-sources.yaml. Three variants select by CONC/KV_OFFLOADING: TP8 c8 native DRAM, TP8 c14 Simple NVMe, TP8 c14 native DRAM+NVMe. NVMe variants use host_setup to create/teardown /scratch/inferencex-kv-$SLURM_JOB_ID and container_mounts with {job_id} templating. Each search-space row in nvidia-master.yaml now carries srt-recipe:, routing to the native-single-node launch path in the b200-nscale launcher. The bash-specific routing, /ix mount override, NVMe directory management, and GPU_MEMORY_UTILIZATION=0.85 export are removed from launch_b200-nscale-slurm.sh (gpu-memory-utilization is now 0.85 in the recipe). A new srt-slurm patch (pr56318-overlay-validation.patch) adds pr56318-overlay-check.sh to configs/patches/, referenced by the recipe's setup_script to verify overlay checksums before engine start. Co-Authored-By: Claude Opus 4.6 --- .../dsv4_fp4_b200_vllm_cache_sources_mtp.sh | 108 --------------- .../dsv4/vllm/b200-fp4-mtp/cache-sources.yaml | 127 ++++++++++++++++++ configs/nvidia-master.yaml | 6 +- infx/tests/srt_slurm/test_srt_single_node.py | 53 ++++++++ runners/launch_b200-nscale-slurm.sh | 36 +---- .../patches/pr56318-overlay-validation.patch | 14 ++ 6 files changed, 199 insertions(+), 145 deletions(-) delete mode 100644 benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml create mode 100644 runners/srt-slurm/patches/pr56318-overlay-validation.patch diff --git a/benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh b/benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh deleted file mode 100644 index 0c6a61e9c3..0000000000 --- a/benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh +++ /dev/null @@ -1,108 +0,0 @@ -#!/usr/bin/env bash -set -eo pipefail -set -x - -# The original DeepSeek-V4-Pro TP8/MTP3 cache-source validation recipe. -# Preserve run 34420875287's cache pressure independently of the 0813 DSpark curve. -source "$(dirname "$0")/../../benchmark_lib.sh" -check_env_vars MODEL MODEL_PATH TP CONC KV_OFFLOADING KV_OFFLOAD_BACKEND -check_env_vars TOTAL_CPU_DRAM_GB RESULT_DIR DURATION PORT GPU_MEMORY_UTILIZATION -check_env_vars EP_SIZE DP_ATTENTION DCP_SIZE PCP_SIZE EVAL_ONLY THINKING_MODE - -if [[ "$TP" != 8 || "$EP_SIZE" != 1 || "$DP_ATTENTION" != false || "$DCP_SIZE" != 1 || "$PCP_SIZE" != 1 ]]; then - echo "Cache-source validation requires TP8 without EP, DP attention, or context parallelism" >&2 - exit 1 -fi -export GPU_COUNT=8 -[[ -d "$MODEL_PATH" ]] || { echo "Missing pre-staged model: $MODEL_PATH" >&2; exit 1; } -nvidia-smi -resolve_trace_source -install_agentic_deps - -export AIPERF_SERVER_METRICS_URLS="http://localhost:$PORT/metrics" -export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="vllm:prompt_tokens_cached_by_source" -export VLLM_ENGINE_READY_TIMEOUT_S=3600 -export VLLM_PREFIX_CACHE_RETENTION_INTERVAL=32768 -export VLLM_USE_V2_MODEL_RUNNER=1 -export VLLM_USE_RUST_FRONTEND=0 -export VLLM_DSV4_MEGA_FP8_COMBINE=1 -export VLLM_RPC_TIMEOUT=600000 -export PYTHONHASHSEED=42 -export TORCH_CUDA_ARCH_LIST=10.0 -export PYTHONNOUSERSITE=1 -export VLLM_FLOAT32_MATMUL_PRECISION=high - -mkdir -p "$RESULT_DIR" -SERVER_LOG="$RESULT_DIR/server.log" -case "$KV_OFFLOAD_BACKEND" in - vllm-native) - require_agentic_kv_offload_backend vllm-native "dram dram+nvme" - SECONDARY_TIERS='[]' - if [[ "$KV_OFFLOADING" == dram+nvme ]]; then - check_env_vars NVME_OFFLOAD_DIR - SECONDARY_TIERS="[{\"type\":\"fs\",\"root_dir\":\"$NVME_OFFLOAD_DIR\",\"locality\":\"LOCAL\"}]" - fi - OFFLOAD_CONFIG="{\"kv_connector\":\"OffloadingConnector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{\"spec_name\":\"TieringOffloadingSpec\",\"cpu_bytes_to_use\":$((TOTAL_CPU_DRAM_GB * 1000000000)),\"secondary_tiers\":$SECONDARY_TIERS}}" - ;; - vllm-simple) - require_agentic_kv_offload_backend vllm-simple nvme - check_env_vars NVME_OFFLOAD_DIR - OFFLOAD_CONFIG="{\"kv_connector\":\"SimpleCPUOffloadConnector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{\"kv_offload_backend\":\"disk\",\"disk_path\":\"$NVME_OFFLOAD_DIR/cache.bin\",\"disk_capacity_bytes\":$((1000000000000 / GPU_COUNT)),\"disk_buffer_slots\":4,\"lazy_offload\":false}}" - ;; - *) echo "Unsupported validation offload backend: $KV_OFFLOAD_BACKEND" >&2; exit 1 ;; -esac - -# Use the same committed golden MTP curve as the shared SRT selector. -SPEC_CONFIG=$(python3 - <<'PY' -import json -import os -from infx.srt_slurm.synthetic_acceptance import GOLDEN_DIR, golden_length - -spec = {"method": "mtp", "num_speculative_tokens": 3} -if os.environ["EVAL_ONLY"] != "true": - spec["rejection_sample_method"] = "synthetic" - spec["synthetic_acceptance_length"] = golden_length( - "dsv4", spec, os.environ["THINKING_MODE"], GOLDEN_DIR - ) -print(json.dumps(spec)) -PY -) -MAX_NUM_SEQS=$((2 * CONC)) -CAPTURE_SIZE_LIST=() -for ((num_seqs = 1; num_seqs <= MAX_NUM_SEQS; num_seqs++)); do - CAPTURE_SIZE_LIST+=("$((num_seqs * 4))") -done -CAPTURE_SIZE_LIST+=(100 200 300 400 500) -CUDA_GRAPH_CAPTURE_SIZES=$(printf '%s\n' "${CAPTURE_SIZE_LIST[@]}" | sort -n -u | paste -sd, -) -COMPILATION_CONFIG="{\"cudagraph_mode\":\"FULL_AND_PIECEWISE\",\"cudagraph_capture_sizes\":[${CUDA_GRAPH_CAPTURE_SIZES}]}" - -{ set +x; } 2>/dev/null -VLLM_CMD=( - vllm serve "$MODEL_PATH" --served-model-name "$MODEL" - --host 0.0.0.0 --port "$PORT" --trust-remote-code - --kv-cache-dtype fp8 --block-size 256 --max-model-len 1048576 - --gpu-memory-utilization "$GPU_MEMORY_UTILIZATION" - --numa-bind --enable-cumem-allocator --no-enable-flashinfer-autotune - --tokenizer-mode deepseek_v4 --tool-call-parser deepseek_v4 - --enable-auto-tool-choice --reasoning-parser deepseek_v4 - --attention-config '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}' - --speculative-config "$SPEC_CONFIG" --no-disable-hybrid-kv-cache-manager - --disable-uvicorn-access-log --compilation-config "$COMPILATION_CONFIG" - --max-num-seqs "$MAX_NUM_SEQS" --tensor-parallel-size "$TP" --data-parallel-size 1 - --kv-transfer-config "$OFFLOAD_CONFIG" -) -printf '%q ' "${VLLM_CMD[@]}" | tee "$RESULT_DIR/vllm_command.txt" -printf '\n' | tee -a "$RESULT_DIR/vllm_command.txt" -sha256sum -c /opt/pr56318-validation/validation-overlay.sha256 | tee "$RESULT_DIR/overlay-verification.log" -"${VLLM_CMD[@]}" > "$SERVER_LOG" 2>&1 & -SERVER_PID=$! -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -if [[ "$EVAL_ONLY" == true ]]; then - run_eval --port "$PORT" -else - build_replay_cmd "$RESULT_DIR" - curl --fail --silent --show-error "$AIPERF_SERVER_METRICS_URLS" > "$RESULT_DIR/metrics-before.prom" - run_agentic_replay_and_write_outputs "$RESULT_DIR" - curl --fail --silent --show-error "$AIPERF_SERVER_METRICS_URLS" > "$RESULT_DIR/metrics-after.prom" -fi diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml new file mode 100644 index 0000000000..9a0af1a64d --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml @@ -0,0 +1,127 @@ +# PR56318 cache-source validation: DeepSeek-V4-Pro TP8/MTP3 on B200 with +# three KV offload modes (native DRAM, Simple NVMe, native DRAM+NVMe). +# One variant per search-space row. +base: + schema: 2 + name: dsv4-fp4-b200-vllm-cache-sources + model: + path: hf:deepseek-ai/DeepSeek-V4-Pro + container: cquil11/vllm-cache-sources@sha256:6f9e476f6a1e1c25ac9bc1a863329538db465812fb4ac18c6553e988eae09a56 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: vllm + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: vllm + connector: null + health_check: + interval_seconds: 10 + max_attempts: 720 + setup_script: pr56318-overlay-check.sh + roles: + agg: + nodes: 1 + workers: 1 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Pro + trust-remote-code: true + kv-cache-dtype: fp8 + block-size: 256 + max-model-len: 1048576 + gpu-memory-utilization: 0.85 + numa-bind: true + enable-cumem-allocator: true + no-enable-flashinfer-autotune: true + tokenizer-mode: deepseek_v4 + tool-call-parser: deepseek_v4 + enable-auto-tool-choice: true + reasoning-parser: deepseek_v4 + attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"indexer_kv_dtype":"mxfp4"}' + speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + no-disable-hybrid-kv-cache-manager: true + disable-uvicorn-access-log: true + tensor-parallel-size: 8 + data-parallel-size: 1 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' + VLLM_USE_V2_MODEL_RUNNER: '1' + VLLM_USE_RUST_FRONTEND: '0' + VLLM_DSV4_MEGA_FP8_COMBINE: '1' + VLLM_RPC_TIMEOUT: '600000' + PYTHONHASHSEED: '42' + TORCH_CUDA_ARCH_LIST: '10.0' + PYTHONNOUSERSITE: '1' + VLLM_FLOAT32_MATMUL_PRECISION: high + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + MODEL: deepseek-ai/DeepSeek-V4-Pro + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:prompt_tokens_cached_by_source' + +# ---------- Variant 1: TP8 c8 native DRAM ---------- +# max-num-seqs = 2 * CONC = 16; capture sizes = {4*n for n=1..16} ∪ {100..500} +override_tp8_c8_dram_native: + roles: + agg: + gpus: 8 + args: + max-num-seqs: 16 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,100,200,300,400,500]}' + kv-transfer-config: '{"kv_connector":"OffloadingConnector","kv_role":"kv_both","kv_connector_extra_config":{"spec_name":"TieringOffloadingSpec","cpu_bytes_to_use":128000000000,"secondary_tiers":[]}}' + benchmark: + env: + CONC: '8' + KV_OFFLOADING: dram + TOTAL_CPU_DRAM_GB: '128' + +# ---------- Variant 2: TP8 c14 Simple NVMe ---------- +# max-num-seqs = 2 * CONC = 28; disk_capacity_bytes = 1e12 / 8 +override_tp8_c14_nvme_simple: + container_mounts: + "/scratch/inferencex-kv-{job_id}": "/kv-offload" + host_setup: + commands: ["mkdir -m 700 /scratch/inferencex-kv-$SLURM_JOB_ID"] + teardown: ["rm -rf /scratch/inferencex-kv-$SLURM_JOB_ID"] + nodes: workers + roles: + agg: + gpus: 8 + args: + max-num-seqs: 28 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,200,300,400,500]}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"kv_offload_backend":"disk","disk_path":"/kv-offload/cache.bin","disk_capacity_bytes":125000000000,"disk_buffer_slots":4,"lazy_offload":false}}' + benchmark: + env: + CONC: '14' + KV_OFFLOADING: nvme + +# ---------- Variant 3: TP8 c14 native DRAM + NVMe ---------- +# max-num-seqs = 2 * CONC = 28; cpu_bytes = 128e9, fs tier at /kv-offload +override_tp8_c14_dramnvme_native: + container_mounts: + "/scratch/inferencex-kv-{job_id}": "/kv-offload" + host_setup: + commands: ["mkdir -m 700 /scratch/inferencex-kv-$SLURM_JOB_ID"] + teardown: ["rm -rf /scratch/inferencex-kv-$SLURM_JOB_ID"] + nodes: workers + roles: + agg: + gpus: 8 + args: + max-num-seqs: 28 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,200,300,400,500]}' + kv-transfer-config: '{"kv_connector":"OffloadingConnector","kv_role":"kv_both","kv_connector_extra_config":{"spec_name":"TieringOffloadingSpec","cpu_bytes_to_use":128000000000,"secondary_tiers":[{"type":"fs","root_dir":"/kv-offload","locality":"LOCAL"}]}}' + benchmark: + env: + CONC: '14' + KV_OFFLOADING: dram+nvme + TOTAL_CPU_DRAM_GB: '128' diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 30d7264832..30a355c896 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8398,12 +8398,12 @@ dsv4-fp4-b200-vllm-agentic-cache-sources-mtp: agentic-coding: - dram-utilization: 0.0592 search-space: - - { tp: 8, kv-offloading: dram, kv-offload-backend: { name: vllm-native }, spec-decoding: mtp, conc-list: [8] } + - { tp: 8, kv-offloading: dram, kv-offload-backend: { name: vllm-native }, spec-decoding: mtp, conc-list: [8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml } - search-space: - - { tp: 8, kv-offloading: nvme, kv-offload-backend: { name: vllm-simple }, spec-decoding: mtp, conc-list: [14] } + - { tp: 8, kv-offloading: nvme, kv-offload-backend: { name: vllm-simple }, spec-decoding: mtp, conc-list: [14], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml } - dram-utilization: 0.0592 search-space: - - { tp: 8, kv-offloading: [dram, nvme], kv-offload-backend: { name: vllm-native }, spec-decoding: mtp, conc-list: [14] } + - { tp: 8, kv-offloading: [dram, nvme], kv-offload-backend: { name: vllm-native }, spec-decoding: mtp, conc-list: [14], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml } dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg: image: cquil11/vllm-cache-sources@sha256:6ce87fcb9ef8b929f7a8270bc6fbba3c219c3440cd77d3bf1ae74db5a1cb5715 diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index d0b160dd5f..2f77ec2c9d 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -453,3 +453,56 @@ def test_b300_keeps_agentic_and_explicit_collector_dispatch(tmp_path, collector) assert calls[-1][-2:] == ["bash", expected] assert "--jobid=42" in calls[-1] assert (tmp_path / "cancelled").read_text() == "42\n" + + +# --------------------------------------------------------------------------- +# Cache-sources recipe: each variant renders the expected kv-transfer-config +# and capture sizes. +# --------------------------------------------------------------------------- + +CACHE_SOURCES_RECIPE = str(ROOT / "benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml") + +CACHE_SOURCES_BASE_ENV = { + "MODEL": "deepseek-ai/DeepSeek-V4-Pro", + "IMAGE": "cquil11/vllm-cache-sources@sha256:6f9e476f6a1e1c25ac9bc1a863329538db465812fb4ac18c6553e988eae09a56", + "PRECISION": "fp4", "FRAMEWORK": "vllm", "TP": "8", "EP_SIZE": "1", + "DP_ATTENTION": "false", "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", + "GPU_COUNT": "8", "IS_AGENTIC": "1", "SPEC_DECODING": "mtp", + "THINKING_MODE": "off", "EVAL_ONLY": "false", "RUN_EVAL": "false", + "MODEL_PREFIX": "dsv4", "DURATION": "3600", "RESULT_FILENAME": "test", + "GPU_MONITOR_INTERVAL": "5", +} + + +@pytest.mark.parametrize( + ("conc", "kv_offloading", "total_dram", "variant_suffix", "connector", "kv_path"), + [ + ("8", "dram", "128", "c8_dram_native", "OffloadingConnector", None), + ("14", "nvme", "0", "c14_nvme_simple", "SimpleCPUOffloadConnector", "/kv-offload/cache.bin"), + ("14", "dram+nvme", "128", "c14_dramnvme_native", "OffloadingConnector", "/kv-offload"), + ], +) +def test_cache_sources_variant_renders_expected_offload( + conc, kv_offloading, total_dram, variant_suffix, connector, kv_path, +): + env = {**CACHE_SOURCES_BASE_ENV, "CONC": conc, "KV_OFFLOADING": kv_offloading, + "TOTAL_CPU_DRAM_GB": total_dram} + config, recipe = select_recipe(CACHE_SOURCES_RECIPE, env) + assert config.endswith(f"override_tp8_{variant_suffix}") + + transfer = json.loads(recipe["roles"]["agg"]["args"]["kv-transfer-config"]) + assert transfer["kv_connector"] == connector + + max_num_seqs = recipe["roles"]["agg"]["args"]["max-num-seqs"] + assert max_num_seqs == 2 * int(conc) + + compilation = json.loads(recipe["roles"]["agg"]["args"]["compilation-config"]) + expected_captures = sorted(set([4 * n for n in range(1, max_num_seqs + 1)] + [100, 200, 300, 400, 500])) + assert compilation["cudagraph_capture_sizes"] == expected_captures + + if kv_path and connector == "SimpleCPUOffloadConnector": + assert transfer["kv_connector_extra_config"]["disk_path"] == kv_path + elif kv_path and connector == "OffloadingConnector": + tiers = transfer["kv_connector_extra_config"]["secondary_tiers"] + assert len(tiers) == 1 + assert tiers[0]["root_dir"] == kv_path diff --git a/runners/launch_b200-nscale-slurm.sh b/runners/launch_b200-nscale-slurm.sh index 291538f32b..1c6c7afc04 100755 --- a/runners/launch_b200-nscale-slurm.sh +++ b/runners/launch_b200-nscale-slurm.sh @@ -786,10 +786,6 @@ run_agentic() { if [[ ! -f "$BENCH_SCRIPT" ]]; then BENCH_SCRIPT="${BENCH_BASE}${FRAMEWORK_SUFFIX}${SPEC_SUFFIX}.sh" fi - if [[ "$MODEL" == "deepseek-ai/DeepSeek-V4-Pro" && "$FRAMEWORK" == vllm && "$SPEC_DECODING" == mtp && "$IMAGE" == cquil11/vllm-cache-sources@sha256:6f9e476f6a1e1c25ac9bc1a863329538db465812fb4ac18c6553e988eae09a56 ]]; then - BENCH_SCRIPT="benchmarks/single_node/agentic/dsv4_fp4_b200_vllm_cache_sources_mtp.sh" - export GPU_MEMORY_UTILIZATION=0.85 - fi LOCK_FILE="${SQUASH_FILE}.lock" # TODO(Cam): lmsysorg/sglang:deepseek-v4-blackwell installs sglang editable at @@ -797,7 +793,7 @@ run_agentic() { # the default $GITHUB_WORKSPACE:/workspace/ bind-mount masks the install and # breaks `import sglang`. Mount this one image at /ix instead; drop the # conditional once the image stops installing editable under /workspace. - if [[ "$IMAGE" == *deepseek-v4-blackwell* || "$BENCH_SCRIPT" == *cache_sources_mtp.sh ]]; then + if [[ "$IMAGE" == *deepseek-v4-blackwell* ]]; then CONTAINER_MOUNT_DIR=/ix else CONTAINER_MOUNT_DIR=/workspace @@ -814,37 +810,9 @@ run_agentic() { else CONTAINER_MOUNTS="$GITHUB_WORKSPACE:$CONTAINER_MOUNT_DIR,$MODEL_PATH:$MODEL_PATH,$AIPERF_MMAP_CACHE_HOST_PATH:/aiperf_mmap_cache" fi - if [[ "$BENCH_SCRIPT" == *cache_sources_mtp.sh ]]; then - export INFMAX_CONTAINER_WORKSPACE="$CONTAINER_MOUNT_DIR" - export RESULT_DIR="$CONTAINER_MOUNT_DIR/results" - fi - salloc --partition=$SLURM_PARTITION --account=$SLURM_ACCOUNT --gres=gpu:$GPU_COUNT --exclusive --mem=0 --time="$SALLOC_TIME_LIMIT" --no-shell --job-name="$RUNNER_NAME" || return 1 + salloc --partition=$SLURM_PARTITION --account=$SLURM_ACCOUNT --gres=gpu:$GPU_COUNT --exclusive --mem=0 --time="$SALLOC_TIME_LIMIT" --no-shell --job-name="$RUNNER_NAME" JOB_ID=$(squeue --name="$RUNNER_NAME" -u "$USER" -h -o %A | head -n1) - [[ "$JOB_ID" =~ ^[0-9]+$ ]] || { echo "Could not resolve Slurm allocation" >&2; return 1; } - - if [[ "$BENCH_SCRIPT" == *cache_sources_mtp.sh ]]; then - NVME_HOST_DIR="" - cleanup_offload_cache() { - local rc=$? - if [[ -n "$NVME_HOST_DIR" ]]; then - timeout --kill-after=10s 300s srun --jobid="$JOB_ID" rm -rf -- "$NVME_HOST_DIR" || rc=1 - fi - scancel "$JOB_ID" || true - exit "$rc" - } - trap cleanup_offload_cache EXIT - trap 'exit 130' INT - trap 'exit 143' TERM - fi - - if [[ "$KV_OFFLOADING" == *nvme* ]]; then - # This directory belongs only to the validated numeric allocation above. - export NVME_HOST_DIR="/scratch/inferencex-kv-$JOB_ID" - srun --jobid="$JOB_ID" mkdir -m 700 "$NVME_HOST_DIR" || return 1 - export NVME_OFFLOAD_DIR=/kv-offload - CONTAINER_MOUNTS+=",$NVME_HOST_DIR:$NVME_OFFLOAD_DIR" - fi # Bench scripts skip `hf download` when MODEL is a local path. export MODEL="$MODEL_PATH" diff --git a/runners/srt-slurm/patches/pr56318-overlay-validation.patch b/runners/srt-slurm/patches/pr56318-overlay-validation.patch new file mode 100644 index 0000000000..6cfa15c934 --- /dev/null +++ b/runners/srt-slurm/patches/pr56318-overlay-validation.patch @@ -0,0 +1,14 @@ +diff --git a/configs/patches/pr56318-overlay-check.sh b/configs/patches/pr56318-overlay-check.sh +new file mode 100644 +index 000000000..000000001 +--- /dev/null ++++ b/configs/patches/pr56318-overlay-check.sh +@@ -0,0 +1,8 @@ ++#!/bin/bash ++# Verify PR56318 validation overlay checksums before engine start. ++# A no-op when run on images that do not carry the overlay. ++set -euo pipefail ++OVERLAY=/opt/pr56318-validation/validation-overlay.sha256 ++if [ -f "$OVERLAY" ]; then ++ sha256sum -c "$OVERLAY" ++fi From ef523df3b3ccb2f51d5ca385804857156fbbc1c3 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Sun, 27 Sep 2026 01:31:19 -0500 Subject: [PATCH 5/7] fix: use current AgentX entrypoint in cache-source validation --- .../gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml index 853ca3f404..baf5f7e97e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml @@ -214,7 +214,7 @@ srun_options: benchmark: type: custom - command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + command: bash /infmax-workspace/benchmarks/srt_agentic.sh env: AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:prompt_tokens_cached_by_source" INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" From 968e80fe593a91e71aa014acbc9b05e78e8087dc Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 1 Oct 2026 12:30:05 -0500 Subject: [PATCH 6/7] test: rerun cache-source sweeps with bounded-range sweep accounting --- .../agentx/cache-sources-dep4-dep16-c256-mtp.yaml | 4 ++-- .../dsv4/vllm/b200-fp4-mtp/cache-sources.yaml | 2 +- inferencex-e2e/configs/nvidia-master.yaml | 4 ++-- inferencex-e2e/perf-changelog.yaml | 11 +++++++++++ 4 files changed, 16 insertions(+), 5 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml index baf5f7e97e..41aa39a9a9 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml @@ -9,14 +9,14 @@ name: "dsv4-vllm-cache-sources-gb300-dep4-dep16-c256" model: path: "deepseek-v4-pro" - container: "cquil11/vllm-cache-sources@sha256:6ce87fcb9ef8b929f7a8270bc6fbba3c219c3440cd77d3bf1ae74db5a1cb5715" + container: "cquil11/vllm-cache-sources@sha256:188445e9835a0ed9cd022f0e70c61dd3fd9d24792008841d4747da1797e7fe4b" precision: "fp4" identity: model: repo: "deepseek-ai/DeepSeek-V4-Pro" container: - image: "cquil11/vllm-cache-sources@sha256:6ce87fcb9ef8b929f7a8270bc6fbba3c219c3440cd77d3bf1ae74db5a1cb5715" + image: "cquil11/vllm-cache-sources@sha256:188445e9835a0ed9cd022f0e70c61dd3fd9d24792008841d4747da1797e7fe4b" frameworks: dynamo: "1.4.0" diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml index 9a0af1a64d..701a48e8d2 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml @@ -6,7 +6,7 @@ base: name: dsv4-fp4-b200-vllm-cache-sources model: path: hf:deepseek-ai/DeepSeek-V4-Pro - container: cquil11/vllm-cache-sources@sha256:6f9e476f6a1e1c25ac9bc1a863329538db465812fb4ac18c6553e988eae09a56 + container: cquil11/vllm-cache-sources@sha256:873b8bcb4cc734cfc3e3b7f9d2c1495be329439a944f76a7b356448be6a61641 precision: fp4 resources: gpu_type: b200 diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 7bb01fedc9..5b8d2710e3 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8385,7 +8385,7 @@ dsv41flash-fp4-b200-sglang-agentic-dspark: # PR56318: rerun the four successful hybrid cache-source validation points. dsv4-fp4-b200-vllm-agentic-cache-sources-mtp: - image: cquil11/vllm-cache-sources@sha256:6f9e476f6a1e1c25ac9bc1a863329538db465812fb4ac18c6553e988eae09a56 + image: cquil11/vllm-cache-sources@sha256:873b8bcb4cc734cfc3e3b7f9d2c1495be329439a944f76a7b356448be6a61641 model: deepseek-ai/DeepSeek-V4-Pro model-prefix: dsv4 runner: cluster:b200-nscale @@ -8404,7 +8404,7 @@ dsv4-fp4-b200-vllm-agentic-cache-sources-mtp: - { tp: 8, kv-offloading: [dram, nvme], kv-offload-backend: { name: vllm-native }, spec-decoding: mtp, conc-list: [14], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b200-fp4-mtp/cache-sources.yaml } dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg: - image: cquil11/vllm-cache-sources@sha256:6ce87fcb9ef8b929f7a8270bc6fbba3c219c3440cd77d3bf1ae74db5a1cb5715 + image: cquil11/vllm-cache-sources@sha256:188445e9835a0ed9cd022f0e70c61dd3fd9d24792008841d4747da1797e7fe4b model: deepseek-ai/DeepSeek-V4-Pro model-prefix: dsv4 runner: cluster:gb300-nv diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 0db4cf1f9a..bd6d05d255 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9192,3 +9192,14 @@ - "Preserve B200 GPU utilization 0.85, native DRAM utilization 0.0592, the dedicated 1 TB Simple NVMe budget, 3600-second AgentX profiles, and disabled evals. GB300 is one DEP4 prefill worker on one node plus one DEP16 decode worker across four nodes: five nodes and twenty GPUs total." - "Keep existing benchmark curves unchanged. Adapt the isolated old-Pro validation recipes to current InferenceX paths and current vLLM flag names, use the Python frontend, and collect all raw backend metrics with the cached-prompt source counter required." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3491 + +- config-keys: + - dsv4-fp4-b200-vllm-agentic-cache-sources-mtp + - dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg + scenario-type: + - agentic-coding + no-evals: true + description: + - "Rerun the four DeepSeek-V4-Pro cache-source points with vLLM PR56318 commit 6ce00ee2ebcf29076d671786c36bf8f8ca71bd0b: B200 native DRAM c8, Simple NVMe c14, native DRAM+NVMe c14, and GB300 NIXL plus Mooncake Store c256." + - "Keep the prior runtimes and serving settings, 3600-second AIPerf AgentX profiles, Python frontend, and Prometheus source collection. Update the attribution overlay for bounded attention ranges and sweep-line accounting." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3491 From 0efad80962a04adbefe8e2e32d372b5520eeb22d Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 1 Oct 2026 12:52:50 -0500 Subject: [PATCH 7/7] fix: preserve B200 DeepSeek Pro validation checkpoint --- .../infx/launch/drivers/srt/checkout.py | 4 +- .../infx/launch/drivers/srt/models.py | 10 +++++ .../infx/tests/launch/test_srt_policy.py | 38 +++++++++++++++++++ inferencex-e2e/perf-changelog.yaml | 10 +++++ 4 files changed, 61 insertions(+), 1 deletion(-) diff --git a/inferencex-e2e/infx/launch/drivers/srt/checkout.py b/inferencex-e2e/infx/launch/drivers/srt/checkout.py index d7c5e12a48..42c030fb39 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/checkout.py +++ b/inferencex-e2e/infx/launch/drivers/srt/checkout.py @@ -87,7 +87,9 @@ def prepare_checkout(run: SrtRun, destination: Path, *, power: bool) -> Checkout if run.request.config_file == ( "recipes/dsv4/vllm/gb300-fp4/agentx/cache-sources-dep4-dep16-c256-mtp.yaml" ): - _git("-C", destination, "apply", run.workspace / "runners/srt-slurm/validation/pr56318.patch") + _git( + "-C", destination, "apply", run.workspace / "runners/srt-slurm/validation/pr56318.patch" + ) head = _git("-C", destination, "rev-parse", "HEAD", capture=True) if head != commit: raise LaunchError(f"srt-slurm checkout is at {head}, expected {commit}") diff --git a/inferencex-e2e/infx/launch/drivers/srt/models.py b/inferencex-e2e/infx/launch/drivers/srt/models.py index bb7f701534..63ff070d23 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/models.py +++ b/inferencex-e2e/infx/launch/drivers/srt/models.py @@ -39,6 +39,16 @@ class Override: OVERRIDES: dict[str, tuple[Override, ...]] = { "b200-nscale": ( + Override( + Match( + any_of("dsv4"), + any_of("fp4"), + any_of("vllm"), + multinode=False, + model_glob="deepseek-ai/DeepSeek-V4-Pro", + ), + entry="DeepSeek-V4-Pro-NVFP4", + ), Override( Match(any_of("glm5.1"), frameworks=any_of("tilert")), entry="GLM-5.1-FP8@shared", diff --git a/inferencex-e2e/infx/tests/launch/test_srt_policy.py b/inferencex-e2e/infx/tests/launch/test_srt_policy.py index cbaf6d97da..4893549428 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_policy.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_policy.py @@ -106,6 +106,44 @@ def test_a_checkpoint_that_must_be_readable_fails_before_submission(tmp_path, mo assert host_path(c, checkpoint(c, request(MODEL="org/S"))) == tmp_path / "shared/s" +@pytest.mark.parametrize( + ("model", "framework", "multinode", "directory"), + [ + ("DeepSeek-V4-Pro", "vllm", "false", "converted"), + ("DeepSeek-V4-Pro-0813", "vllm", "false", "august"), + ("DeepSeek-V4-Pro", "vllm", "true", "original"), + ("DeepSeek-V4-Pro", "sglang", "false", "original"), + ], +) +def test_b200_original_pro_single_node_vllm_keeps_converted_checkpoint( + tmp_path, + model, + framework, + multinode, + directory, +): + c = cluster(tmp_path) + c.bind_id("b200-nscale") + entry = c.models.entries["M"] + entries = { + name: entry.model_copy(update={"dir": path}) + for name, path in { + "DeepSeek-V4-Pro": "original", + "DeepSeek-V4-Pro-NVFP4": "converted", + "DeepSeek-V4-Pro-0813": "august", + }.items() + } + c = c.model_copy(update={"models": c.models.model_copy(update={"entries": entries})}) + point = request( + MODEL=f"deepseek-ai/{model}", + MODEL_PREFIX="dsv4", + PRECISION="fp4", + FRAMEWORK=framework, + IS_MULTINODE=multinode, + ) + assert single_node_model_path(c, point) == str(tmp_path / "shared" / directory) + + def test_a_points_own_model_path_is_what_its_job_serves(tmp_path, monkeypatch): c = cluster(tmp_path) host = request(MODEL="org/M", MODEL_PATH="/host/m") diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index bd6d05d255..31e35e46e7 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9203,3 +9203,13 @@ - "Rerun the four DeepSeek-V4-Pro cache-source points with vLLM PR56318 commit 6ce00ee2ebcf29076d671786c36bf8f8ca71bd0b: B200 native DRAM c8, Simple NVMe c14, native DRAM+NVMe c14, and GB300 NIXL plus Mooncake Store c256." - "Keep the prior runtimes and serving settings, 3600-second AIPerf AgentX profiles, Python frontend, and Prometheus source collection. Update the attribution overlay for bounded attention ranges and sweep-line accounting." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3491 + +- config-keys: + - dsv4-fp4-b200-vllm-agentic-cache-sources-mtp + - dsv4-fp4-gb300-dynamo-vllm-agentic-cache-sources-mtp-disagg + scenario-type: + - agentic-coding + no-evals: true + description: + - "Restore the original B200 single-node vLLM DeepSeek-V4-Pro-NVFP4 checkpoint mapping after the launcher migration. Repeat the four cache-source validation points with unchanged images and serving settings." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3491