diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml index 4cc23aa069..1d3be2acdf 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml @@ -6,7 +6,7 @@ base: name: dsv41flash-fp4-gb300-vllm-agentic model: path: hf:deepseek-ai/DeepSeek-V4.1-Flash - container: vllm/vllm-openai:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d + container: vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7 precision: fp4 resources: gpu_type: gb300 @@ -37,17 +37,21 @@ base: enable-auto-tool-choice: true reasoning-parser: deepseek_v41 engram-config: '{"cpu_offload":true}' - # FlashInfer sparse attention with MXFP4 indexer KV and sparse logits. + kernel-config: '{"enable_flashinfer_autotune":true}' attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' kv-cache-dtype: fp8 # Five-token DSpark with probabilistic drafting. Throughput runs replace # block rejection with the golden acceptance length. speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}' max-model-len: 1048576 - max-num-seqs: 256 + # Piecewise graph capture sizes are multiples of the six-token DSpark + # verification block, denser for small batches. + compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + max-cudagraph-capture-size: 8190 + max-num-batched-tokens: 8192 disable-uvicorn-access-log: true env: - VLLM_DISABLED_KERNELS: FlashInferCutedslMxfp8LinearKernel + VLLM_USE_V2_MODEL_RUNNER: '1' VLLM_USE_RUST_FRONTEND: '1' PYTHONUNBUFFERED: '1' VLLM_ENGINE_READY_TIMEOUT_S: '7200' @@ -58,216 +62,66 @@ base: MODEL: deepseek-ai/DeepSeek-V4.1-Flash AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' -# One variant per point. Piecewise graph capture sizes are multiples of the -# six-token DSpark verification block, denser for small batches, and each -# batched-token limit matches the largest captured graph: CONC <= 4 and TP2 -# CONC 128 capture up to 2046 tokens (the latter at 0.97 memory utilization), -# every other point up to 8190. -override_tp4_c1: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' - max-cudagraph-capture-size: 2046 - max-num-batched-tokens: 2048 - benchmark: - env: - CONC: '1' - -override_tp4_c2: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' - max-cudagraph-capture-size: 2046 - max-num-batched-tokens: 2048 - benchmark: - env: - CONC: '2' - -override_tp4_c4: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' - max-cudagraph-capture-size: 2046 - max-num-batched-tokens: 2048 - benchmark: - env: - CONC: '4' - -override_tp4_c8: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' - max-cudagraph-capture-size: 8190 - max-num-batched-tokens: 8192 - benchmark: - env: - CONC: '8' - -override_tp4_c16: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' - max-cudagraph-capture-size: 8190 - max-num-batched-tokens: 8192 - benchmark: - env: - CONC: '16' - -override_tp4_c32: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' - max-cudagraph-capture-size: 8190 - max-num-batched-tokens: 8192 - benchmark: - env: - CONC: '32' - -override_tp4_c64: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' - max-cudagraph-capture-size: 8190 - max-num-batched-tokens: 8192 - benchmark: - env: - CONC: '64' +# Native zip expansion creates one variant per concurrency. -override_tp4_c128: +# TP4 covers low concurrency with THP Engram tables. +zip_override_tp4: roles: agg: gpus: 4 args: tensor-parallel-size: 4 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' - max-cudagraph-capture-size: 8190 - max-num-batched-tokens: 8192 - benchmark: - env: - CONC: '128' - -override_tp2_c2: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' - max-cudagraph-capture-size: 2046 - max-num-batched-tokens: 2048 - benchmark: - env: - CONC: '2' - -override_tp2_c4: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' - max-cudagraph-capture-size: 2046 - max-num-batched-tokens: 2048 - benchmark: - env: - CONC: '4' - -override_tp2_c8: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' - max-cudagraph-capture-size: 8190 - max-num-batched-tokens: 8192 - benchmark: - env: - CONC: '8' - -override_tp2_c16: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' - max-cudagraph-capture-size: 8190 - max-num-batched-tokens: 8192 - benchmark: - env: - CONC: '16' - -override_tp2_c32: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' - max-cudagraph-capture-size: 8190 - max-num-batched-tokens: 8192 - benchmark: - env: - CONC: '32' - -override_tp2_c64: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' - max-cudagraph-capture-size: 8190 - max-num-batched-tokens: 8192 - benchmark: - env: - CONC: '64' - -override_tp2_c128: - roles: - agg: - gpus: 2 - args: - tensor-parallel-size: 2 - compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}' - max-cudagraph-capture-size: 2046 - max-num-batched-tokens: 2048 - gpu-memory-utilization: 0.97 - benchmark: - env: - CONC: '128' - -override_tp2_c1: + engram-config: '{"cpu_offload":true,"use_thp":true}' + max-num-seqs: [8, 8, 16, 32] + compilation-config: + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}' + max-cudagraph-capture-size: [4092, 4092, 8190, 8190] + max-num-batched-tokens: [4096, 4096, 8192, 8192] + benchmark: + env: + CONC: ['1', '4', '8', '16'] + +# DEP2: TP1 x DP2 + EP2 with MegaMoE behind a consistent-hash router; +# MegaAttention at CONC 128 and above. +zip_override_dep2: + frontend: + type: vllm-router + args: + policy: consistent_hash + prometheus-host: 127.0.0.1 + prometheus-port: 18000 + request-timeout-secs: 14400 + disable-retries: true + setup_script: vllm-router-0.1.14.sh roles: agg: gpus: 2 args: - tensor-parallel-size: 2 - max-cudagraph-capture-size: 64 - max-num-batched-tokens: 4096 - max-num-seqs: 16 - benchmark: - env: - CONC: '1' + tensor-parallel-size: 1 + data-parallel-size: 2 + enable-expert-parallel: true + kernel-config: '{"enable_flashinfer_autotune":true,"moe_backend":"deep_gemm_mega_moe"}' + engram-config: '{"cpu_offload":true,"embedding_across_dp":true}' + max-num-seqs: [16, 16, 32, 64, 128, 192] + attention-config: + - '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHMLA_MEGA_ATTN_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + - '{"backend":"FLASHMLA_MEGA_ATTN_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}' + compilation-config: + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}' + - '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048]}' + max-cudagraph-capture-size: [8190, 8190, 8190, 8190, 8190, 4092] + max-num-batched-tokens: [8192, 8192, 8192, 8192, 16384, 16384] + benchmark: + env: + CONC: ['8', '16', '32', '64', '128', '192'] diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index dcbe00cb46..0e1f65e659 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8534,7 +8534,7 @@ dsv41flash-fp4-b200-sglang-agentic-dspark: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml } dsv41flash-fp4-gb300-vllm-agentic-dspark: - image: vllm/vllm-openai:nightly-ac68c3087215e0a4f3cdfa218508c6aada57235d + image: vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7 model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:gb300-nv @@ -8546,13 +8546,11 @@ dsv41flash-fp4-gb300-vllm-agentic-dspark: - dram-utilization: 0.80 search-space: # Engram weights use UVA DRAM; the KV cache stays GPU-resident. - - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml } - # TP2 halves the GPU count per replica. Weights rise to ~175 GiB on each - # 277 GiB GPU with the Engram tables in pinned host DRAM, so the arm - # takes the same caps as the B200 TP2 arm: batched tokens 4096 (the - # indexer's logits buffer is 32 GiB at the upstream 16384) and graph - # capture stopped at 512, leaving ~36 GiB of KV per GPU. - - { tp: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml } + # TP4 covers low concurrency with FlashInfer attention and THP Engram tables. + - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 4, 8, 16], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml } + # DEP2 is native TP1 x DP2 + EP2 with MegaMoE behind a consistent-hash vLLM + # Router; FlashInfer attention up to conc 64, MegaAttention above. + - { tp: 2, ep: 2, dp-attn: true, kv-offloading: none, spec-decoding: mtp, conc-list: [8, 16, 32, 64, 128, 192], router: { name: vllm-router, version: "0.1.14" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml } # H200 AgentX arm for DeepSeek-V4.1-Flash. Upstream marks h200 verified and says # the GB200 NVL4 TP4 layout becomes TP8 on 8-GPU nodes, so this is TP8. diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 4ca114de6c..a82885b6d1 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -328,7 +328,7 @@ EAGLE3 K3, golden AL 2.78 and indexer CP are unchanged. The GB200 DSpark recipe uses a minimum CUDA graph capture size of 64 tokens to cover concurrent AgentX subagents. This raises c1/c2/c4 from 8/16/32 to 64; c8 and above retain their existing sizes. The full trace, AL 3.51, and Engram UVA settings are preserved; low-concurrency tail latency improvements require CI confirmation. The B200 DSpark recipe uses the same minimum capture size and preserves the same workload settings. -The GB300 DSpark recipe uses the same minimum capture size and preserves the same workload settings. +The GB300 DSpark recipe sets explicit capture sizes per point, described below. The H200 DSpark recipe uses the same minimum capture size and preserves the same workload settings. The B300 DSpark recipe sets explicit capture sizes per point, described below. @@ -342,7 +342,7 @@ weights determine the recipe's `precision: fp4` label. The GPU-specific entry points share the text-only serving behavior, `deepseek_v41` tokenizer and parsers, 1M context, and the shared AgentX trace replay, power, metrics, and eval -helpers. The TP4 concurrency range is 1–128. The shared script sizes graph capture +helpers. Unless noted below, the TP4 concurrency range is 1–128. The shared script sizes graph capture for the six-token DSpark verification block. The srt-slurm single-node path mounts the checkout at `/infmax-workspace`, so AgentX runtime directories are not created under `/workspace`. Cluster model paths and persistent caches are reused. The recipe probes the serving port on the compute node and selects an available @@ -355,6 +355,11 @@ behind a consistent-hash vLLM Router, switching to MegaAttention at 128 and abov `FULL_AND_PIECEWISE` CUDA graphs sized in multiples of the six-token verification block. Other SKUs continue to use the shared script. +The GB300 entry uses `vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7` with FlashInfer +autotuning. TP4 covers concurrency 1–16; DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) covers 8–192 +behind a consistent-hash vLLM Router, switching to MegaAttention at 128 and above. All points use +`FULL_AND_PIECEWISE` CUDA graphs sized in multiples of the six-token verification block. + The GB300 launcher allows 7200 seconds for engine readiness. In [run 34504969146](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34504969146), the Rust frontend exhausted its 3600-second deadline while the engine was still capturing graphs; model loading alone took 18–23 minutes. This extends startup time without changing the benchmark duration or decoding settings. GPU sweep and eval evidence is required before calling any recipe validated. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 3cacd57f63..b0aaba8e65 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -302,7 +302,7 @@ TP4 约 1.23 TB)全部来自节点 0 的 1.5 TB 内存。`runners/srt-slurm/ho GB200 的 DSpark 配方将 CUDA graph 最小捕获范围设为 64 tokens,以覆盖 AgentX 子代理并发。这会将 c1/c2/c4 的上限从 8/16/32 提升至 64;c8 及以上保持原有大小。完整轨迹、AL 3.51 和 Engram UVA 配置保持不变;需通过 CI 验证低并发尾延迟改善。 B200 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工作负载配置。 -GB300 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工作负载配置。 +GB300 的 DSpark 配方按测试点显式设置捕获尺寸,详见下文。 H200 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工作负载配置。 B300 的 DSpark 配方按测试点显式设置捕获尺寸,详见下文。 @@ -315,7 +315,7 @@ B300 的 DSpark 配方按测试点显式设置捕获尺寸,详见下文。 专家权重为 MXFP4,因此配方标记为 `precision: fp4`。 各 GPU 入口共用纯文本服务行为,使用 `deepseek_v41` tokenizer 和解析器、1M 上下文, -以及共享的 AgentX 轨迹回放、功耗、指标和 eval helper。TP4 的并发范围为 1–128。 +以及共享的 AgentX 轨迹回放、功耗、指标和 eval helper。除下文另有说明外,TP4 的并发范围为 1–128。 共享脚本按六 token DSpark 验证块设置 CUDA graph capture。srt-slurm 单节点路径将检出挂载到 `/infmax-workspace`,避免在 `/workspace` 下创建 AgentX 运行目录。沿用集群的模型路径和持久化缓存。配方在计算节点探测服务端口,首选端口被占用时选择可用端口, 服务、回放、指标和 eval 共用同一端点。所有配方都必须获得 GPU sweep 和 eval @@ -326,6 +326,11 @@ TP4 覆盖并发 1–16;DEP2(TP1 x DP2 + EP2,DeepGEMM MegaMoE)覆盖 8 Router,并发 128 及以上改用 MegaAttention。所有测试点使用 `FULL_AND_PIECEWISE` CUDA graph,捕获尺寸为 六 token 验证块的倍数。其他 SKU 继续使用共享脚本。 +GB300 条目使用 `vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7`,开启 FlashInfer autotune。 +TP4 覆盖并发 1–16;DEP2(TP1 x DP2 + EP2,DeepGEMM MegaMoE)覆盖 8–192,前置一致性哈希 vLLM +Router,并发 128 及以上改用 MegaAttention。所有测试点使用 `FULL_AND_PIECEWISE` CUDA graph,捕获尺寸为 +六 token 验证块的倍数。 + GB300 launcher 将引擎就绪等待时间设为 7200 秒。在[运行 34504969146](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34504969146) 中,仅模型加载就耗时 18–23 分钟;Rust frontend 达到 3600 秒期限时,引擎仍在捕获 CUDA graph。此次仅延长启动等待时间,基准测试时长和解码设置保持不变。 来源:[上游配方](https://recipes.vllm.ai/deepseek-ai/DeepSeek-V4.1-Flash)。 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index f813a77ed7..b5f4f5572b 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9266,3 +9266,12 @@ - "Update the B300 DeepSeek-V4.1-Flash vLLM AgentX image from nightly ddd6fbca to nightly-dev-x86_64-cu130-ac9126e58aa7 and enable FlashInfer autotuning." - "Run TP4 at concurrency 1-16 and replace TP2 with DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) at concurrency 8-192 behind a consistent-hash vLLM Router." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3653 + +- config-keys: + - dsv41flash-fp4-gb300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Update the GB300 DeepSeek-V4.1-Flash vLLM AgentX image from nightly ac68c308 to nightly-dev-arm64-cu130-ac9126e58aa7 and enable FlashInfer autotuning." + - "Run TP4 at concurrency 1-16 and replace TP2 with DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) at concurrency 8-192 behind a consistent-hash vLLM Router." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3652