Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ base:
name: dsv41flash-fp4-gb300-vllm-agentic
model:
path: hf:deepseek-ai/DeepSeek-V4.1-Flash
container: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e
container: vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89
precision: fp4
resources:
gpu_type: gb300
Expand Down Expand Up @@ -37,12 +37,23 @@ base:
enable-auto-tool-choice: true
reasoning-parser: deepseek_v41
engram-config: '{"cpu_offload":true}'
# Avoid invalid MXFP8 split-K tactics during DSpark autotune warmup.
kernel-config: '{"enable_flashinfer_autotune":false}'
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}'
kv-cache-dtype: fp8
# Five-token DSpark with probabilistic drafting. Throughput runs replace
# block rejection with the golden acceptance length.
speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}'
max-model-len: 1048576
# Piecewise graph capture sizes are multiples of the six-token DSpark
# verification block, denser for small batches. Each batched-token
# limit matches the largest captured graph.
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
disable-uvicorn-access-log: true
env:
VLLM_USE_V2_MODEL_RUNNER: '1'
VLLM_USE_RUST_FRONTEND: '1'
PYTHONUNBUFFERED: '1'
VLLM_ENGINE_READY_TIMEOUT_S: '7200'
Expand All @@ -53,199 +64,86 @@ base:
MODEL: deepseek-ai/DeepSeek-V4.1-Flash
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:'

# One variant per point. Graph capture starts at 64 tokens and doubles until it
# covers CONC x (1 + 5 drafts), up to 2048. TP2 leaves ~175 GiB of weights on
# each 277 GiB GPU, so it caps batched tokens at 4096 (the indexer's 1M-wide
# logits buffer), bounds the scheduler at 2x CONC within [16, 256] (FlashInfer
# autotune fails below 16) and stops capturing above 512 tokens.
override_tp4_c1:
# TP4 covers low concurrency. CONC <= 4 captures graphs up to 2046 tokens with a
# 2048-token batch budget; higher points keep the base 8190/8192 tier. Engram
# tables use transparent huge pages. AgentX subagents keep up to 3 x CONC
# requests in flight at CONC 1 and under CONC at higher points, so each engine
# admits 2 x CONC sequences, at least 8. Native zip expansion creates one
# variant per concurrency.
zip_override_tp4:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
engram-config: '{"cpu_offload":true,"use_thp":true}'
max-num-seqs: [8, 8, 16, 32]
compilation-config:
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: [2046, 2046, 8190, 8190]
max-num-batched-tokens: [2048, 2048, 8192, 8192]
benchmark:
env:
CONC: '1'
CONC: ['1', '4', '8', '16']

override_tp4_c2:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
benchmark:
env:
CONC: '2'

override_tp4_c4:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
benchmark:
env:
CONC: '4'

override_tp4_c8:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
benchmark:
env:
CONC: '8'

override_tp4_c16:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 128
benchmark:
env:
CONC: '16'

override_tp4_c32:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 256
benchmark:
env:
CONC: '32'

override_tp4_c64:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 512
benchmark:
env:
CONC: '64'

override_tp4_c128:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 1024
benchmark:
env:
CONC: '128'

override_tp2_c1:
# TP2 covers the middle range with the same graph tiers, sequence caps and
# huge-page Engram tables.
zip_override_tp2:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '1'

override_tp2_c2:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '2'

override_tp2_c4:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '4'

override_tp2_c8:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '8'

override_tp2_c16:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 128
max-num-batched-tokens: 4096
max-num-seqs: 32
benchmark:
env:
CONC: '16'

override_tp2_c32:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 256
max-num-batched-tokens: 4096
max-num-seqs: 64
benchmark:
env:
CONC: '32'

override_tp2_c64:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 512
max-num-batched-tokens: 4096
max-num-seqs: 128
benchmark:
env:
CONC: '64'

override_tp2_c128:
engram-config: '{"cpu_offload":true,"use_thp":true}'
max-num-seqs: [8, 8, 16, 32, 64]
compilation-config:
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: [2046, 2046, 8190, 8190, 8190]
max-num-batched-tokens: [2048, 2048, 8192, 8192, 8192]
benchmark:
env:
CONC: ['2', '4', '8', '16', '32']

# DEP2 runs TP1 x DP2 with EP2 on two GPUs, with MegaAttention/MegaMoE and
# Engram embeddings sharded across DP ranks. The consistent-hash router pins
# each conversation to one DP rank. Each DP rank admits CONC sequences, above
# the deployment-wide in-flight peak; CONC is deployment-wide. CONC 128 stops
# graph capture at 2046 tokens.
zip_override_dep2:
frontend:
type: vllm-router
args:
policy: consistent_hash
prometheus-host: 127.0.0.1
prometheus-port: 18000
request-timeout-secs: 14400
disable-retries: true
setup_script: vllm-router-0.1.14.sh
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 512
max-num-batched-tokens: 4096
max-num-seqs: 256
benchmark:
env:
CONC: '128'
tensor-parallel-size: 1
data-parallel-size: 2
max-num-seqs: [16, 32, 48, 64, 128]
compilation-config:
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046]}'
max-cudagraph-capture-size: [8190, 8190, 8190, 8190, 2046]
enable-expert-parallel: true
kernel-config: '{"enable_flashinfer_autotune":false,"moe_backend":"deep_gemm_mega_moe"}'
attention-config: '{"backend":"FLASHMLA_MEGA_ATTN_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}'
engram-config: '{"cpu_offload":true,"embedding_across_dp":true}'
benchmark:
env:
CONC: ['16', '32', '48', '64', '128']
15 changes: 7 additions & 8 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8396,7 +8396,7 @@ dsv41flash-fp4-b200-sglang-agentic-dspark:
- { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/b200-fp4-mtp/agentic.yaml }

dsv41flash-fp4-gb300-vllm-agentic-dspark:
image: vllm/vllm-openai:nightly-af1c01499b289be555c475669ba50a88e96d846e
image: vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89
model: deepseek-ai/DeepSeek-V4.1-Flash
model-prefix: dsv41flash
runner: cluster:gb300-nv
Expand All @@ -8408,13 +8408,12 @@ dsv41flash-fp4-gb300-vllm-agentic-dspark:
- dram-utilization: 0.80
search-space:
# Engram weights use UVA DRAM; the KV cache stays GPU-resident.
- { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml }
# TP2 halves the GPU count per replica. Weights rise to ~175 GiB on each
# 277 GiB GPU with the Engram tables in pinned host DRAM, so the arm
# takes the same caps as the B200 TP2 arm: batched tokens 4096 (the
# indexer's logits buffer is 32 GiB at the upstream 16384) and graph
# capture stopped at 512, leaving ~36 GiB of KV per GPU.
- { tp: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml }
# TP4 covers low concurrency with FlashInfer attention and THP Engram tables.
- { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 4, 8, 16], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml }
# TP2 covers the middle range with the same attention and graph tiers.
- { tp: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [2, 4, 8, 16, 32], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml }
# DEP2 is native TP1 x DP2 + EP2, with MegaAttention/MegaMoE.
- { tp: 2, ep: 2, dp-attn: true, kv-offloading: none, spec-decoding: mtp, conc-list: [16, 32, 48, 64, 128], router: { name: vllm-router, version: "0.1.14" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb300-fp4-mtp/agentic.yaml }

# H200 AgentX arm for DeepSeek-V4.1-Flash. Upstream marks h200 verified and says
# the GB200 NVL4 TP4 layout becomes TP8 on 8-GPU nodes, so this is TP8.
Expand Down
19 changes: 18 additions & 1 deletion inferencex-e2e/docs/configuration-procedures.md
Original file line number Diff line number Diff line change
Expand Up @@ -305,7 +305,7 @@ weights determine the recipe's `precision: fp4` label.

The GPU-specific entry points share the text-only serving behavior, `deepseek_v41` tokenizer and
parsers, 1M context, and the shared AgentX trace replay, power, metrics, and eval
helpers. The TP4 concurrency range is 1–128. The shared script sizes graph capture
helpers. Unless noted below, the TP4 concurrency range is 1–128. The shared script sizes graph capture
for the six-token DSpark verification block. The srt-slurm single-node path mounts the checkout at `/infmax-workspace`, so
AgentX runtime directories are not created under `/workspace`. Cluster model paths and persistent caches are reused.
The recipe probes the serving port on the compute node and selects an available
Expand All @@ -319,6 +319,23 @@ at 2046 or 8190 tokens. It sets `--max-num-batched-tokens` to 2048 for concurren
TP2 concurrency-128 variant also sets `--gpu-memory-utilization 0.97`. Other SKUs
continue to use the shared script.

The GB300 entry ([#3575](https://github.com/SemiAnalysisAI/InferenceX/pull/3575)) uses
`vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89` with FlashInfer sparse
MLA attention, an MXFP4 indexer KV cache with sparse logits, an fp8 KV cache, and FlashInfer
autotuning disabled. It runs TP4 at concurrency 1, 4, 8 and 16, TP2 at 2, 4, 8, 16 and 32,
and DEP2 (TP1 x DP2 with expert parallelism) at 16, 32, 48, 64 and 128. All variants use
`FULL_AND_PIECEWISE` CUDA graphs with capture sizes in multiples of the six-token
verification block: concurrency 1–4 captures up to 2046 tokens with
`--max-num-batched-tokens 2048`, other points capture up to 8190 tokens with 8192, and DEP2
concurrency 128 captures up to 2046 tokens while keeping 8192 batched tokens. TP variants
place Engram tables on transparent huge pages (`"use_thp":true`) and set `--max-num-seqs`
to 2 x concurrency, at least 8, because AgentX subagents keep up to 3 x concurrency
requests in flight at concurrency 1. DEP2 uses MegaAttention (`FLASHMLA_MEGA_ATTN_DSV41`),
DeepGEMM MegaMoE, and Engram embeddings sharded across DP ranks; each DP rank admits
`--max-num-seqs` equal to the deployment-wide concurrency, and a consistent-hash vLLM
Router 0.1.14 pins each conversation to one rank. DEP2 does not use huge pages because
Engram DP shared memory, enabled automatically when DP > 1, excludes them.

The GB300 launcher allows 7200 seconds for engine readiness. In [run 34504969146](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34504969146), the Rust frontend exhausted its 3600-second deadline while the engine was still capturing graphs; model loading alone took 18–23 minutes. This extends startup time without changing the benchmark duration or decoding settings.

GPU sweep and eval evidence is required before calling any recipe validated.
Expand Down
16 changes: 15 additions & 1 deletion inferencex-e2e/docs/configuration-procedures_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -281,7 +281,7 @@ B300 在 c1/c2/c4 使用相同的最小捕获范围。其 c1 CI 对比中,请
专家权重为 MXFP4,因此配方标记为 `precision: fp4`。

各 GPU 入口共用纯文本服务行为,使用 `deepseek_v41` tokenizer 和解析器、1M 上下文,
以及共享的 AgentX 轨迹回放、功耗、指标和 eval helper。TP4 的并发范围为 1–128。
以及共享的 AgentX 轨迹回放、功耗、指标和 eval helper。除下文另有说明外,TP4 的并发范围为 1–128。
共享脚本按六 token DSpark 验证块设置 CUDA graph capture。srt-slurm 单节点路径将检出挂载到 `/infmax-workspace`,避免在 `/workspace`
下创建 AgentX 运行目录。沿用集群的模型路径和持久化缓存。配方在计算节点探测服务端口,首选端口被占用时选择可用端口,
服务、回放、指标和 eval 共用同一端点。所有配方都必须获得 GPU sweep 和 eval
Expand All @@ -293,6 +293,20 @@ TP2 并发 128 使用 `--max-num-batched-tokens 2048`,其余情况使用 8192
`--max-num-seqs` 固定为 256。TP2 并发 128 还设置
`--gpu-memory-utilization 0.97`。其他 SKU 继续使用共享脚本。

GB300 条目([#3575](https://github.com/SemiAnalysisAI/InferenceX/pull/3575))使用
`vllm/vllm-openai:nightly-36768d1bfd39094681cdbc8cb37d4b31c0729c89`,采用 FlashInfer
稀疏 MLA attention、带稀疏 logits 的 MXFP4 indexer KV cache、fp8 KV cache,并关闭
FlashInfer autotune。TP4 覆盖并发 1、4、8、16,TP2 覆盖 2、4、8、16、32,DEP2(TP1 x DP2,
开启专家并行)覆盖 16、32、48、64、128。所有变体使用 `FULL_AND_PIECEWISE` CUDA graph,
捕获尺寸均为六 token 验证块的倍数:并发 1–4 最大捕获 2046 tokens,并使用
`--max-num-batched-tokens 2048`;其余并发最大捕获 8190 tokens,使用 8192;DEP2 并发 128
最大捕获 2046 tokens,但保持 8192 batched tokens。TP 变体将 Engram 表放在透明大页上
(`"use_thp":true`),`--max-num-seqs` 设为 2 x 并发且至少为 8,因为在并发 1 时 AgentX
子 agent 最多会同时保持 3 x 并发个请求。DEP2 使用 MegaAttention(`FLASHMLA_MEGA_ATTN_DSV41`)、
DeepGEMM MegaMoE,并将 Engram 嵌入表切分到各 DP rank;每个 DP rank 的 `--max-num-seqs`
等于整个部署的并发,一致性哈希 vLLM Router 0.1.14 将每个会话固定到一个 rank。DP > 1 时
会自动启用 Engram DP 共享内存,它与透明大页不兼容,因此 DEP2 不使用大页。

GB300 launcher 将引擎就绪等待时间设为 7200 秒。在[运行 34504969146](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34504969146) 中,仅模型加载就耗时 18–23 分钟;Rust frontend 达到 3600 秒期限时,引擎仍在捕获 CUDA graph。此次仅延长启动等待时间,基准测试时长和解码设置保持不变。

来源:[上游配方](https://recipes.vllm.ai/deepseek-ai/DeepSeek-V4.1-Flash)。
Expand Down
Loading
Loading