Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ base:
name: dsv41flash-fp4-gb200-vllm-agentic
model:
path: hf:deepseek-ai/DeepSeek-V4.1-Flash
container: vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3
container: vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7
precision: fp4
resources:
gpu_type: gb200
Expand Down Expand Up @@ -37,215 +37,109 @@ base:
enable-auto-tool-choice: true
reasoning-parser: deepseek_v41
engram-config: '{"cpu_offload":true}'
kernel-config: '{"enable_flashinfer_autotune":true}'
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}'
kv-cache-dtype: fp8
# Five-token DSpark with probabilistic drafting. Throughput runs replace
# block rejection with the golden acceptance length.
speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}'
max-model-len: 1048576
# Piecewise graph capture sizes are multiples of the six-token DSpark
# verification block, denser for small batches.
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
max-cudagraph-capture-size: 8190
max-num-batched-tokens: 8192
gpu-memory-utilization: 0.97
disable-uvicorn-access-log: true
env:
VLLM_USE_V2_MODEL_RUNNER: '1'
VLLM_USE_RUST_FRONTEND: '1'
PYTHONUNBUFFERED: '1'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
VLLM_ENGINE_READY_TIMEOUT_S: '7200'
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
env:
MODEL: deepseek-ai/DeepSeek-V4.1-Flash
AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:'

# One variant per point. Graph capture starts at 64 tokens and doubles until it
# covers CONC x (1 + 5 drafts), up to 2048. TP2 leaves ~145 GiB of weights on
# each 256 GiB GPU, so it caps batched tokens at 4096 (the indexer's 1M-wide
# logits buffer), bounds the scheduler at 2x CONC within [16, 256] (FlashInfer
# autotune fails below 16) and stops capturing above 512 tokens.
override_tp4_c1:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
benchmark:
env:
CONC: '1'

override_tp4_c2:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
benchmark:
env:
CONC: '2'

override_tp4_c4:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
benchmark:
env:
CONC: '4'

override_tp4_c8:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 64
benchmark:
env:
CONC: '8'

override_tp4_c16:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 128
benchmark:
env:
CONC: '16'

override_tp4_c32:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 256
benchmark:
env:
CONC: '32'

override_tp4_c64:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 512
benchmark:
env:
CONC: '64'
# Native zip expansion creates one variant per concurrency.

override_tp4_c128:
# TP4 covers concurrency 1-128 with FlashInfer attention and THP Engram tables.
zip_override_tp4:
roles:
agg:
gpus: 4
args:
tensor-parallel-size: 4
max-cudagraph-capture-size: 1024
benchmark:
env:
CONC: '128'

override_tp2_c1:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '1'

override_tp2_c2:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '2'

override_tp2_c4:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '4'

override_tp2_c8:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 64
max-num-batched-tokens: 4096
max-num-seqs: 16
benchmark:
env:
CONC: '8'

override_tp2_c16:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 128
max-num-batched-tokens: 4096
max-num-seqs: 32
benchmark:
env:
CONC: '16'

override_tp2_c32:
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 256
max-num-batched-tokens: 4096
max-num-seqs: 64
benchmark:
env:
CONC: '32'

override_tp2_c64:
engram-config: '{"cpu_offload":true,"use_thp":true}'
max-num-seqs: [8, 8, 16, 32, 64, 128, 256]
compilation-config:
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}'
- '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}'
max-cudagraph-capture-size: [4092, 4092, 8190, 8190, 8190, 8190, 8190]
max-num-batched-tokens: [4096, 4096, 8192, 8192, 8192, 8192, 8192]
benchmark:
env:
CONC: ['1', '4', '8', '16', '32', '64', '128']

# DEP2: TP1 x DP2 + EP2 with MegaMoE behind a consistent-hash router. Each
# GB200 rank holds ~150 GiB of weights, so batched tokens stay at 4096 and
# graph capture stops at 576 tokens.
zip_override_dep2:
frontend:
type: vllm-router
args:
policy: consistent_hash
prometheus-host: 127.0.0.1
prometheus-port: 18000
request-timeout-secs: 14400
disable-retries: true
setup_script: vllm-router-0.1.14.sh
roles:
agg:
gpus: 2
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 512
tensor-parallel-size: 1
data-parallel-size: 2
enable-expert-parallel: true
kernel-config: '{"enable_flashinfer_autotune":true,"moe_backend":"deep_gemm_mega_moe"}'
engram-config: '{"cpu_offload":true,"embedding_across_dp":true}'
max-num-seqs: [16, 16, 32]
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576]}'
max-cudagraph-capture-size: 576
max-num-batched-tokens: 4096
max-num-seqs: 128
benchmark:
env:
CONC: '64'
CONC: ['8', '16', '32']

override_tp2_c128:
# DEP4: TP1 x DP4 + EP4 with MegaMoE behind a consistent-hash router.
zip_override_dep4:
frontend:
type: vllm-router
args:
policy: consistent_hash
prometheus-host: 127.0.0.1
prometheus-port: 18000
request-timeout-secs: 14400
disable-retries: true
setup_script: vllm-router-0.1.14.sh
roles:
agg:
gpus: 2
gpus: 4
args:
tensor-parallel-size: 2
max-cudagraph-capture-size: 512
max-num-batched-tokens: 4096
max-num-seqs: 256
tensor-parallel-size: 1
data-parallel-size: 4
enable-expert-parallel: true
kernel-config: '{"enable_flashinfer_autotune":true,"moe_backend":"deep_gemm_mega_moe"}'
engram-config: '{"cpu_offload":true,"embedding_across_dp":true}'
max-num-seqs: [64, 128]
compilation-config: '{"mode":"VLLM_COMPILE","cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2046,3072,4092,6144,8190],"decoder_replay_cudagraph_capture_sizes":[6,12,18,24,30,36,48,60,72,96,120,144,192,240,288,384,480,576,768,1020,1536,2048,2304,2560,2816,3072,3328,3584,3840,4096]}'
benchmark:
env:
CONC: '128'
CONC: ['64', '128']
15 changes: 7 additions & 8 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8429,7 +8429,7 @@ dsv41flash-fp4-h100-sglang-agentic-dspark-dpa:
- { tp: 8, ep: 8, dp-attn: true, kv-offloading: none, spec-decoding: mtp, conc-list: [4, 8, 16, 20], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/h100-fp4-mtp/agentic.yaml }

dsv41flash-fp4-gb200-vllm-agentic-dspark:
image: vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3
image: vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7
model: deepseek-ai/DeepSeek-V4.1-Flash
model-prefix: dsv41flash
runner: cluster:gb200-nv
Expand All @@ -8441,13 +8441,12 @@ dsv41flash-fp4-gb200-vllm-agentic-dspark:
- dram-utilization: 0.80
search-space:
# Engram weights use UVA DRAM; the KV cache stays GPU-resident.
- { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml }
# TP2 halves the GPU count per replica. Weights rise to ~145 GiB on each
# 256 GiB GPU with the Engram tables in pinned host DRAM, so the arm
# takes the same caps as the B200 TP2 arm: batched tokens 4096 (the
# indexer's logits buffer is 32 GiB at the upstream 16384) and graph
# capture stopped at 512, leaving ~49 GiB of KV per GPU.
- { tp: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml }
# TP4 covers concurrency 1-128 with FlashInfer attention and THP Engram tables.
- { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml }
# DEP2 is native TP1 x DP2 + EP2 with MegaMoE behind a consistent-hash vLLM Router.
- { tp: 2, ep: 2, dp-attn: true, kv-offloading: none, spec-decoding: mtp, conc-list: [8, 16, 32], router: { name: vllm-router, version: "0.1.14" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml }
# DEP4 is native TP1 x DP4 + EP4 with MegaMoE behind the same router.
- { tp: 4, ep: 4, dp-attn: true, kv-offloading: none, spec-decoding: mtp, conc-list: [64, 128], router: { name: vllm-router, version: "0.1.14" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/gb200-fp4-mtp/agentic.yaml }
Comment on lines +8445 to +8449

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 (optional) DEP4 (tp:4, ep:4, dp-attn:true, conc-list [64,128]) at nvidia-master.yaml:8449 shares tp and concurrency with the TP4 arm at :8445, so both get the identical exp-name dsv41flash_tp4_conc64_kvnone_spec-mtp for two different deployments. generate.py's _agentic_entries exp-name formula (infx/matrix/generate.py:1188-1192) encodes model, tp, conc, kv-offload and spec-decoding but never ep/dp-attn. This breaks the uniqueness convention documented at nvidia-master.yaml:1544 and makes filter_exp_names raise 'matched multiple rows' for that name. Fix: fold ep/dp-attn into exp-name for any arm that can share tp+conc with another.

Why this was flagged

Trigger: generating the agentic matrix for dsv41flash-fp4-gb200-vllm-agentic-dspark (full-sweep, generate_config_matrix, or test-config --exp-names) via infx/matrix/generate.py's _agentic_entries. The TP4 row at nvidia-master.yaml:8445 (tp:4, conc-list [1,4,8,16,32,64,128]) and the new DEP4 row at :8449 (tp:4, ep:4, dp-attn:true, conc-list [64,128]) both produce exp-name dsv41flash_tp4_conc64_kvnone_spec-mtp (and the conc128 variant) because generate.py:1188-1192 omits ep and dp-attn. On base no two rows for this key share tp+conc: GB300's DEP2 sibling uses tp:2 and the sglang EP arms use a different tp than their TP arm, so exp-names were unique; this is the first collision. filter_exp_names then raises 'Experiment name(s) matched multiple rows' for that name, blocking targeted selection, and no test in infx/tests/matrix checks exp-name uniqueness across search-space rows, so the claimed 528 passing tests do not catch it.

Verification: nit. The exp-name collision is real and reachable, but its serious consequences are prevented downstream, so this is a convention/tooling issue, not a correctness regression. generate.py:1188-1192 builds the exp-name from tp, conc, kv-offload and spec-decoding, omitting ep and dp-attn, so TP4 (:8445) and DEP4 (:8449) both emit dsv41flash_tp4_conc64_kvnone_spec-mtp, violating the uniqueness convention at :1544 and making filter_exp_names raise "matched multiple rows".


# SGLang arm for DeepSeek-V4.1-Flash AgentX on GB200, from the SGLang cookbook
# (https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4_1).
Expand Down
10 changes: 8 additions & 2 deletions inferencex-e2e/docs/configuration-procedures.md
Original file line number Diff line number Diff line change
Expand Up @@ -326,7 +326,13 @@ EAGLE3 K3, golden AL 2.78 and indexer CP are unchanged.

### DeepSeek-V4.1-Flash DSpark

The GB200 DSpark recipe uses a minimum CUDA graph capture size of 64 tokens to cover concurrent AgentX subagents. This raises c1/c2/c4 from 8/16/32 to 64; c8 and above retain their existing sizes. The full trace, AL 3.51, and Engram UVA settings are preserved; low-concurrency tail latency improvements require CI confirmation.
The GB200 DSpark recipe sets explicit capture sizes per point. It uses
`vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7` with FlashInfer autotuning. TP4 covers
concurrency 1–128. DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) covers 8–32 and DEP4 (TP1 x DP4 + EP4,
MegaMoE) covers 64–128, both behind a consistent-hash vLLM Router. DEP2 keeps about 150 GiB of weights
per GB200 rank, so it caps batched tokens at 4096 and graph capture at 576 tokens. All GB200 points set
`--gpu-memory-utilization 0.97` and use `FULL_AND_PIECEWISE` CUDA graphs sized in multiples of the
six-token verification block.
The B200 DSpark recipe uses the same minimum capture size and preserves the same workload settings.
The GB300 DSpark recipe sets explicit capture sizes per point, described below.
The H200 DSpark recipe uses the same minimum capture size and preserves the same workload settings.
Expand Down Expand Up @@ -369,7 +375,7 @@ Source: [upstream recipe](https://recipes.vllm.ai/deepseek-ai/DeepSeek-V4.1-Flas
### DeepSeek-V4.1-Flash DSpark on H200

`dsv41flash-fp4-h200-vllm-agentic-dspark` is the H200 AgentX arm of the
DeepSeek-V4.1-Flash recipe. It pins `vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3` (shared with B200 and GB200) and shares the
DeepSeek-V4.1-Flash recipe. It pins `vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3` and shares the
text-only serving settings with the Blackwell arms: `deepseek_v41` tokenizer and parsers,
1M context, native five-token DSpark with probabilistic drafting. Throughput uses the [committed golden AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection and adaptive verification.

Expand Down
8 changes: 6 additions & 2 deletions inferencex-e2e/docs/configuration-procedures_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -300,7 +300,11 @@ TP4 约 1.23 TB)全部来自节点 0 的 1.5 TB 内存。`runners/srt-slurm/ho

### DeepSeek-V4.1-Flash DSpark

GB200 的 DSpark 配方将 CUDA graph 最小捕获范围设为 64 tokens,以覆盖 AgentX 子代理并发。这会将 c1/c2/c4 的上限从 8/16/32 提升至 64;c8 及以上保持原有大小。完整轨迹、AL 3.51 和 Engram UVA 配置保持不变;需通过 CI 验证低并发尾延迟改善。
GB200 的 DSpark 配方按测试点显式设置捕获尺寸,使用 `vllm/vllm-openai:nightly-dev-arm64-cu130-ac9126e58aa7`,
开启 FlashInfer autotune。TP4 覆盖并发 1–128;DEP2(TP1 x DP2 + EP2,DeepGEMM MegaMoE)覆盖 8–32,
DEP4(TP1 x DP4 + EP4,DeepGEMM MegaMoE)覆盖 64–128,两者均前置一致性哈希 vLLM Router。DEP2 每个 GB200 rank
约有 150 GiB 权重,因此 batched tokens 上限为 4096,CUDA graph 捕获上限为 576 tokens。所有 GB200 测试点设置
`--gpu-memory-utilization 0.97`,使用 `FULL_AND_PIECEWISE` CUDA graph,捕获尺寸为六 token 验证块的倍数。
B200 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工作负载配置。
GB300 的 DSpark 配方按测试点显式设置捕获尺寸,详见下文。
H200 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工作负载配置。
Expand Down Expand Up @@ -338,7 +342,7 @@ GB300 launcher 将引擎就绪等待时间设为 7200 秒。在[运行 345049691
### H200 上的 DeepSeek-V4.1-Flash DSpark

`dsv41flash-fp4-h200-vllm-agentic-dspark` 是 DeepSeek-V4.1-Flash 配方的 H200 AgentX
分支。它固定使用 `vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3`(与 B200、GB200 相同),并与 Blackwell 分支共用纯文本服务
分支。它固定使用 `vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3`,并与 Blackwell 分支共用纯文本服务
设置:`deepseek_v41` tokenizer 和解析器、1M 上下文、原生五 token DSpark(概率采样草稿)。吞吐测试使用[已提交的黄金 AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样和自适应验证。

该分支使用 **TP8**,而非上游的 TP4。上游在一个 GB200 NVL4 tray 上验证 TP4,并说明在
Expand Down
9 changes: 9 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9275,3 +9275,12 @@
- "Update the GB300 DeepSeek-V4.1-Flash vLLM AgentX image from nightly ac68c308 to nightly-dev-arm64-cu130-ac9126e58aa7 and enable FlashInfer autotuning."
- "Run TP4 at concurrency 1-16 and replace TP2 with DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) at concurrency 8-192 behind a consistent-hash vLLM Router."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3652

- config-keys:
- dsv41flash-fp4-gb200-vllm-agentic-dspark
scenario-type:
- agentic-coding
description:
- "Update the GB200 DeepSeek-V4.1-Flash vLLM AgentX image from nightly cd10ed6f to nightly-dev-arm64-cu130-ac9126e58aa7 and enable FlashInfer autotuning."
- "Run TP4 at concurrency 1-128 and replace TP2 with DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) at concurrency 8-32 and DEP4 (TP1 x DP4 + EP4, MegaMoE) at concurrency 64-128, both behind a consistent-hash vLLM Router. All points set gpu-memory-utilization 0.97."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3687
Loading