Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ base:
name: dsv41flash-fp4-b300-vllm-agentic
model:
path: hf:deepseek-ai/DeepSeek-V4.1-Flash
container: vllm/vllm-openai:deepseekv41-flash-0909
container: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4
precision: fp4
resources:
gpu_type: b300
Expand All @@ -24,7 +24,7 @@ base:
# The engine may take the full VLLM_ENGINE_READY_TIMEOUT_S to become ready.
health_check:
interval_seconds: 10
max_attempts: 360
max_attempts: 720
roles:
agg:
nodes: 1
Expand All @@ -37,6 +37,10 @@ base:
enable-auto-tool-choice: true
reasoning-parser: deepseek_v41
engram-config: '{"cpu_offload":true}'
# FlashInfer sparse attention with the merged Blackwell sparse indexer
# (MXFP4 indexer KV, sparse logits) at both TP sizes; fp8 KV cache.
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV41","indexer_kv_dtype":"mxfp4","indexer_sparse_logits":true}'
kv-cache-dtype: fp8
# Five-token DSpark with probabilistic drafting. Throughput runs replace
# block rejection with the golden acceptance length.
speculative-config: '{"method":"dspark","num_speculative_tokens":5,"draft_sample_method":"probabilistic","rejection_sample_method":"block","enable_adaptive_verification":false}'
Expand All @@ -46,7 +50,7 @@ base:
env:
VLLM_USE_RUST_FRONTEND: '1'
PYTHONUNBUFFERED: '1'
VLLM_ENGINE_READY_TIMEOUT_S: '3600'
VLLM_ENGINE_READY_TIMEOUT_S: '7200'
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/srt_agentic.sh
Expand Down
2 changes: 1 addition & 1 deletion inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8324,7 +8324,7 @@ dsv41flash-fp4-gb200-sglang-agentic-dspark:
- { tp: 2, ep: 1, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb200-fp4-mtp/agentic.yaml }

dsv41flash-fp4-b300-vllm-agentic-dspark:
image: vllm/vllm-openai:deepseekv41-flash-0909
image: vllm/vllm-openai:nightly-ddd6fbca148a867aad1fcab7ec72f582b9977db4
model: deepseek-ai/DeepSeek-V4.1-Flash
model-prefix: dsv41flash
runner: cluster:b300-dsxe
Expand Down
6 changes: 6 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9012,3 +9012,9 @@
- "Update the GLM-5.2 MXFP4 MI355X SGLang AgentX image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260923 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260924 (digest sha256:baadaba198e23c46c1dd651b5edefae8000bd5430943bd10b246b563905b3ecf); keep all serving flags, HiCache settings, and sweep points unchanged."
- "将 GLM-5.2 MXFP4 MI355X SGLang AgentX 镜像从 lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260923 更新到 lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260924(digest sha256:baadaba198e23c46c1dd651b5edefae8000bd5430943bd10b246b563905b3ecf);其余服务参数、HiCache 设置与 sweep 点保持不变。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3446

- config-keys:
- dsv41flash-fp4-b300-vllm-agentic-dspark
description:
- "Pin B300 DeepSeek-V4.1-Flash vLLM to nightly ddd6fbca with FlashInfer sparse attention, MXFP4 indexer KV, sparse logits, fp8 KV cache, and a 7200 s readiness timeout."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3458
Loading