Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
# DeepSeek-V4-Flash 8k1k on MI355X at TP2, following SGLang's MI355X Flash FP4
# low-latency recipe (TP8 upstream) with bundled MTP through EAGLE.
# https://github.com/sgl-project/sglang/blob/8ca82118e0e1a0a1b49f85f675843b19ebb388ba/docs/src/snippets/configs/deepseek-ai/deepseek-v4.jsx
base:
schema: 2
name: dsv4flash-fp4-mi355x-sglang-mtp-8k1k
model:
path: hf:deepseek-ai/DeepSeek-V4-Flash
container: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260926
precision: fp4
resources:
gpu_type: mi355x
gpus_per_node: 8
frontend:
type: sglang
enable_multiple_frontends: false
observability:
enabled: false
tachometer:
enabled: false
engine: sglang
roles:
agg:
nodes: 1
workers: 1
gpus: 2
args:
served-model-name: deepseek-ai/DeepSeek-V4-Flash
trust-remote-code: true
tensor-parallel-size: 2
data-parallel-size: 1
expert-parallel-size: 1
attention-backend: dsv4
page-size: 256
kv-cache-dtype: fp8_e4m3
enforce-shared-experts-fusion: true
disable-radix-cache: true
mem-fraction-static: 0.9
swa-full-tokens-ratio: 0.15
chunked-prefill-size: 16384
context-length: 9236
reasoning-parser: deepseek-v4
tool-call-parser: deepseekv4
speculative-algorithm: EAGLE
speculative-num-steps: 2
speculative-eagle-topk: 1
speculative-num-draft-tokens: 3
# Eager decode and speculative verify/draft; no graph capture.
cuda-graph-backend-decode: disabled
cuda-graph-backend-prefill: disabled
watchdog-timeout: 3600
enable-metrics: true
env:
PYTHONNOUSERSITE: '1'
PYTHONUNBUFFERED: '1'
SGLANG_USE_ROCM700A: '0'
TORCH_BLAS_PREFER_HIPBLASLT: '1'
SGLANG_HACK_FLASHMLA_BACKEND: unified_kv_triton
AITER_BF16_FP8_MOE_BOUND: '0'
# aiter batched GEMM for the absorbed MLA projections.
SGLANG_OPT_USE_AITER_BATCHED_GEMM: 'true'
benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --dsv4
env:
MODEL: deepseek-ai/DeepSeek-V4-Flash
ISL: '8192'
OSL: '1024'
RANDOM_RANGE_RATIO: '0.8'
USE_CHAT_TEMPLATE: 'true'

zip_override_tp2:
roles:
agg:
args:
max-running-requests: [1, 2, 4, 8, 16, 32, 64, 128]
benchmark:
env:
CONC: ['1', '2', '4', '8', '16', '32', '64', '128']
2 changes: 1 addition & 1 deletion benchmarks/single_node/srt_fixed_sequence.sh
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL"
CLIENT_ARGS=()
for argument in "$@"; do
case "$argument" in
--trust-remote-code) CLIENT_ARGS+=("$argument") ;;
--trust-remote-code|--dsv4) CLIENT_ARGS+=("$argument") ;;
*) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;;
esac
done
Expand Down
15 changes: 15 additions & 0 deletions configs/amd-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -1469,3 +1469,18 @@ dsv41flash-fp4-mi355x-sglang-agentic-dspark:
- dram-utilization: 0.60
search-space:
- { tp: 4, ep: 4, dp-attn: false, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/mi355x-fp4-mtp/agentic.yaml }

dsv4flash-fp4-mi355x-sglang-mtp:
image: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260926
model: deepseek-ai/DeepSeek-V4-Flash
model-prefix: dsv4flash
runner: mi355x
precision: fp4
framework: sglang
multinode: false
scenarios:
fixed-seq-len:
- isl: 8192
osl: 1024
search-space:
- { tp: 2, ep: 1, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml }
3 changes: 3 additions & 0 deletions docs/configuration-procedures.md
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,9 @@ router image does not require changing the worker image. TRT-LLM recipes use nat
`engine.served_model_name`, without duplicating that flag in `roles.agg.extra_args`.
The former fork's direct ATOM frontend is not required.

DeepSeek-V4 fixed-sequence recipes pass `--dsv4` to `srt_fixed_sequence.sh` and set
`USE_CHAT_TEMPLATE: 'true'` to use the shared DeepSeek-V4 encoder rather than a missing HF chat template.

### Cluster profiles

Launchers that use srt-slurm keep their cluster configuration in
Expand Down
3 changes: 3 additions & 0 deletions docs/configuration-procedures_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,9 @@ frontend、一个聚合 worker,并设置 `enable_multiple_frontends: false`。
镜像。TRT-LLM 配方使用原生 `engine.served_model_name`,不再通过 `roles.agg.extra_args`
重复传入该参数。不再依赖此前分叉中的 ATOM 直连 frontend。

DeepSeek-V4 固定序列长度配方向 `srt_fixed_sequence.sh` 传入 `--dsv4`,并设置
`USE_CHAT_TEMPLATE: 'true'`,使用共享 DeepSeek-V4 编码器,而不是 checkpoint 中缺失的 HF chat template。

## 规程索引

1. [准备 worktree](#准备-worktree)
Expand Down
79 changes: 79 additions & 0 deletions infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
"""Exercise the fixed-sequence client's DeepSeek-V4 encoder forwarding."""

import os
import subprocess
from pathlib import Path

ROOT = Path(__file__).resolve().parents[3]


def test_dsv4_client_forwards_encoder_and_workload(tmp_path: Path) -> None:
harness = tmp_path / "harness.sh"
harness.write_text(
"""
source() {
if [[ "$1" == */benchmark_lib.sh && "$2" != --validation-only ]]; then
start_gpu_monitor() { :; }
stop_gpu_monitor() { :; }
run_benchmark_serving() { printf '%s\\n' "$@" > "$CAPTURE"; }
else
builtin source "$@"
fi
}
pip3() { :; }
"""
)
capture = tmp_path / "arguments"
result = subprocess.run(
["bash", str(ROOT / "benchmarks/single_node/srt_fixed_sequence.sh"), "--dsv4"],
env={
**os.environ,
"BASH_ENV": str(harness),
"CAPTURE": str(capture),
"INFERENCEX_REPO_ROOT": str(ROOT),
"MODEL": "test/model",
"FRAMEWORK": "sglang",
"CONC": "2",
"ISL": "256",
"OSL": "64",
"RANDOM_RANGE_RATIO": "0.8",
"RESULT_FILENAME": "point",
"RESULT_DIR": str(tmp_path),
"SRT_FRONTEND_HOST": "127.0.0.1",
"SRT_FRONTEND_PORT": "8000",
"RUN_EVAL": "false",
"EVAL_ONLY": "false",
"GPU_MONITOR_INTERVAL": "3",
"USE_CHAT_TEMPLATE": "true",
},
capture_output=True,
text=True,
check=False,
)
assert result.returncode == 0, result.stdout + result.stderr
assert capture.read_text().splitlines() == [
"--model",
"test/model",
"--port",
"8000",
"--base-url",
"http://127.0.0.1:8000",
"--backend",
"vllm",
"--input-len",
"256",
"--output-len",
"64",
"--random-range-ratio",
"0.8",
"--num-prompts",
"20",
"--max-concurrency",
"2",
"--result-filename",
"point",
"--result-dir",
str(tmp_path),
"--dsv4",
"--use-chat-template",
]
9 changes: 9 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8986,3 +8986,12 @@
- "Hopper 使用 FA3,Blackwell FA4 fp8 descale 问题不适用,EAGLE3 draft 保持 attention_backend FLASH_ATTN。"
- "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3503

- config-keys:
- dsv4flash-fp4-mi355x-sglang-mtp
scenario-type:
- fixed-seq-len
description:
- "Add DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 on lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926, with bundled MTP via EAGLE (2 steps, top-k 1, 3 draft tokens) with CUDA/HIP graphs disabled for decode, prefill, and speculative verify/draft, real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder. Flags follow SGLang's MI355X Flash FP4 low-latency recipe at TP2."
- "新增 DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,镜像为 lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926;通过 EAGLE 使用原生 MTP(2 steps、top-k 1、3 draft tokens),decode、prefill 及投机验证/draft 均关闭 CUDA/HIP graph,采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。参数沿用 SGLang MI355X Flash FP4 低延迟配方,改为 TP2。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3518
Loading