From d3eebca2d2b694efd52a786a97f4d82a6730f5fd Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sun, 27 Sep 2026 22:33:40 +0800 Subject: [PATCH 1/3] feat: add MI355X TP2 V4 Flash SGLang 8k1k MTP sweep Mirror the B300/B200 DeepSeek-V4-Flash fixed-sequence recipes on MI355X with SGLang's MI355X Flash FP4 low-latency flags at TP2 and EAGLE 2/1/3. --- .../dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml | 77 ++++++++++++++++++ benchmarks/single_node/srt_fixed_sequence.sh | 2 +- configs/amd-master.yaml | 15 ++++ docs/configuration-procedures.md | 3 + docs/configuration-procedures_zh.md | 3 + .../test_dsv4_fixed_sequence_client.py | 79 +++++++++++++++++++ perf-changelog.yaml | 9 +++ 7 files changed, 187 insertions(+), 1 deletion(-) create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml create mode 100644 infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..b3ef319876 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml @@ -0,0 +1,77 @@ +# DeepSeek-V4-Flash 8k1k on MI355X at TP2, following SGLang's MI355X Flash FP4 +# low-latency recipe (TP8 upstream) with bundled MTP through EAGLE. +# https://github.com/sgl-project/sglang/blob/8ca82118e0e1a0a1b49f85f675843b19ebb388ba/docs/src/snippets/configs/deepseek-ai/deepseek-v4.jsx +base: + schema: 2 + name: dsv4flash-fp4-mi355x-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-V4-Flash + container: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260926 + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Flash + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + attention-backend: dsv4 + page-size: 256 + kv-cache-dtype: fp8_e4m3 + enforce-shared-experts-fusion: true + disable-radix-cache: true + mem-fraction-static: 0.9 + swa-full-tokens-ratio: 0.15 + chunked-prefill-size: 16384 + context-length: 9236 + reasoning-parser: deepseek-v4 + tool-call-parser: deepseekv4 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + watchdog-timeout: 3600 + enable-metrics: true + env: + PYTHONNOUSERSITE: '1' + PYTHONUNBUFFERED: '1' + SGLANG_USE_ROCM700A: '0' + TORCH_BLAS_PREFER_HIPBLASLT: '1' + SGLANG_HACK_FLASHMLA_BACKEND: unified_kv_triton + AITER_BF16_FP8_MOE_BOUND: '0' + # aiter batched GEMM for the absorbed MLA projections. + SGLANG_OPT_USE_AITER_BATCHED_GEMM: 'true' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --dsv4 + env: + MODEL: deepseek-ai/DeepSeek-V4-Flash + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' + +zip_override_tp2: + roles: + agg: + args: + max-running-requests: [1, 2, 4, 8, 16, 32, 64, 128] + cuda-graph-max-bs-decode: [1, 2, 4, 8, 16, 32, 64, 128] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index 0bc79b9ab4..0a4c511444 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -20,7 +20,7 @@ SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" CLIENT_ARGS=() for argument in "$@"; do case "$argument" in - --trust-remote-code) CLIENT_ARGS+=("$argument") ;; + --trust-remote-code|--dsv4) CLIENT_ARGS+=("$argument") ;; *) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;; esac done diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index 74c709949e..c2e08aa93e 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -1469,3 +1469,18 @@ dsv41flash-fp4-mi355x-sglang-agentic-dspark: - dram-utilization: 0.60 search-space: - { tp: 4, ep: 4, dp-attn: false, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/mi355x-fp4-mtp/agentic.yaml } + +dsv4flash-fp4-mi355x-sglang-mtp: + image: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260926 + model: deepseek-ai/DeepSeek-V4-Flash + model-prefix: dsv4flash + runner: mi355x + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 2, ep: 1, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml } diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index d480a4bfd7..cae948c781 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -44,6 +44,9 @@ router image does not require changing the worker image. TRT-LLM recipes use nat `engine.served_model_name`, without duplicating that flag in `roles.agg.extra_args`. The former fork's direct ATOM frontend is not required. +DeepSeek-V4 fixed-sequence recipes pass `--dsv4` to `srt_fixed_sequence.sh` and set +`USE_CHAT_TEMPLATE: 'true'` to use the shared DeepSeek-V4 encoder rather than a missing HF chat template. + ### Cluster profiles Launchers that use srt-slurm keep their cluster configuration in diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 47812b4a67..32408f27fd 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -42,6 +42,9 @@ frontend、一个聚合 worker,并设置 `enable_multiple_frontends: false`。 镜像。TRT-LLM 配方使用原生 `engine.served_model_name`,不再通过 `roles.agg.extra_args` 重复传入该参数。不再依赖此前分叉中的 ATOM 直连 frontend。 +DeepSeek-V4 固定序列长度配方向 `srt_fixed_sequence.sh` 传入 `--dsv4`,并设置 +`USE_CHAT_TEMPLATE: 'true'`,使用共享 DeepSeek-V4 编码器,而不是 checkpoint 中缺失的 HF chat template。 + ## 规程索引 1. [准备 worktree](#准备-worktree) diff --git a/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py b/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py new file mode 100644 index 0000000000..bacb33261e --- /dev/null +++ b/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py @@ -0,0 +1,79 @@ +"""Exercise the fixed-sequence client's DeepSeek-V4 encoder forwarding.""" + +import os +import subprocess +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[3] + + +def test_dsv4_client_forwards_encoder_and_workload(tmp_path: Path) -> None: + harness = tmp_path / "harness.sh" + harness.write_text( + """ +source() { + if [[ "$1" == */benchmark_lib.sh && "$2" != --validation-only ]]; then + start_gpu_monitor() { :; } + stop_gpu_monitor() { :; } + run_benchmark_serving() { printf '%s\\n' "$@" > "$CAPTURE"; } + else + builtin source "$@" + fi +} +pip3() { :; } +""" + ) + capture = tmp_path / "arguments" + result = subprocess.run( + ["bash", str(ROOT / "benchmarks/single_node/srt_fixed_sequence.sh"), "--dsv4"], + env={ + **os.environ, + "BASH_ENV": str(harness), + "CAPTURE": str(capture), + "INFERENCEX_REPO_ROOT": str(ROOT), + "MODEL": "test/model", + "FRAMEWORK": "sglang", + "CONC": "2", + "ISL": "256", + "OSL": "64", + "RANDOM_RANGE_RATIO": "0.8", + "RESULT_FILENAME": "point", + "RESULT_DIR": str(tmp_path), + "SRT_FRONTEND_HOST": "127.0.0.1", + "SRT_FRONTEND_PORT": "8000", + "RUN_EVAL": "false", + "EVAL_ONLY": "false", + "GPU_MONITOR_INTERVAL": "3", + "USE_CHAT_TEMPLATE": "true", + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert capture.read_text().splitlines() == [ + "--model", + "test/model", + "--port", + "8000", + "--base-url", + "http://127.0.0.1:8000", + "--backend", + "vllm", + "--input-len", + "256", + "--output-len", + "64", + "--random-range-ratio", + "0.8", + "--num-prompts", + "20", + "--max-concurrency", + "2", + "--result-filename", + "point", + "--result-dir", + str(tmp_path), + "--dsv4", + "--use-chat-template", + ] diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 352c61fc0d..3693b72bbf 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8986,3 +8986,12 @@ - "Hopper 使用 FA3,Blackwell FA4 fp8 descale 问题不适用,EAGLE3 draft 保持 attention_backend FLASH_ATTN。" - "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3503 + +- config-keys: + - dsv4flash-fp4-mi355x-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Add DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 on lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926, with bundled MTP via EAGLE (2 steps, top-k 1, 3 draft tokens), real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder. Flags follow SGLang's MI355X Flash FP4 low-latency recipe at TP2." + - "新增 DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,镜像为 lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926;通过 EAGLE 使用原生 MTP(2 steps、top-k 1、3 draft tokens),采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。参数沿用 SGLang MI355X Flash FP4 低延迟配方,改为 TP2。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/TBD From 533540b8440fb8d42a5306493d21948989d001c1 Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sun, 27 Sep 2026 22:33:49 +0800 Subject: [PATCH 2/3] chore: link perf-changelog entry to #3518 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 3693b72bbf..77614fa8f5 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8994,4 +8994,4 @@ description: - "Add DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 on lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926, with bundled MTP via EAGLE (2 steps, top-k 1, 3 draft tokens), real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder. Flags follow SGLang's MI355X Flash FP4 low-latency recipe at TP2." - "新增 DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,镜像为 lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926;通过 EAGLE 使用原生 MTP(2 steps、top-k 1、3 draft tokens),采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。参数沿用 SGLang MI355X Flash FP4 低延迟配方,改为 TP2。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3518 From 455ddc40737167ee39d1ade01da1c3e473f52808 Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sun, 27 Sep 2026 23:10:05 +0800 Subject: [PATCH 3/3] perf: run V4 Flash decode and speculative steps without graphs Set cuda-graph-backend-{decode,prefill}=disabled, which also skips target verify and draft capture, and drop the per-concurrency decode graph batch. --- .../dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml | 4 +++- perf-changelog.yaml | 4 ++-- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml index b3ef319876..732ab46675 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/mi355x-fp4-mtp/8k1k.yaml @@ -45,6 +45,9 @@ base: speculative-num-steps: 2 speculative-eagle-topk: 1 speculative-num-draft-tokens: 3 + # Eager decode and speculative verify/draft; no graph capture. + cuda-graph-backend-decode: disabled + cuda-graph-backend-prefill: disabled watchdog-timeout: 3600 enable-metrics: true env: @@ -71,7 +74,6 @@ zip_override_tp2: agg: args: max-running-requests: [1, 2, 4, 8, 16, 32, 64, 128] - cuda-graph-max-bs-decode: [1, 2, 4, 8, 16, 32, 64, 128] benchmark: env: CONC: ['1', '2', '4', '8', '16', '32', '64', '128'] diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 77614fa8f5..6d298c9f43 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8992,6 +8992,6 @@ scenario-type: - fixed-seq-len description: - - "Add DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 on lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926, with bundled MTP via EAGLE (2 steps, top-k 1, 3 draft tokens), real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder. Flags follow SGLang's MI355X Flash FP4 low-latency recipe at TP2." - - "新增 DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,镜像为 lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926;通过 EAGLE 使用原生 MTP(2 steps、top-k 1、3 draft tokens),采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。参数沿用 SGLang MI355X Flash FP4 低延迟配方,改为 TP2。" + - "Add DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 on lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926, with bundled MTP via EAGLE (2 steps, top-k 1, 3 draft tokens) with CUDA/HIP graphs disabled for decode, prefill, and speculative verify/draft, real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder. Flags follow SGLang's MI355X Flash FP4 low-latency recipe at TP2." + - "新增 DeepSeek-V4-Flash MI355X TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,镜像为 lmsysorg/sglang-rocm v0.5.20-rocm720-mi35x-20260926;通过 EAGLE 使用原生 MTP(2 steps、top-k 1、3 draft tokens),decode、prefill 及投机验证/draft 均关闭 CUDA/HIP graph,采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。参数沿用 SGLang MI355X Flash FP4 低延迟配方,改为 TP2。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3518