diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..ed7f590936 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml @@ -0,0 +1,69 @@ +base: + schema: 2 + name: dsv4flash-fp4-b300-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-V4-Flash + container: lmsysorg/sglang:nightly-dev-cu13-20260925-8ca82118@sha256:baea7ec5ea86412ce59837f17011028b940f9777fa15129254de02bd6a842559 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Flash + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_mxfp4 + disable-flashinfer-autotune: true + disable-radix-cache: true + mem-fraction-static: 0.85 + swa-full-tokens-ratio: 0.1 + chunked-prefill-size: 8192 + context-length: 9236 + reasoning-parser: deepseek-v4 + tool-call-parser: deepseekv4 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 3 + # Eager decode and speculative verify/draft; no CUDA graph capture. + cuda-graph-backend-decode: disabled + cuda-graph-backend-prefill: disabled + watchdog-timeout: 3600 + enable-metrics: true + env: + PYTHONNOUSERSITE: '1' + PYTHONUNBUFFERED: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --dsv4 + env: + MODEL: deepseek-ai/DeepSeek-V4-Flash + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' + +zip_override_tp2: + roles: + agg: + args: + max-running-requests: [1, 2, 4, 8, 16, 32, 64, 128] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index 0bc79b9ab4..0a4c511444 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -20,7 +20,7 @@ SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" CLIENT_ARGS=() for argument in "$@"; do case "$argument" in - --trust-remote-code) CLIENT_ARGS+=("$argument") ;; + --trust-remote-code|--dsv4) CLIENT_ARGS+=("$argument") ;; *) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;; esac done diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index b0fc9ef451..6dc1079d88 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8928,3 +8928,18 @@ dsv41flash-fp4-gb300-sglang-agentic-dspark: search-space: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } + +dsv4flash-fp4-b300-sglang-mtp: + image: lmsysorg/sglang:nightly-dev-cu13-20260925-8ca82118@sha256:baea7ec5ea86412ce59837f17011028b940f9777fa15129254de02bd6a842559 + model: deepseek-ai/DeepSeek-V4-Flash + model-prefix: dsv4flash + runner: cluster:b300-dsxe + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 2, ep: 1, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml } diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index d480a4bfd7..cae948c781 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -44,6 +44,9 @@ router image does not require changing the worker image. TRT-LLM recipes use nat `engine.served_model_name`, without duplicating that flag in `roles.agg.extra_args`. The former fork's direct ATOM frontend is not required. +DeepSeek-V4 fixed-sequence recipes pass `--dsv4` to `srt_fixed_sequence.sh` and set +`USE_CHAT_TEMPLATE: 'true'` to use the shared DeepSeek-V4 encoder rather than a missing HF chat template. + ### Cluster profiles Launchers that use srt-slurm keep their cluster configuration in diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 47812b4a67..32408f27fd 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -42,6 +42,9 @@ frontend、一个聚合 worker,并设置 `enable_multiple_frontends: false`。 镜像。TRT-LLM 配方使用原生 `engine.served_model_name`,不再通过 `roles.agg.extra_args` 重复传入该参数。不再依赖此前分叉中的 ATOM 直连 frontend。 +DeepSeek-V4 固定序列长度配方向 `srt_fixed_sequence.sh` 传入 `--dsv4`,并设置 +`USE_CHAT_TEMPLATE: 'true'`,使用共享 DeepSeek-V4 编码器,而不是 checkpoint 中缺失的 HF chat template。 + ## 规程索引 1. [准备 worktree](#准备-worktree) diff --git a/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py b/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py new file mode 100644 index 0000000000..bacb33261e --- /dev/null +++ b/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py @@ -0,0 +1,79 @@ +"""Exercise the fixed-sequence client's DeepSeek-V4 encoder forwarding.""" + +import os +import subprocess +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[3] + + +def test_dsv4_client_forwards_encoder_and_workload(tmp_path: Path) -> None: + harness = tmp_path / "harness.sh" + harness.write_text( + """ +source() { + if [[ "$1" == */benchmark_lib.sh && "$2" != --validation-only ]]; then + start_gpu_monitor() { :; } + stop_gpu_monitor() { :; } + run_benchmark_serving() { printf '%s\\n' "$@" > "$CAPTURE"; } + else + builtin source "$@" + fi +} +pip3() { :; } +""" + ) + capture = tmp_path / "arguments" + result = subprocess.run( + ["bash", str(ROOT / "benchmarks/single_node/srt_fixed_sequence.sh"), "--dsv4"], + env={ + **os.environ, + "BASH_ENV": str(harness), + "CAPTURE": str(capture), + "INFERENCEX_REPO_ROOT": str(ROOT), + "MODEL": "test/model", + "FRAMEWORK": "sglang", + "CONC": "2", + "ISL": "256", + "OSL": "64", + "RANDOM_RANGE_RATIO": "0.8", + "RESULT_FILENAME": "point", + "RESULT_DIR": str(tmp_path), + "SRT_FRONTEND_HOST": "127.0.0.1", + "SRT_FRONTEND_PORT": "8000", + "RUN_EVAL": "false", + "EVAL_ONLY": "false", + "GPU_MONITOR_INTERVAL": "3", + "USE_CHAT_TEMPLATE": "true", + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert capture.read_text().splitlines() == [ + "--model", + "test/model", + "--port", + "8000", + "--base-url", + "http://127.0.0.1:8000", + "--backend", + "vllm", + "--input-len", + "256", + "--output-len", + "64", + "--random-range-ratio", + "0.8", + "--num-prompts", + "20", + "--max-concurrency", + "2", + "--result-filename", + "point", + "--result-dir", + str(tmp_path), + "--dsv4", + "--use-chat-template", + ] diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 352c61fc0d..2f6218f838 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8986,3 +8986,12 @@ - "Hopper 使用 FA3,Blackwell FA4 fp8 descale 问题不适用,EAGLE3 draft 保持 attention_backend FLASH_ATTN。" - "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3503 + +- config-keys: + - dsv4flash-fp4-b300-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Add DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled MTP via EAGLE (2 steps, top-k 1, 3 draft tokens) with CUDA/HIP graphs disabled for decode, prefill, and speculative verify/draft, real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder. Weights load from the node-local /scratch/models copy; mmap page faults on the shared /data copy read at ~10 MB/s and outlast the 1800 s health timeout." + - "新增 DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128;通过 EAGLE 使用原生 MTP(2 steps、top-k 1、3 draft tokens),decode、prefill 及投机验证/draft 均关闭 CUDA/HIP graph,采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。权重从节点本地 /scratch/models 副本加载;共享 /data 副本的 mmap 缺页读取仅约 10 MB/s,会超过 1800 s 健康检查超时。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index efd0506131..6769a3c1aa 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -61,6 +61,7 @@ WRITABLE_MODELS_DIR="/data/home/sa-gha-runner/models" STAGED_MODELS=( DeepSeek-R1-0528 DeepSeek-R1-0528-NVFP4-v2 + DeepSeek-V4-Flash DeepSeek-V4-Pro DeepSeek-V4-Pro-0813 DeepSeek-V4-Pro-NVFP4