From 4bde51e8d4103551c30dd099466c56b4e74a3aae Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 06:43:45 +0000 Subject: [PATCH 1/7] feat: add B300 TP2 DeepSeek V4 Flash serving sweep MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 新增 B300 TP2 DeepSeek V4 Flash SGLang 8k1k serving 扫描,使用原生 EAGLE 三步 MTP。 --- .../dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml | 68 ++++++++++++++++ benchmarks/single_node/srt_fixed_sequence.sh | 2 +- configs/nvidia-master.yaml | 15 ++++ docs/configuration-procedures.md | 3 + docs/configuration-procedures_zh.md | 3 + .../test_dsv4_fixed_sequence_client.py | 79 +++++++++++++++++++ perf-changelog.yaml | 9 +++ 7 files changed, 178 insertions(+), 1 deletion(-) create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml create mode 100644 infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..01d6a0c111 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml @@ -0,0 +1,68 @@ +base: + schema: 2 + name: dsv4flash-fp4-b300-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-V4-Flash + container: lmsysorg/sglang:nightly-dev-cu13-20260925-8ca82118@sha256:baea7ec5ea86412ce59837f17011028b940f9777fa15129254de02bd6a842559 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Flash + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_mxfp4 + disable-flashinfer-autotune: true + disable-radix-cache: true + mem-fraction-static: 0.85 + swa-full-tokens-ratio: 0.1 + chunked-prefill-size: 8192 + context-length: 9236 + reasoning-parser: deepseek-v4 + tool-call-parser: deepseekv4 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + watchdog-timeout: 3600 + enable-metrics: true + env: + PYTHONNOUSERSITE: '1' + PYTHONUNBUFFERED: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --dsv4 + env: + MODEL: deepseek-ai/DeepSeek-V4-Flash + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' + KV_OFFLOADING: none + +zip_override_tp2: + roles: + agg: + args: + max-running-requests: [1, 2, 4, 8, 16, 32, 64, 128] + cuda-graph-max-bs-decode: [1, 2, 4, 8, 16, 32, 64, 128] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index 0bc79b9ab4..0a4c511444 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -20,7 +20,7 @@ SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" CLIENT_ARGS=() for argument in "$@"; do case "$argument" in - --trust-remote-code) CLIENT_ARGS+=("$argument") ;; + --trust-remote-code|--dsv4) CLIENT_ARGS+=("$argument") ;; *) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;; esac done diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index b0fc9ef451..6dc1079d88 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8928,3 +8928,18 @@ dsv41flash-fp4-gb300-sglang-agentic-dspark: search-space: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } + +dsv4flash-fp4-b300-sglang-mtp: + image: lmsysorg/sglang:nightly-dev-cu13-20260925-8ca82118@sha256:baea7ec5ea86412ce59837f17011028b940f9777fa15129254de02bd6a842559 + model: deepseek-ai/DeepSeek-V4-Flash + model-prefix: dsv4flash + runner: cluster:b300-dsxe + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 2, ep: 1, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml } diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index d480a4bfd7..cae948c781 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -44,6 +44,9 @@ router image does not require changing the worker image. TRT-LLM recipes use nat `engine.served_model_name`, without duplicating that flag in `roles.agg.extra_args`. The former fork's direct ATOM frontend is not required. +DeepSeek-V4 fixed-sequence recipes pass `--dsv4` to `srt_fixed_sequence.sh` and set +`USE_CHAT_TEMPLATE: 'true'` to use the shared DeepSeek-V4 encoder rather than a missing HF chat template. + ### Cluster profiles Launchers that use srt-slurm keep their cluster configuration in diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 47812b4a67..32408f27fd 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -42,6 +42,9 @@ frontend、一个聚合 worker,并设置 `enable_multiple_frontends: false`。 镜像。TRT-LLM 配方使用原生 `engine.served_model_name`,不再通过 `roles.agg.extra_args` 重复传入该参数。不再依赖此前分叉中的 ATOM 直连 frontend。 +DeepSeek-V4 固定序列长度配方向 `srt_fixed_sequence.sh` 传入 `--dsv4`,并设置 +`USE_CHAT_TEMPLATE: 'true'`,使用共享 DeepSeek-V4 编码器,而不是 checkpoint 中缺失的 HF chat template。 + ## 规程索引 1. [准备 worktree](#准备-worktree) diff --git a/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py b/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py new file mode 100644 index 0000000000..bacb33261e --- /dev/null +++ b/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py @@ -0,0 +1,79 @@ +"""Exercise the fixed-sequence client's DeepSeek-V4 encoder forwarding.""" + +import os +import subprocess +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[3] + + +def test_dsv4_client_forwards_encoder_and_workload(tmp_path: Path) -> None: + harness = tmp_path / "harness.sh" + harness.write_text( + """ +source() { + if [[ "$1" == */benchmark_lib.sh && "$2" != --validation-only ]]; then + start_gpu_monitor() { :; } + stop_gpu_monitor() { :; } + run_benchmark_serving() { printf '%s\\n' "$@" > "$CAPTURE"; } + else + builtin source "$@" + fi +} +pip3() { :; } +""" + ) + capture = tmp_path / "arguments" + result = subprocess.run( + ["bash", str(ROOT / "benchmarks/single_node/srt_fixed_sequence.sh"), "--dsv4"], + env={ + **os.environ, + "BASH_ENV": str(harness), + "CAPTURE": str(capture), + "INFERENCEX_REPO_ROOT": str(ROOT), + "MODEL": "test/model", + "FRAMEWORK": "sglang", + "CONC": "2", + "ISL": "256", + "OSL": "64", + "RANDOM_RANGE_RATIO": "0.8", + "RESULT_FILENAME": "point", + "RESULT_DIR": str(tmp_path), + "SRT_FRONTEND_HOST": "127.0.0.1", + "SRT_FRONTEND_PORT": "8000", + "RUN_EVAL": "false", + "EVAL_ONLY": "false", + "GPU_MONITOR_INTERVAL": "3", + "USE_CHAT_TEMPLATE": "true", + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert capture.read_text().splitlines() == [ + "--model", + "test/model", + "--port", + "8000", + "--base-url", + "http://127.0.0.1:8000", + "--backend", + "vllm", + "--input-len", + "256", + "--output-len", + "64", + "--random-range-ratio", + "0.8", + "--num-prompts", + "20", + "--max-concurrency", + "2", + "--result-filename", + "point", + "--result-dir", + str(tmp_path), + "--dsv4", + "--use-chat-template", + ] diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 352c61fc0d..7ab7ed8fcf 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8986,3 +8986,12 @@ - "Hopper 使用 FA3,Blackwell FA4 fp8 descale 问题不适用,EAGLE3 draft 保持 attention_backend FLASH_ATTN。" - "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3503 + +- config-keys: + - dsv4flash-fp4-b300-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Add DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled MTP via EAGLE (3 steps, top-k 1, 4 draft tokens), real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder." + - "新增 DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128;通过 EAGLE 使用原生 MTP(3 steps、top-k 1、4 draft tokens),采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX From db0df130b7ee3f2961ade612487d67a769bfbe81 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 06:44:06 +0000 Subject: [PATCH 2/7] docs: link V4 Flash sweep changelog to PR 3516 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将新增 V4 Flash 扫描记录关联到 PR #3516。 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 7ab7ed8fcf..b40e7036c6 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8994,4 +8994,4 @@ description: - "Add DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled MTP via EAGLE (3 steps, top-k 1, 4 draft tokens), real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder." - "新增 DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128;通过 EAGLE 使用原生 MTP(3 steps、top-k 1、4 draft tokens),采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 From 8766bfd1e1219c484a2f11f47ecbbbd56c61dc43 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 06:48:52 +0000 Subject: [PATCH 3/7] fix: match V4 Flash recipe to fixed-sequence CI metadata MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 移除仅适用于 AgentX 的 KV_OFFLOADING 元数据,修复固定序列 CI 配方选择失败。 --- .../dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml | 1 - perf-changelog.yaml | 9 +++++++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml index 01d6a0c111..fba3f44057 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml @@ -55,7 +55,6 @@ base: OSL: '1024' RANDOM_RANGE_RATIO: '0.8' USE_CHAT_TEMPLATE: 'true' - KV_OFFLOADING: none zip_override_tp2: roles: diff --git a/perf-changelog.yaml b/perf-changelog.yaml index b40e7036c6..35b5e9a25a 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8995,3 +8995,12 @@ - "Add DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled MTP via EAGLE (3 steps, top-k 1, 4 draft tokens), real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder." - "新增 DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128;通过 EAGLE 使用原生 MTP(3 steps、top-k 1、4 draft tokens),采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 + +- config-keys: + - dsv4flash-fp4-b300-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Remove AgentX-only KV_OFFLOADING metadata from the fixed-sequence recipe: CI passes an empty value for this scenario, and the explicit 'none' value rejected recipe selection before server launch. Serving settings are unchanged." + - "移除固定序列配方中仅适用于 AgentX 的 KV_OFFLOADING 元数据:该场景的 CI 传入空值,显式设置 'none' 导致配方选择在服务启动前失败。Serving 设置保持不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 From cc1c168bc3401ebbade872106c7d0dd243104f45 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 06:54:30 +0000 Subject: [PATCH 4/7] fix: download V4 Flash into B300 runner home before submission MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 提交前将 V4 Flash 下载到 B300 runner 的 home 可写目录,保留其他模型路径,并测试下载失败时停止提交。 --- docs/configuration-procedures.md | 5 +++ docs/configuration-procedures_zh.md | 4 +++ infx/tests/srt_slurm/test_srt_single_node.py | 34 ++++++++++++++++++-- perf-changelog.yaml | 9 ++++++ runners/launch_b300-dsxe.sh | 5 +++ 5 files changed, 54 insertions(+), 3 deletions(-) diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index cae948c781..1971249cc1 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -195,6 +195,11 @@ Concurrent cells serialize draft staging with a per-model lock. Each cell lets `hf download` validate or resume the existing cache before serving; a nonempty directory is not a completion signal. +The B300 native single-node launcher downloads `deepseek-ai/DeepSeek-V4-Flash` +with `hf download --local-dir` into `/data/home/sa-gha-runner/models/DeepSeek-V4-Flash` +before SRT submission. Every launch validates or resumes the local download; a failure +stops submission. This checkpoint does not require a pre-staged `/data/models` copy. + ## Native TileRT power TileRT's shared importer preserves Docker Hub image names and converts explicit registries such as `ghcr.io/team/image:tag` to Enroot's `docker://ghcr.io#team/image:tag` syntax. Existing `#` references are preserved. Valid cached squash images are reused without importing; a cache hit does not validate the registry import path. Invalid cached images are removed under the import lock before retrying the import. diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 32408f27fd..5e5709de15 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -144,6 +144,10 @@ B300 DSXE 的 Kimi-K3 AgentX 路径在 `/scratch/models` 下挂载预置目标 并发任务通过模型专用锁串行准备草稿权重。每个任务在启动服务前由 `hf download` 校验或续传现有缓存;目录非空不代表下载完成。 +B300 原生单节点启动器在 SRT 提交前,通过 `hf download --local-dir` 将 +`deepseek-ai/DeepSeek-V4-Flash` 下载至 `/data/home/sa-gha-runner/models/DeepSeek-V4-Flash`。 +每次启动都会验证或续传本地下载,下载失败则停止提交,不要求 `/data/models` 中存在预置副本。 + ## TileRT 原生功耗 TileRT 的共享导入器保留 Docker Hub 镜像名称,并将 `ghcr.io/team/image:tag` 等显式仓库地址转换为 Enroot 的 `docker://ghcr.io#team/image:tag` 格式。已有的 `#` 地址保持不变。有效的缓存 squash 镜像会直接复用;命中缓存不能证明仓库导入路径有效。无效的缓存镜像会在持有导入锁时删除,再重新导入。 diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index d0b160dd5f..c705c64bfa 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -260,6 +260,7 @@ def test_submission_manifest(tmp_path, record, expected): ("b200-cw", "none"), ("b200-nb", "none"), ("b200-nscale-slurm", "none"), ("b200-nscale-slurm", "agentic"), ("b300-dsxe", "none"), + ("b300-dsxe", "download"), ("b300-dsxe", "download-failure"), ("mi300x-amd", "none"), ("mi325x-amds", "none"), ("mi355x-amds", "none"), ] + [(pool, "missing-recipe") for pool in ( "b200-cw", "b200-nb", "b200-nscale-slurm", "b300-dsxe", "h100-cw", @@ -267,7 +268,12 @@ def test_submission_manifest(tmp_path, record, expected): "mi325x-amds", "mi355x-amds", )]) def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, pool, failure): - path, _, point_env = point + path, recipe, point_env = point + if failure.startswith("download"): + point_env = {**point_env, "MODEL": "deepseek-ai/DeepSeek-V4-Flash"} + recipe["model"]["path"] = "hf:deepseek-ai/DeepSeek-V4-Flash" + recipe["benchmark"]["env"]["MODEL"] = "deepseek-ai/DeepSeek-V4-Flash" + path.write_text(yaml.safe_dump({"base": recipe})) binaries = tmp_path / "bin" binaries.mkdir() model = tmp_path / "model" @@ -279,7 +285,15 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, # setup/profile/acceptance helpers, binder, and artifact collection. scripts = { "git": 'if [[ " $* " == *" clone "* ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', - "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', + "uv": """ +if [[ "$1" == tool ]]; then + printf '%s\\n' "$@" > "$DOWNLOAD_CAPTURE" + [[ "$TEST_FAILURE" != download-failure ]] || exit 17 +elif [[ "$1" == venv ]]; then + mkdir -p .venv/bin + echo ":" > .venv/bin/activate +fi +""", "make": '[[ "$TEST_FAILURE" == bootstrap ]] && exit 13; mkdir -p bin; touch bin/uv', "squeue": '[[ "$TEST_FAILURE" == submission || "$TEST_FAILURE" == agentic ]] && echo "42"; exit 0', "salloc": 'echo "Granted job allocation 42"', @@ -325,6 +339,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "B300_HF_CACHE_CONTAINER_DIR": "/hf", "ENROOT_IMPORT_TIME_LIMIT": "10", "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), + "DOWNLOAD_CAPTURE": str(tmp_path / "download-args"), "SRUN_CAPTURE": str(tmp_path / "srun.jsonl"), "KEEP_LOGS": "0", } @@ -340,7 +355,16 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, ["bash", str(ROOT / f"runners/launch_{pool}.sh")], cwd=tmp_path, env=env, capture_output=True, text=True, timeout=30, ) - assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0}[failure], result.stderr + assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0, "download": 0, "download-failure": 1}[failure], result.stderr + if failure.startswith("download"): + assert Path(env["DOWNLOAD_CAPTURE"]).read_text().splitlines() == [ + "tool", "run", "--from", "huggingface-hub>=0.34,<2", "hf", "download", + "deepseek-ai/DeepSeek-V4-Flash", "--local-dir", + "/data/home/sa-gha-runner/models/DeepSeek-V4-Flash", + ] + if failure == "download-failure": + assert not (tmp_path / "srt-single-node-submission.json").exists() + return if failure == "agentic": calls = [json.loads(line) for line in Path(env["SRUN_CAPTURE"]).read_text().splitlines()] assert calls[-1][-2:] == ["bash", "benchmarks/single_node/agentic/fixture_fp8_b200.sh"] @@ -364,6 +388,10 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, cluster_config = yaml.safe_load(next(tmp_path.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) assert cluster_config["containers"]["test:tag"] == "test:tag" assert cluster_config["use_exclusive_sbatch_directive"] is True + if failure == "download": + assert cluster_config["model_paths"]["hf:deepseek-ai/DeepSeek-V4-Flash"] == ( + "/data/home/sa-gha-runner/models/DeepSeek-V4-Flash" + ) assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 35b5e9a25a..0f9b70cb10 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -9004,3 +9004,12 @@ - "Remove AgentX-only KV_OFFLOADING metadata from the fixed-sequence recipe: CI passes an empty value for this scenario, and the explicit 'none' value rejected recipe selection before server launch. Serving settings are unchanged." - "移除固定序列配方中仅适用于 AgentX 的 KV_OFFLOADING 元数据:该场景的 CI 传入空值,显式设置 'none' 导致配方选择在服务启动前失败。Serving 设置保持不变。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 + +- config-keys: + - dsv4flash-fp4-b300-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Download DeepSeek-V4-Flash into the B300 runner's writable home models directory before native SRT submission instead of requiring a missing /data/models copy. Resume or validate the download on every launch and stop before submission on download failure." + - "在原生 SRT 提交前,将 DeepSeek-V4-Flash 下载到 B300 runner 的 home 可写模型目录,不再要求不存在的 /data/models 副本。每次启动均续传或验证下载,下载失败时停止提交。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index efd0506131..137096e38e 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -137,6 +137,11 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then SRT_MODEL_PATH="$MODEL_ROOT/${MODEL##*/}" if [[ "$MODEL" == nvidia/DeepSeek-R1-0528-FP4-V2 ]]; then SRT_MODEL_PATH="$MODEL_ROOT/DeepSeek-R1-0528-NVFP4-v2" + elif [[ "$MODEL" == deepseek-ai/DeepSeek-V4-Flash ]]; then + SRT_MODEL_PATH="$WRITABLE_MODELS_DIR/${MODEL##*/}" + # Resume and verify the download before SRT checks the local model path. + uv tool run --from 'huggingface-hub>=0.34,<2' hf download "$MODEL" \ + --local-dir "$SRT_MODEL_PATH" || exit 1 elif [[ " ${STAGED_MODELS[*]} " != *" ${MODEL##*/} "* || "${MODEL##*/}" == DeepSeek-V4-Pro-0813 ]]; then # Not staged on every node's NVMe; read the shared copy. SRT_MODEL_PATH="$SHARED_MODEL_ROOT/${MODEL##*/}" From 32720374252a81f7e4923569f4cc010982a7c847 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 07:03:10 +0000 Subject: [PATCH 5/7] fix: isolate host model download cache from container paths MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 主机下载显式使用 home 下的 HF 和 Xet 缓存,避免继承容器路径导致权限错误。 --- docs/configuration-procedures.md | 3 +++ docs/configuration-procedures_zh.md | 3 +++ infx/tests/srt_slurm/test_srt_single_node.py | 8 ++++++++ perf-changelog.yaml | 9 +++++++++ runners/launch_b300-dsxe.sh | 3 +++ 5 files changed, 26 insertions(+) diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 1971249cc1..c7778e80ae 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -199,6 +199,9 @@ The B300 native single-node launcher downloads `deepseek-ai/DeepSeek-V4-Flash` with `hf download --local-dir` into `/data/home/sa-gha-runner/models/DeepSeek-V4-Flash` before SRT submission. Every launch validates or resumes the local download; a failure stops submission. This checkpoint does not require a pre-staged `/data/models` copy. +The host download explicitly sets `HF_HOME`, `HF_HUB_CACHE`, and `HF_XET_CACHE` +under `B300_HF_CACHE_HOST_DIR`; container-only cache paths must not leak into the +host downloader. These overrides are command-scoped so SRT retains its container cache mount. ## Native TileRT power diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 5e5709de15..0a605e0e6f 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -147,6 +147,9 @@ B300 DSXE 的 Kimi-K3 AgentX 路径在 `/scratch/models` 下挂载预置目标 B300 原生单节点启动器在 SRT 提交前,通过 `hf download --local-dir` 将 `deepseek-ai/DeepSeek-V4-Flash` 下载至 `/data/home/sa-gha-runner/models/DeepSeek-V4-Flash`。 每次启动都会验证或续传本地下载,下载失败则停止提交,不要求 `/data/models` 中存在预置副本。 +主机下载命令显式将 `HF_HOME`、`HF_HUB_CACHE` 和 `HF_XET_CACHE` 设置到 +`B300_HF_CACHE_HOST_DIR` 下,防止继承仅适用于容器的缓存路径。这些覆盖仅作用于下载命令, +不会改变 SRT 的容器缓存挂载。 ## TileRT 原生功耗 diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index c705c64bfa..f0daf014ef 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -288,6 +288,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "uv": """ if [[ "$1" == tool ]]; then printf '%s\\n' "$@" > "$DOWNLOAD_CAPTURE" + printf '%s\\n' "$HF_HOME" "$HF_HUB_CACHE" "$HF_XET_CACHE" > "$DOWNLOAD_ENV_CAPTURE" [[ "$TEST_FAILURE" != download-failure ]] || exit 17 elif [[ "$1" == venv ]]; then mkdir -p .venv/bin @@ -340,12 +341,15 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), "DOWNLOAD_CAPTURE": str(tmp_path / "download-args"), + "DOWNLOAD_ENV_CAPTURE": str(tmp_path / "download-env"), "SRUN_CAPTURE": str(tmp_path / "srun.jsonl"), "KEEP_LOGS": "0", } env.pop("AIPERF_DRAIN_TIMEOUT_SECONDS", None) env.pop("AIPERF_DRAIN_POLL_SECONDS", None) env.pop("BENCH_SCRIPT_OVERRIDE", None) + if failure.startswith("download"): + env.update(HF_HOME="/data/models", HF_HUB_CACHE="/mnt/hf_hub_cache", HF_XET_CACHE="/data/models/xet") if failure == "missing-recipe": env.pop("SRT_RECIPE") if failure == "agentic": @@ -357,6 +361,9 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, ) assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0, "download": 0, "download-failure": 1}[failure], result.stderr if failure.startswith("download"): + assert Path(env["DOWNLOAD_ENV_CAPTURE"]).read_text().splitlines() == [ + str(tmp_path), str(tmp_path / "hub"), str(tmp_path / "xet"), + ] assert Path(env["DOWNLOAD_CAPTURE"]).read_text().splitlines() == [ "tool", "run", "--from", "huggingface-hub>=0.34,<2", "hf", "download", "deepseek-ai/DeepSeek-V4-Flash", "--local-dir", @@ -392,6 +399,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, assert cluster_config["model_paths"]["hf:deepseek-ai/DeepSeek-V4-Flash"] == ( "/data/home/sa-gha-runner/models/DeepSeek-V4-Flash" ) + assert cluster_config["default_mounts"][str(tmp_path / "hub")] == "/mnt/hf_hub_cache" assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 0f9b70cb10..35315178d1 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -9013,3 +9013,12 @@ - "Download DeepSeek-V4-Flash into the B300 runner's writable home models directory before native SRT submission instead of requiring a missing /data/models copy. Resume or validate the download on every launch and stop before submission on download failure." - "在原生 SRT 提交前,将 DeepSeek-V4-Flash 下载到 B300 runner 的 home 可写模型目录,不再要求不存在的 /data/models 副本。每次启动均续传或验证下载,下载失败时停止提交。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 + +- config-keys: + - dsv4flash-fp4-b300-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Scope HF_HOME, HF_HUB_CACHE, and HF_XET_CACHE to the runner's writable home cache during the host-side V4 Flash download. Preserve the container cache environment and mounts for subsequent SRT execution." + - "在主机下载 V4 Flash 时,将 HF_HOME、HF_HUB_CACHE 和 HF_XET_CACHE 限定到 runner 的 home 可写缓存;保留后续 SRT 执行所需的容器缓存环境和挂载。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index 137096e38e..9f7214a411 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -140,6 +140,9 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then elif [[ "$MODEL" == deepseek-ai/DeepSeek-V4-Flash ]]; then SRT_MODEL_PATH="$WRITABLE_MODELS_DIR/${MODEL##*/}" # Resume and verify the download before SRT checks the local model path. + HF_HOME="$B300_HF_CACHE_HOST_DIR" \ + HF_HUB_CACHE="$HF_HUB_CACHE_MOUNT" \ + HF_XET_CACHE="$B300_HF_CACHE_HOST_DIR/xet" \ uv tool run --from 'huggingface-hub>=0.34,<2' hf download "$MODEL" \ --local-dir "$SRT_MODEL_PATH" || exit 1 elif [[ " ${STAGED_MODELS[*]} " != *" ${MODEL##*/} "* || "${MODEL##*/}" == DeepSeek-V4-Pro-0813 ]]; then From e98dc7dec3a6030c32ed9d8f3542e72416f7d84f Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sun, 27 Sep 2026 18:44:30 +0800 Subject: [PATCH 6/7] fix: load V4 Flash from node-local NVMe and use EAGLE 2/1/3 The shared Lustre copy is read through mmap page faults at ~10 MB/s, so TP2 weight loading outlasted the 1800 s health timeout. DeepSeek-V4-Flash is now staged under /scratch/models on every B300 node; drop the host-side download path and list it in STAGED_MODELS. Switch MTP to 2 steps, top-k 1, 3 draft tokens. --- .../dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml | 4 +- docs/configuration-procedures.md | 8 ---- docs/configuration-procedures_zh.md | 7 ---- infx/tests/srt_slurm/test_srt_single_node.py | 42 ++----------------- perf-changelog.yaml | 31 +------------- runners/launch_b300-dsxe.sh | 9 +--- 6 files changed, 8 insertions(+), 93 deletions(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml index fba3f44057..d86ac85958 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml @@ -37,9 +37,9 @@ base: reasoning-parser: deepseek-v4 tool-call-parser: deepseekv4 speculative-algorithm: EAGLE - speculative-num-steps: 3 + speculative-num-steps: 2 speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 + speculative-num-draft-tokens: 3 watchdog-timeout: 3600 enable-metrics: true env: diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index c7778e80ae..cae948c781 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -195,14 +195,6 @@ Concurrent cells serialize draft staging with a per-model lock. Each cell lets `hf download` validate or resume the existing cache before serving; a nonempty directory is not a completion signal. -The B300 native single-node launcher downloads `deepseek-ai/DeepSeek-V4-Flash` -with `hf download --local-dir` into `/data/home/sa-gha-runner/models/DeepSeek-V4-Flash` -before SRT submission. Every launch validates or resumes the local download; a failure -stops submission. This checkpoint does not require a pre-staged `/data/models` copy. -The host download explicitly sets `HF_HOME`, `HF_HUB_CACHE`, and `HF_XET_CACHE` -under `B300_HF_CACHE_HOST_DIR`; container-only cache paths must not leak into the -host downloader. These overrides are command-scoped so SRT retains its container cache mount. - ## Native TileRT power TileRT's shared importer preserves Docker Hub image names and converts explicit registries such as `ghcr.io/team/image:tag` to Enroot's `docker://ghcr.io#team/image:tag` syntax. Existing `#` references are preserved. Valid cached squash images are reused without importing; a cache hit does not validate the registry import path. Invalid cached images are removed under the import lock before retrying the import. diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 0a605e0e6f..32408f27fd 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -144,13 +144,6 @@ B300 DSXE 的 Kimi-K3 AgentX 路径在 `/scratch/models` 下挂载预置目标 并发任务通过模型专用锁串行准备草稿权重。每个任务在启动服务前由 `hf download` 校验或续传现有缓存;目录非空不代表下载完成。 -B300 原生单节点启动器在 SRT 提交前,通过 `hf download --local-dir` 将 -`deepseek-ai/DeepSeek-V4-Flash` 下载至 `/data/home/sa-gha-runner/models/DeepSeek-V4-Flash`。 -每次启动都会验证或续传本地下载,下载失败则停止提交,不要求 `/data/models` 中存在预置副本。 -主机下载命令显式将 `HF_HOME`、`HF_HUB_CACHE` 和 `HF_XET_CACHE` 设置到 -`B300_HF_CACHE_HOST_DIR` 下,防止继承仅适用于容器的缓存路径。这些覆盖仅作用于下载命令, -不会改变 SRT 的容器缓存挂载。 - ## TileRT 原生功耗 TileRT 的共享导入器保留 Docker Hub 镜像名称,并将 `ghcr.io/team/image:tag` 等显式仓库地址转换为 Enroot 的 `docker://ghcr.io#team/image:tag` 格式。已有的 `#` 地址保持不变。有效的缓存 squash 镜像会直接复用;命中缓存不能证明仓库导入路径有效。无效的缓存镜像会在持有导入锁时删除,再重新导入。 diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index f0daf014ef..d0b160dd5f 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -260,7 +260,6 @@ def test_submission_manifest(tmp_path, record, expected): ("b200-cw", "none"), ("b200-nb", "none"), ("b200-nscale-slurm", "none"), ("b200-nscale-slurm", "agentic"), ("b300-dsxe", "none"), - ("b300-dsxe", "download"), ("b300-dsxe", "download-failure"), ("mi300x-amd", "none"), ("mi325x-amds", "none"), ("mi355x-amds", "none"), ] + [(pool, "missing-recipe") for pool in ( "b200-cw", "b200-nb", "b200-nscale-slurm", "b300-dsxe", "h100-cw", @@ -268,12 +267,7 @@ def test_submission_manifest(tmp_path, record, expected): "mi325x-amds", "mi355x-amds", )]) def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, pool, failure): - path, recipe, point_env = point - if failure.startswith("download"): - point_env = {**point_env, "MODEL": "deepseek-ai/DeepSeek-V4-Flash"} - recipe["model"]["path"] = "hf:deepseek-ai/DeepSeek-V4-Flash" - recipe["benchmark"]["env"]["MODEL"] = "deepseek-ai/DeepSeek-V4-Flash" - path.write_text(yaml.safe_dump({"base": recipe})) + path, _, point_env = point binaries = tmp_path / "bin" binaries.mkdir() model = tmp_path / "model" @@ -285,16 +279,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, # setup/profile/acceptance helpers, binder, and artifact collection. scripts = { "git": 'if [[ " $* " == *" clone "* ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', - "uv": """ -if [[ "$1" == tool ]]; then - printf '%s\\n' "$@" > "$DOWNLOAD_CAPTURE" - printf '%s\\n' "$HF_HOME" "$HF_HUB_CACHE" "$HF_XET_CACHE" > "$DOWNLOAD_ENV_CAPTURE" - [[ "$TEST_FAILURE" != download-failure ]] || exit 17 -elif [[ "$1" == venv ]]; then - mkdir -p .venv/bin - echo ":" > .venv/bin/activate -fi -""", + "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', "make": '[[ "$TEST_FAILURE" == bootstrap ]] && exit 13; mkdir -p bin; touch bin/uv', "squeue": '[[ "$TEST_FAILURE" == submission || "$TEST_FAILURE" == agentic ]] && echo "42"; exit 0', "salloc": 'echo "Granted job allocation 42"', @@ -340,16 +325,12 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "B300_HF_CACHE_CONTAINER_DIR": "/hf", "ENROOT_IMPORT_TIME_LIMIT": "10", "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), - "DOWNLOAD_CAPTURE": str(tmp_path / "download-args"), - "DOWNLOAD_ENV_CAPTURE": str(tmp_path / "download-env"), "SRUN_CAPTURE": str(tmp_path / "srun.jsonl"), "KEEP_LOGS": "0", } env.pop("AIPERF_DRAIN_TIMEOUT_SECONDS", None) env.pop("AIPERF_DRAIN_POLL_SECONDS", None) env.pop("BENCH_SCRIPT_OVERRIDE", None) - if failure.startswith("download"): - env.update(HF_HOME="/data/models", HF_HUB_CACHE="/mnt/hf_hub_cache", HF_XET_CACHE="/data/models/xet") if failure == "missing-recipe": env.pop("SRT_RECIPE") if failure == "agentic": @@ -359,19 +340,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, ["bash", str(ROOT / f"runners/launch_{pool}.sh")], cwd=tmp_path, env=env, capture_output=True, text=True, timeout=30, ) - assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0, "download": 0, "download-failure": 1}[failure], result.stderr - if failure.startswith("download"): - assert Path(env["DOWNLOAD_ENV_CAPTURE"]).read_text().splitlines() == [ - str(tmp_path), str(tmp_path / "hub"), str(tmp_path / "xet"), - ] - assert Path(env["DOWNLOAD_CAPTURE"]).read_text().splitlines() == [ - "tool", "run", "--from", "huggingface-hub>=0.34,<2", "hf", "download", - "deepseek-ai/DeepSeek-V4-Flash", "--local-dir", - "/data/home/sa-gha-runner/models/DeepSeek-V4-Flash", - ] - if failure == "download-failure": - assert not (tmp_path / "srt-single-node-submission.json").exists() - return + assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0}[failure], result.stderr if failure == "agentic": calls = [json.loads(line) for line in Path(env["SRUN_CAPTURE"]).read_text().splitlines()] assert calls[-1][-2:] == ["bash", "benchmarks/single_node/agentic/fixture_fp8_b200.sh"] @@ -395,11 +364,6 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, cluster_config = yaml.safe_load(next(tmp_path.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) assert cluster_config["containers"]["test:tag"] == "test:tag" assert cluster_config["use_exclusive_sbatch_directive"] is True - if failure == "download": - assert cluster_config["model_paths"]["hf:deepseek-ai/DeepSeek-V4-Flash"] == ( - "/data/home/sa-gha-runner/models/DeepSeek-V4-Flash" - ) - assert cluster_config["default_mounts"][str(tmp_path / "hub")] == "/mnt/hf_hub_cache" assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 35315178d1..2a32a37af1 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8992,33 +8992,6 @@ scenario-type: - fixed-seq-len description: - - "Add DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled MTP via EAGLE (3 steps, top-k 1, 4 draft tokens), real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder." - - "新增 DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128;通过 EAGLE 使用原生 MTP(3 steps、top-k 1、4 draft tokens),采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 - -- config-keys: - - dsv4flash-fp4-b300-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - "Remove AgentX-only KV_OFFLOADING metadata from the fixed-sequence recipe: CI passes an empty value for this scenario, and the explicit 'none' value rejected recipe selection before server launch. Serving settings are unchanged." - - "移除固定序列配方中仅适用于 AgentX 的 KV_OFFLOADING 元数据:该场景的 CI 传入空值,显式设置 'none' 导致配方选择在服务启动前失败。Serving 设置保持不变。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 - -- config-keys: - - dsv4flash-fp4-b300-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - "Download DeepSeek-V4-Flash into the B300 runner's writable home models directory before native SRT submission instead of requiring a missing /data/models copy. Resume or validate the download on every launch and stop before submission on download failure." - - "在原生 SRT 提交前,将 DeepSeek-V4-Flash 下载到 B300 runner 的 home 可写模型目录,不再要求不存在的 /data/models 副本。每次启动均续传或验证下载,下载失败时停止提交。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 - -- config-keys: - - dsv4flash-fp4-b300-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - "Scope HF_HOME, HF_HUB_CACHE, and HF_XET_CACHE to the runner's writable home cache during the host-side V4 Flash download. Preserve the container cache environment and mounts for subsequent SRT execution." - - "在主机下载 V4 Flash 时,将 HF_HOME、HF_HUB_CACHE 和 HF_XET_CACHE 限定到 runner 的 home 可写缓存;保留后续 SRT 执行所需的容器缓存环境和挂载。" + - "Add DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled MTP via EAGLE (2 steps, top-k 1, 3 draft tokens), real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder. Weights load from the node-local /scratch/models copy; mmap page faults on the shared /data copy read at ~10 MB/s and outlast the 1800 s health timeout." + - "新增 DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128;通过 EAGLE 使用原生 MTP(2 steps、top-k 1、3 draft tokens),采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。权重从节点本地 /scratch/models 副本加载;共享 /data 副本的 mmap 缺页读取仅约 10 MB/s,会超过 1800 s 健康检查超时。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516 diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index 9f7214a411..6769a3c1aa 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -61,6 +61,7 @@ WRITABLE_MODELS_DIR="/data/home/sa-gha-runner/models" STAGED_MODELS=( DeepSeek-R1-0528 DeepSeek-R1-0528-NVFP4-v2 + DeepSeek-V4-Flash DeepSeek-V4-Pro DeepSeek-V4-Pro-0813 DeepSeek-V4-Pro-NVFP4 @@ -137,14 +138,6 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then SRT_MODEL_PATH="$MODEL_ROOT/${MODEL##*/}" if [[ "$MODEL" == nvidia/DeepSeek-R1-0528-FP4-V2 ]]; then SRT_MODEL_PATH="$MODEL_ROOT/DeepSeek-R1-0528-NVFP4-v2" - elif [[ "$MODEL" == deepseek-ai/DeepSeek-V4-Flash ]]; then - SRT_MODEL_PATH="$WRITABLE_MODELS_DIR/${MODEL##*/}" - # Resume and verify the download before SRT checks the local model path. - HF_HOME="$B300_HF_CACHE_HOST_DIR" \ - HF_HUB_CACHE="$HF_HUB_CACHE_MOUNT" \ - HF_XET_CACHE="$B300_HF_CACHE_HOST_DIR/xet" \ - uv tool run --from 'huggingface-hub>=0.34,<2' hf download "$MODEL" \ - --local-dir "$SRT_MODEL_PATH" || exit 1 elif [[ " ${STAGED_MODELS[*]} " != *" ${MODEL##*/} "* || "${MODEL##*/}" == DeepSeek-V4-Pro-0813 ]]; then # Not staged on every node's NVMe; read the shared copy. SRT_MODEL_PATH="$SHARED_MODEL_ROOT/${MODEL##*/}" From c9bb03d9a0346f633f08ac80f34d54fe307ec93a Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sun, 27 Sep 2026 23:10:00 +0800 Subject: [PATCH 7/7] perf: run V4 Flash decode and speculative steps without graphs Set cuda-graph-backend-{decode,prefill}=disabled, which also skips target verify and draft capture, and drop the per-concurrency decode graph batch. --- .../srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml | 4 +++- perf-changelog.yaml | 4 ++-- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml index d86ac85958..ed7f590936 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b300-fp4-mtp/8k1k.yaml @@ -40,6 +40,9 @@ base: speculative-num-steps: 2 speculative-eagle-topk: 1 speculative-num-draft-tokens: 3 + # Eager decode and speculative verify/draft; no CUDA graph capture. + cuda-graph-backend-decode: disabled + cuda-graph-backend-prefill: disabled watchdog-timeout: 3600 enable-metrics: true env: @@ -61,7 +64,6 @@ zip_override_tp2: agg: args: max-running-requests: [1, 2, 4, 8, 16, 32, 64, 128] - cuda-graph-max-bs-decode: [1, 2, 4, 8, 16, 32, 64, 128] benchmark: env: CONC: ['1', '2', '4', '8', '16', '32', '64', '128'] diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 2a32a37af1..2f6218f838 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8992,6 +8992,6 @@ scenario-type: - fixed-seq-len description: - - "Add DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled MTP via EAGLE (2 steps, top-k 1, 3 draft tokens), real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder. Weights load from the node-local /scratch/models copy; mmap page faults on the shared /data copy read at ~10 MB/s and outlast the 1800 s health timeout." - - "新增 DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128;通过 EAGLE 使用原生 MTP(2 steps、top-k 1、3 draft tokens),采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。权重从节点本地 /scratch/models 副本加载;共享 /data 副本的 mmap 缺页读取仅约 10 MB/s,会超过 1800 s 健康检查超时。" + - "Add DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled MTP via EAGLE (2 steps, top-k 1, 3 draft tokens) with CUDA/HIP graphs disabled for decode, prefill, and speculative verify/draft, real verification, GPU-resident weights/KV, and the DeepSeek-V4 chat encoder. Weights load from the node-local /scratch/models copy; mmap page faults on the shared /data copy read at ~10 MB/s and outlast the 1800 s health timeout." + - "新增 DeepSeek-V4-Flash B300 TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128;通过 EAGLE 使用原生 MTP(2 steps、top-k 1、3 draft tokens),decode、prefill 及投机验证/draft 均关闭 CUDA/HIP graph,采用真实验证、GPU 常驻权重/KV 和 DeepSeek-V4 chat 编码器。权重从节点本地 /scratch/models 副本加载;共享 /data 副本的 mmap 缺页读取仅约 10 MB/s,会超过 1800 s 健康检查超时。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3516