From d24d369ed6bddd0391a77db9185129125f8c4c51 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 07:08:13 +0000 Subject: [PATCH 1/6] feat: add B200 TP2 DeepSeek V4 Flash serving sweep MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 新增 B200 TP2 DeepSeek V4 Flash serving 扫描,在共享可写目录下载模型并隔离 HF/Xet 缓存。 --- .../dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml | 67 ++++++++++++++++ benchmarks/single_node/srt_fixed_sequence.sh | 2 +- configs/nvidia-master.yaml | 15 ++++ docs/configuration-procedures.md | 11 +++ docs/configuration-procedures_zh.md | 9 +++ .../test_dsv4_fixed_sequence_client.py | 79 +++++++++++++++++++ infx/tests/srt_slurm/test_srt_single_node.py | 43 +++++++++- perf-changelog.yaml | 9 +++ runners/launch_b200-nscale-slurm.sh | 9 +++ 9 files changed, 240 insertions(+), 4 deletions(-) create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml create mode 100644 infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..5d15a30a20 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml @@ -0,0 +1,67 @@ +base: + schema: 2 + name: dsv4flash-fp4-b200-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-V4-Flash + container: lmsysorg/sglang:nightly-dev-cu13-20260925-8ca82118@sha256:baea7ec5ea86412ce59837f17011028b940f9777fa15129254de02bd6a842559 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + served-model-name: deepseek-ai/DeepSeek-V4-Flash + trust-remote-code: true + tensor-parallel-size: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_mxfp4 + disable-flashinfer-autotune: true + disable-radix-cache: true + mem-fraction-static: 0.85 + swa-full-tokens-ratio: 0.1 + chunked-prefill-size: 8192 + context-length: 9236 + reasoning-parser: deepseek-v4 + tool-call-parser: deepseekv4 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + watchdog-timeout: 3600 + enable-metrics: true + env: + PYTHONNOUSERSITE: '1' + PYTHONUNBUFFERED: '1' + SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --dsv4 + env: + MODEL: deepseek-ai/DeepSeek-V4-Flash + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' + +zip_override_tp2: + roles: + agg: + args: + max-running-requests: [1, 2, 4, 8, 16, 32, 64, 128] + cuda-graph-max-bs-decode: [1, 2, 4, 8, 16, 32, 64, 128] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index 0bc79b9ab4..0a4c511444 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -20,7 +20,7 @@ SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" CLIENT_ARGS=() for argument in "$@"; do case "$argument" in - --trust-remote-code) CLIENT_ARGS+=("$argument") ;; + --trust-remote-code|--dsv4) CLIENT_ARGS+=("$argument") ;; *) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;; esac done diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index b0fc9ef451..1410fbb7cd 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8928,3 +8928,18 @@ dsv41flash-fp4-gb300-sglang-agentic-dspark: search-space: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } + +dsv4flash-fp4-b200-sglang-mtp: + image: lmsysorg/sglang:nightly-dev-cu13-20260925-8ca82118@sha256:baea7ec5ea86412ce59837f17011028b940f9777fa15129254de02bd6a842559 + model: deepseek-ai/DeepSeek-V4-Flash + model-prefix: dsv4flash + runner: cluster:b200-nscale + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 2, ep: 1, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml } diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index d480a4bfd7..8eee6d72c0 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -46,6 +46,17 @@ The former fork's direct ATOM frontend is not required. ### Cluster profiles +The B200 Nscale fixed-sequence `deepseek-ai/DeepSeek-V4-Flash` launcher stages the +checkpoint with `hf download --local-dir` under +`/data/home/sa-shared/gharunners/models/DeepSeek-V4-Flash` before SRT submission. +Each launch resumes or validates that download; failure stops submission. This +shared writable location does not require a pre-staged `/scratch/models` copy. +The download process uses explicit writable `HF_HOME`, `HF_HUB_CACHE`, and +`HF_XET_CACHE` paths beneath the checkpoint's `.cache/huggingface` directory; +the serving process retains its container cache settings. +The recipe uses bundled MTP through `EAGLE` (3 steps, top-k 1, 4 draft tokens), +with the DeepSeek-V4 chat encoder selected by the client's `--dsv4` option. + Launchers that use srt-slurm keep their cluster configuration in [`runners/srt-slurm/.yaml`](../runners/srt-slurm/). The native settings (GPU count, scheduling directives, aliases, and mounts) are separate from workload recipes. diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 47812b4a67..eabc5c28c9 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -42,6 +42,15 @@ frontend、一个聚合 worker,并设置 `enable_multiple_frontends: false`。 镜像。TRT-LLM 配方使用原生 `engine.served_model_name`,不再通过 `roles.agg.extra_args` 重复传入该参数。不再依赖此前分叉中的 ATOM 直连 frontend。 +B200 Nscale 的固定序列 `deepseek-ai/DeepSeek-V4-Flash` 启动器在提交 SRT 前, +通过 `hf download --local-dir` 将 checkpoint 下载至 +`/data/home/sa-shared/gharunners/models/DeepSeek-V4-Flash`。 +每次启动均续传或验证下载,失败时停止提交;此共享可写目录不依赖 +`/scratch/models` 中的预置副本。下载进程显式将 `HF_HOME`、`HF_HUB_CACHE` +和 `HF_XET_CACHE` 指向 checkpoint 的 `.cache/huggingface` 下的可写路径, +服务进程仍保留容器缓存设置。配方通过 `EAGLE` 使用原生 MTP +(3 steps、top-k 1、4 draft tokens),客户端以 `--dsv4` 选择 DeepSeek-V4 chat 编码器。 + ## 规程索引 1. [准备 worktree](#准备-worktree) diff --git a/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py b/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py new file mode 100644 index 0000000000..bacb33261e --- /dev/null +++ b/infx/tests/srt_slurm/test_dsv4_fixed_sequence_client.py @@ -0,0 +1,79 @@ +"""Exercise the fixed-sequence client's DeepSeek-V4 encoder forwarding.""" + +import os +import subprocess +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[3] + + +def test_dsv4_client_forwards_encoder_and_workload(tmp_path: Path) -> None: + harness = tmp_path / "harness.sh" + harness.write_text( + """ +source() { + if [[ "$1" == */benchmark_lib.sh && "$2" != --validation-only ]]; then + start_gpu_monitor() { :; } + stop_gpu_monitor() { :; } + run_benchmark_serving() { printf '%s\\n' "$@" > "$CAPTURE"; } + else + builtin source "$@" + fi +} +pip3() { :; } +""" + ) + capture = tmp_path / "arguments" + result = subprocess.run( + ["bash", str(ROOT / "benchmarks/single_node/srt_fixed_sequence.sh"), "--dsv4"], + env={ + **os.environ, + "BASH_ENV": str(harness), + "CAPTURE": str(capture), + "INFERENCEX_REPO_ROOT": str(ROOT), + "MODEL": "test/model", + "FRAMEWORK": "sglang", + "CONC": "2", + "ISL": "256", + "OSL": "64", + "RANDOM_RANGE_RATIO": "0.8", + "RESULT_FILENAME": "point", + "RESULT_DIR": str(tmp_path), + "SRT_FRONTEND_HOST": "127.0.0.1", + "SRT_FRONTEND_PORT": "8000", + "RUN_EVAL": "false", + "EVAL_ONLY": "false", + "GPU_MONITOR_INTERVAL": "3", + "USE_CHAT_TEMPLATE": "true", + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert capture.read_text().splitlines() == [ + "--model", + "test/model", + "--port", + "8000", + "--base-url", + "http://127.0.0.1:8000", + "--backend", + "vllm", + "--input-len", + "256", + "--output-len", + "64", + "--random-range-ratio", + "0.8", + "--num-prompts", + "20", + "--max-concurrency", + "2", + "--result-filename", + "point", + "--result-dir", + str(tmp_path), + "--dsv4", + "--use-chat-template", + ] diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index d0b160dd5f..3b9d265aba 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -259,6 +259,7 @@ def test_submission_manifest(tmp_path, record, expected): ("h200-cw", "none"), ("h100-cw", "none"), ("h100-dgxc-slurm", "none"), ("b200-cw", "none"), ("b200-nb", "none"), ("b200-nscale-slurm", "none"), ("b200-nscale-slurm", "agentic"), + ("b200-nscale-slurm", "download"), ("b200-nscale-slurm", "download-failure"), ("b300-dsxe", "none"), ("mi300x-amd", "none"), ("mi325x-amds", "none"), ("mi355x-amds", "none"), ] + [(pool, "missing-recipe") for pool in ( @@ -267,7 +268,12 @@ def test_submission_manifest(tmp_path, record, expected): "mi325x-amds", "mi355x-amds", )]) def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, pool, failure): - path, _, point_env = point + path, recipe, point_env = point + if failure.startswith("download"): + point_env = {**point_env, "MODEL": "deepseek-ai/DeepSeek-V4-Flash"} + recipe["model"]["path"] = "hf:deepseek-ai/DeepSeek-V4-Flash" + recipe["benchmark"]["env"]["MODEL"] = "deepseek-ai/DeepSeek-V4-Flash" + path.write_text(yaml.safe_dump({"base": recipe})) binaries = tmp_path / "bin" binaries.mkdir() model = tmp_path / "model" @@ -279,7 +285,16 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, # setup/profile/acceptance helpers, binder, and artifact collection. scripts = { "git": 'if [[ " $* " == *" clone "* ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', - "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', + "uv": """ +if [[ "$1" == tool ]]; then + printf '%s\\n' "$@" > "$DOWNLOAD_CAPTURE" + printf '%s\\n' "$HF_HOME" "$HF_HUB_CACHE" "$HF_XET_CACHE" > "$DOWNLOAD_CACHE_CAPTURE" + [[ "$TEST_FAILURE" != download-failure ]] || exit 17 +elif [[ "$1" == venv ]]; then + mkdir -p .venv/bin + echo ":" > .venv/bin/activate +fi +""", "make": '[[ "$TEST_FAILURE" == bootstrap ]] && exit 13; mkdir -p bin; touch bin/uv', "squeue": '[[ "$TEST_FAILURE" == submission || "$TEST_FAILURE" == agentic ]] && echo "42"; exit 0', "salloc": 'echo "Granted job allocation 42"', @@ -325,6 +340,9 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "B300_HF_CACHE_CONTAINER_DIR": "/hf", "ENROOT_IMPORT_TIME_LIMIT": "10", "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), + "DOWNLOAD_CAPTURE": str(tmp_path / "download-args"), + "DOWNLOAD_CACHE_CAPTURE": str(tmp_path / "download-cache"), + "HF_HOME": "/read-only/hf", "HF_XET_CACHE": "/read-only/xet", "SRUN_CAPTURE": str(tmp_path / "srun.jsonl"), "KEEP_LOGS": "0", } @@ -340,7 +358,21 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, ["bash", str(ROOT / f"runners/launch_{pool}.sh")], cwd=tmp_path, env=env, capture_output=True, text=True, timeout=30, ) - assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0}[failure], result.stderr + assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0, "download": 0, "download-failure": 1}[failure], result.stderr + if failure.startswith("download"): + assert Path(env["DOWNLOAD_CAPTURE"]).read_text().splitlines() == [ + "tool", "run", "--from", "huggingface-hub>=0.34,<2", "hf", "download", + "deepseek-ai/DeepSeek-V4-Flash", "--local-dir", + "/data/home/sa-shared/gharunners/models/DeepSeek-V4-Flash", + ] + assert Path(env["DOWNLOAD_CACHE_CAPTURE"]).read_text().splitlines() == [ + "/data/home/sa-shared/gharunners/models/DeepSeek-V4-Flash/.cache/huggingface", + "/data/home/sa-shared/gharunners/models/DeepSeek-V4-Flash/.cache/huggingface/hub", + "/data/home/sa-shared/gharunners/models/DeepSeek-V4-Flash/.cache/huggingface/xet", + ] + if failure == "download-failure": + assert not (tmp_path / "srt-single-node-submission.json").exists() + return if failure == "agentic": calls = [json.loads(line) for line in Path(env["SRUN_CAPTURE"]).read_text().splitlines()] assert calls[-1][-2:] == ["bash", "benchmarks/single_node/agentic/fixture_fp8_b200.sh"] @@ -364,6 +396,11 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, cluster_config = yaml.safe_load(next(tmp_path.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) assert cluster_config["containers"]["test:tag"] == "test:tag" assert cluster_config["use_exclusive_sbatch_directive"] is True + if failure == "download": + assert cluster_config["model_paths"]["hf:deepseek-ai/DeepSeek-V4-Flash"] == ( + "/data/home/sa-shared/gharunners/models/DeepSeek-V4-Flash" + ) + assert cluster_config["default_mounts"]["/data/home/sa-shared/gharunners/hf-hub-cache"] == "/hf" assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 352c61fc0d..75c330e277 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8986,3 +8986,12 @@ - "Hopper 使用 FA3,Blackwell FA4 fp8 descale 问题不适用,EAGLE3 draft 保持 attention_backend FLASH_ATTN。" - "This change does not alter the EAGLE3 draft model data type. The draft loads unmodified from the published Inferact/MiniMax-M3-EAGLE3-GQA checkpoint via --speculative-config (method=eagle3). kv-cache-dtype fp8 sets KV-cache storage precision, not the draft weights, and no flag overrides or re-quantizes the draft weights." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3503 + +- config-keys: + - dsv4flash-fp4-b200-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Add DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled EAGLE MTP (3 steps, top-k 1, 4 draft tokens), real verification, and GPU-resident weights/KV. Stage the unstaged checkpoint and HF/Xet download caches on writable shared storage before SRT submission; stop on download failure." + - "新增 DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,采用原生 EAGLE MTP(3 steps、top-k 1、4 draft tokens)、真实验证和 GPU 常驻权重/KV。SRT 提交前在共享可写目录下载 checkpoint 并设置 HF/Xet 下载缓存,下载失败时停止提交。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX diff --git a/runners/launch_b200-nscale-slurm.sh b/runners/launch_b200-nscale-slurm.sh index 1c6c7afc04..4f10c19f6b 100755 --- a/runners/launch_b200-nscale-slurm.sh +++ b/runners/launch_b200-nscale-slurm.sh @@ -90,6 +90,15 @@ if [[ "$LAUNCH_PATH" == "native-srt" ]]; then export SRT_SLURM_MODEL_PREFIX="glm5.1-fp8" ;; esac +elif [[ "$LAUNCH_PATH" == native-single-node && "$MODEL" == deepseek-ai/DeepSeek-V4-Flash ]]; then + export MODEL_PATH="/data/home/sa-shared/gharunners/models/${MODEL##*/}" + # Stage on writable shared storage before SRT validates the local path. + # Host downloads must not inherit container-only or read-only cache paths. + HF_HOME="$MODEL_PATH/.cache/huggingface" \ + HF_HUB_CACHE="$MODEL_PATH/.cache/huggingface/hub" \ + HF_XET_CACHE="$MODEL_PATH/.cache/huggingface/xet" \ + uv tool run --from 'huggingface-hub>=0.34,<2' hf download "$MODEL" \ + --local-dir "$MODEL_PATH" || exit 1 elif [[ "$MODEL_PREFIX" == "dsv41flash" && "$PRECISION" == "fp4" && ( "$FRAMEWORK" == "vllm" || "$FRAMEWORK" == "sglang" ) && "$IS_MULTINODE" != "true" ]]; then export MODEL_PATH="$MODEL" export HF_HUB_CACHE_HOST_PATH="/data/home/sa-shared/gharunners/hf-hub-cache" From c359fcf6220d47e99c6c4a8f1250b4266bb66171 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 07:08:37 +0000 Subject: [PATCH 2/6] docs: link B200 sweep changelog to PR 3517 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 B200 扫描变更记录关联至 PR #3517。 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 75c330e277..07f9100d20 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8994,4 +8994,4 @@ description: - "Add DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled EAGLE MTP (3 steps, top-k 1, 4 draft tokens), real verification, and GPU-resident weights/KV. Stage the unstaged checkpoint and HF/Xet download caches on writable shared storage before SRT submission; stop on download failure." - "新增 DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,采用原生 EAGLE MTP(3 steps、top-k 1、4 draft tokens)、真实验证和 GPU 常驻权重/KV。SRT 提交前在共享可写目录下载 checkpoint 并设置 HF/Xet 下载缓存,下载失败时停止提交。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3517 From 840c69f1f8e22df939214c06b98ed4404db61f3f Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 27 Sep 2026 07:24:10 +0000 Subject: [PATCH 3/6] fix: normalize B200 Flash image digest for Pyxis MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 B200 Flash 镜像 digest 转为 Pyxis/Enroot 原生语法,保留固定 digest 和有效 squash 缓存优先级。 --- docs/configuration-procedures.md | 3 ++ docs/configuration-procedures_zh.md | 3 ++ infx/tests/srt_slurm/test_srt_single_node.py | 23 ++++++++++++--- perf-changelog.yaml | 9 ++++++ runners/launch_b200-nscale-slurm.sh | 31 +++++++++++++------- 5 files changed, 54 insertions(+), 15 deletions(-) diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 8eee6d72c0..06785960ad 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -54,6 +54,9 @@ shared writable location does not require a pre-staged `/scratch/models` copy. The download process uses explicit writable `HF_HOME`, `HF_HUB_CACHE`, and `HF_XET_CACHE` paths beneath the checkpoint's `.cache/huggingface` directory; the serving process retains its container cache settings. +For an uncached digest-pinned image, this lane maps Docker's `repo:tag@sha256:...` +to Pyxis/Enroot's `repo:sha256:...` manifest reference. The recorded image and +digest stay unchanged; a valid cached squash image still takes precedence. The recipe uses bundled MTP through `EAGLE` (3 steps, top-k 1, 4 draft tokens), with the DeepSeek-V4 chat encoder selected by the client's `--dsv4` option. diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index eabc5c28c9..f7894acf9e 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -50,6 +50,9 @@ B200 Nscale 的固定序列 `deepseek-ai/DeepSeek-V4-Flash` 启动器在提交 S 和 `HF_XET_CACHE` 指向 checkpoint 的 `.cache/huggingface` 下的可写路径, 服务进程仍保留容器缓存设置。配方通过 `EAGLE` 使用原生 MTP (3 steps、top-k 1、4 draft tokens),客户端以 `--dsv4` 选择 DeepSeek-V4 chat 编码器。 +对于尚未缓存的 digest 固定镜像,此路径将 Docker 的 `repo:tag@sha256:...` +转换为 Pyxis/Enroot 的 `repo:sha256:...` manifest 引用。记录的镜像和 digest +保持不变,已有有效 squash 镜像仍优先使用。 ## 规程索引 diff --git a/infx/tests/srt_slurm/test_srt_single_node.py b/infx/tests/srt_slurm/test_srt_single_node.py index 3b9d265aba..501de0eb20 100644 --- a/infx/tests/srt_slurm/test_srt_single_node.py +++ b/infx/tests/srt_slurm/test_srt_single_node.py @@ -260,6 +260,7 @@ def test_submission_manifest(tmp_path, record, expected): ("b200-cw", "none"), ("b200-nb", "none"), ("b200-nscale-slurm", "none"), ("b200-nscale-slurm", "agentic"), ("b200-nscale-slurm", "download"), ("b200-nscale-slurm", "download-failure"), + ("b200-nscale-slurm", "download-cached"), ("b300-dsxe", "none"), ("mi300x-amd", "none"), ("mi325x-amds", "none"), ("mi355x-amds", "none"), ] + [(pool, "missing-recipe") for pool in ( @@ -270,10 +271,16 @@ def test_submission_manifest(tmp_path, record, expected): def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, pool, failure): path, recipe, point_env = point if failure.startswith("download"): - point_env = {**point_env, "MODEL": "deepseek-ai/DeepSeek-V4-Flash"} + point_env = { + **point_env, "MODEL": "deepseek-ai/DeepSeek-V4-Flash", + "IMAGE": f"example/server:nightly@sha256:{'a' * 64}", + } recipe["model"]["path"] = "hf:deepseek-ai/DeepSeek-V4-Flash" + recipe["model"]["container"] = point_env["IMAGE"] recipe["benchmark"]["env"]["MODEL"] = "deepseek-ai/DeepSeek-V4-Flash" path.write_text(yaml.safe_dump({"base": recipe})) + if failure == "download-cached": + (tmp_path / f"example_server_nightly_sha256_{'a' * 64}.sqsh").write_text("cache") binaries = tmp_path / "bin" binaries.mkdir() model = tmp_path / "model" @@ -301,6 +308,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "sacct": 'if [[ "$TEST_FAILURE" == allocation ]]; then echo "FAILED|1:0"; else echo "COMPLETED|0:0"; fi', "scancel": 'printf "%s\\n" "$@" >> "$CANCEL_CAPTURE"', "tail": 'exit 0', + "unsquashfs": '[[ "$TEST_FAILURE" == download-cached ]]', } for name, script in scripts.items(): binary = binaries / name @@ -358,7 +366,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, ["bash", str(ROOT / f"runners/launch_{pool}.sh")], cwd=tmp_path, env=env, capture_output=True, text=True, timeout=30, ) - assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0, "download": 0, "download-failure": 1}[failure], result.stderr + assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0, "download": 0, "download-failure": 1, "download-cached": 0}[failure], result.stderr if failure.startswith("download"): assert Path(env["DOWNLOAD_CAPTURE"]).read_text().splitlines() == [ "tool", "run", "--from", "huggingface-hub>=0.34,<2", "hf", "download", @@ -394,9 +402,16 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, assert json.loads((tmp_path / "gpu_metrics_context.json").read_text()) == {"device_count": 4} assert (tmp_path / "srt-single-node-logs.tar.gz").stat().st_size > 0 cluster_config = yaml.safe_load(next(tmp_path.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) - assert cluster_config["containers"]["test:tag"] == "test:tag" - assert cluster_config["use_exclusive_sbatch_directive"] is True if failure == "download": + assert cluster_config["containers"][point_env["IMAGE"]] == f"example/server:sha256:{'a' * 64}" + elif failure == "download-cached": + assert cluster_config["containers"][point_env["IMAGE"]] == str( + tmp_path / f"example_server_nightly_sha256_{'a' * 64}.sqsh" + ) + else: + assert cluster_config["containers"]["test:tag"] == "test:tag" + assert cluster_config["use_exclusive_sbatch_directive"] is True + if failure in {"download", "download-cached"}: assert cluster_config["model_paths"]["hf:deepseek-ai/DeepSeek-V4-Flash"] == ( "/data/home/sa-shared/gharunners/models/DeepSeek-V4-Flash" ) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 07f9100d20..cb7cf9087e 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8995,3 +8995,12 @@ - "Add DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled EAGLE MTP (3 steps, top-k 1, 4 draft tokens), real verification, and GPU-resident weights/KV. Stage the unstaged checkpoint and HF/Xet download caches on writable shared storage before SRT submission; stop on download failure." - "新增 DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,采用原生 EAGLE MTP(3 steps、top-k 1、4 draft tokens)、真实验证和 GPU 常驻权重/KV。SRT 提交前在共享可写目录下载 checkpoint 并设置 HF/Xet 下载缓存,下载失败时停止提交。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3517 + +- config-keys: + - dsv4flash-fp4-b200-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Fix B200 V4 Flash single-node Pyxis image import by converting uncached Docker @sha256 references to Enroot manifest-tag syntax. Preserve the pinned image digest and prefer valid cached squash images. No serving settings change." + - "修复 B200 V4 Flash 单节点 Pyxis 镜像导入:将尚未缓存的 Docker @sha256 引用转换为 Enroot manifest-tag 语法,保留固定镜像 digest 并优先使用有效 squash 缓存,不更改 serving 设置。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3517 diff --git a/runners/launch_b200-nscale-slurm.sh b/runners/launch_b200-nscale-slurm.sh index 4f10c19f6b..78b39b34f7 100755 --- a/runners/launch_b200-nscale-slurm.sh +++ b/runners/launch_b200-nscale-slurm.sh @@ -159,17 +159,6 @@ else exit 1 fi -if [[ "$LAUNCH_PATH" == native-single-node ]]; then - HF_HUB_CACHE_MOUNT=/data/home/sa-shared/gharunners/hf-hub-cache - SRT_MODEL_PATH="$MODEL_PATH" - # Models not staged locally resolve through the Hugging Face cache mount. - [[ "$SRT_MODEL_PATH" == /* ]] || SRT_MODEL_PATH="hf:$MODEL" - SRT_SQUASH_FILE="$B200_SQUASH_DIR/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" - launch_srt_single_node b200-nscale-slurm \ - --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" - exit $? -fi - # --------------------------------------------------------------------------- # Container import helpers shared by both srt-slurm paths # --------------------------------------------------------------------------- @@ -202,6 +191,26 @@ enroot_uri_for_image() { fi } +if [[ "$LAUNCH_PATH" == native-single-node ]]; then + HF_HUB_CACHE_MOUNT=/data/home/sa-shared/gharunners/hf-hub-cache + SRT_MODEL_PATH="$MODEL_PATH" + # Models not staged locally resolve through the Hugging Face cache mount. + [[ "$SRT_MODEL_PATH" == /* ]] || SRT_MODEL_PATH="hf:$MODEL" + SRT_SQUASH_FILE="$B200_SQUASH_DIR/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + SRT_IMAGE_ARGS=() + if [[ "$MODEL" == deepseek-ai/DeepSeek-V4-Flash && "$IMAGE" == *@sha256:* ]] && + ! { [[ -r "$SRT_SQUASH_FILE" ]] && unsquashfs -s "$SRT_SQUASH_FILE" >/dev/null 2>&1; }; then + # Pyxis interprets Docker's @digest as credentials; preserve the digest + # using Enroot's manifest-tag syntax, without its docker:// prefix. + SRT_IMAGE_URI=$(enroot_uri_for_image "$IMAGE") || exit 1 + SRT_IMAGE_ARGS=(--container "$IMAGE" "${SRT_IMAGE_URI#docker://}") + fi + launch_srt_single_node b200-nscale-slurm \ + --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" \ + "${SRT_IMAGE_ARGS[@]}" + exit $? +fi + # Import containers via enroot, serialized so concurrent runners on this # cluster don't race on the same squash file. import_squash() { From 5f6084fd5e6fb6889b40149d572acd2045d80f7b Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sun, 27 Sep 2026 18:44:32 +0800 Subject: [PATCH 4/6] feat: use EAGLE MTP 2 steps, top-k 1, 3 draft tokens --- .../srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml | 4 ++-- perf-changelog.yaml | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml index 5d15a30a20..600906e395 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml @@ -37,9 +37,9 @@ base: reasoning-parser: deepseek-v4 tool-call-parser: deepseekv4 speculative-algorithm: EAGLE - speculative-num-steps: 3 + speculative-num-steps: 2 speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 + speculative-num-draft-tokens: 3 watchdog-timeout: 3600 enable-metrics: true env: diff --git a/perf-changelog.yaml b/perf-changelog.yaml index cb7cf9087e..6f21b10050 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8992,8 +8992,8 @@ scenario-type: - fixed-seq-len description: - - "Add DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled EAGLE MTP (3 steps, top-k 1, 4 draft tokens), real verification, and GPU-resident weights/KV. Stage the unstaged checkpoint and HF/Xet download caches on writable shared storage before SRT submission; stop on download failure." - - "新增 DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,采用原生 EAGLE MTP(3 steps、top-k 1、4 draft tokens)、真实验证和 GPU 常驻权重/KV。SRT 提交前在共享可写目录下载 checkpoint 并设置 HF/Xet 下载缓存,下载失败时停止提交。" + - "Add DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled EAGLE MTP (2 steps, top-k 1, 3 draft tokens), real verification, and GPU-resident weights/KV. Stage the unstaged checkpoint and HF/Xet download caches on writable shared storage before SRT submission; stop on download failure." + - "新增 DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,采用原生 EAGLE MTP(2 steps、top-k 1、3 draft tokens)、真实验证和 GPU 常驻权重/KV。SRT 提交前在共享可写目录下载 checkpoint 并设置 HF/Xet 下载缓存,下载失败时停止提交。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3517 - config-keys: From c5ad47ad32a62187e27b88245d9e3fea114426a5 Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sun, 27 Sep 2026 22:32:34 +0800 Subject: [PATCH 5/6] docs: record EAGLE 2/1/3 for the B200 V4 Flash recipe --- docs/configuration-procedures.md | 2 +- docs/configuration-procedures_zh.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 06785960ad..5e79d69428 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -57,7 +57,7 @@ the serving process retains its container cache settings. For an uncached digest-pinned image, this lane maps Docker's `repo:tag@sha256:...` to Pyxis/Enroot's `repo:sha256:...` manifest reference. The recorded image and digest stay unchanged; a valid cached squash image still takes precedence. -The recipe uses bundled MTP through `EAGLE` (3 steps, top-k 1, 4 draft tokens), +The recipe uses bundled MTP through `EAGLE` (2 steps, top-k 1, 3 draft tokens), with the DeepSeek-V4 chat encoder selected by the client's `--dsv4` option. Launchers that use srt-slurm keep their cluster configuration in diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index f7894acf9e..c566c9da96 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -49,7 +49,7 @@ B200 Nscale 的固定序列 `deepseek-ai/DeepSeek-V4-Flash` 启动器在提交 S `/scratch/models` 中的预置副本。下载进程显式将 `HF_HOME`、`HF_HUB_CACHE` 和 `HF_XET_CACHE` 指向 checkpoint 的 `.cache/huggingface` 下的可写路径, 服务进程仍保留容器缓存设置。配方通过 `EAGLE` 使用原生 MTP -(3 steps、top-k 1、4 draft tokens),客户端以 `--dsv4` 选择 DeepSeek-V4 chat 编码器。 +(2 steps、top-k 1、3 draft tokens),客户端以 `--dsv4` 选择 DeepSeek-V4 chat 编码器。 对于尚未缓存的 digest 固定镜像,此路径将 Docker 的 `repo:tag@sha256:...` 转换为 Pyxis/Enroot 的 `repo:sha256:...` manifest 引用。记录的镜像和 digest 保持不变,已有有效 squash 镜像仍优先使用。 From 2e425b6e40d5a082013e3c2866482e3b2e5b83ad Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Sun, 27 Sep 2026 23:10:02 +0800 Subject: [PATCH 6/6] perf: run V4 Flash decode and speculative steps without graphs Set cuda-graph-backend-{decode,prefill}=disabled, which also skips target verify and draft capture, and drop the per-concurrency decode graph batch. --- .../srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml | 4 +++- perf-changelog.yaml | 4 ++-- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml index 600906e395..053e84466b 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsv4flash/sglang/b200-fp4-mtp/8k1k.yaml @@ -40,6 +40,9 @@ base: speculative-num-steps: 2 speculative-eagle-topk: 1 speculative-num-draft-tokens: 3 + # Eager decode and speculative verify/draft; no CUDA graph capture. + cuda-graph-backend-decode: disabled + cuda-graph-backend-prefill: disabled watchdog-timeout: 3600 enable-metrics: true env: @@ -61,7 +64,6 @@ zip_override_tp2: agg: args: max-running-requests: [1, 2, 4, 8, 16, 32, 64, 128] - cuda-graph-max-bs-decode: [1, 2, 4, 8, 16, 32, 64, 128] benchmark: env: CONC: ['1', '2', '4', '8', '16', '32', '64', '128'] diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 6f21b10050..edc8572d4c 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8992,8 +8992,8 @@ scenario-type: - fixed-seq-len description: - - "Add DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled EAGLE MTP (2 steps, top-k 1, 3 draft tokens), real verification, and GPU-resident weights/KV. Stage the unstaged checkpoint and HF/Xet download caches on writable shared storage before SRT submission; stop on download failure." - - "新增 DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,采用原生 EAGLE MTP(2 steps、top-k 1、3 draft tokens)、真实验证和 GPU 常驻权重/KV。SRT 提交前在共享可写目录下载 checkpoint 并设置 HF/Xet 下载缓存,下载失败时停止提交。" + - "Add DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving at concurrency 1, 2, 4, 8, 16, 32, 64, 128 with bundled EAGLE MTP (2 steps, top-k 1, 3 draft tokens) with CUDA/HIP graphs disabled for decode, prefill, and speculative verify/draft, real verification, and GPU-resident weights/KV. Stage the unstaged checkpoint and HF/Xet download caches on writable shared storage before SRT submission; stop on download failure." + - "新增 DeepSeek-V4-Flash B200 Nscale TP2 SGLang 8k1k serving,并发为 1、2、4、8、16、32、64、128,采用原生 EAGLE MTP(2 steps、top-k 1、3 draft tokens),decode、prefill 及投机验证/draft 均关闭 CUDA/HIP graph、真实验证和 GPU 常驻权重/KV。SRT 提交前在共享可写目录下载 checkpoint 并设置 HF/Xet 下载缓存,下载失败时停止提交。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3517 - config-keys: