From eafd4be3c02f05a8cf2dd882702fee5f3dab3008 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Mon, 21 Sep 2026 17:23:42 -0500 Subject: [PATCH 01/29] refactor: start native single-node SRT-Slurm migration MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 开始原生单节点 SRT-Slurm 迁移,添加 H200 SGLang 8k1k 并行候选配方并复用现有基准客户端;生产路由保持不变。 --- benchmarks/benchmark_lib.sh | 11 +- .../dsr1/sglang/h200-fp8/8k1k.yaml | 54 ++++++ benchmarks/single_node/srt_fixed_sequence.sh | 47 ++++++ docs/configuration-procedures.md | 4 + docs/configuration-procedures_zh.md | 2 + docs/single-node-srt-migration.md | 109 ++++++++++++ docs/single-node-srt-migration_zh.md | 63 +++++++ perf-changelog.yaml | 7 + utils/test_srt_fixed_sequence.py | 156 ++++++++++++++++++ 9 files changed, 452 insertions(+), 1 deletion(-) create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt_fixed_sequence.sh create mode 100644 docs/single-node-srt-migration.md create mode 100644 docs/single-node-srt-migration_zh.md create mode 100644 utils/test_srt_fixed_sequence.py diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index 60bdfecead..214d7ca7cb 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -800,6 +800,7 @@ run_benchmark_serving() { local model="" local port="" local backend="" + local base_url="" local endpoint="" local input_len="" local output_len="" @@ -834,6 +835,10 @@ run_benchmark_serving() { endpoint="$2" shift 2 ;; + --base-url) + base_url="$2" + shift 2 + ;; --input-len) input_len="$2" shift 2 @@ -954,12 +959,16 @@ run_benchmark_serving() { num_prompts="$max_concurrency" fi + if [[ -z "$base_url" ]]; then + base_url="http://0.0.0.0:$port" + fi + local benchmark_cmd=( env PYTHONPATH="$workspace_dir${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench_serving.benchmark_serving --model "$model" --backend "$backend" - --base-url "http://0.0.0.0:$port" + --base-url "$base_url" --dataset-name random --random-input-len "$input_len" --random-output-len "$output_len" diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml new file mode 100644 index 0000000000..6611a6bf9f --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml @@ -0,0 +1,54 @@ +# Parallel replacement for dsr1-fp8-h200-sglang; not selected by production yet. +# Each concurrency is a separate native SRT job, matching the current sweep. +base: + schema: 2 + name: dsr1-fp8-h200-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + PYTHONNOUSERSITE: "1" + TORCH_CUDA_ARCH_LIST: "9.0" + args: + tensor-parallel-size: 8 + data-parallel-size: 1 + trust-remote-code: true + disable-radix-cache: true + max-running-requests: 256 + cuda-graph-max-bs: 256 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + mem-fraction-static: 0.82 + attention-backend: flashinfer + stream-interval: 10 + decode-log-interval: 1 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: "8192" + OSL: "1024" + RANDOM_RANGE_RATIO: "0.8" + +zip_override_concurrency: + benchmark: + env: + CONC: ["4", "8", "16", "32", "64"] diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh new file mode 100644 index 0000000000..7972d16a45 --- /dev/null +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -0,0 +1,47 @@ +#!/usr/bin/env bash + +# SRT owns the server lifecycle; retain the existing InferenceX client and sampler. +set -eo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only +check_env_vars MODEL CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME RESULT_DIR \ + SRT_FRONTEND_HOST SRT_FRONTEND_PORT RUN_EVAL EVAL_ONLY GPU_MONITOR_INTERVAL +SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" + +for name in CONC ISL OSL SRT_FRONTEND_PORT GPU_MONITOR_INTERVAL; do + if [[ ! "${!name}" =~ ^[1-9][0-9]*$ ]]; then + echo "ERROR: $name must be a positive integer" >&2 + exit 1 + fi +done + +# The initial parallel port supports throughput only. Eval context and artifact +# forwarding must be connected before this becomes a workflow execution path. +if [[ "$RUN_EVAL" != false || "$EVAL_ONLY" != false ]]; then + echo "ERROR: the single-node SRT pilot does not support evals yet" >&2 + exit 1 +fi + +if [[ ! -d "$RESULT_DIR" ]]; then + echo "ERROR: RESULT_DIR must be an existing runtime-provided directory" >&2 + exit 1 +fi + +source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" +cd "$INFERENCEX_REPO_ROOT" +pip3 install --user --break-system-packages sentencepiece + +start_gpu_monitor --output "$RESULT_DIR/gpu_metrics.csv" --interval "$SRT_MONITOR_INTERVAL" +trap 'rc=$?; stop_gpu_monitor; exit "$rc"' EXIT + +run_benchmark_serving \ + --model "$MODEL" \ + --port "$SRT_FRONTEND_PORT" \ + --base-url "http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" \ + --backend vllm \ + --input-len "$ISL" \ + --output-len "$OSL" \ + --random-range-ratio "$RANDOM_RANGE_RATIO" \ + --num-prompts "$((CONC * 10))" \ + --max-concurrency "$CONC" \ + --result-filename "$RESULT_FILENAME" \ + --result-dir "$RESULT_DIR" diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index de9f1cda5c..385651bdba 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -171,6 +171,10 @@ Only fixed 8192/1024 `glm5.1-fp8-b200-tilert` requires native power. TileRT runs ## Register an srt-slurm recipe +For the parallel single-node migration and its activation checklist, see +[Single-node SRT-Slurm migration](./single-node-srt-migration.md). Its initial +recipe is not selected by the production master config or launcher yet. + Mapping source: [`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchmarks/multi_node/srt-slurm-recipes/RECIPES.md). Checked-in recipes: [`benchmarks/multi_node/srt-slurm-recipes/`](../benchmarks/multi_node/srt-slurm-recipes/). 1. Locate the exact upstream [NVIDIA/srt-slurm](https://github.com/NVIDIA/srt-slurm) recipe and record its commit-pinned source path. diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index b2e0ce2d60..8300757c79 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -148,6 +148,8 @@ B200 Nscale 的 GLM-5.1 可用 `MODEL_PATH` 指定已有共享权重,覆盖默 ## 注册 srt-slurm 配方 +并行构建的单节点迁移及其启用清单见[单节点 SRT-Slurm 迁移](./single-node-srt-migration_zh.md)。首个候选配方尚未接入生产主配置或启动器。 + 映射来源:[`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchmarks/multi_node/srt-slurm-recipes/RECIPES.md)。检入的配方:[`benchmarks/multi_node/srt-slurm-recipes/`](../benchmarks/multi_node/srt-slurm-recipes/)。 1. 定位精确的上游 [NVIDIA/srt-slurm](https://github.com/NVIDIA/srt-slurm) 配方,并记录固定到 commit 的来源路径。 diff --git a/docs/single-node-srt-migration.md b/docs/single-node-srt-migration.md new file mode 100644 index 0000000000..1cf14c8e41 --- /dev/null +++ b/docs/single-node-srt-migration.md @@ -0,0 +1,109 @@ +# Single-node SRT-Slurm migration + +**English** | [中文](./single-node-srt-migration_zh.md) + +This draft moves active single-node serving settings into native SRT-Slurm YAML. +The first candidate is built alongside the existing route; it is not a production +cutover. The existing master config, runner selection, dependency pin, and result +ingestion remain in use until the replacement has runtime evidence. + +## Ownership and scope + +Alec owns the single-node recipe migration. Cam owns the fork pin, AMD support, +and AMD runtime cleanup. Keep reusable execution changes small enough to send +upstream; the intended destination is NVIDIA SRT-Slurm, with the SemiAnalysisAI +fork as an intermediate dependency. Do not import the separate prepared-job or +publication systems from the earlier H100 pilot into this work. + +At InferenceX main `4ab85c1e`, the active master files contain 110 single-node +configurations; deprecated archives are excluded: + +| Vendor | SGLang | vLLM | TRT-LLM | ATOM | Total | +| --- | ---: | ---: | ---: | ---: | ---: | +| NVIDIA | 47 | 13 | 10 | 0 | 70 | +| AMD | 21 | 8 | 0 | 11 | 40 | + +This is an inventory of migration work, not a claim that all these paths are +supported by the current SRT fork. The AMD stack and non-Slurm execution need +their own capability checks. + +## First native recipe + +[`8k1k.yaml`](../benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml) +ports `dsr1-fp8-h200-sglang` from +[`dsr1_fp8_h200.sh`](../benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh). +It uses native schema 2 and native `zip_override_concurrency` expansion: one +TP8 aggregate worker on one H200 node, with separate jobs at concurrency +4, 8, 16, 32, and 64. It preserves the model, image, 8k1k workload, server +environment, and explicit serving flags from the active recipe. + +SRT owns allocation, container startup, endpoint selection, readiness, and server +cleanup. Account, partition, mounts, image caches, and exclusive allocation belong +to the cluster/launcher integration, not the workload YAML. The candidate disables +SRT observability to avoid adding a second telemetry workload beside the existing +GPU sampler; runtime qualification must still compare effective native defaults. + +The native `custom` benchmark invokes +[`srt_fixed_sequence.sh`](../benchmarks/single_node/srt_fixed_sequence.sh). +This client uses the existing `run_benchmark_serving` helper and GPU sampler. +It preserves `10 * concurrency` requests, `2 * concurrency` warmups, random length +variation, completions API behavior, and the existing JSON result format. The +helper now accepts an explicit base URL so the client reaches the endpoint SRT +selected. Existing callers retain their previous local endpoint. + +The client runs in the serving image and retains the legacy `sentencepiece` +installation. This is client compatibility glue, not a second server launcher. +The initial client rejects eval requests explicitly. Eval-only context handling, +eval artifacts, and cancellation qualification are required before cutover. + +## Runtime inputs + +The recipe provides model/workload inputs, including concurrency through native +override expansion. SRT provides `SRT_FRONTEND_HOST` and `SRT_FRONTEND_PORT`. +The launch integration must export `INFMAX_WORKSPACE` before submission; SRT +mounts it at `/infmax-workspace`. It must also supply these native overrides: + +| Native override | Caller-owned value | +| --- | --- | +| `benchmark.env.RESULT_FILENAME` | Existing InferenceX result basename | +| `benchmark.env.RESULT_DIR` | `/logs`, SRT's existing per-job artifact mount | +| `benchmark.env.GPU_MONITOR_INTERVAL` | Explicit sampling interval in seconds | +| `benchmark.env.RUN_EVAL` | `"false"` for the throughput pilot | +| `benchmark.env.EVAL_ONLY` | `"false"` for the throughput pilot | + +All environment values above are strings; use quoted YAML values with native +`--set`. Missing inputs fail before the client runs. No fallback runtime settings +are hidden in the workload recipe. Submit through the shared `apply_srt_recipe` +entrypoint so speculative ports later retain automatic golden-AL handling. + +For a local allocation render, with SRT dependencies installed: + +```bash +INFMAX_WORKSPACE="$PWD" PYTHONPATH=utils/srt-slurm/src srtctl dry-run \ + -f 'benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[0]' +python -m pytest utils/test_srt_fixed_sequence.py +``` + +Without a cluster profile this renders SRT's generic scheduling defaults. It +validates configuration structure; it does not qualify a cluster or benchmark. + +## Before enabling the replacement + +- Add first-class recipe selection to the master/matrix/workflow contract and + consume it in the existing pool launcher; retain one launcher per pool. +- Preserve both H200 runner paths: the current `h200` label includes + `h200-dgxc-slurm` and `h200-cw`. Do not silently drop CoreWeave or pretend its + Docker execution is already covered by a Slurm recipe. +- Stage the candidate in the job-local checkout and bind caller inputs using + native `--set`; retain `nodes:1` and the existing result filename/metadata. +- Connect eval context, real-verification evals, result/eval artifact staging, + and GPU power collection to the workflow without changing publication format. +- Compare legacy and native commands, then qualify startup, throughput, accuracy, + power, cancellation, and cleanup on the same image/model/hardware. Coordinate + existing smoke/vendor evaluation work rather than duplicating it. +- Expand to the other active single-node recipes after this path is qualified, + including speculative decoding and KV offload. Coordinate AMD capabilities + with Cam's fork work. Retire legacy scripts only after their callers move. + +Keep the migration PR draft. Local schema checks and stubbed client tests do not +establish performance parity or GPU qualification. diff --git a/docs/single-node-srt-migration_zh.md b/docs/single-node-srt-migration_zh.md new file mode 100644 index 0000000000..854c3e9182 --- /dev/null +++ b/docs/single-node-srt-migration_zh.md @@ -0,0 +1,63 @@ +# 单节点 SRT-Slurm 迁移 + +[English](./single-node-srt-migration.md) | **中文** + +本草稿将仍在使用的单节点服务设置迁移到原生 SRT-Slurm YAML。首个候选实现与现有路径并行构建,尚未切换生产路由。在取得实际运行证据之前,主配置、runner 选择、依赖固定版本和结果导入流程继续沿用现有实现。 + +## 分工与范围 + +Alec 负责单节点配方迁移。Cam 负责固定分叉版本、AMD 支持及 AMD 运行时清理。可复用的执行改动应保持小而独立,便于提交上游;最终目标是使用 NVIDIA SRT-Slurm,SemiAnalysisAI 分叉只是过渡依赖。本项工作不引入此前 H100 试点的独立 prepared-job 或发布系统。 + +在 InferenceX main `4ab85c1e` 中,排除 deprecated 归档后,活跃主配置包含 110 个单节点配置: + +| 厂商 | SGLang | vLLM | TRT-LLM | ATOM | 合计 | +| --- | ---: | ---: | ---: | ---: | ---: | +| NVIDIA | 47 | 13 | 10 | 0 | 70 | +| AMD | 21 | 8 | 0 | 11 | 40 | + +这是待迁移清单,不代表当前 SRT 分叉已支持所有路径。AMD 功能和非 Slurm 执行需要分别确认。 + +## 首个原生配方 + +[`8k1k.yaml`](../benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml) 对应活跃配置 `dsr1-fp8-h200-sglang`,来源为 [`dsr1_fp8_h200.sh`](../benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh)。它使用原生 schema 2 和 `zip_override_concurrency` 展开:在一个 H200 节点上启动一个 TP8 聚合 worker,并发 4、8、16、32、64 分别运行独立作业。模型、镜像、8k1k 工作负载、服务环境和显式服务参数均取自现有配方。 + +SRT 负责资源分配、容器启动、端点选择、就绪检查及服务清理。账号、分区、挂载、镜像缓存和独占分配属于集群与启动器集成,不放入工作负载 YAML。候选配方关闭 SRT observability,避免在现有 GPU 采样器之外增加一套遥测负载;验收时仍须比较实际生效的原生默认设置。 + +原生 `custom` benchmark 调用 [`srt_fixed_sequence.sh`](../benchmarks/single_node/srt_fixed_sequence.sh),复用现有 `run_benchmark_serving` 和 GPU 采样器。它保留 `10 * concurrency` 个请求、`2 * concurrency` 次预热、随机长度变化、completions API 行为和现有 JSON 结果格式。共享 helper 新增显式 base URL 参数,让客户端连接 SRT 选定的端点;现有调用仍使用原来的本地端点。 + +客户端在服务镜像中运行,并保留原有 `sentencepiece` 安装步骤。这是客户端兼容胶水,不是第二套服务启动器。初版客户端明确拒绝 eval 请求;切换前必须完成 eval-only 上下文处理、评测产物接入和取消验收。 + +## 运行时输入 + +配方提供模型及工作负载参数,并通过原生覆盖展开并发。SRT 提供 `SRT_FRONTEND_HOST` 和 `SRT_FRONTEND_PORT`。启动集成须在提交前导出 `INFMAX_WORKSPACE`,由 SRT 挂载到 `/infmax-workspace`,并提供以下原生覆盖: + +| 原生覆盖字段 | 调用方负责的值 | +| --- | --- | +| `benchmark.env.RESULT_FILENAME` | 现有 InferenceX 结果文件基本名称 | +| `benchmark.env.RESULT_DIR` | `/logs`,SRT 已创建的作业产物挂载 | +| `benchmark.env.GPU_MONITOR_INTERVAL` | 显式指定的采样间隔秒数 | +| `benchmark.env.RUN_EVAL` | 吞吐试点使用 `"false"` | +| `benchmark.env.EVAL_ONLY` | 吞吐试点使用 `"false"` | + +上述环境值均为字符串;通过原生 `--set` 传递时使用带引号的 YAML 值。缺失输入会在客户端启动前报错,工作负载配方不隐藏运行时回退设置。提交应经过共享 `apply_srt_recipe`,确保后续推测解码迁移保留自动 golden-AL 选择。 + +安装 SRT 依赖后,可在本地渲染资源分配: + +```bash +INFMAX_WORKSPACE="$PWD" PYTHONPATH=utils/srt-slurm/src srtctl dry-run \ + -f 'benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[0]' +python -m pytest utils/test_srt_fixed_sequence.py +``` + +未提供集群 profile 时,该命令使用 SRT 的通用调度默认值。它验证配置结构,不代表集群或基准已经验收。 + +## 启用替代路径之前 + +- 在主配置、矩阵和工作流中增加一等配方选择字段,并由现有池启动器消费,保持每个池只有一个启动器。 +- 保留两个 H200 runner 路径:当前 `h200` 标签同时包含 `h200-dgxc-slurm` 和 `h200-cw`。不能静默删除 CoreWeave 覆盖,也不能把其 Docker 执行视为已被 Slurm 配方覆盖。 +- 在作业独立检出目录中准备候选配方,通过原生 `--set` 绑定调用方输入,并保留 `nodes:1`、现有结果文件名和元数据。 +- 接入 eval 上下文、真实验证评测、结果及评测产物准备和 GPU 功耗收集,不改变发布格式。 +- 比较旧路径与原生路径的命令,在相同镜像、模型和硬件上验收启动、吞吐、准确性、功耗、取消及清理。协调现有 smoke/vendor 评测工作,避免重复执行。 +- 首条路径验收后,再扩展到其他活跃单节点配方,包括推测解码和 KV offload;AMD 能力与 Cam 的分叉工作协调。只有调用方完成迁移后才删除旧脚本。 + +迁移 PR 保持草稿。本地 schema 检查和使用外部进程桩的客户端测试不能证明性能一致或 GPU 验收通过。 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 515e02bd84..d9b45a3c34 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8542,3 +8542,10 @@ - "为 GB300 vLLM DeepSeek-V4.1-Flash AgentX 配方在现有 TP4 臂旁新增 TP2 臂,Engram 表继续通过 --engram-config cpu_offload 放在固定页主机 DRAM;每张 277 GiB GPU 的权重升至约 175 GiB" - "在 dsv41flash_fp4_vllm_mtp.sh 中将 B200 TP2 的上限推广到所有 TP2 臂:--max-num-batched-tokens 4096(上游 16384 时 indexer 的 [batched-tokens, 1M] fp8 缓冲区达 32 GiB)、--max-num-seqs 为并发的两倍(16-256)、CUDA graph 捕获上限 512,为每张 GPU 留出约 36 GiB KV;TP4 与 TP8 臂沿用上游默认值" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3321 + +- config-keys: + - dsr1-fp8-h200-sglang + description: + - "Begin the parallel single-node SRT-Slurm port with native H200 SGLang 8k1k serving settings and the existing InferenceX benchmark client; production routing is unchanged and GPU parity remains unqualified" + - "开始并行构建单节点 SRT-Slurm 迁移:采用原生 H200 SGLang 8k1k 服务设置并复用现有 InferenceX 基准客户端;生产路由未变,GPU 性能一致性尚未验收" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX diff --git a/utils/test_srt_fixed_sequence.py b/utils/test_srt_fixed_sequence.py new file mode 100644 index 0000000000..470e33fa2b --- /dev/null +++ b/utils/test_srt_fixed_sequence.py @@ -0,0 +1,156 @@ +"""Run the shared client against stubbed external benchmark/GPU processes.""" + +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +CLIENT = ROOT / "benchmarks/single_node/srt_fixed_sequence.sh" + + +@pytest.fixture +def client_environment(tmp_path): + binaries = tmp_path / "bin" + binaries.mkdir() + benchmark = binaries / "python3" + benchmark.write_text( + f"#!{sys.executable}\n" + "import json, os, pathlib, sys\n" + "pathlib.Path(os.environ['CAPTURE']).write_text(json.dumps(sys.argv[1:]))\n" + "sys.exit(int(os.environ['CLIENT_EXIT']))\n" + ) + benchmark.chmod(0o755) + for name, body in { + "pip3": "exit 0\n", + "nvidia-smi": "printf 'timestamp,index,power.draw\\n'\n", + }.items(): + binary = binaries / name + binary.write_text(f"#!/bin/bash\n{body}") + binary.chmod(0o755) + env = { + **os.environ, + "PATH": f"{binaries}:{os.environ['PATH']}", + "MODEL": "test/model", + "CONC": "3", + "ISL": "128", + "OSL": "64", + "RANDOM_RANGE_RATIO": "0.5", + "RESULT_FILENAME": "test-result", + "RESULT_DIR": str(tmp_path), + "SRT_FRONTEND_HOST": "10.2.3.4", + "SRT_FRONTEND_PORT": "9444", + "RUN_EVAL": "false", + "EVAL_ONLY": "false", + "GPU_MONITOR_INTERVAL": "2", + "IS_AGENTIC": "0", + "SCENARIO_TYPE": "fixed-seq-len", + "CLIENT_EXIT": "0", + "CAPTURE": str(tmp_path / "argv.json"), + } + for key in ("PROFILE", "INFERENCEX_SERVER_PID", "INFERENCEX_SERVER_STATE"): + env.pop(key, None) + return env + + +@pytest.mark.parametrize("exit_code", [0, 7]) +def test_native_endpoint_preserves_client_settings_and_failure( + client_environment, exit_code +): + env = {**client_environment, "CLIENT_EXIT": str(exit_code)} + result = subprocess.run( + ["bash", str(CLIENT)], env=env, capture_output=True, text=True + ) + assert result.returncode == exit_code, result.stderr + argv = json.loads(Path(env["CAPTURE"]).read_text()) + assert argv == [ + "-m", + "infx.bench_serving.benchmark_serving", + "--model", + "test/model", + "--backend", + "vllm", + "--base-url", + "http://10.2.3.4:9444", + "--dataset-name", + "random", + "--random-input-len", + "128", + "--random-output-len", + "64", + "--random-range-ratio", + "0.5", + "--num-prompts", + "30", + "--max-concurrency", + "3", + "--request-rate", + "inf", + "--ignore-eos", + "--save-result", + "--num-warmups", + "6", + "--percentile-metrics", + "ttft,tpot,itl,e2el", + "--result-dir", + env["RESULT_DIR"], + "--result-filename", + "test-result.json", + ] + assert ( + (Path(env["RESULT_DIR"]) / "gpu_metrics.csv") + .read_text() + .startswith("timestamp") + ) + + +@pytest.mark.parametrize( + ("key", "value", "error"), + [ + ("MODEL", None, "MODEL"), + ("GPU_MONITOR_INTERVAL", None, "GPU_MONITOR_INTERVAL"), + ("CONC", "0", "CONC must be a positive integer"), + ("RUN_EVAL", "true", "does not support evals yet"), + ("EVAL_ONLY", "true", "does not support evals yet"), + ], +) +def test_invalid_runtime_inputs_fail_before_the_client( + client_environment, key, value, error +): + env = dict(client_environment) + if value is None: + env.pop(key) + else: + env[key] = value + result = subprocess.run( + ["bash", str(CLIENT)], env=env, capture_output=True, text=True + ) + assert result.returncode != 0 + assert error in result.stdout + result.stderr + assert not Path(env["CAPTURE"]).exists() + + +def test_legacy_client_keeps_its_local_endpoint(client_environment): + env = client_environment + result = subprocess.run( + [ + "bash", + "-c", + """source "$1/benchmarks/benchmark_lib.sh" +run_benchmark_serving --model test/model --port 8888 --backend vllm \\ + --input-len 128 --output-len 64 --random-range-ratio 0.5 \\ + --num-prompts 30 --max-concurrency 3 --result-filename old --result-dir "$RESULT_DIR" +""", + "bash", + str(ROOT), + ], + env=env, + capture_output=True, + text=True, + ) + assert result.returncode == 0, result.stderr + argv = json.loads(Path(env["CAPTURE"]).read_text()) + assert argv[argv.index("--base-url") + 1] == "http://0.0.0.0:8888" From 81f65a0852c6edb2623e802c1418baad39a3cf66 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Mon, 21 Sep 2026 17:25:18 -0500 Subject: [PATCH 02/29] docs: link single-node migration changelog to draft PR MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为单节点迁移的 changelog 条目补充草稿 PR 链接。 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index d9b45a3c34..4a2d3ce7b9 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8548,4 +8548,4 @@ description: - "Begin the parallel single-node SRT-Slurm port with native H200 SGLang 8k1k serving settings and the existing InferenceX benchmark client; production routing is unchanged and GPU parity remains unqualified" - "开始并行构建单节点 SRT-Slurm 迁移:采用原生 H200 SGLang 8k1k 服务设置并复用现有 InferenceX 基准客户端;生产路由未变,GPU 性能一致性尚未验收" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 From 80a4da382f46baf42f6e9068b908e685b64f2b7d Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:05:15 -0500 Subject: [PATCH 03/29] feat: wire native H200 SRT pilot into end-to-end workflow MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将原生 H200 SRT 试点接入端到端工作流,显式校验配方、保留结果和功耗产物,并保持生产路由不变。 --- .github/workflows/benchmark-tmpl.yml | 8 ++ .../dsr1/sglang/h200-fp8/8k1k.yaml | 2 + benchmarks/single_node/srt_fixed_sequence.sh | 2 +- configs/pilots/h200-srt.yaml | 17 +++ docs/single-node-srt-migration.md | 41 ++++++- docs/single-node-srt-migration_zh.md | 23 +++- infx/matrix/generate.py | 2 + infx/matrix/validation.py | 3 + infx/srt_slurm/cluster_config.py | 3 + infx/srt_slurm/single_node.py | 106 ++++++++++++++++++ infx/srt_slurm/synthetic_acceptance.py | 1 + perf-changelog.yaml | 7 ++ runners/launch_h200-dgxc-slurm.sh | 81 ++++++++++++- runners/runtime_settings.sh | 1 + .../test_generate_sweep_configs.py | 14 +++ utils/test_srt_cluster_config.py | 14 +++ utils/test_srt_single_node.py | 95 ++++++++++++++++ 17 files changed, 409 insertions(+), 11 deletions(-) create mode 100644 configs/pilots/h200-srt.yaml create mode 100644 infx/srt_slurm/single_node.py create mode 100644 utils/test_srt_single_node.py diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index 11b14aadec..565d61d32f 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -110,6 +110,7 @@ env: HF_HUB_CACHE: '/mnt/hf_hub_cache/' EXP_NAME: ${{ fromJSON(inputs.config).exp-name }} RECIPE_FINGERPRINT: ${{ fromJSON(inputs.config).recipe-fingerprint || '' }} + SRT_RECIPE: ${{ fromJSON(inputs.config).srt-recipe || '' }} MODEL: ${{ fromJSON(inputs.config).model }} THINKING_MODE: thinking_on MODEL_PREFIX: ${{ fromJSON(inputs.config).model-prefix }} @@ -305,6 +306,10 @@ jobs: if [[ -f runners/runtime_settings.sh ]]; then source runners/runtime_settings.sh fi + if [[ -n "$SRT_RECIPE" && "${RUNNER_NAME%%_*}" != h200-dgxc-slurm ]]; then + echo "The native single-node SRT pilot requires the H200 DGXC Slurm pool" >&2 + exit 1 + fi bash "./runners/launch_${RUNNER_NAME%%_*}.sh" if [ "${EVAL_ONLY}" = "true" ]; then @@ -383,6 +388,9 @@ jobs: server.log results/*.log results/*_config.json + srt-single-node-logs.tar.gz + srt-single-node-submission.json + srt-slurm-sha.txt if-no-files-found: ignore - name: Upload GPU metrics diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml index 6611a6bf9f..39498a632c 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml @@ -30,6 +30,8 @@ base: tensor-parallel-size: 8 data-parallel-size: 1 trust-remote-code: true + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false disable-radix-cache: true max-running-requests: 256 cuda-graph-max-bs: 256 diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index 7972d16a45..9d55137805 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -15,7 +15,7 @@ for name in CONC ISL OSL SRT_FRONTEND_PORT GPU_MONITOR_INTERVAL; do done # The initial parallel port supports throughput only. Eval context and artifact -# forwarding must be connected before this becomes a workflow execution path. +# forwarding must be connected before production cutover. if [[ "$RUN_EVAL" != false || "$EVAL_ONLY" != false ]]; then echo "ERROR: the single-node SRT pilot does not support evals yet" >&2 exit 1 diff --git a/configs/pilots/h200-srt.yaml b/configs/pilots/h200-srt.yaml new file mode 100644 index 0000000000..d0ca0a71d4 --- /dev/null +++ b/configs/pilots/h200-srt.yaml @@ -0,0 +1,17 @@ +# Opt-in native SRT pilot. Production coverage remains in nvidia-master.yaml. +dsr1-fp8-h200-sglang: + image: lmsysorg/sglang:v0.5.12-cu130 + model: deepseek-ai/DeepSeek-R1-0528 + model-prefix: dsr1 + runner: cluster:h200-dgxc + precision: fp8 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - tp: 8 + conc-list: [4] + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml:base diff --git a/docs/single-node-srt-migration.md b/docs/single-node-srt-migration.md index 1cf14c8e41..ec345c3979 100644 --- a/docs/single-node-srt-migration.md +++ b/docs/single-node-srt-migration.md @@ -87,17 +87,46 @@ python -m pytest utils/test_srt_fixed_sequence.py Without a cluster profile this renders SRT's generic scheduling defaults. It validates configuration structure; it does not qualify a cluster or benchmark. +## Opt-in workflow pilot + +[`configs/pilots/h200-srt.yaml`](../configs/pilots/h200-srt.yaml) selects only +`cluster:h200-dgxc`, 8k1k, TP8, concurrency 4. The search-space `srt-recipe` field +passes a native file/selector through the matrix and workflow to the existing +H200 pool launcher. Production `h200` coverage, including CoreWeave, is unchanged. + +The launcher checks the recipe's model, image, precision, topology, and workload +against matrix metadata before submission. It resolves the model and requested +image to their existing staged cluster assets, requests an exclusive node, and +binds concurrency and artifact inputs with native `--set`. Missing assets fail +readiness instead of starting a second download or using a different image. +Plain `sglang` submissions use the shared automatic acceptance connector. + +Submission uses native JSON output. The launcher waits for a successful Slurm +allocation exit, preserves the result basename, and stages raw results and GPU +sampling sidecars for the existing processor and uploads. A native log archive, +submission manifest, and SRT commit identify the run. Failed jobs retain available +artifacts; cancellation targets only the submitted job. + +Dispatch the workflow definition from the draft branch, with `ref` set to the +exact pushed commit: + +```bash +gh workflow run e2e-tests.yml --ref codex/single-node-srt-slurm \ + -f ref= -f test-name='native H200 SRT pilot' \ + -f generate-cli-command='test-config --config-keys dsr1-fp8-h200-sglang --config-file configs/pilots/h200-srt.yaml --no-evals' \ + -f require-power=true +``` + +The pilot requires `--no-evals`: evals are still rejected explicitly. A passing +throughput run alone is not accuracy or performance-parity qualification. + ## Before enabling the replacement -- Add first-class recipe selection to the master/matrix/workflow contract and - consume it in the existing pool launcher; retain one launcher per pool. - Preserve both H200 runner paths: the current `h200` label includes `h200-dgxc-slurm` and `h200-cw`. Do not silently drop CoreWeave or pretend its Docker execution is already covered by a Slurm recipe. -- Stage the candidate in the job-local checkout and bind caller inputs using - native `--set`; retain `nodes:1` and the existing result filename/metadata. -- Connect eval context, real-verification evals, result/eval artifact staging, - and GPU power collection to the workflow without changing publication format. +- Connect eval context, real-verification evals, and eval artifact staging; + qualify the wired result and GPU power paths without changing publication format. - Compare legacy and native commands, then qualify startup, throughput, accuracy, power, cancellation, and cleanup on the same image/model/hardware. Coordinate existing smoke/vendor evaluation work rather than duplicating it. diff --git a/docs/single-node-srt-migration_zh.md b/docs/single-node-srt-migration_zh.md index 854c3e9182..94f536c07b 100644 --- a/docs/single-node-srt-migration_zh.md +++ b/docs/single-node-srt-migration_zh.md @@ -51,12 +51,29 @@ python -m pytest utils/test_srt_fixed_sequence.py 未提供集群 profile 时,该命令使用 SRT 的通用调度默认值。它验证配置结构,不代表集群或基准已经验收。 +## 显式启用的工作流试点 + +[`configs/pilots/h200-srt.yaml`](../configs/pilots/h200-srt.yaml) 仅选择 `cluster:h200-dgxc`、8k1k、TP8 和并发 4。搜索空间的 `srt-recipe` 字段将原生文件及选择器经矩阵和工作流传给现有 H200 池启动器。生产 `h200` 覆盖(包括 CoreWeave)保持不变。 + +启动器在提交前核对配方与矩阵中的模型、镜像、精度、拓扑和工作负载。它使用集群已暂存的模型及指定镜像,申请独占节点,并通过原生 `--set` 绑定并发与产物参数。资源缺失会在就绪检查时失败,不会重复下载或换用其他镜像。普通 `sglang` 提交也经过共享的自动 acceptance 连接器。 + +提交使用原生 JSON 输出。启动器验证 Slurm 分配成功结束,保留结果文件名,并将原始结果和 GPU 采样附属文件交给现有处理及上传流程。原生日志归档、提交清单和 SRT commit 用于追踪运行来源。失败时保留已有产物,取消操作仅针对本次提交的作业。 + +从草稿分支触发工作流,将 `ref` 设为已推送的准确 commit: + +```bash +gh workflow run e2e-tests.yml --ref codex/single-node-srt-slurm \ + -f ref= -f test-name='native H200 SRT pilot' \ + -f generate-cli-command='test-config --config-keys dsr1-fp8-h200-sglang --config-file configs/pilots/h200-srt.yaml --no-evals' \ + -f require-power=true +``` + +试点必须传 `--no-evals`,目前仍明确拒绝 eval。一次吞吐运行通过不代表准确性或性能一致性已验收。 + ## 启用替代路径之前 -- 在主配置、矩阵和工作流中增加一等配方选择字段,并由现有池启动器消费,保持每个池只有一个启动器。 - 保留两个 H200 runner 路径:当前 `h200` 标签同时包含 `h200-dgxc-slurm` 和 `h200-cw`。不能静默删除 CoreWeave 覆盖,也不能把其 Docker 执行视为已被 Slurm 配方覆盖。 -- 在作业独立检出目录中准备候选配方,通过原生 `--set` 绑定调用方输入,并保留 `nodes:1`、现有结果文件名和元数据。 -- 接入 eval 上下文、真实验证评测、结果及评测产物准备和 GPU 功耗收集,不改变发布格式。 +- 接入 eval 上下文、真实验证评测和评测产物准备,并验收已接入的结果与 GPU 功耗路径,不改变发布格式。 - 比较旧路径与原生路径的命令,在相同镜像、模型和硬件上验收启动、吞吐、准确性、功耗、取消及清理。协调现有 smoke/vendor 评测工作,避免重复执行。 - 首条路径验收后,再扩展到其他活跃单节点配方,包括推测解码和 KV offload;AMD 能力与 Cam 的分叉工作协调。只有调用方完成迁移后才删除旧脚本。 diff --git a/infx/matrix/generate.py b/infx/matrix/generate.py index 5bbad46d3c..af7ec486fd 100644 --- a/infx/matrix/generate.py +++ b/infx/matrix/generate.py @@ -850,6 +850,8 @@ def _fixed_sequence_entries( Fields.SPEC_DECODING.value: spec_decoding, } ) + if benchmark.get(Fields.SRT_RECIPE.value) is not None: + entry[Fields.SRT_RECIPE.value] = benchmark[Fields.SRT_RECIPE.value] entry.update( { Fields.EXP_NAME.value: f"{model_code}_{seq_len_to_str(isl, osl)}", diff --git a/infx/matrix/validation.py b/infx/matrix/validation.py index d3503c7f2f..7f774570b9 100644 --- a/infx/matrix/validation.py +++ b/infx/matrix/validation.py @@ -45,6 +45,7 @@ class Fields(Enum): # Search-space/benchmark fields TP = "tp" + SRT_RECIPE = "srt-recipe" PP = "pp" DCP_SIZE = "dcp-size" PCP_SIZE = "pcp-size" @@ -162,6 +163,7 @@ class SingleNodeMatrixEntry(BaseModel): model_config = ConfigDict(extra="forbid", populate_by_name=True) image: str + srt_recipe: str | None = Field(default=None, alias=Fields.SRT_RECIPE.value, min_length=1) model: str model_prefix: str = Field(alias=Fields.MODEL_PREFIX.value) precision: str @@ -534,6 +536,7 @@ class SingleNodeSearchSpaceEntry(BaseModel): model_config = ConfigDict(extra="forbid", populate_by_name=True) tp: int + srt_recipe: str | None = Field(default=None, alias=Fields.SRT_RECIPE.value, min_length=1) pp: int = Field(default=1, gt=0, strict=True) dcp_size: int = Field(default=1, alias=Fields.DCP_SIZE.value, gt=0, strict=True) pcp_size: int = Field(default=1, alias=Fields.PCP_SIZE.value, gt=0, strict=True) diff --git a/infx/srt_slurm/cluster_config.py b/infx/srt_slurm/cluster_config.py index 4ce1154400..4990692541 100644 --- a/infx/srt_slurm/cluster_config.py +++ b/infx/srt_slurm/cluster_config.py @@ -37,6 +37,7 @@ def main() -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("profile", type=Path) parser.add_argument("output", type=Path) + parser.add_argument("--exclusive", action="store_true", help="Request exclusive Slurm nodes") parser.add_argument("--var", nargs=2, action="append", default=[], metavar=("NAME", "VALUE")) parser.add_argument("--model", nargs=2, action="append", default=[], metavar=("ALIAS", "PATH")) parser.add_argument( @@ -55,6 +56,8 @@ def main() -> None: dict(args.var), {"model_paths": args.model, "containers": args.container, "default_mounts": args.mount}, ) + if args.exclusive: + config["use_exclusive_sbatch_directive"] = True args.output.write_text(yaml.safe_dump(config, sort_keys=False)) except (OSError, ValueError, KeyError, yaml.YAMLError) as exc: parser.error(str(exc)) diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py new file mode 100644 index 0000000000..d0b0f5c6d3 --- /dev/null +++ b/infx/srt_slurm/single_node.py @@ -0,0 +1,106 @@ +"""Bind a native single-node SRT recipe to one fixed-sequence matrix point.""" + +from __future__ import annotations + +import argparse +import json +import os +from collections.abc import Mapping +from pathlib import Path + +import yaml + +from infx.srt_slurm.synthetic_acceptance import selected_recipes + + +def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: + """Reject mismatched metadata before allocation; retain recipe-owned server settings.""" + path, _, selector = config.partition(":") + raw = yaml.safe_load(Path(path).read_text()) + if not isinstance(raw, dict): + raise ValueError("Recipe must be a mapping") + recipes = selected_recipes(raw, selector or None) + if len(recipes) != 1: + raise ValueError("A single-node matrix point must select exactly one SRT recipe") + recipe = recipes[0][1] + role = recipe["roles"]["agg"] + args = role["args"] + benchmark = recipe["benchmark"] + workload = benchmark["env"] + expected = { + "engine": (recipe["engine"], environment["FRAMEWORK"]), + "model": (recipe["model"]["path"], f"hf:{environment['MODEL']}"), + "image": (recipe["model"]["container"], environment["IMAGE"]), + "precision": (recipe["model"]["precision"], environment["PRECISION"]), + "tensor-parallel-size": (args["tensor-parallel-size"], int(environment["TP"])), + "data-parallel-size": (args["data-parallel-size"], 1), + "gpus": (role["gpus"], int(environment["GPU_COUNT"])), + "nodes": (role["nodes"], 1), + "workers": (role["workers"], 1), + "roles": (set(recipe["roles"]), {"agg"}), + "benchmark type": (benchmark["type"], "custom"), + "benchmark MODEL": (workload["MODEL"], environment["MODEL"]), + } + for name in ("ISL", "OSL", "RANDOM_RANGE_RATIO"): + expected[name] = (str(workload[name]), environment[name]) + # Other topology/eval paths remain on their current launchers until ported. + for name, value in { + "FRAMEWORK": "sglang", + "PP_SIZE": "1", + "DCP_SIZE": "1", + "PCP_SIZE": "1", + "EP_SIZE": "1", + "DP_ATTENTION": "false", + "IS_AGENTIC": "0", + "SPEC_DECODING": "none", + "RUN_EVAL": "false", + "EVAL_ONLY": "false", + }.items(): + expected[name] = (environment[name], value) + for name, (actual, wanted) in expected.items(): + if actual != wanted: + raise ValueError(f"Single-node SRT {name}: recipe/matrix {actual!r} != {wanted!r}") + overrides = [] + for name in ("CONC", "RESULT_FILENAME", "GPU_MONITOR_INTERVAL", "RUN_EVAL", "EVAL_ONLY"): + value = environment[name] + if not value: + raise ValueError(f"Missing runtime input: {name}") + overrides += ["--set", f"benchmark.env.{name}={json.dumps(value)}"] + return [*overrides, "--set", 'benchmark.env.RESULT_DIR="/logs"'] + + +def submission_fields(path: Path) -> tuple[str, str]: + """Accept exactly one successful native JSON submission, never scrape prose.""" + record = json.loads(path.read_text()) + if record.get("status") != "submitted": + raise ValueError("SRT did not submit a job") + job_id = str(record["slurm_job_id"]) + output = str(record["output_dir"]) + if not job_id.isascii() or not job_id.isdecimal() or int(job_id) <= 0: + raise ValueError("Invalid SRT Slurm job ID") + if not Path(output).is_absolute() or "\n" in output: + raise ValueError("SRT output directory must be absolute") + return job_id, output + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + commands = parser.add_subparsers(dest="command", required=True) + prepare = commands.add_parser("prepare") + prepare.add_argument("recipe") + prepare.add_argument("output", type=Path) + submitted = commands.add_parser("submission") + submitted.add_argument("manifest", type=Path) + parsed = parser.parse_args() + try: + if parsed.command == "prepare": + arguments = runtime_arguments(parsed.recipe, os.environ) + parsed.output.write_bytes("\0".join([*arguments, ""]).encode()) + else: + print("\n".join(submission_fields(parsed.manifest))) + except (OSError, ValueError, KeyError, TypeError, yaml.YAMLError) as exc: + parser.error(str(exc)) + + +if __name__ == "__main__": + main() diff --git a/infx/srt_slurm/synthetic_acceptance.py b/infx/srt_slurm/synthetic_acceptance.py index 95d06b7763..edd6990a51 100644 --- a/infx/srt_slurm/synthetic_acceptance.py +++ b/infx/srt_slurm/synthetic_acceptance.py @@ -19,6 +19,7 @@ GOLDEN_DIR = Path(__file__).resolve().parents[2] / "golden_al_distribution" ENGINES = { + "sglang": "sglang", "vllm": "vllm", "dynamo-vllm": "vllm", "dynamo-sglang": "sglang", diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 4a2d3ce7b9..ea7d1439d4 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8549,3 +8549,10 @@ - "Begin the parallel single-node SRT-Slurm port with native H200 SGLang 8k1k serving settings and the existing InferenceX benchmark client; production routing is unchanged and GPU parity remains unqualified" - "开始并行构建单节点 SRT-Slurm 迁移:采用原生 H200 SGLang 8k1k 服务设置并复用现有 InferenceX 基准客户端;生产路由未变,GPU 性能一致性尚未验收" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp8-h200-sglang + description: + - "Wire the opt-in H200 native SRT single-node pilot through the matrix, launcher, result and GPU power paths; keep production routing unchanged pending qualification" + - "将显式启用的 H200 原生 SRT 单节点试点接入矩阵、启动器、结果及 GPU 功耗流程;验收完成前保留现有生产路由" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index 36405594c2..ad95379635 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -15,7 +15,86 @@ set -x source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 -if [[ "$IS_MULTINODE" == "true" ]]; then +EXECUTION_PATH=legacy-single-node +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi + +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + check_env_vars GITHUB_WORKSPACE SRT_RECIPE FRAMEWORK MODEL MODEL_PREFIX IMAGE PRECISION \ + TP PP_SIZE DCP_SIZE PCP_SIZE EP_SIZE DP_ATTENTION GPU_COUNT IS_AGENTIC SPEC_DECODING \ + CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME GPU_MONITOR_INTERVAL DSR1_FP8_MODEL_PATH HF_HUB_CACHE + if [[ "$MODEL_PREFIX" != dsr1 || "$PRECISION" != fp8 ]]; then + echo "ERROR: the native H200 pilot currently supports dsr1/fp8 only" >&2 + exit 1 + fi + SRT_PILOT_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-single.XXXXXX") + SRTCTL_ROOT="$SRT_PILOT_ROOT/checkout" + export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" + setup_srt_slurm "$SRTCTL_ROOT" "$FRAMEWORK" 0 + if ! command -v uv >/dev/null; then + curl -LsSf https://astral.sh/uv/install.sh | sh + source "$HOME/.local/bin/env" + fi + uv venv .venv + source .venv/bin/activate + uv pip install -e . + export PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" + + python3 -m infx.srt_slurm.single_node prepare "$GITHUB_WORKSPACE/$SRT_RECIPE" "$SRT_PILOT_ROOT/arguments" + mapfile -d '' -t SRT_RUNTIME_ARGS < "$SRT_PILOT_ROOT/arguments" + SQUASH_FILE="/data/containers/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + check_staged_srt_assets "$DSR1_FP8_MODEL_PATH" "$SQUASH_FILE" + NGINX_SQUASH_FILE=/data/containers/nginx+1.27.4.sqsh + write_srt_cluster_config h200-dgxc-slurm srtslurm.yaml 0 \ + --var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" \ + --var AIPERF_MMAP_CACHE_HOST_PATH "$AIPERF_MMAP_CACHE_HOST_PATH" \ + --var HF_HUB_CACHE_MOUNT "$HF_HUB_CACHE_MOUNT" --var CONTAINER_KEY "$IMAGE" \ + --model "hf:$MODEL" "$DSR1_FP8_MODEL_PATH" \ + --mount "$HF_HUB_CACHE_MOUNT" "$HF_HUB_CACHE" --exclusive + + SRT_JOB_ID="" + SRT_JOB_OUTPUT="" + finish_native_single_node() { + local rc=$? artifact + trap - EXIT + # Submission may succeed immediately before cancellation or a client error. + if [[ -z "$SRT_JOB_ID" ]] && python3 -m infx.srt_slurm.single_node submission \ + "$GITHUB_WORKSPACE/srt-single-node-submission.json" > "$SRT_PILOT_ROOT/submission-fields" 2>/dev/null; then + mapfile -t SRT_SUBMISSION < "$SRT_PILOT_ROOT/submission-fields" + SRT_JOB_ID="${SRT_SUBMISSION[0]}" + SRT_JOB_OUTPUT="${SRT_SUBMISSION[1]}" + fi + if [[ -n "$SRT_JOB_ID" ]] && slurm_job_is_active "$SRT_JOB_ID"; then + scancel "$SRT_JOB_ID" || true + fi + if [[ -n "$SRT_JOB_OUTPUT" && -d "$SRT_JOB_OUTPUT" ]]; then + bundle_server_logs "$SRT_JOB_OUTPUT" "$GITHUB_WORKSPACE/srt-single-node-logs.tar.gz" + for artifact in "$SRT_JOB_OUTPUT/logs/$RESULT_FILENAME.json" "$SRT_JOB_OUTPUT"/logs/gpu_metrics*; do + [[ -f "$artifact" ]] || continue + copy_to_workspace "$artifact" "$GITHUB_WORKSPACE/$(basename "$artifact")" || rc=1 + done + fi + exit "$rc" + } + trap finish_native_single_node EXIT + trap 'exit 130' INT + trap 'exit 143' TERM + apply_srt_recipe "$GITHUB_WORKSPACE/$SRT_RECIPE" "$FRAMEWORK" \ + --json --yes --output "$SRT_PILOT_ROOT/outputs" "${SRT_RUNTIME_ARGS[@]}" \ + > "$GITHUB_WORKSPACE/srt-single-node-submission.json" + python3 -m infx.srt_slurm.single_node submission "$GITHUB_WORKSPACE/srt-single-node-submission.json" \ + > "$SRT_PILOT_ROOT/submission-fields" + mapfile -t SRT_SUBMISSION < "$SRT_PILOT_ROOT/submission-fields" + SRT_JOB_ID="${SRT_SUBMISSION[0]}" + SRT_JOB_OUTPUT="${SRT_SUBMISSION[1]}" + stream_slurm_job_log "$SRT_JOB_ID" "$SRT_JOB_OUTPUT/logs/sweep_${SRT_JOB_ID}.log" + verify_slurm_job_status "$SRT_JOB_ID" + test -s "$SRT_JOB_OUTPUT/logs/$RESULT_FILENAME.json" + +elif [[ "$EXECUTION_PATH" == multinode ]]; then if [[ -z "${CONFIG_FILE:-}" ]]; then echo "Error: CONFIG_FILE is not set. The srt-slurm path requires a CONFIG_FILE in additional-settings." >&2 diff --git a/runners/runtime_settings.sh b/runners/runtime_settings.sh index 09de8f25e5..de00d288d7 100644 --- a/runners/runtime_settings.sh +++ b/runners/runtime_settings.sh @@ -28,6 +28,7 @@ case "${RUNNER_NAME%%_*}" in ;; gb300-nv) export SLURM_PARTITION=batch_1 ;; h200-dgxc-slurm) + export DSR1_FP8_MODEL_PATH=/models/DeepSeek-R1-0528 export HF_HUB_CACHE_MOUNT=/models/gharunners/hf-hub-cache export AIPERF_MMAP_CACHE_HOST_PATH=/home/sa-shared/gharunners/ai-perf-cache export DSV4_MODEL_PATH="$HF_HUB_CACHE_MOUNT/DeepSeek-V4-Pro" diff --git a/utils/matrix_logic/test_generate_sweep_configs.py b/utils/matrix_logic/test_generate_sweep_configs.py index 4d16b62bfd..a09db34998 100644 --- a/utils/matrix_logic/test_generate_sweep_configs.py +++ b/utils/matrix_logic/test_generate_sweep_configs.py @@ -190,6 +190,20 @@ def test_multinode_node_count_prefers_recipe_roles( # Test Fixtures # ============================================================================= + +@pytest.mark.parametrize("command", ["full-sweep", "test-config"]) +def test_srt_recipe_selection_stays_with_its_scenario( + sample_single_node_config, sample_runner_config, full_sweep_args_both, command, +): + key, config = next(iter(sample_single_node_config.items())) + config["scenarios"]["fixed-seq-len"][0]["search-space"][0]["srt-recipe"] = "pilot.yaml:base" + vars(full_sweep_args_both).update(config_keys=[key], no_evals=True) + generate = generate_full_sweep if command == "full-sweep" else generate_test_config_sweep + rows = generate(full_sweep_args_both, sample_single_node_config, sample_runner_config) + assert rows + assert {row.get("srt-recipe") for row in rows if row["isl"] == 1024} == {"pilot.yaml:base"} + assert {row.get("srt-recipe") for row in rows if row["isl"] == 8192} == {None} + @pytest.fixture def sample_single_node_config(): """Single node config based on dsr1-fp8-mi300x-sglang.""" diff --git a/utils/test_srt_cluster_config.py b/utils/test_srt_cluster_config.py index a6229e28b8..ebfa78f46e 100644 --- a/utils/test_srt_cluster_config.py +++ b/utils/test_srt_cluster_config.py @@ -13,6 +13,20 @@ ROOT = Path(__file__).resolve().parents[1] +def test_exclusive_allocation_override(tmp_path): + profile = tmp_path / "profile.yaml" + output = tmp_path / "cluster.yaml" + profile.write_text("use_exclusive_sbatch_directive: false\ndefault_partition: test\n") + result = subprocess.run( + [sys.executable, "-m", "infx.srt_slurm.cluster_config", str(profile), str(output), "--exclusive"], + cwd=ROOT, capture_output=True, text=True, + ) + assert result.returncode == 0, result.stderr + assert yaml.safe_load(output.read_text()) == { + "use_exclusive_sbatch_directive": True, "default_partition": "test", + } + + @pytest.mark.parametrize("power", ["0", "1", "missing-exporter"]) def test_launcher_writes_job_local_cluster_config(tmp_path: Path, power: str) -> None: runner_dir = tmp_path / "runners" diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py new file mode 100644 index 0000000000..1a432d6b85 --- /dev/null +++ b/utils/test_srt_single_node.py @@ -0,0 +1,95 @@ +"""Behavioral checks for binding a matrix point to a native SRT recipe.""" + +import copy +import json +import sys +from pathlib import Path + +import pytest +import yaml + +from infx.srt_slurm.single_node import runtime_arguments, submission_fields +from infx.srt_slurm.synthetic_acceptance import plan_commands + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) +from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides + + +@pytest.fixture +def point(tmp_path): + recipe = { + "engine": "sglang", + "model": {"path": "hf:test/model", "container": "test:tag", "precision": "fp8"}, + "roles": {"agg": { + "nodes": 1, "workers": 1, "gpus": 4, + "args": {"tensor-parallel-size": 4, "data-parallel-size": 1, "max-running-requests": 32}, + }}, + "benchmark": {"type": "custom", "env": { + "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", + }}, + } + path = tmp_path / "recipe.yaml" + path.write_text(yaml.safe_dump({"base": recipe, "zip_override_conc": { + "benchmark": {"env": {"CONC": ["2", "4"]}}, + }})) + env = { + "FRAMEWORK": "sglang", "MODEL": "test/model", "IMAGE": "test:tag", "PRECISION": "fp8", + "TP": "4", "GPU_COUNT": "4", "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", + "EP_SIZE": "1", "DP_ATTENTION": "false", "SPEC_DECODING": "none", "IS_AGENTIC": "0", + "RUN_EVAL": "false", "EVAL_ONLY": "false", "ISL": "256", "OSL": "64", + "RANDOM_RANGE_RATIO": "0.5", "CONC": "2", "RESULT_FILENAME": "point-identity", + "GPU_MONITOR_INTERVAL": "3", "MODEL_PREFIX": "test", + } + return path, recipe, env + + +def test_native_binding_submits_one_point_and_keeps_server_settings(point): + path, recipe, env = point + argv = runtime_arguments(f"{path}:base", env) + overrides = parse_overrides(argv[1::2], []) + actual = copy.deepcopy(recipe) + apply_overrides_to_recipe(actual, overrides) + assert actual["benchmark"]["env"] == { + "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", + "CONC": "2", "RESULT_FILENAME": "point-identity", "GPU_MONITOR_INTERVAL": "3", + "RUN_EVAL": "false", "EVAL_ONLY": "false", "RESULT_DIR": "/logs", + } + assert actual["roles"]["agg"]["args"] == { + "tensor-parallel-size": 4, "data-parallel-size": 1, "max-running-requests": 32, + } + commands = plan_commands(f"{path}:base", "sglang", ["--json", "--yes", *argv], env) + assert commands == [["srtctl", "apply", "--json", "--yes", *argv, "--file", f"{path}:base"]] + + +@pytest.mark.parametrize("field,value,message", [ + ("TP", "2", "tensor-parallel-size"), ("IMAGE", "other:tag", "image"), + ("ISL", "128", "ISL"), ("RUN_EVAL", "true", "RUN_EVAL"), + ("PP_SIZE", "2", "PP_SIZE"), ("RESULT_FILENAME", "", "Missing runtime input"), +]) +def test_mismatched_point_fails_before_submission(point, field, value, message): + path, _, env = point + with pytest.raises(ValueError, match=message): + runtime_arguments(f"{path}:base", {**env, field: value}) + + +def test_multi_variant_selection_is_rejected(point): + path, _, env = point + with pytest.raises(ValueError, match="exactly one"): + runtime_arguments(str(path), env) + + +@pytest.mark.parametrize("record,expected", [ + ({"status": "submitted", "slurm_job_id": "42", "output_dir": "/shared/42"}, ("42", "/shared/42")), + ({"status": "error"}, None), + ({"status": "submitted", "slurm_job_id": "42;43", "output_dir": "/shared/42"}, None), + ({"status": "submitted", "slurm_job_id": "42", "output_dir": "relative"}, None), +]) +def test_submission_manifest(tmp_path, record, expected): + path = tmp_path / "submission.json" + path.write_text(json.dumps(record)) + if expected is None: + with pytest.raises(ValueError): + submission_fields(path) + else: + assert submission_fields(path) == expected From 42eecdb764f6d65b7d5de74d649744d8dfdb1bfb Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:11:18 -0500 Subject: [PATCH 04/29] test: cover native single-node job failures and artifacts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 覆盖原生单节点作业的提交失败、Slurm 失败、产物保留及定向取消行为。 --- utils/test_srt_single_node.py | 65 +++++++++++++++++++++++++++++++++++ 1 file changed, 65 insertions(+) diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 1a432d6b85..e2d523f618 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -2,6 +2,8 @@ import copy import json +import os +import subprocess import sys from pathlib import Path @@ -93,3 +95,66 @@ def test_submission_manifest(tmp_path, record, expected): submission_fields(path) else: assert submission_fields(path) == expected + + +@pytest.mark.parametrize("failure", ["none", "allocation", "submission"]) +def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, failure): + path, _, point_env = point + binaries = tmp_path / "bin" + binaries.mkdir() + model = tmp_path / "model" + model.mkdir() + (model / "config.json").write_text("{}") + (tmp_path / "benchmarks").symlink_to(ROOT / "benchmarks", target_is_directory=True) + capture = tmp_path / "cancelled" + # Only external executables are stubbed; run the real pool launcher, shared + # setup/profile/acceptance helpers, binder, and artifact collection. + scripts = { + "git": 'if [[ "$1" == clone ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', + "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', + "unsquashfs": "exit 0", + "squeue": '[[ "$TEST_FAILURE" == submission ]] && echo "42 RUNNING"; exit 0', + "sacct": 'if [[ "$TEST_FAILURE" == allocation ]]; then echo "FAILED|1:0"; else echo "COMPLETED|0:0"; fi', + "scancel": 'printf "%s\\n" "$@" >> "$CANCEL_CAPTURE"', + "tail": 'exit 0', + } + for name, script in scripts.items(): + binary = binaries / name + binary.write_text(f"#!/usr/bin/env bash\n{script}\n") + binary.chmod(0o755) + srtctl = binaries / "srtctl" + srtctl.write_text( + f"#!{sys.executable}\n" + "import json, os, pathlib, sys\n" + "output = pathlib.Path(sys.argv[sys.argv.index('--output') + 1]) / '42'\n" + "logs = output / 'logs'\n" + "logs.mkdir(parents=True)\n" + "(logs / 'sweep_42.log').write_text('benchmark complete\\n')\n" + "(logs / (os.environ['RESULT_FILENAME'] + '.json')).write_text('{\"completed\":2}')\n" + "(logs / 'gpu_metrics.csv').write_text('gpu,power\\n0,300\\n')\n" + "(logs / 'gpu_metrics_context.json').write_text('{\"device_count\":4}')\n" + "print(json.dumps({'status':'submitted', 'slurm_job_id':'42', 'output_dir':str(output)}))\n" + "sys.exit(7 if os.environ['TEST_FAILURE'] == 'submission' else 0)\n" + ) + srtctl.chmod(0o755) + env = { + **os.environ, **point_env, + "PATH": f"{binaries}:{Path(sys.executable).parent}:{os.environ['PATH']}", + "PYTHONPATH": f"{ROOT}:{ROOT / 'utils/srt-slurm/src'}", + "GITHUB_WORKSPACE": str(tmp_path), "SRT_RECIPE": f"{path.name}:base", + "IS_MULTINODE": "false", "REQUIRE_POWER": "1", "SALLOC_TIME_LIMIT": "10", + "HF_HUB_CACHE_MOUNT": str(tmp_path), "AIPERF_MMAP_CACHE_HOST_PATH": str(tmp_path), + "HF_HUB_CACHE": "/hf", "DSR1_FP8_MODEL_PATH": str(model), "MODEL_PREFIX": "dsr1", + "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "AIPERF_DRAIN_TIMEOUT_SECONDS": "1", + "AIPERF_DRAIN_POLL_SECONDS": "1", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), + } + result = subprocess.run( + ["bash", str(ROOT / "runners/launch_h200-dgxc-slurm.sh")], cwd=tmp_path, + env=env, capture_output=True, text=True, timeout=30, + ) + assert result.returncode == {"none": 0, "allocation": 1, "submission": 7}[failure], result.stderr + assert json.loads((tmp_path / "point-identity.json").read_text()) == {"completed": 2} + assert (tmp_path / "gpu_metrics.csv").read_text() == "gpu,power\n0,300\n" + assert json.loads((tmp_path / "gpu_metrics_context.json").read_text()) == {"device_count": 4} + assert (tmp_path / "srt-single-node-logs.tar.gz").stat().st_size > 0 + assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") From 74687e35a7015132a9245608c265f094312f51b0 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:14:50 -0500 Subject: [PATCH 05/29] fix: drop unused AIPerf inputs from SRT setup MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 删除 SRT 准备阶段未使用的 AIPerf 排空参数要求,使固定序列工作流能进入实际提交;新增缺省参数回归覆盖。 --- runners/slurm_utils.sh | 2 +- utils/test_srt_single_node.py | 6 ++++-- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 67a6eac780..49180eea53 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -39,7 +39,7 @@ setup_srt_slurm() { return 1 fi local destination="$1" framework="$2" uses_power="$3" - check_env_vars INFERENCEX_RUNTIME_ENV_VARS AIPERF_DRAIN_TIMEOUT_SECONDS AIPERF_DRAIN_POLL_SECONDS EVAL_ONLY + check_env_vars INFERENCEX_RUNTIME_ENV_VARS EVAL_ONLY local eval_passthrough eval_passthrough=$(python3 - <<'PYENV' import json diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index e2d523f618..4ba04a15b6 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -145,9 +145,11 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "IS_MULTINODE": "false", "REQUIRE_POWER": "1", "SALLOC_TIME_LIMIT": "10", "HF_HUB_CACHE_MOUNT": str(tmp_path), "AIPERF_MMAP_CACHE_HOST_PATH": str(tmp_path), "HF_HUB_CACHE": "/hf", "DSR1_FP8_MODEL_PATH": str(model), "MODEL_PREFIX": "dsr1", - "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "AIPERF_DRAIN_TIMEOUT_SECONDS": "1", - "AIPERF_DRAIN_POLL_SECONDS": "1", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), + "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", + "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), } + env.pop("AIPERF_DRAIN_TIMEOUT_SECONDS", None) + env.pop("AIPERF_DRAIN_POLL_SECONDS", None) result = subprocess.run( ["bash", str(ROOT / "runners/launch_h200-dgxc-slurm.sh")], cwd=tmp_path, env=env, capture_output=True, text=True, timeout=30, From 443ff63921c5f0327ddcdb02e45204ebade145b4 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:23:27 -0500 Subject: [PATCH 06/29] fix: let native SRT stage the pilot container image MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将配方指定镜像直接交给原生 SRT/Pyxis,移除试点对旧 squash 缓存就绪状态的依赖,保留模型预检查。 --- docs/single-node-srt-migration.md | 9 +++++---- docs/single-node-srt-migration_zh.md | 2 +- runners/launch_h200-dgxc-slurm.sh | 6 +++++- utils/test_srt_single_node.py | 4 +++- 4 files changed, 14 insertions(+), 7 deletions(-) diff --git a/docs/single-node-srt-migration.md b/docs/single-node-srt-migration.md index ec345c3979..6038ef30f2 100644 --- a/docs/single-node-srt-migration.md +++ b/docs/single-node-srt-migration.md @@ -95,10 +95,11 @@ passes a native file/selector through the matrix and workflow to the existing H200 pool launcher. Production `h200` coverage, including CoreWeave, is unchanged. The launcher checks the recipe's model, image, precision, topology, and workload -against matrix metadata before submission. It resolves the model and requested -image to their existing staged cluster assets, requests an exclusive node, and -binds concurrency and artifact inputs with native `--set`. Missing assets fail -readiness instead of starting a second download or using a different image. +against matrix metadata before submission. It resolves the model to its staged +cluster path and passes the exact recipe image URI to native SRT/Pyxis container +startup. It requests an exclusive node and binds concurrency and artifact inputs +with native `--set`. Missing model assets fail before submission; the pilot does +not depend on the legacy launcher's separately managed squash cache. Plain `sglang` submissions use the shared automatic acceptance connector. Submission uses native JSON output. The launcher waits for a successful Slurm diff --git a/docs/single-node-srt-migration_zh.md b/docs/single-node-srt-migration_zh.md index 94f536c07b..ccbbc10b49 100644 --- a/docs/single-node-srt-migration_zh.md +++ b/docs/single-node-srt-migration_zh.md @@ -55,7 +55,7 @@ python -m pytest utils/test_srt_fixed_sequence.py [`configs/pilots/h200-srt.yaml`](../configs/pilots/h200-srt.yaml) 仅选择 `cluster:h200-dgxc`、8k1k、TP8 和并发 4。搜索空间的 `srt-recipe` 字段将原生文件及选择器经矩阵和工作流传给现有 H200 池启动器。生产 `h200` 覆盖(包括 CoreWeave)保持不变。 -启动器在提交前核对配方与矩阵中的模型、镜像、精度、拓扑和工作负载。它使用集群已暂存的模型及指定镜像,申请独占节点,并通过原生 `--set` 绑定并发与产物参数。资源缺失会在就绪检查时失败,不会重复下载或换用其他镜像。普通 `sglang` 提交也经过共享的自动 acceptance 连接器。 +启动器在提交前核对配方与矩阵中的模型、镜像、精度、拓扑和工作负载。它使用集群已暂存的模型路径,将配方指定的准确镜像 URI 交给原生 SRT/Pyxis 启动容器,申请独占节点,并通过原生 `--set` 绑定并发与产物参数。模型资源缺失会在提交前失败;试点不依赖旧启动器单独管理的 squash 缓存。普通 `sglang` 提交也经过共享的自动 acceptance 连接器。 提交使用原生 JSON 输出。启动器验证 Slurm 分配成功结束,保留结果文件名,并将原始结果和 GPU 采样附属文件交给现有处理及上传流程。原生日志归档、提交清单和 SRT commit 用于追踪运行来源。失败时保留已有产物,取消操作仅针对本次提交的作业。 diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index ad95379635..908f3a38a8 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -46,13 +46,17 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then python3 -m infx.srt_slurm.single_node prepare "$GITHUB_WORKSPACE/$SRT_RECIPE" "$SRT_PILOT_ROOT/arguments" mapfile -d '' -t SRT_RUNTIME_ARGS < "$SRT_PILOT_ROOT/arguments" SQUASH_FILE="/data/containers/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" - check_staged_srt_assets "$DSR1_FP8_MODEL_PATH" "$SQUASH_FILE" + if [[ ! -r "$DSR1_FP8_MODEL_PATH/config.json" ]]; then + echo "ERROR: staged model config is unavailable: $DSR1_FP8_MODEL_PATH" >&2 + exit 1 + fi NGINX_SQUASH_FILE=/data/containers/nginx+1.27.4.sqsh write_srt_cluster_config h200-dgxc-slurm srtslurm.yaml 0 \ --var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" \ --var AIPERF_MMAP_CACHE_HOST_PATH "$AIPERF_MMAP_CACHE_HOST_PATH" \ --var HF_HUB_CACHE_MOUNT "$HF_HUB_CACHE_MOUNT" --var CONTAINER_KEY "$IMAGE" \ --model "hf:$MODEL" "$DSR1_FP8_MODEL_PATH" \ + --container "$IMAGE" "$IMAGE" \ --mount "$HF_HUB_CACHE_MOUNT" "$HF_HUB_CACHE" --exclusive SRT_JOB_ID="" diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 4ba04a15b6..7ee2d60f6e 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -112,7 +112,6 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, scripts = { "git": 'if [[ "$1" == clone ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', - "unsquashfs": "exit 0", "squeue": '[[ "$TEST_FAILURE" == submission ]] && echo "42 RUNNING"; exit 0', "sacct": 'if [[ "$TEST_FAILURE" == allocation ]]; then echo "FAILED|1:0"; else echo "COMPLETED|0:0"; fi', "scancel": 'printf "%s\\n" "$@" >> "$CANCEL_CAPTURE"', @@ -159,4 +158,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, assert (tmp_path / "gpu_metrics.csv").read_text() == "gpu,power\n0,300\n" assert json.loads((tmp_path / "gpu_metrics_context.json").read_text()) == {"device_count": 4} assert (tmp_path / "srt-single-node-logs.tar.gz").stat().st_size > 0 + cluster_config = yaml.safe_load(next(tmp_path.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) + assert cluster_config["containers"]["test:tag"] == "test:tag" + assert cluster_config["use_exclusive_sbatch_directive"] is True assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") From 348f773eb6fba29cdb4216d41240cff122a4522e Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:31:59 -0500 Subject: [PATCH 07/29] fix: bootstrap native SRT binaries before pilot submission MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 在试点提交前执行原生 SRT 二进制准备,并覆盖准备失败时不提交作业的行为。 --- runners/launch_h200-dgxc-slurm.sh | 1 + utils/test_srt_single_node.py | 10 ++++++++-- 2 files changed, 9 insertions(+), 2 deletions(-) diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index 908f3a38a8..67172cb39e 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -58,6 +58,7 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then --model "hf:$MODEL" "$DSR1_FP8_MODEL_PATH" \ --container "$IMAGE" "$IMAGE" \ --mount "$HF_HUB_CACHE_MOUNT" "$HF_HUB_CACHE" --exclusive + make setup ARCH=x86_64 SRT_JOB_ID="" SRT_JOB_OUTPUT="" diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 7ee2d60f6e..8155315255 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -97,7 +97,7 @@ def test_submission_manifest(tmp_path, record, expected): assert submission_fields(path) == expected -@pytest.mark.parametrize("failure", ["none", "allocation", "submission"]) +@pytest.mark.parametrize("failure", ["none", "allocation", "submission", "bootstrap"]) def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, failure): path, _, point_env = point binaries = tmp_path / "bin" @@ -112,6 +112,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, scripts = { "git": 'if [[ "$1" == clone ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', + "make": '[[ "$TEST_FAILURE" == bootstrap ]] && exit 13; mkdir -p bin; touch bin/uv', "squeue": '[[ "$TEST_FAILURE" == submission ]] && echo "42 RUNNING"; exit 0', "sacct": 'if [[ "$TEST_FAILURE" == allocation ]]; then echo "FAILED|1:0"; else echo "COMPLETED|0:0"; fi', "scancel": 'printf "%s\\n" "$@" >> "$CANCEL_CAPTURE"', @@ -125,6 +126,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, srtctl.write_text( f"#!{sys.executable}\n" "import json, os, pathlib, sys\n" + "assert pathlib.Path('bin/uv').is_file(), 'native bootstrap was skipped'\n" "output = pathlib.Path(sys.argv[sys.argv.index('--output') + 1]) / '42'\n" "logs = output / 'logs'\n" "logs.mkdir(parents=True)\n" @@ -153,7 +155,11 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, ["bash", str(ROOT / "runners/launch_h200-dgxc-slurm.sh")], cwd=tmp_path, env=env, capture_output=True, text=True, timeout=30, ) - assert result.returncode == {"none": 0, "allocation": 1, "submission": 7}[failure], result.stderr + assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13}[failure], result.stderr + if failure == "bootstrap": + assert not (tmp_path / "srt-single-node-submission.json").exists() + assert not capture.exists() + return assert json.loads((tmp_path / "point-identity.json").read_text()) == {"completed": 2} assert (tmp_path / "gpu_metrics.csv").read_text() == "gpu,power\n0,300\n" assert json.loads((tmp_path / "gpu_metrics_context.json").read_text()) == {"device_count": 4} From a88a27a7cdf04d9f7b9e87492e6abdeadd38fc2c Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 08:52:48 -0500 Subject: [PATCH 08/29] feat: port H200 MTP and Qwen recipes to native SRT MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 H200 DeepSeek-R1 MTP 与 Qwen3.5 EP8 配方迁移到并行原生 SRT 试点,保留聊天模板及随并发变化的图捕获,并使用 4/16/64 逐点验证回归。 --- .../dsr1/sglang/h200-fp8-mtp/8k1k.yaml | 62 +++++++++++++++++ .../dsr1/sglang/h200-fp8/8k1k.yaml | 1 + .../qwen3.5/sglang/h200-fp8/8k1k.yaml | 66 +++++++++++++++++++ benchmarks/single_node/srt_fixed_sequence.sh | 13 +++- configs/pilots/h200-srt.yaml | 47 ++++++++++++- docs/single-node-srt-migration.md | 21 ++++-- docs/single-node-srt-migration_zh.md | 8 ++- infx/srt_slurm/single_node.py | 16 ++++- perf-changelog.yaml | 9 +++ runners/launch_h200-dgxc-slurm.sh | 12 ++-- runners/runtime_settings.sh | 5 +- utils/test_srt_fixed_sequence.py | 10 +-- utils/test_srt_single_node.py | 44 ++++++++++++- 13 files changed, 287 insertions(+), 27 deletions(-) create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..b4a49010ee --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml @@ -0,0 +1,62 @@ +# Parallel port of dsr1_fp8_h200_mtp.sh; real verification, embedded MTP head. +base: + schema: 2 + name: dsr1-fp8-h200-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + PYTHONNOUSERSITE: "1" + TORCH_CUDA_ARCH_LIST: "9.0" + SGLANG_ENABLE_SPEC_V2: "1" + args: + tensor-parallel-size: 8 + data-parallel-size: 1 + ep-size: 1 + trust-remote-code: true + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + disable-radix-cache: true + max-running-requests: 256 + cuda-graph-max-bs: 256 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + mem-fraction-static: 0.82 + attention-backend: flashinfer + stream-interval: 10 + decode-log-interval: 1 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-num-draft-tokens: 3 + speculative-eagle-topk: 1 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: "8192" + OSL: "1024" + RANDOM_RANGE_RATIO: "0.8" + USE_CHAT_TEMPLATE: "true" + +zip_override_concurrency: + benchmark: + env: + CONC: ["4", "8", "16", "32", "64"] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml index 39498a632c..b87ce4db55 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml @@ -49,6 +49,7 @@ base: ISL: "8192" OSL: "1024" RANDOM_RANGE_RATIO: "0.8" + USE_CHAT_TEMPLATE: "false" zip_override_concurrency: benchmark: diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml new file mode 100644 index 0000000000..8ce11d7c78 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml @@ -0,0 +1,66 @@ +# Parallel port of qwen3.5_fp8_h200.sh; graph capture follows concurrency. +base: + schema: 2 + name: qwen3.5-fp8-h200-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 8 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-flashinfer-allreduce-fusion: true + max-running-requests: 128 + chunked-prefill-size: 16384 + decode-log-interval: 1 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 4 + context-length: 9236 + kv-cache-dtype: fp8_e4m3 + quantization: fp8 + attention-backend: flashinfer + stream-interval: 50 + tokenizer-worker-num: 6 + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + trust-remote-code: true + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: "8192" + OSL: "1024" + RANDOM_RANGE_RATIO: "0.8" + USE_CHAT_TEMPLATE: "false" + CONC: "4" + +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ["4", "8", "16", "32", "64"] diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index 9d55137805..ac77ebe4ed 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -4,8 +4,14 @@ set -eo pipefail source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only check_env_vars MODEL CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME RESULT_DIR \ - SRT_FRONTEND_HOST SRT_FRONTEND_PORT RUN_EVAL EVAL_ONLY GPU_MONITOR_INTERVAL + SRT_FRONTEND_HOST SRT_FRONTEND_PORT RUN_EVAL EVAL_ONLY GPU_MONITOR_INTERVAL USE_CHAT_TEMPLATE SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" +CLIENT_ARGS=() +case "$USE_CHAT_TEMPLATE" in + true) CLIENT_ARGS+=(--use-chat-template) ;; + false) ;; + *) echo "ERROR: USE_CHAT_TEMPLATE must be true or false" >&2; exit 1 ;; +esac for name in CONC ISL OSL SRT_FRONTEND_PORT GPU_MONITOR_INTERVAL; do if [[ ! "${!name}" =~ ^[1-9][0-9]*$ ]]; then @@ -28,7 +34,7 @@ fi source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" cd "$INFERENCEX_REPO_ROOT" -pip3 install --user --break-system-packages sentencepiece +pip3 install --user --break-system-packages sentencepiece datasets pandas start_gpu_monitor --output "$RESULT_DIR/gpu_metrics.csv" --interval "$SRT_MONITOR_INTERVAL" trap 'rc=$?; stop_gpu_monitor; exit "$rc"' EXIT @@ -44,4 +50,5 @@ run_benchmark_serving \ --num-prompts "$((CONC * 10))" \ --max-concurrency "$CONC" \ --result-filename "$RESULT_FILENAME" \ - --result-dir "$RESULT_DIR" + --result-dir "$RESULT_DIR" \ + "${CLIENT_ARGS[@]}" diff --git a/configs/pilots/h200-srt.yaml b/configs/pilots/h200-srt.yaml index d0ca0a71d4..bffc2206cf 100644 --- a/configs/pilots/h200-srt.yaml +++ b/configs/pilots/h200-srt.yaml @@ -13,5 +13,50 @@ dsr1-fp8-h200-sglang: osl: 1024 search-space: - tp: 8 - conc-list: [4] + conc-list: [4, 16, 64] srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml:base + +dsr1-fp8-h200-sglang-mtp: + image: lmsysorg/sglang:v0.5.19-cu130 + model: deepseek-ai/DeepSeek-R1-0528 + model-prefix: dsr1 + runner: cluster:h200-dgxc + precision: fp8 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - tp: 8 + ep: 1 + spec-decoding: mtp + conc-list: [4, 16, 64] + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml:base + +qwen3.5-fp8-h200-sglang: + image: lmsysorg/sglang:v0.5.14-cu130 + model: Qwen/Qwen3.5-397B-A17B-FP8 + model-prefix: qwen3.5 + runner: cluster:h200-dgxc + precision: fp8 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - tp: 8 + ep: 8 + conc-list: [4] + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[0] + - tp: 8 + ep: 8 + conc-list: [16] + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[2] + - tp: 8 + ep: 8 + conc-list: [64] + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[4] diff --git a/docs/single-node-srt-migration.md b/docs/single-node-srt-migration.md index 6038ef30f2..9d7e6e0778 100644 --- a/docs/single-node-srt-migration.md +++ b/docs/single-node-srt-migration.md @@ -51,8 +51,9 @@ variation, completions API behavior, and the existing JSON result format. The helper now accepts an explicit base URL so the client reaches the endpoint SRT selected. Existing callers retain their previous local endpoint. -The client runs in the serving image and retains the legacy `sentencepiece` -installation. This is client compatibility glue, not a second server launcher. +The client runs in the serving image and installs its `sentencepiece`, `datasets`, +and `pandas` dependencies before measurement. Recipe-owned `USE_CHAT_TEMPLATE` +retains the legacy client behavior, including chat formatting for MTP. The initial client rejects eval requests explicitly. Eval-only context handling, eval artifacts, and cancellation qualification are required before cutover. @@ -90,7 +91,13 @@ validates configuration structure; it does not qualify a cluster or benchmark. ## Opt-in workflow pilot [`configs/pilots/h200-srt.yaml`](../configs/pilots/h200-srt.yaml) selects only -`cluster:h200-dgxc`, 8k1k, TP8, concurrency 4. The search-space `srt-recipe` field +`cluster:h200-dgxc`, 8k1k, TP8, and concurrencies 4, 16, and 64. +It now includes DeepSeek-R1 FP8 without speculation, DeepSeek-R1 FP8 MTP, +and Qwen3.5 FP8 with EP8. The MTP recipe preserves the embedded head, +EAGLE with two steps/three draft tokens, and real verification. Qwen preserves +its 9236-token context, FP8 KV cache, and graph capture size equal to concurrency; +native zipped overrides bind the graph size and client concurrency together. +The binder rejects a selector whose concurrency disagrees with the matrix. The search-space `srt-recipe` field passes a native file/selector through the matrix and workflow to the existing H200 pool launcher. Production `h200` coverage, including CoreWeave, is unchanged. @@ -114,10 +121,16 @@ exact pushed commit: ```bash gh workflow run e2e-tests.yml --ref codex/single-node-srt-slurm \ -f ref= -f test-name='native H200 SRT pilot' \ - -f generate-cli-command='test-config --config-keys dsr1-fp8-h200-sglang --config-file configs/pilots/h200-srt.yaml --no-evals' \ + -f generate-cli-command='test-config --config-keys dsr1-fp8-h200-sglang --config-file configs/pilots/h200-srt.yaml --conc 4 --no-evals' \ -f require-power=true ``` +Submit only one recipe and one `--conc` value per E2E run, and wait for that +allocation to finish before submitting the next. Check cluster load first. +Use the lowest, middle, and highest configured points (4/16/64 here) instead +of a full sweep; compare existing matching baselines, then run a fresh legacy +point only if a difference needs investigation. This keeps GPU usage bounded. + The pilot requires `--no-evals`: evals are still rejected explicitly. A passing throughput run alone is not accuracy or performance-parity qualification. diff --git a/docs/single-node-srt-migration_zh.md b/docs/single-node-srt-migration_zh.md index ccbbc10b49..cbbbe2eda0 100644 --- a/docs/single-node-srt-migration_zh.md +++ b/docs/single-node-srt-migration_zh.md @@ -25,7 +25,7 @@ SRT 负责资源分配、容器启动、端点选择、就绪检查及服务清 原生 `custom` benchmark 调用 [`srt_fixed_sequence.sh`](../benchmarks/single_node/srt_fixed_sequence.sh),复用现有 `run_benchmark_serving` 和 GPU 采样器。它保留 `10 * concurrency` 个请求、`2 * concurrency` 次预热、随机长度变化、completions API 行为和现有 JSON 结果格式。共享 helper 新增显式 base URL 参数,让客户端连接 SRT 选定的端点;现有调用仍使用原来的本地端点。 -客户端在服务镜像中运行,并保留原有 `sentencepiece` 安装步骤。这是客户端兼容胶水,不是第二套服务启动器。初版客户端明确拒绝 eval 请求;切换前必须完成 eval-only 上下文处理、评测产物接入和取消验收。 +客户端在服务镜像中运行,在测量前安装 `sentencepiece`、`datasets` 和 `pandas` 依赖。配方中的 `USE_CHAT_TEMPLATE` 保留原有客户端行为,包括 MTP 的聊天模板格式。初版客户端明确拒绝 eval 请求;切换前必须完成 eval-only 上下文处理、评测产物接入和取消验收。 ## 运行时输入 @@ -53,7 +53,7 @@ python -m pytest utils/test_srt_fixed_sequence.py ## 显式启用的工作流试点 -[`configs/pilots/h200-srt.yaml`](../configs/pilots/h200-srt.yaml) 仅选择 `cluster:h200-dgxc`、8k1k、TP8 和并发 4。搜索空间的 `srt-recipe` 字段将原生文件及选择器经矩阵和工作流传给现有 H200 池启动器。生产 `h200` 覆盖(包括 CoreWeave)保持不变。 +[`configs/pilots/h200-srt.yaml`](../configs/pilots/h200-srt.yaml) 仅选择 `cluster:h200-dgxc`、8k1k、TP8 和并发 4、16、64。现包含无推测的 DeepSeek-R1 FP8、DeepSeek-R1 FP8 MTP,以及 EP8 的 Qwen3.5 FP8。MTP 保留内置 draft head、EAGLE 的两步/三个 draft token 及真实验证。Qwen 保留 9236-token 上下文、FP8 KV cache,以及等于并发数的图捕获大小;原生 zipped overrides 同时绑定图大小和客户端并发。若选择器的并发与矩阵不一致,绑定器会拒绝提交。搜索空间的 `srt-recipe` 字段将原生文件及选择器经矩阵和工作流传给现有 H200 池启动器。生产 `h200` 覆盖(包括 CoreWeave)保持不变。 启动器在提交前核对配方与矩阵中的模型、镜像、精度、拓扑和工作负载。它使用集群已暂存的模型路径,将配方指定的准确镜像 URI 交给原生 SRT/Pyxis 启动容器,申请独占节点,并通过原生 `--set` 绑定并发与产物参数。模型资源缺失会在提交前失败;试点不依赖旧启动器单独管理的 squash 缓存。普通 `sglang` 提交也经过共享的自动 acceptance 连接器。 @@ -64,10 +64,12 @@ python -m pytest utils/test_srt_fixed_sequence.py ```bash gh workflow run e2e-tests.yml --ref codex/single-node-srt-slurm \ -f ref= -f test-name='native H200 SRT pilot' \ - -f generate-cli-command='test-config --config-keys dsr1-fp8-h200-sglang --config-file configs/pilots/h200-srt.yaml --no-evals' \ + -f generate-cli-command='test-config --config-keys dsr1-fp8-h200-sglang --config-file configs/pilots/h200-srt.yaml --conc 4 --no-evals' \ -f require-power=true ``` +每次 E2E 仅提交一个配方和一个 `--conc` 值,等待该分配结束后再提交下一点,并在提交前检查集群负载。选择配置中的最低、中间和最高并发(此处为 4/16/64),不运行完整扫描;先比较已有匹配基线,仅在差异需要调查时重新运行对应的旧路径点,以限制 GPU 使用量。 + 试点必须传 `--no-evals`,目前仍明确拒绝 eval。一次吞吐运行通过不代表准确性或性能一致性已验收。 ## 启用替代路径之前 diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index d0b0f5c6d3..6c355eb704 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -10,7 +10,7 @@ import yaml -from infx.srt_slurm.synthetic_acceptance import selected_recipes +from infx.srt_slurm.synthetic_acceptance import selected_recipes, spec_parameters def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: @@ -27,6 +27,10 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: args = role["args"] benchmark = recipe["benchmark"] workload = benchmark["env"] + spec = spec_parameters(role, "sglang") + if spec and spec["method"] not in {"eagle", "nextn"}: + raise ValueError("Single-node SRT pilot supports only native MTP or no speculation") + speculation = "mtp" if spec else "none" expected = { "engine": (recipe["engine"], environment["FRAMEWORK"]), "model": (recipe["model"]["path"], f"hf:{environment['MODEL']}"), @@ -34,13 +38,21 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: "precision": (recipe["model"]["precision"], environment["PRECISION"]), "tensor-parallel-size": (args["tensor-parallel-size"], int(environment["TP"])), "data-parallel-size": (args["data-parallel-size"], 1), + "expert-parallel-size": ( + args.get("expert-parallel-size", args.get("ep-size", 1)), + int(environment["EP_SIZE"]), + ), "gpus": (role["gpus"], int(environment["GPU_COUNT"])), "nodes": (role["nodes"], 1), "workers": (role["workers"], 1), "roles": (set(recipe["roles"]), {"agg"}), "benchmark type": (benchmark["type"], "custom"), "benchmark MODEL": (workload["MODEL"], environment["MODEL"]), + "SPEC_DECODING": (speculation, environment["SPEC_DECODING"]), + "USE_CHAT_TEMPLATE": (workload["USE_CHAT_TEMPLATE"], "true" if spec else "false"), } + if "CONC" in workload: + expected["CONC"] = (str(workload["CONC"]), environment["CONC"]) for name in ("ISL", "OSL", "RANDOM_RANGE_RATIO"): expected[name] = (str(workload[name]), environment[name]) # Other topology/eval paths remain on their current launchers until ported. @@ -49,10 +61,8 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", - "EP_SIZE": "1", "DP_ATTENTION": "false", "IS_AGENTIC": "0", - "SPEC_DECODING": "none", "RUN_EVAL": "false", "EVAL_ONLY": "false", }.items(): diff --git a/perf-changelog.yaml b/perf-changelog.yaml index ea7d1439d4..a58a4c62e8 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8556,3 +8556,12 @@ - "Wire the opt-in H200 native SRT single-node pilot through the matrix, launcher, result and GPU power paths; keep production routing unchanged pending qualification" - "将显式启用的 H200 原生 SRT 单节点试点接入矩阵、启动器、结果及 GPU 功耗流程;验收完成前保留现有生产路由" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp8-h200-sglang + - dsr1-fp8-h200-sglang-mtp + - qwen3.5-fp8-h200-sglang + description: + - "Extend the opt-in native H200 SRT pilot to DeepSeek-R1 MTP and Qwen3.5 EP8, preserving real verification, chat formatting, and concurrency-dependent graph capture; select 4/16/64 for sequential regression checks without production cutover" + - "将显式启用的原生 H200 SRT 试点扩展到 DeepSeek-R1 MTP 和 Qwen3.5 EP8,保留真实验证、聊天模板及随并发变化的图捕获;选择 4/16/64 逐点检查回归,尚不切换生产路由" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index 67172cb39e..36889a423c 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -25,9 +25,9 @@ fi if [[ "$EXECUTION_PATH" == native-single-node ]]; then check_env_vars GITHUB_WORKSPACE SRT_RECIPE FRAMEWORK MODEL MODEL_PREFIX IMAGE PRECISION \ TP PP_SIZE DCP_SIZE PCP_SIZE EP_SIZE DP_ATTENTION GPU_COUNT IS_AGENTIC SPEC_DECODING \ - CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME GPU_MONITOR_INTERVAL DSR1_FP8_MODEL_PATH HF_HUB_CACHE - if [[ "$MODEL_PREFIX" != dsr1 || "$PRECISION" != fp8 ]]; then - echo "ERROR: the native H200 pilot currently supports dsr1/fp8 only" >&2 + CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME GPU_MONITOR_INTERVAL SRT_MODEL_PATH HF_HUB_CACHE + if [[ ( "$MODEL_PREFIX" != dsr1 && "$MODEL_PREFIX" != qwen3.5 ) || "$PRECISION" != fp8 ]]; then + echo "ERROR: the native H200 pilot supports dsr1/fp8 and qwen3.5/fp8 only" >&2 exit 1 fi SRT_PILOT_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-single.XXXXXX") @@ -46,8 +46,8 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then python3 -m infx.srt_slurm.single_node prepare "$GITHUB_WORKSPACE/$SRT_RECIPE" "$SRT_PILOT_ROOT/arguments" mapfile -d '' -t SRT_RUNTIME_ARGS < "$SRT_PILOT_ROOT/arguments" SQUASH_FILE="/data/containers/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" - if [[ ! -r "$DSR1_FP8_MODEL_PATH/config.json" ]]; then - echo "ERROR: staged model config is unavailable: $DSR1_FP8_MODEL_PATH" >&2 + if [[ ! -r "$SRT_MODEL_PATH/config.json" ]]; then + echo "ERROR: staged model config is unavailable: $SRT_MODEL_PATH" >&2 exit 1 fi NGINX_SQUASH_FILE=/data/containers/nginx+1.27.4.sqsh @@ -55,7 +55,7 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then --var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" \ --var AIPERF_MMAP_CACHE_HOST_PATH "$AIPERF_MMAP_CACHE_HOST_PATH" \ --var HF_HUB_CACHE_MOUNT "$HF_HUB_CACHE_MOUNT" --var CONTAINER_KEY "$IMAGE" \ - --model "hf:$MODEL" "$DSR1_FP8_MODEL_PATH" \ + --model "hf:$MODEL" "$SRT_MODEL_PATH" \ --container "$IMAGE" "$IMAGE" \ --mount "$HF_HUB_CACHE_MOUNT" "$HF_HUB_CACHE" --exclusive make setup ARCH=x86_64 diff --git a/runners/runtime_settings.sh b/runners/runtime_settings.sh index de00d288d7..06d50bdfa7 100644 --- a/runners/runtime_settings.sh +++ b/runners/runtime_settings.sh @@ -28,8 +28,11 @@ case "${RUNNER_NAME%%_*}" in ;; gb300-nv) export SLURM_PARTITION=batch_1 ;; h200-dgxc-slurm) - export DSR1_FP8_MODEL_PATH=/models/DeepSeek-R1-0528 export HF_HUB_CACHE_MOUNT=/models/gharunners/hf-hub-cache + case "$MODEL_PREFIX/$PRECISION" in + dsr1/fp8) export SRT_MODEL_PATH=/models/DeepSeek-R1-0528 ;; + qwen3.5/fp8) export SRT_MODEL_PATH="$HF_HUB_CACHE_MOUNT/Qwen3.5-397B-A17B-FP8" ;; + esac export AIPERF_MMAP_CACHE_HOST_PATH=/home/sa-shared/gharunners/ai-perf-cache export DSV4_MODEL_PATH="$HF_HUB_CACHE_MOUNT/DeepSeek-V4-Pro" export GLM52_FP8_MODEL_PATH=/models/GLM-5.2-FP8 diff --git a/utils/test_srt_fixed_sequence.py b/utils/test_srt_fixed_sequence.py index 470e33fa2b..81dc2f64b4 100644 --- a/utils/test_srt_fixed_sequence.py +++ b/utils/test_srt_fixed_sequence.py @@ -46,6 +46,7 @@ def client_environment(tmp_path): "RUN_EVAL": "false", "EVAL_ONLY": "false", "GPU_MONITOR_INTERVAL": "2", + "USE_CHAT_TEMPLATE": "false", "IS_AGENTIC": "0", "SCENARIO_TYPE": "fixed-seq-len", "CLIENT_EXIT": "0", @@ -56,11 +57,11 @@ def client_environment(tmp_path): return env -@pytest.mark.parametrize("exit_code", [0, 7]) +@pytest.mark.parametrize("exit_code,chat_template", [(0, "false"), (7, "false"), (0, "true")]) def test_native_endpoint_preserves_client_settings_and_failure( - client_environment, exit_code + client_environment, exit_code, chat_template ): - env = {**client_environment, "CLIENT_EXIT": str(exit_code)} + env = {**client_environment, "CLIENT_EXIT": str(exit_code), "USE_CHAT_TEMPLATE": chat_template} result = subprocess.run( ["bash", str(CLIENT)], env=env, capture_output=True, text=True ) @@ -99,7 +100,7 @@ def test_native_endpoint_preserves_client_settings_and_failure( env["RESULT_DIR"], "--result-filename", "test-result.json", - ] + ] + (["--use-chat-template"] if chat_template == "true" else []) assert ( (Path(env["RESULT_DIR"]) / "gpu_metrics.csv") .read_text() @@ -112,6 +113,7 @@ def test_native_endpoint_preserves_client_settings_and_failure( [ ("MODEL", None, "MODEL"), ("GPU_MONITOR_INTERVAL", None, "GPU_MONITOR_INTERVAL"), + ("USE_CHAT_TEMPLATE", "yes", "USE_CHAT_TEMPLATE must be true or false"), ("CONC", "0", "CONC must be a positive integer"), ("RUN_EVAL", "true", "does not support evals yet"), ("EVAL_ONLY", "true", "does not support evals yet"), diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 8155315255..6725f7e78d 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -11,7 +11,7 @@ import yaml from infx.srt_slurm.single_node import runtime_arguments, submission_fields -from infx.srt_slurm.synthetic_acceptance import plan_commands +from infx.srt_slurm.synthetic_acceptance import plan_commands, selected_recipes ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) @@ -29,6 +29,7 @@ def point(tmp_path): }}, "benchmark": {"type": "custom", "env": { "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", + "USE_CHAT_TEMPLATE": "false", }}, } path = tmp_path / "recipe.yaml" @@ -54,6 +55,7 @@ def test_native_binding_submits_one_point_and_keeps_server_settings(point): apply_overrides_to_recipe(actual, overrides) assert actual["benchmark"]["env"] == { "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", + "USE_CHAT_TEMPLATE": "false", "CONC": "2", "RESULT_FILENAME": "point-identity", "GPU_MONITOR_INTERVAL": "3", "RUN_EVAL": "false", "EVAL_ONLY": "false", "RESULT_DIR": "/logs", } @@ -68,6 +70,7 @@ def test_native_binding_submits_one_point_and_keeps_server_settings(point): ("TP", "2", "tensor-parallel-size"), ("IMAGE", "other:tag", "image"), ("ISL", "128", "ISL"), ("RUN_EVAL", "true", "RUN_EVAL"), ("PP_SIZE", "2", "PP_SIZE"), ("RESULT_FILENAME", "", "Missing runtime input"), + ("EP_SIZE", "2", "expert-parallel-size"), ("SPEC_DECODING", "mtp", "SPEC_DECODING"), ]) def test_mismatched_point_fails_before_submission(point, field, value, message): path, _, env = point @@ -81,6 +84,43 @@ def test_multi_variant_selection_is_rejected(point): runtime_arguments(str(path), env) +def test_mtp_binding_uses_real_verification_and_preserves_expert_parallelism(point): + path, recipe, env = point + recipe["roles"]["agg"]["args"].update({ + "expert-parallel-size": 4, "speculative-algorithm": "EAGLE", + "speculative-num-steps": 2, "speculative-num-draft-tokens": 3, + }) + recipe["roles"]["agg"]["env"] = {"SGLANG_SIMULATE_ACC_LEN": "2.5"} + recipe["benchmark"]["env"]["USE_CHAT_TEMPLATE"] = "true" + path.write_text(yaml.safe_dump({"base": recipe})) + env = {**env, "EP_SIZE": "4", "SPEC_DECODING": "mtp"} + argv = runtime_arguments(f"{path}:base", env) + commands = plan_commands(f"{path}:base", "sglang", ["--json", *argv], env) + assert commands == [[ + "srtctl", "apply", "--json", *argv, "--file", f"{path}:base", + "--unset", "roles.agg.env.SGLANG_SIMULATE_ACC_LEN", + ]] + recipe["benchmark"]["env"]["USE_CHAT_TEMPLATE"] = "false" + path.write_text(yaml.safe_dump({"base": recipe})) + with pytest.raises(ValueError, match="USE_CHAT_TEMPLATE"): + runtime_arguments(f"{path}:base", env) + + +def test_concurrency_selector_keeps_graph_capture_coupled_to_client(point): + path, recipe, env = point + path.write_text(yaml.safe_dump({"base": recipe, "zip_override_conc": { + "roles": {"agg": {"args": {"cuda-graph-max-bs": [2, 4]}}}, + "benchmark": {"env": {"CONC": ["2", "4"]}}, + }})) + argv = runtime_arguments(f"{path}:zip_override_conc[1]", {**env, "CONC": "4"}) + actual = selected_recipes(yaml.safe_load(path.read_text()), "zip_override_conc[1]")[0][1] + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual["roles"]["agg"]["args"]["cuda-graph-max-bs"] == 4 + assert actual["benchmark"]["env"]["CONC"] == "4" + with pytest.raises(ValueError, match="CONC"): + runtime_arguments(f"{path}:zip_override_conc[1]", env) + + @pytest.mark.parametrize("record,expected", [ ({"status": "submitted", "slurm_job_id": "42", "output_dir": "/shared/42"}, ("42", "/shared/42")), ({"status": "error"}, None), @@ -145,7 +185,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "GITHUB_WORKSPACE": str(tmp_path), "SRT_RECIPE": f"{path.name}:base", "IS_MULTINODE": "false", "REQUIRE_POWER": "1", "SALLOC_TIME_LIMIT": "10", "HF_HUB_CACHE_MOUNT": str(tmp_path), "AIPERF_MMAP_CACHE_HOST_PATH": str(tmp_path), - "HF_HUB_CACHE": "/hf", "DSR1_FP8_MODEL_PATH": str(model), "MODEL_PREFIX": "dsr1", + "HF_HUB_CACHE": "/hf", "SRT_MODEL_PATH": str(model), "MODEL_PREFIX": "dsr1", "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), } From b9757d9ae7b637fb18f340a1a4010a43ccdc69ee Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:01:59 -0500 Subject: [PATCH 09/29] docs: remove migration guide MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 移除独立迁移指南及入口链接,迁移范围与验证证据保留在 PR 中。 --- docs/configuration-procedures.md | 4 - docs/configuration-procedures_zh.md | 2 - docs/single-node-srt-migration.md | 152 --------------------------- docs/single-node-srt-migration_zh.md | 82 --------------- 4 files changed, 240 deletions(-) delete mode 100644 docs/single-node-srt-migration.md delete mode 100644 docs/single-node-srt-migration_zh.md diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 385651bdba..de9f1cda5c 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -171,10 +171,6 @@ Only fixed 8192/1024 `glm5.1-fp8-b200-tilert` requires native power. TileRT runs ## Register an srt-slurm recipe -For the parallel single-node migration and its activation checklist, see -[Single-node SRT-Slurm migration](./single-node-srt-migration.md). Its initial -recipe is not selected by the production master config or launcher yet. - Mapping source: [`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchmarks/multi_node/srt-slurm-recipes/RECIPES.md). Checked-in recipes: [`benchmarks/multi_node/srt-slurm-recipes/`](../benchmarks/multi_node/srt-slurm-recipes/). 1. Locate the exact upstream [NVIDIA/srt-slurm](https://github.com/NVIDIA/srt-slurm) recipe and record its commit-pinned source path. diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 8300757c79..b2e0ce2d60 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -148,8 +148,6 @@ B200 Nscale 的 GLM-5.1 可用 `MODEL_PATH` 指定已有共享权重,覆盖默 ## 注册 srt-slurm 配方 -并行构建的单节点迁移及其启用清单见[单节点 SRT-Slurm 迁移](./single-node-srt-migration_zh.md)。首个候选配方尚未接入生产主配置或启动器。 - 映射来源:[`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchmarks/multi_node/srt-slurm-recipes/RECIPES.md)。检入的配方:[`benchmarks/multi_node/srt-slurm-recipes/`](../benchmarks/multi_node/srt-slurm-recipes/)。 1. 定位精确的上游 [NVIDIA/srt-slurm](https://github.com/NVIDIA/srt-slurm) 配方,并记录固定到 commit 的来源路径。 diff --git a/docs/single-node-srt-migration.md b/docs/single-node-srt-migration.md deleted file mode 100644 index 9d7e6e0778..0000000000 --- a/docs/single-node-srt-migration.md +++ /dev/null @@ -1,152 +0,0 @@ -# Single-node SRT-Slurm migration - -**English** | [中文](./single-node-srt-migration_zh.md) - -This draft moves active single-node serving settings into native SRT-Slurm YAML. -The first candidate is built alongside the existing route; it is not a production -cutover. The existing master config, runner selection, dependency pin, and result -ingestion remain in use until the replacement has runtime evidence. - -## Ownership and scope - -Alec owns the single-node recipe migration. Cam owns the fork pin, AMD support, -and AMD runtime cleanup. Keep reusable execution changes small enough to send -upstream; the intended destination is NVIDIA SRT-Slurm, with the SemiAnalysisAI -fork as an intermediate dependency. Do not import the separate prepared-job or -publication systems from the earlier H100 pilot into this work. - -At InferenceX main `4ab85c1e`, the active master files contain 110 single-node -configurations; deprecated archives are excluded: - -| Vendor | SGLang | vLLM | TRT-LLM | ATOM | Total | -| --- | ---: | ---: | ---: | ---: | ---: | -| NVIDIA | 47 | 13 | 10 | 0 | 70 | -| AMD | 21 | 8 | 0 | 11 | 40 | - -This is an inventory of migration work, not a claim that all these paths are -supported by the current SRT fork. The AMD stack and non-Slurm execution need -their own capability checks. - -## First native recipe - -[`8k1k.yaml`](../benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml) -ports `dsr1-fp8-h200-sglang` from -[`dsr1_fp8_h200.sh`](../benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh). -It uses native schema 2 and native `zip_override_concurrency` expansion: one -TP8 aggregate worker on one H200 node, with separate jobs at concurrency -4, 8, 16, 32, and 64. It preserves the model, image, 8k1k workload, server -environment, and explicit serving flags from the active recipe. - -SRT owns allocation, container startup, endpoint selection, readiness, and server -cleanup. Account, partition, mounts, image caches, and exclusive allocation belong -to the cluster/launcher integration, not the workload YAML. The candidate disables -SRT observability to avoid adding a second telemetry workload beside the existing -GPU sampler; runtime qualification must still compare effective native defaults. - -The native `custom` benchmark invokes -[`srt_fixed_sequence.sh`](../benchmarks/single_node/srt_fixed_sequence.sh). -This client uses the existing `run_benchmark_serving` helper and GPU sampler. -It preserves `10 * concurrency` requests, `2 * concurrency` warmups, random length -variation, completions API behavior, and the existing JSON result format. The -helper now accepts an explicit base URL so the client reaches the endpoint SRT -selected. Existing callers retain their previous local endpoint. - -The client runs in the serving image and installs its `sentencepiece`, `datasets`, -and `pandas` dependencies before measurement. Recipe-owned `USE_CHAT_TEMPLATE` -retains the legacy client behavior, including chat formatting for MTP. -The initial client rejects eval requests explicitly. Eval-only context handling, -eval artifacts, and cancellation qualification are required before cutover. - -## Runtime inputs - -The recipe provides model/workload inputs, including concurrency through native -override expansion. SRT provides `SRT_FRONTEND_HOST` and `SRT_FRONTEND_PORT`. -The launch integration must export `INFMAX_WORKSPACE` before submission; SRT -mounts it at `/infmax-workspace`. It must also supply these native overrides: - -| Native override | Caller-owned value | -| --- | --- | -| `benchmark.env.RESULT_FILENAME` | Existing InferenceX result basename | -| `benchmark.env.RESULT_DIR` | `/logs`, SRT's existing per-job artifact mount | -| `benchmark.env.GPU_MONITOR_INTERVAL` | Explicit sampling interval in seconds | -| `benchmark.env.RUN_EVAL` | `"false"` for the throughput pilot | -| `benchmark.env.EVAL_ONLY` | `"false"` for the throughput pilot | - -All environment values above are strings; use quoted YAML values with native -`--set`. Missing inputs fail before the client runs. No fallback runtime settings -are hidden in the workload recipe. Submit through the shared `apply_srt_recipe` -entrypoint so speculative ports later retain automatic golden-AL handling. - -For a local allocation render, with SRT dependencies installed: - -```bash -INFMAX_WORKSPACE="$PWD" PYTHONPATH=utils/srt-slurm/src srtctl dry-run \ - -f 'benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[0]' -python -m pytest utils/test_srt_fixed_sequence.py -``` - -Without a cluster profile this renders SRT's generic scheduling defaults. It -validates configuration structure; it does not qualify a cluster or benchmark. - -## Opt-in workflow pilot - -[`configs/pilots/h200-srt.yaml`](../configs/pilots/h200-srt.yaml) selects only -`cluster:h200-dgxc`, 8k1k, TP8, and concurrencies 4, 16, and 64. -It now includes DeepSeek-R1 FP8 without speculation, DeepSeek-R1 FP8 MTP, -and Qwen3.5 FP8 with EP8. The MTP recipe preserves the embedded head, -EAGLE with two steps/three draft tokens, and real verification. Qwen preserves -its 9236-token context, FP8 KV cache, and graph capture size equal to concurrency; -native zipped overrides bind the graph size and client concurrency together. -The binder rejects a selector whose concurrency disagrees with the matrix. The search-space `srt-recipe` field -passes a native file/selector through the matrix and workflow to the existing -H200 pool launcher. Production `h200` coverage, including CoreWeave, is unchanged. - -The launcher checks the recipe's model, image, precision, topology, and workload -against matrix metadata before submission. It resolves the model to its staged -cluster path and passes the exact recipe image URI to native SRT/Pyxis container -startup. It requests an exclusive node and binds concurrency and artifact inputs -with native `--set`. Missing model assets fail before submission; the pilot does -not depend on the legacy launcher's separately managed squash cache. -Plain `sglang` submissions use the shared automatic acceptance connector. - -Submission uses native JSON output. The launcher waits for a successful Slurm -allocation exit, preserves the result basename, and stages raw results and GPU -sampling sidecars for the existing processor and uploads. A native log archive, -submission manifest, and SRT commit identify the run. Failed jobs retain available -artifacts; cancellation targets only the submitted job. - -Dispatch the workflow definition from the draft branch, with `ref` set to the -exact pushed commit: - -```bash -gh workflow run e2e-tests.yml --ref codex/single-node-srt-slurm \ - -f ref= -f test-name='native H200 SRT pilot' \ - -f generate-cli-command='test-config --config-keys dsr1-fp8-h200-sglang --config-file configs/pilots/h200-srt.yaml --conc 4 --no-evals' \ - -f require-power=true -``` - -Submit only one recipe and one `--conc` value per E2E run, and wait for that -allocation to finish before submitting the next. Check cluster load first. -Use the lowest, middle, and highest configured points (4/16/64 here) instead -of a full sweep; compare existing matching baselines, then run a fresh legacy -point only if a difference needs investigation. This keeps GPU usage bounded. - -The pilot requires `--no-evals`: evals are still rejected explicitly. A passing -throughput run alone is not accuracy or performance-parity qualification. - -## Before enabling the replacement - -- Preserve both H200 runner paths: the current `h200` label includes - `h200-dgxc-slurm` and `h200-cw`. Do not silently drop CoreWeave or pretend its - Docker execution is already covered by a Slurm recipe. -- Connect eval context, real-verification evals, and eval artifact staging; - qualify the wired result and GPU power paths without changing publication format. -- Compare legacy and native commands, then qualify startup, throughput, accuracy, - power, cancellation, and cleanup on the same image/model/hardware. Coordinate - existing smoke/vendor evaluation work rather than duplicating it. -- Expand to the other active single-node recipes after this path is qualified, - including speculative decoding and KV offload. Coordinate AMD capabilities - with Cam's fork work. Retire legacy scripts only after their callers move. - -Keep the migration PR draft. Local schema checks and stubbed client tests do not -establish performance parity or GPU qualification. diff --git a/docs/single-node-srt-migration_zh.md b/docs/single-node-srt-migration_zh.md deleted file mode 100644 index cbbbe2eda0..0000000000 --- a/docs/single-node-srt-migration_zh.md +++ /dev/null @@ -1,82 +0,0 @@ -# 单节点 SRT-Slurm 迁移 - -[English](./single-node-srt-migration.md) | **中文** - -本草稿将仍在使用的单节点服务设置迁移到原生 SRT-Slurm YAML。首个候选实现与现有路径并行构建,尚未切换生产路由。在取得实际运行证据之前,主配置、runner 选择、依赖固定版本和结果导入流程继续沿用现有实现。 - -## 分工与范围 - -Alec 负责单节点配方迁移。Cam 负责固定分叉版本、AMD 支持及 AMD 运行时清理。可复用的执行改动应保持小而独立,便于提交上游;最终目标是使用 NVIDIA SRT-Slurm,SemiAnalysisAI 分叉只是过渡依赖。本项工作不引入此前 H100 试点的独立 prepared-job 或发布系统。 - -在 InferenceX main `4ab85c1e` 中,排除 deprecated 归档后,活跃主配置包含 110 个单节点配置: - -| 厂商 | SGLang | vLLM | TRT-LLM | ATOM | 合计 | -| --- | ---: | ---: | ---: | ---: | ---: | -| NVIDIA | 47 | 13 | 10 | 0 | 70 | -| AMD | 21 | 8 | 0 | 11 | 40 | - -这是待迁移清单,不代表当前 SRT 分叉已支持所有路径。AMD 功能和非 Slurm 执行需要分别确认。 - -## 首个原生配方 - -[`8k1k.yaml`](../benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml) 对应活跃配置 `dsr1-fp8-h200-sglang`,来源为 [`dsr1_fp8_h200.sh`](../benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh)。它使用原生 schema 2 和 `zip_override_concurrency` 展开:在一个 H200 节点上启动一个 TP8 聚合 worker,并发 4、8、16、32、64 分别运行独立作业。模型、镜像、8k1k 工作负载、服务环境和显式服务参数均取自现有配方。 - -SRT 负责资源分配、容器启动、端点选择、就绪检查及服务清理。账号、分区、挂载、镜像缓存和独占分配属于集群与启动器集成,不放入工作负载 YAML。候选配方关闭 SRT observability,避免在现有 GPU 采样器之外增加一套遥测负载;验收时仍须比较实际生效的原生默认设置。 - -原生 `custom` benchmark 调用 [`srt_fixed_sequence.sh`](../benchmarks/single_node/srt_fixed_sequence.sh),复用现有 `run_benchmark_serving` 和 GPU 采样器。它保留 `10 * concurrency` 个请求、`2 * concurrency` 次预热、随机长度变化、completions API 行为和现有 JSON 结果格式。共享 helper 新增显式 base URL 参数,让客户端连接 SRT 选定的端点;现有调用仍使用原来的本地端点。 - -客户端在服务镜像中运行,在测量前安装 `sentencepiece`、`datasets` 和 `pandas` 依赖。配方中的 `USE_CHAT_TEMPLATE` 保留原有客户端行为,包括 MTP 的聊天模板格式。初版客户端明确拒绝 eval 请求;切换前必须完成 eval-only 上下文处理、评测产物接入和取消验收。 - -## 运行时输入 - -配方提供模型及工作负载参数,并通过原生覆盖展开并发。SRT 提供 `SRT_FRONTEND_HOST` 和 `SRT_FRONTEND_PORT`。启动集成须在提交前导出 `INFMAX_WORKSPACE`,由 SRT 挂载到 `/infmax-workspace`,并提供以下原生覆盖: - -| 原生覆盖字段 | 调用方负责的值 | -| --- | --- | -| `benchmark.env.RESULT_FILENAME` | 现有 InferenceX 结果文件基本名称 | -| `benchmark.env.RESULT_DIR` | `/logs`,SRT 已创建的作业产物挂载 | -| `benchmark.env.GPU_MONITOR_INTERVAL` | 显式指定的采样间隔秒数 | -| `benchmark.env.RUN_EVAL` | 吞吐试点使用 `"false"` | -| `benchmark.env.EVAL_ONLY` | 吞吐试点使用 `"false"` | - -上述环境值均为字符串;通过原生 `--set` 传递时使用带引号的 YAML 值。缺失输入会在客户端启动前报错,工作负载配方不隐藏运行时回退设置。提交应经过共享 `apply_srt_recipe`,确保后续推测解码迁移保留自动 golden-AL 选择。 - -安装 SRT 依赖后,可在本地渲染资源分配: - -```bash -INFMAX_WORKSPACE="$PWD" PYTHONPATH=utils/srt-slurm/src srtctl dry-run \ - -f 'benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[0]' -python -m pytest utils/test_srt_fixed_sequence.py -``` - -未提供集群 profile 时,该命令使用 SRT 的通用调度默认值。它验证配置结构,不代表集群或基准已经验收。 - -## 显式启用的工作流试点 - -[`configs/pilots/h200-srt.yaml`](../configs/pilots/h200-srt.yaml) 仅选择 `cluster:h200-dgxc`、8k1k、TP8 和并发 4、16、64。现包含无推测的 DeepSeek-R1 FP8、DeepSeek-R1 FP8 MTP,以及 EP8 的 Qwen3.5 FP8。MTP 保留内置 draft head、EAGLE 的两步/三个 draft token 及真实验证。Qwen 保留 9236-token 上下文、FP8 KV cache,以及等于并发数的图捕获大小;原生 zipped overrides 同时绑定图大小和客户端并发。若选择器的并发与矩阵不一致,绑定器会拒绝提交。搜索空间的 `srt-recipe` 字段将原生文件及选择器经矩阵和工作流传给现有 H200 池启动器。生产 `h200` 覆盖(包括 CoreWeave)保持不变。 - -启动器在提交前核对配方与矩阵中的模型、镜像、精度、拓扑和工作负载。它使用集群已暂存的模型路径,将配方指定的准确镜像 URI 交给原生 SRT/Pyxis 启动容器,申请独占节点,并通过原生 `--set` 绑定并发与产物参数。模型资源缺失会在提交前失败;试点不依赖旧启动器单独管理的 squash 缓存。普通 `sglang` 提交也经过共享的自动 acceptance 连接器。 - -提交使用原生 JSON 输出。启动器验证 Slurm 分配成功结束,保留结果文件名,并将原始结果和 GPU 采样附属文件交给现有处理及上传流程。原生日志归档、提交清单和 SRT commit 用于追踪运行来源。失败时保留已有产物,取消操作仅针对本次提交的作业。 - -从草稿分支触发工作流,将 `ref` 设为已推送的准确 commit: - -```bash -gh workflow run e2e-tests.yml --ref codex/single-node-srt-slurm \ - -f ref= -f test-name='native H200 SRT pilot' \ - -f generate-cli-command='test-config --config-keys dsr1-fp8-h200-sglang --config-file configs/pilots/h200-srt.yaml --conc 4 --no-evals' \ - -f require-power=true -``` - -每次 E2E 仅提交一个配方和一个 `--conc` 值,等待该分配结束后再提交下一点,并在提交前检查集群负载。选择配置中的最低、中间和最高并发(此处为 4/16/64),不运行完整扫描;先比较已有匹配基线,仅在差异需要调查时重新运行对应的旧路径点,以限制 GPU 使用量。 - -试点必须传 `--no-evals`,目前仍明确拒绝 eval。一次吞吐运行通过不代表准确性或性能一致性已验收。 - -## 启用替代路径之前 - -- 保留两个 H200 runner 路径:当前 `h200` 标签同时包含 `h200-dgxc-slurm` 和 `h200-cw`。不能静默删除 CoreWeave 覆盖,也不能把其 Docker 执行视为已被 Slurm 配方覆盖。 -- 接入 eval 上下文、真实验证评测和评测产物准备,并验收已接入的结果与 GPU 功耗路径,不改变发布格式。 -- 比较旧路径与原生路径的命令,在相同镜像、模型和硬件上验收启动、吞吐、准确性、功耗、取消及清理。协调现有 smoke/vendor 评测工作,避免重复执行。 -- 首条路径验收后,再扩展到其他活跃单节点配方,包括推测解码和 KV offload;AMD 能力与 Cam 的分叉工作协调。只有调用方完成迁移后才删除旧脚本。 - -迁移 PR 保持草稿。本地 schema 检查和使用外部进程桩的客户端测试不能证明性能一致或 GPU 验收通过。 From 2763e8d002efd1786d7c52e30c73410c0d0803c2 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:36:43 -0500 Subject: [PATCH 10/29] feat: migrate fixed-sequence SGLang recipes to native SRT MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 H100、H200、B200 和 B300 的 21 个活跃 SGLang 定长配方及全部 181 个配置点切换到原生 SRT-Slurm。共享提交、eval 和产物处理,并保留逐点拓扑与推测解码设置。 --- .github/workflows/benchmark-tmpl.yml | 4 - .../dsr1/sglang/b200-fp4-mtp/8k1k.yaml | 84 +++++++ .../dsr1/sglang/b200-fp4/8k1k.yaml | 80 ++++++ .../dsr1/sglang/b200-fp8-mtp/8k1k.yaml | 69 ++++++ .../dsr1/sglang/b200-fp8/8k1k.yaml | 74 ++++++ .../dsr1/sglang/b300-fp4/8k1k.yaml | 74 ++++++ .../dsr1/sglang/b300-fp8-mtp/8k1k.yaml | 69 ++++++ .../dsr1/sglang/b300-fp8/8k1k.yaml | 74 ++++++ .../dsr1/sglang/h200-fp8-mtp/8k1k.yaml | 29 ++- .../dsr1/sglang/h200-fp8/8k1k.yaml | 27 +- .../qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml | 96 ++++++++ .../qwen3.5/sglang/b200-fp4/8k1k.yaml | 73 ++++++ .../qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml | 82 +++++++ .../qwen3.5/sglang/b200-fp8/8k1k.yaml | 78 ++++++ .../qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml | 91 +++++++ .../qwen3.5/sglang/b300-fp4/8k1k.yaml | 87 +++++++ .../qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml | 74 ++++++ .../qwen3.5/sglang/b300-fp8/8k1k.yaml | 69 ++++++ .../qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml | 70 ++++++ .../qwen3.5/sglang/h100-fp8/8k1k.yaml | 77 ++++++ .../qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml | 69 ++++++ .../qwen3.5/sglang/h200-fp8/8k1k.yaml | 20 +- benchmarks/single_node/srt_eval.sh | 27 ++ benchmarks/single_node/srt_fixed_sequence.sh | 13 +- configs/nvidia-master.yaml | 230 +++++++++++++++--- configs/pilots/h200-srt.yaml | 62 ----- infx/srt_slurm/single_node.py | 61 ++++- perf-changelog.yaml | 29 +++ runners/launch_b200-cw.sh | 13 + runners/launch_b200-nb.sh | 14 ++ runners/launch_b200-nscale-slurm.sh | 11 + runners/launch_b300-dsxe.sh | 23 +- runners/launch_h100-cw.sh | 13 + runners/launch_h100-dgxc-slurm.sh | 16 +- runners/launch_h200-cw.sh | 13 + runners/launch_h200-dgxc-slurm.sh | 78 +----- runners/slurm_utils.sh | 100 +++++++- runners/srt-slurm/b200-cw.yaml | 6 + runners/srt-slurm/b200-nb.yaml | 6 + runners/srt-slurm/h100-cw.yaml | 6 + runners/srt-slurm/h200-cw.yaml | 6 + utils/test_srt_fixed_sequence.py | 47 +++- utils/test_srt_single_node.py | 70 +++++- 43 files changed, 2066 insertions(+), 248 deletions(-) create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt_eval.sh delete mode 100644 configs/pilots/h200-srt.yaml create mode 100644 runners/srt-slurm/b200-cw.yaml create mode 100644 runners/srt-slurm/b200-nb.yaml create mode 100644 runners/srt-slurm/h100-cw.yaml create mode 100644 runners/srt-slurm/h200-cw.yaml diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index 565d61d32f..2318c19b3d 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -306,10 +306,6 @@ jobs: if [[ -f runners/runtime_settings.sh ]]; then source runners/runtime_settings.sh fi - if [[ -n "$SRT_RECIPE" && "${RUNNER_NAME%%_*}" != h200-dgxc-slurm ]]; then - echo "The native single-node SRT pilot requires the H200 DGXC Slurm pool" >&2 - exit 1 - fi bash "./runners/launch_${RUNNER_NAME%%_*}.sh" if [ "${EVAL_ONLY}" = "true" ]; then diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..d006f22922 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml @@ -0,0 +1,84 @@ +# Serving settings from dsr1_fp4_b200_mtp.sh. +base: + schema: 2 + name: dsr1-fp4-b200-sglang-mtp-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: lmsysorg/sglang:v0.5.16-cu130 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + cuda-graph-max-bs: 256 + max-running-requests: 256 + mem-fraction-static: 0.85 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + expert-parallel-size: 1 + quantization: modelopt_fp4 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-piecewise-cuda-graph: true + attention-backend: trtllm_mla + moe-runner-backend: flashinfer_trtllm + stream-interval: 10 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-num-draft-tokens: 3 + speculative-eagle-topk: 1 + served-model-name: nvidia/DeepSeek-R1-0528-FP4-V2 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + SGLANG_RADIX_FORCE_MISS: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + chunked-prefill-size: 32768 + data-parallel-size: 4 + enable-dp-attention: true + enable-dp-attention-local-control-broadcast: true + enable-dp-lm-head: true + enable-prefill-delayer: true + expert-parallel-size: 4 + schedule-conservativeness: 3.33 + scheduler-recv-interval: 1 + benchmark: + env: + CONC: ['64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml new file mode 100644 index 0000000000..2fcd1bb81e --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml @@ -0,0 +1,80 @@ +# Serving settings from dsr1_fp4_b200.sh. +base: + schema: 2 + name: dsr1-fp4-b200-sglang-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + cuda-graph-max-bs: 256 + max-running-requests: 256 + mem-fraction-static: 0.85 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + expert-parallel-size: 1 + quantization: modelopt_fp4 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + enable-symm-mem: true + disable-piecewise-cuda-graph: true + attention-backend: trtllm_mla + moe-runner-backend: flashinfer_trtllm + stream-interval: 10 + served-model-name: nvidia/DeepSeek-R1-0528-FP4-V2 + enable-metrics: false + env: + SGLANG_RADIX_FORCE_MISS: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + chunked-prefill-size: 32768 + data-parallel-size: 4 + enable-dp-attention: true + enable-dp-attention-local-control-broadcast: true + enable-dp-lm-head: true + enable-prefill-delayer: true + expert-parallel-size: 4 + schedule-conservativeness: 3.33 + scheduler-recv-interval: 1 + benchmark: + env: + CONC: ['64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..aa2eb02bf7 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml @@ -0,0 +1,69 @@ +# Serving settings from dsr1_fp8_b200_mtp.sh. +base: + schema: 2 + name: dsr1-fp8-b200-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + data-parallel-size: 1 + cuda-graph-max-bs: 512 + max-running-requests: 512 + mem-fraction-static: 0.82 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-radix-cache: true + fp8-gemm-backend: flashinfer_trtllm + attention-backend: trtllm_mla + stream-interval: 30 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_trtllm + quantization: fp8 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-num-draft-tokens: 3 + speculative-eagle-topk: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30, 30, 30, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128', '256', '512'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml new file mode 100644 index 0000000000..06af0e4743 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml @@ -0,0 +1,74 @@ +# Serving settings from dsr1_fp8_b200.sh. +base: + schema: 2 + name: dsr1-fp8-b200-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + data-parallel-size: 1 + cuda-graph-max-bs: 128 + max-running-requests: 128 + mem-fraction-static: 0.82 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-radix-cache: true + attention-backend: trtllm_mla + stream-interval: 30 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_trtllm + quantization: fp8 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1: + benchmark: + env: + CONC: ['1', '2', '4'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + chunked-prefill-size: 8192 + cuda-graph-max-bs: 32 + max-prefill-tokens: 8192 + max-running-requests: 32 + mem-fraction-static: 0.95 + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml new file mode 100644 index 0000000000..935a533ac0 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml @@ -0,0 +1,74 @@ +# Serving settings from dsr1_fp4_b300.sh. +base: + schema: 2 + name: dsr1-fp4-b300-sglang-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: nvidia/DeepSeek-R1-0528-FP4-V2 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + cuda-graph-max-bs: 256 + max-running-requests: 256 + mem-fraction-static: 0.85 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + expert-parallel-size: 4 + quantization: modelopt_fp4 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + enable-symm-mem: true + disable-radix-cache: true + attention-backend: trtllm_mla + moe-runner-backend: flashinfer_trtllm + stream-interval: 10 + enable-metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep4: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128'] +zip_override_tp8_ep8: + roles: + agg: + gpus: 8 + args: + expert-parallel-size: 8 + scheduler-recv-interval: [10, 10, 10, 10, 30] + tensor-parallel-size: 8 + benchmark: + env: + CONC: ['1', '2', '4', '8', '16'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..cb8887141b --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml @@ -0,0 +1,69 @@ +# Serving settings from dsr1_fp8_b300_mtp.sh. +base: + schema: 2 + name: dsr1-fp8-b300-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.15.post1-cu130 + precision: fp8 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-R1-0528 + tensor-parallel-size: 8 + data-parallel-size: 1 + cuda-graph-max-bs: 512 + max-running-requests: 512 + mem-fraction-static: 0.82 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-radix-cache: true + fp8-gemm-backend: flashinfer_trtllm + attention-backend: trtllm_mla + stream-interval: 30 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_trtllm + quantization: fp8 + speculative-algorithm: EAGLE + speculative-num-steps: 2 + speculative-num-draft-tokens: 3 + speculative-eagle-topk: 1 + enable-metrics: false + env: + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + scheduler-recv-interval: [10, 10, 10, 10, 30, 30, 30, 30, 30, 30] + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32', '64', '128', '256', '512'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml new file mode 100644 index 0000000000..4fdbff56bb --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml @@ -0,0 +1,74 @@ +# Serving settings from dsr1_fp8_b300.sh. +base: + schema: 2 + name: dsr1-fp8-b300-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + served-model-name: deepseek-ai/DeepSeek-R1-0528 + tensor-parallel-size: 8 + data-parallel-size: 1 + cuda-graph-max-bs: 128 + max-running-requests: 128 + mem-fraction-static: 0.82 + kv-cache-dtype: fp8_e4m3 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + enable-flashinfer-allreduce-fusion: true + scheduler-recv-interval: 10 + disable-radix-cache: true + attention-backend: trtllm_mla + stream-interval: 30 + expert-parallel-size: 1 + moe-runner-backend: flashinfer_trtllm + quantization: fp8 + enable-metrics: false + env: + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1: + benchmark: + env: + CONC: ['1', '2', '4'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + chunked-prefill-size: 8192 + cuda-graph-max-bs: 32 + max-prefill-tokens: 8192 + max-running-requests: 32 + mem-fraction-static: 0.95 + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['1', '2', '4', '8', '16', '32'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml index b4a49010ee..ab5b935067 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml @@ -1,4 +1,4 @@ -# Parallel port of dsr1_fp8_h200_mtp.sh; real verification, embedded MTP head. +# Serving settings from dsr1_fp8_h200_mtp.sh. base: schema: 2 name: dsr1-fp8-h200-sglang-mtp-8k1k @@ -22,17 +22,11 @@ base: nodes: 1 workers: 1 gpus: 8 - env: - PYTHONNOUSERSITE: "1" - TORCH_CUDA_ARCH_LIST: "9.0" - SGLANG_ENABLE_SPEC_V2: "1" args: + trust-remote-code: true tensor-parallel-size: 8 data-parallel-size: 1 - ep-size: 1 - trust-remote-code: true - served-model-name: deepseek-ai/DeepSeek-R1-0528 - enable-metrics: false + expert-parallel-size: 1 disable-radix-cache: true max-running-requests: 256 cuda-graph-max-bs: 256 @@ -46,17 +40,22 @@ base: speculative-num-steps: 2 speculative-num-draft-tokens: 3 speculative-eagle-topk: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + TORCH_CUDA_ARCH_LIST: '9.0' + PYTHONNOUSERSITE: '1' benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh env: MODEL: deepseek-ai/DeepSeek-R1-0528 - ISL: "8192" - OSL: "1024" - RANDOM_RANGE_RATIO: "0.8" - USE_CHAT_TEMPLATE: "true" - + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' zip_override_concurrency: benchmark: env: - CONC: ["4", "8", "16", "32", "64"] + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml index b87ce4db55..bee0edbea2 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml @@ -1,5 +1,4 @@ -# Parallel replacement for dsr1-fp8-h200-sglang; not selected by production yet. -# Each concurrency is a separate native SRT job, matching the current sweep. +# Serving settings from dsr1_fp8_h200.sh. base: schema: 2 name: dsr1-fp8-h200-sglang-8k1k @@ -23,15 +22,10 @@ base: nodes: 1 workers: 1 gpus: 8 - env: - PYTHONNOUSERSITE: "1" - TORCH_CUDA_ARCH_LIST: "9.0" args: + trust-remote-code: true tensor-parallel-size: 8 data-parallel-size: 1 - trust-remote-code: true - served-model-name: deepseek-ai/DeepSeek-R1-0528 - enable-metrics: false disable-radix-cache: true max-running-requests: 256 cuda-graph-max-bs: 256 @@ -41,17 +35,22 @@ base: attention-backend: flashinfer stream-interval: 10 decode-log-interval: 1 + expert-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + TORCH_CUDA_ARCH_LIST: '9.0' + PYTHONNOUSERSITE: '1' benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh env: MODEL: deepseek-ai/DeepSeek-R1-0528 - ISL: "8192" - OSL: "1024" - RANDOM_RANGE_RATIO: "0.8" - USE_CHAT_TEMPLATE: "false" - + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' zip_override_concurrency: benchmark: env: - CONC: ["4", "8", "16", "32", "64"] + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..cb84cd7218 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml @@ -0,0 +1,96 @@ +# Serving settings from qwen3.5_fp4_b200_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp4-b200-sglang-mtp-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs: 4 + max-running-requests: 4 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + context-length: 9236 + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + benchmark: + env: + CONC: ['4'] +zip_override_tp2_ep1: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + max-running-requests: [4, 8, 16, 32, 64] + scheduler-recv-interval: [10, 30, 30, 30, 30] + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] +zip_override_tp2_ep2: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [16, 32, 64] + expert-parallel-size: 2 + linear-attn-backend: triton + linear-attn-decode-backend: flashinfer + linear-attn-prefill-backend: flashinfer + max-running-requests: [16, 32, 64] + scheduler-recv-interval: 30 + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml new file mode 100644 index 0000000000..3589bf278a --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml @@ -0,0 +1,73 @@ +# Serving settings from qwen3.5_fp4_b200.sh. +base: + schema: 2 + name: qwen3.5-fp4-b200-sglang-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: modelopt_fp4 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs: 4 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + context-length: 9236 + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + enable-metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + benchmark: + env: + CONC: ['4'] +zip_override_tp2_ep1: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + scheduler-recv-interval: [10, 30, 30, 30, 30, 30] + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..bf219a8c68 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml @@ -0,0 +1,82 @@ +# Serving settings from qwen3.5_fp8_b200_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp8-b200-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs-decode: 4 + max-running-requests: 4 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + mamba-full-memory-ratio: 0.37 + linear-attn-prefill-backend: flashinfer + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: Qwen/Qwen3.5-397B-A17B-FP8 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + context-length: 9236 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp8_ep1: + benchmark: + env: + CONC: ['4'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + cuda-graph-max-bs-decode: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + scheduler-recv-interval: [10, 30, 30, 30, 30, 30, 30] + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml new file mode 100644 index 0000000000..1236c91d90 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml @@ -0,0 +1,78 @@ +# Serving settings from qwen3.5_fp8_b200.sh. +base: + schema: 2 + name: qwen3.5-fp8-b200-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-full-memory-ratio: 0.37 + linear-attn-prefill-backend: flashinfer + attention-backend: trtllm_mha + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs-decode: 1 + max-prefill-tokens: 32768 + chunked-prefill-size: 32768 + mem-fraction-static: 0.86 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + context-length: 9236 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1: + roles: + agg: + args: + cuda-graph-max-bs-decode: [1, 2, 4] + benchmark: + env: + CONC: ['1', '2', '4'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + cuda-graph-max-bs-decode: [2, 4, 8, 16, 32, 64, 128, 256, 512, 320, 384, 448, 640] + scheduler-recv-interval: [10, 10, 30, 30, 30, 30, 30, 30, 30, 30, 30, 30, 30] + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['2', '4', '8', '16', '32', '64', '128', '256', '512', '320', '384', '448', '640'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..21096c3047 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml @@ -0,0 +1,91 @@ +# Serving settings from qwen3.5_fp4_b300_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp4-b300-sglang-mtp-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + mamba-scheduler-strategy: no_buffer + quantization: modelopt_fp4 + fp4-gemm-backend: flashinfer_cutlass + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + cuda-graph-max-bs: 4 + max-running-requests: 128 + mem-fraction-static: 0.8 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + context-length: 9236 + disable-radix-cache: true + attention-backend: trtllm_mha + mm-attention-backend: triton_attn + moe-runner-backend: flashinfer_trtllm + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + stream-interval: 30 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + enable-metrics: false + env: + NCCL_NVLS_ENABLE: '1' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONUNBUFFERED: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + scheduler-recv-interval: [10, 10, 30, 30, 30, 30] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] +zip_override_tp2_ep2: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + expert-parallel-size: 2 + scheduler-recv-interval: [10, 10, 30, 30, 30, 30] + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml new file mode 100644 index 0000000000..7484259bfa --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml @@ -0,0 +1,87 @@ +# Serving settings from qwen3.5_fp4_b300.sh. +base: + schema: 2 + name: qwen3.5-fp4-b300-sglang-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp4 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + mamba-scheduler-strategy: no_buffer + quantization: modelopt_fp4 + fp4-gemm-backend: flashinfer_cutlass + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + cuda-graph-max-bs: 4 + max-running-requests: 128 + mem-fraction-static: 0.8 + chunked-prefill-size: 32768 + max-prefill-tokens: 32768 + context-length: 9236 + disable-radix-cache: true + attention-backend: trtllm_mha + mm-attention-backend: triton_attn + moe-runner-backend: flashinfer_trtllm + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + stream-interval: 30 + enable-metrics: false + env: + NCCL_NVLS_ENABLE: '1' + SGL_ENABLE_JIT_DEEPGEMM: 'false' + SGLANG_ENABLE_FLASHINFER_GEMM: 'true' + PYTHONUNBUFFERED: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + scheduler-recv-interval: [10, 10, 30, 30, 30, 30] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] +zip_override_tp2_ep2: + roles: + agg: + gpus: 2 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + expert-parallel-size: 2 + scheduler-recv-interval: [10, 10, 30, 30, 30, 30] + tensor-parallel-size: 2 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..5489fe7106 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml @@ -0,0 +1,74 @@ +# Serving settings from qwen3.5_fp8_b300_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp8-b300-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + mm-attention-backend: triton_attn + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs: 4 + max-running-requests: 4 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: Qwen/Qwen3.5-397B-A17B-FP8 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + context-length: 9236 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml new file mode 100644 index 0000000000..789199c51d --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml @@ -0,0 +1,69 @@ +# Serving settings from qwen3.5_fp8_b300.sh. +base: + schema: 2 + name: qwen3.5-fp8-b300-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-cu130 + precision: fp8 + resources: + gpu_type: b300 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + enable-symm-mem: true + disable-radix-cache: true + quantization: fp8 + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + attention-backend: trtllm_mha + mm-attention-backend: triton_attn + moe-runner-backend: flashinfer_trtllm + cuda-graph-max-bs: 4 + max-running-requests: 4 + max-prefill-tokens: 16384 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + stream-interval: 50 + scheduler-recv-interval: 10 + tokenizer-worker-num: 6 + tokenizer-path: Qwen/Qwen3.5-397B-A17B-FP8 + context-length: 9236 + enable-metrics: false + env: + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..0bd0858956 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml @@ -0,0 +1,70 @@ +# Serving settings from qwen3.5_fp8_h100_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp8-h100-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.19-cu130 + precision: fp8 + resources: + gpu_type: h100 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + expert-parallel-size: 8 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-flashinfer-allreduce-fusion: true + max-running-requests: 64 + chunked-prefill-size: 8192 + decode-log-interval: 1 + mem-fraction-static: 0.75 + cuda-graph-max-bs: 4 + context-length: 9236 + kv-cache-dtype: fp8_e4m3 + quantization: fp8 + attention-backend: flashinfer + stream-interval: 50 + tokenizer-worker-num: 6 + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + trust-remote-code: true + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32] + benchmark: + env: + CONC: ['4', '8', '16', '32'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml new file mode 100644 index 0000000000..b1b117d767 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml @@ -0,0 +1,77 @@ +# Serving settings from qwen3.5_fp8_h100.sh. +base: + schema: 2 + name: qwen3.5-fp8-h100-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp8 + resources: + gpu_type: h100 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-flashinfer-allreduce-fusion: true + max-running-requests: 256 + chunked-prefill-size: 16384 + decode-log-interval: 1 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 1 + context-length: 9236 + kv-cache-dtype: fp8_e4m3 + quantization: fp8 + attention-backend: flashinfer + stream-interval: 50 + tokenizer-worker-num: 6 + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + enable-symm-mem: true + trust-remote-code: true + scheduler-recv-interval: 2 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1: + roles: + agg: + args: + cuda-graph-max-bs: [1, 2, 4, 8] + scheduler-recv-interval: [2, 2, 2, 60] + benchmark: + env: + CONC: ['1', '2', '4', '8'] +zip_override_tp8_ep8: + roles: + agg: + args: + cuda-graph-max-bs: [16, 32, 64, 128, 256] + expert-parallel-size: 8 + scheduler-recv-interval: [30, 1200, 600, 1920, 1920] + benchmark: + env: + CONC: ['16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..fd74c5af14 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml @@ -0,0 +1,69 @@ +# Serving settings from qwen3.5_fp8_h200_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp8-h200-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.14-cu130 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + tensor-parallel-size: 8 + expert-parallel-size: 8 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + enable-flashinfer-allreduce-fusion: true + max-running-requests: 128 + chunked-prefill-size: 16384 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 4 + context-length: 9472 + kv-cache-dtype: fp8_e4m3 + quantization: fp8 + attention-backend: flashinfer + stream-interval: 50 + tokenizer-worker-num: 6 + mamba-ssm-dtype: bfloat16 + disable-radix-cache: true + trust-remote-code: true + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-num-draft-tokens: 4 + speculative-eagle-topk: 1 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_ENABLE_SPEC_V2: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml index 8ce11d7c78..41d490f7cb 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml @@ -1,4 +1,4 @@ -# Parallel port of qwen3.5_fp8_h200.sh; graph capture follows concurrency. +# Serving settings from qwen3.5_fp8_h200.sh. base: schema: 2 name: qwen3.5-fp8-h200-sglang-8k1k @@ -24,10 +24,7 @@ base: gpus: 8 args: tensor-parallel-size: 8 - data-parallel-size: 1 expert-parallel-size: 8 - served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 - enable-metrics: false reasoning-parser: qwen3 tool-call-parser: qwen3_coder enable-flashinfer-allreduce-fusion: true @@ -45,17 +42,18 @@ base: mamba-ssm-dtype: bfloat16 disable-radix-cache: true trust-remote-code: true + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh env: MODEL: Qwen/Qwen3.5-397B-A17B-FP8 - ISL: "8192" - OSL: "1024" - RANDOM_RANGE_RATIO: "0.8" - USE_CHAT_TEMPLATE: "false" - CONC: "4" - + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' zip_override_concurrency: roles: agg: @@ -63,4 +61,4 @@ zip_override_concurrency: cuda-graph-max-bs: [4, 8, 16, 32, 64] benchmark: env: - CONC: ["4", "8", "16", "32", "64"] + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt_eval.sh b/benchmarks/single_node/srt_eval.sh new file mode 100644 index 0000000000..e6c8af8af4 --- /dev/null +++ b/benchmarks/single_node/srt_eval.sh @@ -0,0 +1,27 @@ +#!/usr/bin/env bash + +# SRT owns readiness and lifecycle; InferenceX owns evaluation and its artifacts. +set -eo pipefail +if [[ $# != 2 || -z "$1" || -z "$2" ]]; then + echo "Usage: $0 endpoint status-file" >&2 + exit 1 +fi +SRT_EVAL_STATUS_FILE="$2" +trap 'rc=$?; printf "%s\n" "$rc" > "$SRT_EVAL_STATUS_FILE"' EXIT + +source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" +check_env_vars MODEL MODEL_NAME CONC TP EP_SIZE DP_ATTENTION IS_MULTINODE MAX_MODEL_LEN +export PORT="${1##*:}" +if [[ ! "$PORT" =~ ^[1-9][0-9]*$ || "$IS_MULTINODE" != false ]]; then + echo "ERROR: single-node eval requires a local endpoint and single-node metadata" >&2 + exit 1 +fi +cd "$INFERENCEX_REPO_ROOT" +if [[ -d /model ]]; then + export MODEL_PATH=/model +fi + +eval_rc=0 +run_eval --framework lm-eval --port "$PORT" || eval_rc=$? +append_lm_eval_summary || eval_rc=1 +exit "$eval_rc" diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index ac77ebe4ed..3f2f081d74 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -5,6 +5,12 @@ set -eo pipefail source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only check_env_vars MODEL CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME RESULT_DIR \ SRT_FRONTEND_HOST SRT_FRONTEND_PORT RUN_EVAL EVAL_ONLY GPU_MONITOR_INTERVAL USE_CHAT_TEMPLATE +for name in RUN_EVAL EVAL_ONLY; do + if [[ "${!name}" != true && "${!name}" != false ]]; then + echo "ERROR: $name must be true or false" >&2 + exit 1 + fi +done SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" CLIENT_ARGS=() case "$USE_CHAT_TEMPLATE" in @@ -20,13 +26,6 @@ for name in CONC ISL OSL SRT_FRONTEND_PORT GPU_MONITOR_INTERVAL; do fi done -# The initial parallel port supports throughput only. Eval context and artifact -# forwarding must be connected before production cutover. -if [[ "$RUN_EVAL" != false || "$EVAL_ONLY" != false ]]; then - echo "ERROR: the single-node SRT pilot does not support evals yet" >&2 - exit 1 -fi - if [[ ! -d "$RESULT_DIR" ]]; then echo "ERROR: RESULT_DIR must be an existing runtime-provided directory" >&2 exit 1 diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 9cecfd0c92..19e90681ce 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -872,8 +872,17 @@ dsr1-fp4-b200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 1, conc-end: 32 } - - { tp: 4, ep: 4, dp-attn: true, conc-start: 64, conc-end: 256 } + - tp: 4 + ep: 1 + conc-start: 1 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-start: 64 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml # agentic-coding: temporarily disabled — blocked by e2e-tests.yml artifact # name mismatch (downloads `agentic_*` but benchmark-tmpl.yml uploads as # `bmk_agentic_*`). Re-enable once that workflow is aligned. @@ -896,8 +905,19 @@ dsr1-fp4-b200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 1, conc-end: 32, spec-decoding: mtp } - - { tp: 4, ep: 4, dp-attn: true, conc-start: 64, conc-end: 256, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 1 + conc-end: 32 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-start: 64 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml dsv4-fp4-b200-sglang-agentic-hicache-mtp: image: lmsysorg/sglang:v0.5.19-cu130 @@ -948,8 +968,16 @@ dsr1-fp4-b300-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 4, conc-start: 1, conc-end: 128 } - - { tp: 8, ep: 8, conc-start: 1, conc-end: 16 } + - tp: 4 + ep: 4 + conc-start: 1 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml + - tp: 8 + ep: 8 + conc-start: 1 + conc-end: 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml dsr1-fp4-b200-trt: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -1007,8 +1035,16 @@ dsr1-fp8-b200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 4 } - - { tp: 4, ep: 1, conc-start: 1, conc-end: 32 } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 1 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml # NOTE: At the time of submission, https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 # does not have a B300-specific recipe, so this config reuses the existing DSR1 FP8 @@ -1026,8 +1062,16 @@ dsr1-fp8-b300-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 4 } - - { tp: 4, ep: 1, conc-start: 1, conc-end: 32 } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 1 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml dsv4-fp4-b300-sglang-agentic-hicache-mtp: image: lmsysorg/sglang:nightly-dev-20260901-07c8f729 @@ -1069,9 +1113,23 @@ qwen3.5-fp8-b200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 1, conc-end: 4 } - - { tp: 4, ep: 1, conc-start: 2, conc-end: 512 } - - { tp: 4, ep: 1, conc-list: [320, 384, 448, 640] } + - tp: 8 + conc-start: 1 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 2 + conc-end: 512 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-list: + - 320 + - 384 + - 448 + - 640 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml qwen3.5-fp8-b200-sglang-agentic-mtp: image: lmsysorg/sglang:nightly-dev-cu13-20260914-4358a161 @@ -1103,8 +1161,16 @@ qwen3.5-fp4-b200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 4 } - - { tp: 2, ep: 1, conc-start: 4, conc-end: 128 } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml + - tp: 2 + ep: 1 + conc-start: 4 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml qwen3.5-fp4-b200-sglang-mtp: image: lmsysorg/sglang:v0.5.14-cu130 @@ -1119,9 +1185,26 @@ qwen3.5-fp4-b200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 4, spec-decoding: mtp } - - { tp: 2, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } - - { tp: 2, ep: 2, conc-list: [16, 32, 64], spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 4 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml + - tp: 2 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml + - tp: 2 + ep: 2 + conc-list: + - 16 + - 32 + - 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml qwen3.5-fp8-b200-sglang-mtp: image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 @@ -1136,8 +1219,18 @@ qwen3.5-fp8-b200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 4, spec-decoding: mtp } - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 4 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml qwen3.5-fp8-b300-sglang-mtp: image: lmsysorg/sglang:v0.5.12-cu130 @@ -1152,7 +1245,12 @@ qwen3.5-fp8-b300-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml qwen3.5-fp8-b300-sglang: image: lmsysorg/sglang:v0.5.12-cu130 @@ -1167,7 +1265,11 @@ qwen3.5-fp8-b300-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256 } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml qwen3.5-fp4-b300-sglang: image: lmsysorg/sglang:v0.5.14-cu130 @@ -1182,8 +1284,16 @@ qwen3.5-fp4-b300-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 128 } - - { tp: 2, ep: 2, conc-start: 4, conc-end: 128 } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml + - tp: 2 + ep: 2 + conc-start: 4 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml # Qwen3.5-397B-A17B NVFP4 single-node SGLang sweep using 4 of 8 RTX PRO # 6000 Blackwell GPUs. Both arms use ordinary NCCL collectives on PCIe. @@ -1234,8 +1344,18 @@ qwen3.5-fp4-b300-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 128, spec-decoding: mtp } - - { tp: 2, ep: 2, conc-start: 4, conc-end: 128, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 128 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml + - tp: 2 + ep: 2 + conc-start: 4 + conc-end: 128 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml # Kimi K3 is a 2.8T MXFP4 MoE served on H200 nodes. # These are the three aggregated strategy classes from the official H200 @@ -1343,7 +1463,12 @@ dsr1-fp8-b200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 512, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 512 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml # NOTE: At the time of submission, https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 # does not have a B300-specific recipe, so this config reuses the existing DSR1 FP8 @@ -1361,7 +1486,12 @@ dsr1-fp8-b300-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 512, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 512 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml kimik3-fp4-b300-vllm-agentic-dspark: # TP8 x DCP8 with Mooncake as the external KV tier. The recipe drafts with @@ -1438,7 +1568,10 @@ dsr1-fp8-h200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml dsr1-fp8-h200-sglang-mtp: image: lmsysorg/sglang:v0.5.19-cu130 @@ -1453,7 +1586,12 @@ dsr1-fp8-h200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml # DeepSeek-V4-Pro AgentX on one aggregated TP8 H200 worker. Keep the serving # topology fixed and sweep only concurrency to produce the Pareto curve. @@ -1519,7 +1657,11 @@ qwen3.5-fp8-h200-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml qwen3.5-fp8-h200-sglang-mtp: image: lmsysorg/sglang:v0.5.14-cu130 @@ -1534,7 +1676,12 @@ qwen3.5-fp8-h200-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 8, conc-start: 4, conc-end: 128, spec-decoding: mtp } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 128 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml dsr1-fp8-h200-trt: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -3950,8 +4097,16 @@ qwen3.5-fp8-h100-sglang: osl: 1024 require-power: true search-space: - - { tp: 8, ep: 1, conc-start: 1, conc-end: 8 } - - { tp: 8, ep: 8, conc-start: 16, conc-end: 256 } + - tp: 8 + ep: 1 + conc-start: 1 + conc-end: 8 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml + - tp: 8 + ep: 8 + conc-start: 16 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml qwen3.5-fp8-h100-sglang-mtp: image: lmsysorg/sglang:v0.5.19-cu130 @@ -3967,7 +4122,12 @@ qwen3.5-fp8-h100-sglang-mtp: osl: 1024 require-power: true search-space: - - { tp: 8, ep: 8, conc-start: 4, conc-end: 32, spec-decoding: mtp } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 32 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml qwen3.5-fp4-gb300-dynamo-sglang: image: lmsysorg/sglang:v0.5.14-cu130 diff --git a/configs/pilots/h200-srt.yaml b/configs/pilots/h200-srt.yaml deleted file mode 100644 index bffc2206cf..0000000000 --- a/configs/pilots/h200-srt.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Opt-in native SRT pilot. Production coverage remains in nvidia-master.yaml. -dsr1-fp8-h200-sglang: - image: lmsysorg/sglang:v0.5.12-cu130 - model: deepseek-ai/DeepSeek-R1-0528 - model-prefix: dsr1 - runner: cluster:h200-dgxc - precision: fp8 - framework: sglang - multinode: false - scenarios: - fixed-seq-len: - - isl: 8192 - osl: 1024 - search-space: - - tp: 8 - conc-list: [4, 16, 64] - srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml:base - -dsr1-fp8-h200-sglang-mtp: - image: lmsysorg/sglang:v0.5.19-cu130 - model: deepseek-ai/DeepSeek-R1-0528 - model-prefix: dsr1 - runner: cluster:h200-dgxc - precision: fp8 - framework: sglang - multinode: false - scenarios: - fixed-seq-len: - - isl: 8192 - osl: 1024 - search-space: - - tp: 8 - ep: 1 - spec-decoding: mtp - conc-list: [4, 16, 64] - srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml:base - -qwen3.5-fp8-h200-sglang: - image: lmsysorg/sglang:v0.5.14-cu130 - model: Qwen/Qwen3.5-397B-A17B-FP8 - model-prefix: qwen3.5 - runner: cluster:h200-dgxc - precision: fp8 - framework: sglang - multinode: false - scenarios: - fixed-seq-len: - - isl: 8192 - osl: 1024 - search-space: - - tp: 8 - ep: 8 - conc-list: [4] - srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[0] - - tp: 8 - ep: 8 - conc-list: [16] - srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[2] - - tp: 8 - ep: 8 - conc-list: [64] - srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml:zip_override_concurrency[4] diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index 6c355eb704..2fc1233383 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -7,29 +7,44 @@ import os from collections.abc import Mapping from pathlib import Path +from typing import Any import yaml from infx.srt_slurm.synthetic_acceptance import selected_recipes, spec_parameters -def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: - """Reject mismatched metadata before allocation; retain recipe-owned server settings.""" +def select_recipe(config: str, environment: Mapping[str, str]) -> tuple[str, dict[str, Any]]: + """Resolve a matrix point to one native variant, never submit an entire sweep.""" path, _, selector = config.partition(":") raw = yaml.safe_load(Path(path).read_text()) if not isinstance(raw, dict): raise ValueError("Recipe must be a mapping") recipes = selected_recipes(raw, selector or None) - if len(recipes) != 1: - raise ValueError("A single-node matrix point must select exactly one SRT recipe") - recipe = recipes[0][1] + matches = [] + errors = [] + for name, recipe in recipes: + try: + validate_recipe(recipe, environment) + except ValueError as exc: + errors.append(f"{name}: {exc}") + else: + matches.append((f"{path}:{name}" if name else path, recipe)) + if len(matches) != 1: + detail = "; ".join(errors) if not matches else ", ".join(name for name, _ in matches) + raise ValueError(f"Expected exactly one matching single-node SRT recipe; {detail}") + return matches[0] + + +def validate_recipe(recipe: dict[str, Any], environment: Mapping[str, str]) -> None: + """Reject metadata mismatches without overwriting recipe-owned server settings.""" role = recipe["roles"]["agg"] args = role["args"] benchmark = recipe["benchmark"] workload = benchmark["env"] spec = spec_parameters(role, "sglang") if spec and spec["method"] not in {"eagle", "nextn"}: - raise ValueError("Single-node SRT pilot supports only native MTP or no speculation") + raise ValueError("Single-node SRT supports only native MTP or no speculation") speculation = "mtp" if spec else "none" expected = { "engine": (recipe["engine"], environment["FRAMEWORK"]), @@ -37,7 +52,14 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: "image": (recipe["model"]["container"], environment["IMAGE"]), "precision": (recipe["model"]["precision"], environment["PRECISION"]), "tensor-parallel-size": (args["tensor-parallel-size"], int(environment["TP"])), - "data-parallel-size": (args["data-parallel-size"], 1), + "data-parallel-size": ( + args.get("data-parallel-size", 1), + int(environment["TP"]) if environment["DP_ATTENTION"] == "true" else 1, + ), + "DP_ATTENTION": ( + args.get("enable-dp-attention", False), + environment["DP_ATTENTION"] == "true", + ), "expert-parallel-size": ( args.get("expert-parallel-size", args.get("ep-size", 1)), int(environment["EP_SIZE"]), @@ -55,27 +77,41 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: expected["CONC"] = (str(workload["CONC"]), environment["CONC"]) for name in ("ISL", "OSL", "RANDOM_RANGE_RATIO"): expected[name] = (str(workload[name]), environment[name]) - # Other topology/eval paths remain on their current launchers until ported. + # Multi-node and AgentX workloads use their existing connector. for name, value in { "FRAMEWORK": "sglang", "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", - "DP_ATTENTION": "false", "IS_AGENTIC": "0", - "RUN_EVAL": "false", - "EVAL_ONLY": "false", }.items(): expected[name] = (environment[name], value) for name, (actual, wanted) in expected.items(): if actual != wanted: raise ValueError(f"Single-node SRT {name}: recipe/matrix {actual!r} != {wanted!r}") + + +def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: + """Bind only runtime-owned values after validating the selected recipe.""" + _, recipe = select_recipe(config, environment) + for name in ("RUN_EVAL", "EVAL_ONLY", "DP_ATTENTION"): + if environment[name] not in {"true", "false"}: + raise ValueError(f"{name} must be true or false") overrides = [] for name in ("CONC", "RESULT_FILENAME", "GPU_MONITOR_INTERVAL", "RUN_EVAL", "EVAL_ONLY"): value = environment[name] if not value: raise ValueError(f"Missing runtime input: {name}") + # Native --set broadcasts into zip groups. CONC already matched above; + # replacing its list could collapse the selected variant's index. + if name == "CONC" and name in recipe["benchmark"]["env"]: + continue overrides += ["--set", f"benchmark.env.{name}={json.dumps(value)}"] + if environment["EVAL_ONLY"] == "true": + context = int(environment["MAX_MODEL_LEN"]) + if context <= 0: + raise ValueError("MAX_MODEL_LEN must be positive") + overrides += ["--set", f"roles.agg.args.context-length={context}"] return [*overrides, "--set", 'benchmark.env.RESULT_DIR="/logs"'] @@ -104,8 +140,9 @@ def main() -> None: parsed = parser.parse_args() try: if parsed.command == "prepare": + config, _ = select_recipe(parsed.recipe, os.environ) arguments = runtime_arguments(parsed.recipe, os.environ) - parsed.output.write_bytes("\0".join([*arguments, ""]).encode()) + parsed.output.write_bytes("\0".join([config, *arguments, ""]).encode()) else: print("\n".join(submission_fields(parsed.manifest))) except (OSError, ValueError, KeyError, TypeError, yaml.YAMLError) as exc: diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 60fafd80eb..8dc1cacdc0 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8610,3 +8610,32 @@ - "Extend the opt-in native H200 SRT pilot to DeepSeek-R1 MTP and Qwen3.5 EP8, preserving real verification, chat formatting, and concurrency-dependent graph capture; select 4/16/64 for sequential regression checks without production cutover" - "将显式启用的原生 H200 SRT 试点扩展到 DeepSeek-R1 MTP 和 Qwen3.5 EP8,保留真实验证、聊天模板及随并发变化的图捕获;选择 4/16/64 逐点检查回归,尚不切换生产路由" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-b200-sglang + - dsr1-fp4-b200-sglang-mtp + - dsr1-fp4-b300-sglang + - dsr1-fp8-b200-sglang + - dsr1-fp8-b300-sglang + - qwen3.5-fp8-b200-sglang + - qwen3.5-fp4-b200-sglang + - qwen3.5-fp4-b200-sglang-mtp + - qwen3.5-fp8-b200-sglang-mtp + - qwen3.5-fp8-b300-sglang-mtp + - qwen3.5-fp8-b300-sglang + - qwen3.5-fp4-b300-sglang + - qwen3.5-fp4-b300-sglang-mtp + - dsr1-fp8-b200-sglang-mtp + - dsr1-fp8-b300-sglang-mtp + - dsr1-fp8-h200-sglang + - dsr1-fp8-h200-sglang-mtp + - qwen3.5-fp8-h200-sglang + - qwen3.5-fp8-h200-sglang-mtp + - qwen3.5-fp8-h100-sglang + - qwen3.5-fp8-h100-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Cut over 21 active H100/H200/B200/B300 SGLang fixed-sequence recipes to native SRT-Slurm across their Slurm pools, retaining all 181 configured points, per-point serving settings, real MTP verification, shared eval dispatch, and existing result/power artifacts" + - "将 21 个活跃的 H100/H200/B200/B300 SGLang 定长配方切换到各自 Slurm 池的原生 SRT-Slurm 路径,保留全部 181 个配置点、逐点服务参数、真实 MTP 验证、共享 eval 调度及现有结果和功耗产物" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/runners/launch_b200-cw.sh b/runners/launch_b200-cw.sh index 8be2bfd9dd..576f2ef026 100644 --- a/runners/launch_b200-cw.sh +++ b/runners/launch_b200-cw.sh @@ -3,6 +3,19 @@ source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 check_env_vars IS_MULTINODE +EXECUTION_PATH=legacy-single-node +if [[ "$IS_MULTINODE" != true && -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + HF_HUB_CACHE_MOUNT=/tmp/gharunner/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + SRT_SQUASH_FILE="/tmp/gharunner/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node b200-cw + exit $? +fi + export HF_HUB_CACHE_MOUNT="/tmp/gharunner/hf-hub-cache" export PORT=8888 diff --git a/runners/launch_b200-nb.sh b/runners/launch_b200-nb.sh index 5ea7691e39..92f1e62e69 100644 --- a/runners/launch_b200-nb.sh +++ b/runners/launch_b200-nb.sh @@ -3,6 +3,20 @@ source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 check_env_vars IS_MULTINODE +EXECUTION_PATH=legacy-single-node +if [[ "$IS_MULTINODE" != true && -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + HF_HUB_CACHE_MOUNT=/mnt/data/gharunners/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + unset SRT_SQUASH_FILE + export UCX_NET_DEVICES=eth0 + launch_srt_single_node b200-nb + exit $? +fi + HF_HUB_CACHE_MOUNT="/mnt/data/gharunners/hf-hub-cache/" PARTITION="main" FRAMEWORK_SUFFIX=$([[ "$FRAMEWORK" == "trt" ]] && printf '_trt' || printf '') diff --git a/runners/launch_b200-nscale-slurm.sh b/runners/launch_b200-nscale-slurm.sh index 4ef5fcc1d3..86f1825aaf 100755 --- a/runners/launch_b200-nscale-slurm.sh +++ b/runners/launch_b200-nscale-slurm.sh @@ -49,6 +49,8 @@ if uses_native_srt_lane; then LAUNCH_PATH="native-srt" elif [[ "$IS_MULTINODE" == "true" ]]; then LAUNCH_PATH="multinode-srt" +elif [[ -n "${SRT_RECIPE:-}" ]]; then + LAUNCH_PATH="native-single-node" else LAUNCH_PATH="single-node" fi @@ -143,6 +145,15 @@ else exit 1 fi +if [[ "$LAUNCH_PATH" == native-single-node ]]; then + HF_HUB_CACHE_MOUNT=/data/home/sa-shared/gharunners/hf-hub-cache + SRT_MODEL_PATH="$MODEL_PATH" + SRT_SQUASH_FILE="$B200_SQUASH_DIR/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node b200-nscale-slurm \ + --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" + exit $? +fi + # --------------------------------------------------------------------------- # Container import helpers shared by both srt-slurm paths # --------------------------------------------------------------------------- diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index 6d563ddf45..a841fa2fcb 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -45,7 +45,6 @@ STAGED_MODELS=( Qwen3.8-2.4T-A95B-FP8 ) -mkdir -p "$SQUASH_DIR" set -x # Keep this definition above the IS_MULTINODE branch: both paths call it, and @@ -62,6 +61,8 @@ import_squash_image() { local sqsh="$2" local lock="${2}.lock" + mkdir -p "$SQUASH_DIR" + if unsquashfs -l "$sqsh" > /dev/null 2>&1; then echo "Squash file already present, skipping import: $sqsh" return 0 @@ -83,7 +84,25 @@ import_squash_image() { test -r "$sqsh" || { echo "Error: squash file not readable: $sqsh" >&2; exit 1; } } -if [[ "$IS_MULTINODE" == "true" ]]; then +EXECUTION_PATH=legacy-single-node +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi + +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + check_env_vars B300_HF_CACHE_HOST_DIR + HF_HUB_CACHE_MOUNT="$B300_HF_CACHE_HOST_DIR/hub" + SRT_MODEL_PATH="$MODEL_ROOT/${MODEL##*/}" + if [[ "$MODEL" == nvidia/DeepSeek-R1-0528-FP4-V2 ]]; then + SRT_MODEL_PATH="$MODEL_ROOT/DeepSeek-R1-0528-NVFP4-v2" + fi + SRT_SQUASH_FILE="$SQUASH_DIR/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node b300-dsxe \ + --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" \ + --var MODEL_ROOT "$MODEL_ROOT" +elif [[ "$EXECUTION_PATH" == multinode ]]; then if [[ $FRAMEWORK != "dynamo-sglang" && $FRAMEWORK != "dynamo-trt" && $FRAMEWORK != "dynamo-vllm" ]]; then echo "Unsupported framework: $FRAMEWORK. Supported frameworks are: dynamo-trt, dynamo-sglang, dynamo-vllm" diff --git a/runners/launch_h100-cw.sh b/runners/launch_h100-cw.sh index 26c4c2ff11..7252f11362 100644 --- a/runners/launch_h100-cw.sh +++ b/runners/launch_h100-cw.sh @@ -3,6 +3,19 @@ source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 check_env_vars IS_MULTINODE +EXECUTION_PATH=legacy-single-node +if [[ "$IS_MULTINODE" != true && -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + HF_HUB_CACHE_MOUNT=/mnt/vast/gharunner/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + SRT_SQUASH_FILE="/mnt/vast/gharunner/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node h100-cw + exit $? +fi + export HF_HUB_CACHE_MOUNT="/mnt/vast/gharunner/hf-hub-cache" PARTITION="h100" SQUASH_FILE="/mnt/vast/gharunner/squash/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" diff --git a/runners/launch_h100-dgxc-slurm.sh b/runners/launch_h100-dgxc-slurm.sh index 3eccf3eba5..8040a9ea0e 100644 --- a/runners/launch_h100-dgxc-slurm.sh +++ b/runners/launch_h100-dgxc-slurm.sh @@ -14,7 +14,21 @@ SPEC_SUFFIX=$([[ "$SPEC_DECODING" == "mtp" ]] && printf '_mtp' || printf '') set -x -if [[ "$IS_MULTINODE" == "true" ]]; then +EXECUTION_PATH=legacy-single-node +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi + +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + HF_HUB_CACHE_MOUNT=/mnt/nfs/sa-shared/gharunners/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + SRT_SQUASH_FILE="/mnt/nfs/lustre/containers/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node h100-dgxc-slurm \ + --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" \ + --var CONTAINER_KEY "$IMAGE" +elif [[ "$EXECUTION_PATH" == multinode ]]; then # Recipes name HF model IDs; resolve them to pre-staged paths so the shared # cluster does not re-download. SRT_SLURM_MODEL_PREFIX must match the diff --git a/runners/launch_h200-cw.sh b/runners/launch_h200-cw.sh index 2ee0b76f6e..ff91570102 100644 --- a/runners/launch_h200-cw.sh +++ b/runners/launch_h200-cw.sh @@ -3,6 +3,19 @@ source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 check_env_vars IS_MULTINODE +EXECUTION_PATH=legacy-single-node +if [[ "$IS_MULTINODE" != true && -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + HF_HUB_CACHE_MOUNT=/mnt/vast/gharunner/hf-hub-cache + SRT_MODEL_PATH="hf:$MODEL" + SRT_SQUASH_FILE="/mnt/vast/gharunner/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node h200-cw + exit $? +fi + export HF_HUB_CACHE_MOUNT="/mnt/vast/gharunner/hf-hub-cache" export AIPERF_MMAP_CACHE_HOST_PATH="/mnt/vast/gharunner/ai-perf-cache" export PORT=8888 diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index 36889a423c..c7f6127711 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -23,81 +23,11 @@ elif [[ -n "${SRT_RECIPE:-}" ]]; then fi if [[ "$EXECUTION_PATH" == native-single-node ]]; then - check_env_vars GITHUB_WORKSPACE SRT_RECIPE FRAMEWORK MODEL MODEL_PREFIX IMAGE PRECISION \ - TP PP_SIZE DCP_SIZE PCP_SIZE EP_SIZE DP_ATTENTION GPU_COUNT IS_AGENTIC SPEC_DECODING \ - CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME GPU_MONITOR_INTERVAL SRT_MODEL_PATH HF_HUB_CACHE - if [[ ( "$MODEL_PREFIX" != dsr1 && "$MODEL_PREFIX" != qwen3.5 ) || "$PRECISION" != fp8 ]]; then - echo "ERROR: the native H200 pilot supports dsr1/fp8 and qwen3.5/fp8 only" >&2 - exit 1 - fi - SRT_PILOT_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-single.XXXXXX") - SRTCTL_ROOT="$SRT_PILOT_ROOT/checkout" - export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" - setup_srt_slurm "$SRTCTL_ROOT" "$FRAMEWORK" 0 - if ! command -v uv >/dev/null; then - curl -LsSf https://astral.sh/uv/install.sh | sh - source "$HOME/.local/bin/env" - fi - uv venv .venv - source .venv/bin/activate - uv pip install -e . - export PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" - - python3 -m infx.srt_slurm.single_node prepare "$GITHUB_WORKSPACE/$SRT_RECIPE" "$SRT_PILOT_ROOT/arguments" - mapfile -d '' -t SRT_RUNTIME_ARGS < "$SRT_PILOT_ROOT/arguments" - SQUASH_FILE="/data/containers/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" - if [[ ! -r "$SRT_MODEL_PATH/config.json" ]]; then - echo "ERROR: staged model config is unavailable: $SRT_MODEL_PATH" >&2 - exit 1 - fi - NGINX_SQUASH_FILE=/data/containers/nginx+1.27.4.sqsh - write_srt_cluster_config h200-dgxc-slurm srtslurm.yaml 0 \ - --var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" \ + SRT_SQUASH_FILE="/data/containers/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node h200-dgxc-slurm \ + --var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" \ --var AIPERF_MMAP_CACHE_HOST_PATH "$AIPERF_MMAP_CACHE_HOST_PATH" \ - --var HF_HUB_CACHE_MOUNT "$HF_HUB_CACHE_MOUNT" --var CONTAINER_KEY "$IMAGE" \ - --model "hf:$MODEL" "$SRT_MODEL_PATH" \ - --container "$IMAGE" "$IMAGE" \ - --mount "$HF_HUB_CACHE_MOUNT" "$HF_HUB_CACHE" --exclusive - make setup ARCH=x86_64 - - SRT_JOB_ID="" - SRT_JOB_OUTPUT="" - finish_native_single_node() { - local rc=$? artifact - trap - EXIT - # Submission may succeed immediately before cancellation or a client error. - if [[ -z "$SRT_JOB_ID" ]] && python3 -m infx.srt_slurm.single_node submission \ - "$GITHUB_WORKSPACE/srt-single-node-submission.json" > "$SRT_PILOT_ROOT/submission-fields" 2>/dev/null; then - mapfile -t SRT_SUBMISSION < "$SRT_PILOT_ROOT/submission-fields" - SRT_JOB_ID="${SRT_SUBMISSION[0]}" - SRT_JOB_OUTPUT="${SRT_SUBMISSION[1]}" - fi - if [[ -n "$SRT_JOB_ID" ]] && slurm_job_is_active "$SRT_JOB_ID"; then - scancel "$SRT_JOB_ID" || true - fi - if [[ -n "$SRT_JOB_OUTPUT" && -d "$SRT_JOB_OUTPUT" ]]; then - bundle_server_logs "$SRT_JOB_OUTPUT" "$GITHUB_WORKSPACE/srt-single-node-logs.tar.gz" - for artifact in "$SRT_JOB_OUTPUT/logs/$RESULT_FILENAME.json" "$SRT_JOB_OUTPUT"/logs/gpu_metrics*; do - [[ -f "$artifact" ]] || continue - copy_to_workspace "$artifact" "$GITHUB_WORKSPACE/$(basename "$artifact")" || rc=1 - done - fi - exit "$rc" - } - trap finish_native_single_node EXIT - trap 'exit 130' INT - trap 'exit 143' TERM - apply_srt_recipe "$GITHUB_WORKSPACE/$SRT_RECIPE" "$FRAMEWORK" \ - --json --yes --output "$SRT_PILOT_ROOT/outputs" "${SRT_RUNTIME_ARGS[@]}" \ - > "$GITHUB_WORKSPACE/srt-single-node-submission.json" - python3 -m infx.srt_slurm.single_node submission "$GITHUB_WORKSPACE/srt-single-node-submission.json" \ - > "$SRT_PILOT_ROOT/submission-fields" - mapfile -t SRT_SUBMISSION < "$SRT_PILOT_ROOT/submission-fields" - SRT_JOB_ID="${SRT_SUBMISSION[0]}" - SRT_JOB_OUTPUT="${SRT_SUBMISSION[1]}" - stream_slurm_job_log "$SRT_JOB_ID" "$SRT_JOB_OUTPUT/logs/sweep_${SRT_JOB_ID}.log" - verify_slurm_job_status "$SRT_JOB_ID" - test -s "$SRT_JOB_OUTPUT/logs/$RESULT_FILENAME.json" + --var HF_HUB_CACHE_MOUNT "$HF_HUB_CACHE_MOUNT" --var CONTAINER_KEY "$IMAGE" elif [[ "$EXECUTION_PATH" == multinode ]]; then diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 49180eea53..5e86363556 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -40,8 +40,7 @@ setup_srt_slurm() { fi local destination="$1" framework="$2" uses_power="$3" check_env_vars INFERENCEX_RUNTIME_ENV_VARS EVAL_ONLY - local eval_passthrough - eval_passthrough=$(python3 - <<'PYENV' + SRT_EVAL_PASSTHROUGH=$(python3 - <<'PYENV' import json import os @@ -49,11 +48,12 @@ names = [ "EVAL_FRAMEWORK", "EVAL_CONC", "EVAL_LIMIT", "EVAL_SUITE", "SWEBENCH_GEN_MODE", "SWEBENCH_USE_MODAL", "MODAL_TOKEN_ID", "MODAL_TOKEN_SECRET", "IS_AGENTIC", "SCENARIO_TYPE", + "TP", "EP_SIZE", "DP_ATTENTION", "PP_SIZE", "DCP_SIZE", "PCP_SIZE", "CONC", ] print(json.dumps(names + os.environ["INFERENCEX_RUNTIME_ENV_VARS"].split())) PYENV ) || return 1 - SRTCTL_EVAL_ARGS+=(--set "post_eval.passthrough_env=$eval_passthrough") + SRTCTL_EVAL_ARGS+=(--set "post_eval.passthrough_env=$SRT_EVAL_PASSTHROUGH") # Custom benchmarks inherit exported workflow settings through sbatch/srun; # native recipe environment and benchmark.env retain their override priority. local source="$INFERENCEX_SLURM_UTILS_DIR/../utils/srt-slurm" @@ -127,6 +127,100 @@ apply_srt_recipe() { "$config" "$framework" -- "$@" } +# One native submission per fixed-sequence matrix point, shared across Slurm pools. +launch_srt_single_node() { + set -eo pipefail + local profile="$1" + shift + check_env_vars GITHUB_WORKSPACE SRT_RECIPE FRAMEWORK MODEL MODEL_PREFIX IMAGE PRECISION \ + TP PP_SIZE DCP_SIZE PCP_SIZE EP_SIZE DP_ATTENTION GPU_COUNT IS_AGENTIC SPEC_DECODING \ + CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME GPU_MONITOR_INTERVAL SRT_MODEL_PATH \ + HF_HUB_CACHE_MOUNT HF_HUB_CACHE SALLOC_TIME_LIMIT + SRT_SINGLE_NODE_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-single.XXXXXX") + SRTCTL_ROOT="$SRT_SINGLE_NODE_ROOT/checkout" + export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" + setup_srt_slurm "$SRTCTL_ROOT" "$FRAMEWORK" 0 + if ! command -v uv >/dev/null; then + curl -LsSf https://astral.sh/uv/install.sh | sh + source "$HOME/.local/bin/env" + fi + uv venv .venv + source .venv/bin/activate + uv pip install -e . + export PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" + + python3 -m infx.srt_slurm.single_node prepare "$GITHUB_WORKSPACE/$SRT_RECIPE" "$SRT_SINGLE_NODE_ROOT/arguments" + mapfile -d '' -t SRT_RUNTIME_ARGS < "$SRT_SINGLE_NODE_ROOT/arguments" + SRT_SELECTED_RECIPE="${SRT_RUNTIME_ARGS[0]}" + SRT_RUNTIME_ARGS=("${SRT_RUNTIME_ARGS[@]:1}") + SRT_RUNTIME_ARGS+=( + --set 'post_eval.command=["bash", "{infmax_workspace}/benchmarks/single_node/srt_eval.sh", "{endpoint}", "/logs/infx-eval-exit-code"]' + --set "post_eval.passthrough_env=$SRT_EVAL_PASSTHROUGH" + ) + # Reuse only a valid cache for this exact image. Missing caches are imported + # by native Pyxis inside the same benchmark allocation. + SRT_CONTAINER="$IMAGE" + if [[ -n "${SRT_SQUASH_FILE:-}" && -r "$SRT_SQUASH_FILE" ]] && unsquashfs -s "$SRT_SQUASH_FILE" >/dev/null 2>&1; then + SRT_CONTAINER="$SRT_SQUASH_FILE" + fi + python3 -m infx.srt_slurm.cluster_config \ + "$INFERENCEX_SLURM_UTILS_DIR/srt-slurm/${profile}.yaml" srtslurm.yaml \ + --var SRTCTL_ROOT "$SRTCTL_ROOT" --var SQUASH_FILE "$SRT_CONTAINER" \ + --var IMAGE "$IMAGE" --var NGINX_SQUASH_FILE nginx:1.27.4 \ + --var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" \ + --model "hf:$MODEL" "$SRT_MODEL_PATH" --container "$IMAGE" "$SRT_CONTAINER" \ + --mount "$HF_HUB_CACHE_MOUNT" "$HF_HUB_CACHE" --exclusive "$@" + make setup ARCH=x86_64 + + SRT_JOB_ID="" + SRT_JOB_OUTPUT="" + finish_native_single_node() { + local rc=$? artifact + trap - EXIT + # Submission may succeed immediately before cancellation or a client error. + if [[ -z "$SRT_JOB_ID" ]] && python3 -m infx.srt_slurm.single_node submission \ + "$GITHUB_WORKSPACE/srt-single-node-submission.json" > "$SRT_SINGLE_NODE_ROOT/submission-fields" 2>/dev/null; then + mapfile -t SRT_SUBMISSION < "$SRT_SINGLE_NODE_ROOT/submission-fields" + SRT_JOB_ID="${SRT_SUBMISSION[0]}" + SRT_JOB_OUTPUT="${SRT_SUBMISSION[1]}" + fi + if [[ -n "$SRT_JOB_ID" ]] && slurm_job_is_active "$SRT_JOB_ID"; then + scancel "$SRT_JOB_ID" || true + fi + if [[ -n "$SRT_JOB_OUTPUT" && -d "$SRT_JOB_OUTPUT" ]]; then + bundle_server_logs "$SRT_JOB_OUTPUT" "$GITHUB_WORKSPACE/srt-single-node-logs.tar.gz" + for artifact in "$SRT_JOB_OUTPUT/logs/$RESULT_FILENAME.json" "$SRT_JOB_OUTPUT"/logs/gpu_metrics*; do + [[ -f "$artifact" ]] || continue + copy_to_workspace "$artifact" "$GITHUB_WORKSPACE/$(basename "$artifact")" || rc=1 + done + fi + exit "$rc" + } + trap finish_native_single_node EXIT + trap 'exit 130' INT + trap 'exit 143' TERM + apply_srt_recipe "$SRT_SELECTED_RECIPE" "$FRAMEWORK" \ + --json --yes --output "$SRT_SINGLE_NODE_ROOT/outputs" "${SRT_RUNTIME_ARGS[@]}" \ + > "$GITHUB_WORKSPACE/srt-single-node-submission.json" + python3 -m infx.srt_slurm.single_node submission "$GITHUB_WORKSPACE/srt-single-node-submission.json" \ + > "$SRT_SINGLE_NODE_ROOT/submission-fields" + mapfile -t SRT_SUBMISSION < "$SRT_SINGLE_NODE_ROOT/submission-fields" + SRT_JOB_ID="${SRT_SUBMISSION[0]}" + SRT_JOB_OUTPUT="${SRT_SUBMISSION[1]}" + stream_slurm_job_log "$SRT_JOB_ID" "$SRT_JOB_OUTPUT/logs/sweep_${SRT_JOB_ID}.log" + verify_slurm_job_status "$SRT_JOB_ID" + # Native SRT treats post-throughput eval failure as non-fatal. InferenceX + # requires every requested eval to finish successfully, including staging. + if [[ "$RUN_EVAL" == true || "$EVAL_ONLY" == true ]]; then + test -f "$SRT_JOB_OUTPUT/logs/infx-eval-exit-code" + test "$(cat "$SRT_JOB_OUTPUT/logs/infx-eval-exit-code")" = 0 + fi + if [[ "$EVAL_ONLY" != true ]]; then + test -s "$SRT_JOB_OUTPUT/logs/$RESULT_FILENAME.json" + fi + +} + slurm_job_is_active() { local job_id="$1" squeue -j "$job_id" --noheader 2>/dev/null | grep -q "$job_id" diff --git a/runners/srt-slurm/b200-cw.yaml b/runners/srt-slurm/b200-cw.yaml new file mode 100644 index 0000000000..881d227fef --- /dev/null +++ b/runners/srt-slurm/b200-cw.yaml @@ -0,0 +1,6 @@ +default_partition: b200 +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: "" +srtctl_root: ${SRTCTL_ROOT} +use_exclusive_sbatch_directive: true diff --git a/runners/srt-slurm/b200-nb.yaml b/runners/srt-slurm/b200-nb.yaml new file mode 100644 index 0000000000..7ad1a06fdf --- /dev/null +++ b/runners/srt-slurm/b200-nb.yaml @@ -0,0 +1,6 @@ +default_partition: main +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: "" +srtctl_root: ${SRTCTL_ROOT} +use_exclusive_sbatch_directive: true diff --git a/runners/srt-slurm/h100-cw.yaml b/runners/srt-slurm/h100-cw.yaml new file mode 100644 index 0000000000..85f05a44b0 --- /dev/null +++ b/runners/srt-slurm/h100-cw.yaml @@ -0,0 +1,6 @@ +default_partition: h100 +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: "" +srtctl_root: ${SRTCTL_ROOT} +use_exclusive_sbatch_directive: true diff --git a/runners/srt-slurm/h200-cw.yaml b/runners/srt-slurm/h200-cw.yaml new file mode 100644 index 0000000000..ac2631ccc2 --- /dev/null +++ b/runners/srt-slurm/h200-cw.yaml @@ -0,0 +1,6 @@ +default_partition: h200 +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: "" +srtctl_root: ${SRTCTL_ROOT} +use_exclusive_sbatch_directive: true diff --git a/utils/test_srt_fixed_sequence.py b/utils/test_srt_fixed_sequence.py index 81dc2f64b4..352db5b8e2 100644 --- a/utils/test_srt_fixed_sequence.py +++ b/utils/test_srt_fixed_sequence.py @@ -2,6 +2,7 @@ import json import os +import shutil import subprocess import sys from pathlib import Path @@ -115,8 +116,8 @@ def test_native_endpoint_preserves_client_settings_and_failure( ("GPU_MONITOR_INTERVAL", None, "GPU_MONITOR_INTERVAL"), ("USE_CHAT_TEMPLATE", "yes", "USE_CHAT_TEMPLATE must be true or false"), ("CONC", "0", "CONC must be a positive integer"), - ("RUN_EVAL", "true", "does not support evals yet"), - ("EVAL_ONLY", "true", "does not support evals yet"), + ("RUN_EVAL", "yes", "RUN_EVAL must be true or false"), + ("EVAL_ONLY", "yes", "EVAL_ONLY must be true or false"), ], ) def test_invalid_runtime_inputs_fail_before_the_client( @@ -156,3 +157,45 @@ def test_legacy_client_keeps_its_local_endpoint(client_environment): assert result.returncode == 0, result.stderr argv = json.loads(Path(env["CAPTURE"]).read_text()) assert argv[argv.index("--base-url") + 1] == "http://0.0.0.0:8888" + + +@pytest.mark.parametrize("eval_exit", [0, 7]) +def test_native_post_eval_preserves_results_topology_and_failure(client_environment, tmp_path, eval_exit): + workspace = tmp_path / "repo" + scripts = workspace / "benchmarks/single_node" + scripts.mkdir(parents=True) + shutil.copyfile(ROOT / "benchmarks/benchmark_lib.sh", scripts.parent / "benchmark_lib.sh") + shutil.copyfile(ROOT / "benchmarks/single_node/srt_eval.sh", scripts / "srt_eval.sh") + python = tmp_path / "bin/python3" + python.write_text( + f"#!{sys.executable}\n" + "import json, os, pathlib, sys\n" + "args = sys.argv[1:]\n" + "assert args[:2] == ['-m', 'lm_eval']\n" + "pathlib.Path(os.environ['CAPTURE']).write_text(json.dumps(args))\n" + "output = pathlib.Path(args[args.index('--output_path') + 1])\n" + "output.mkdir(parents=True, exist_ok=True)\n" + "(output / 'results_fixture.json').write_text('{\"score\":0.75}')\n" + "sys.exit(int(os.environ['CLIENT_EXIT']))\n" + ) + env = { + **client_environment, "CLIENT_EXIT": str(eval_exit), "MODEL_NAME": "served-model", + "TP": "4", "EP_SIZE": "4", "DP_ATTENTION": "true", "IS_MULTINODE": "false", + "MAX_MODEL_LEN": "8192", "EVAL_MAX_MODEL_LEN": "8192", "OPENAI_API_KEY": "EMPTY", + "INFERENCEX_LM_EVAL_RUNTIME_READY": "true", "EVAL_ONLY": "true", "RUN_EVAL": "true", + "EVAL_RESULT_DIR": str(tmp_path / "eval-output"), "FRAMEWORK": "sglang", "PRECISION": "fp8", + } + status = tmp_path / "eval-status" + result = subprocess.run( + ["bash", str(scripts / "srt_eval.sh"), "http://localhost:9444", str(status)], + env=env, capture_output=True, text=True, + ) + assert result.returncode == eval_exit, result.stderr + assert status.read_text() == f"{eval_exit}\n" + assert json.loads((workspace / "results_fixture.json").read_text()) == {"score": 0.75} + metadata = json.loads((workspace / "meta_env.json").read_text()) + assert (metadata["tp"], metadata["ep"], metadata["dp_attention"], metadata["conc"]) == (4, 4, True, 3) + argv = json.loads(Path(env["CAPTURE"]).read_text()) + model_args = argv[argv.index("--model_args") + 1] + assert "model=served-model,base_url=http://0.0.0.0:9444/v1/chat/completions" in model_args + assert "num_concurrent=3" in model_args diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 6725f7e78d..bcd15efb5d 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -10,7 +10,7 @@ import pytest import yaml -from infx.srt_slurm.single_node import runtime_arguments, submission_fields +from infx.srt_slurm.single_node import runtime_arguments, select_recipe, submission_fields from infx.srt_slurm.synthetic_acceptance import plan_commands, selected_recipes ROOT = Path(__file__).resolve().parents[1] @@ -68,7 +68,7 @@ def test_native_binding_submits_one_point_and_keeps_server_settings(point): @pytest.mark.parametrize("field,value,message", [ ("TP", "2", "tensor-parallel-size"), ("IMAGE", "other:tag", "image"), - ("ISL", "128", "ISL"), ("RUN_EVAL", "true", "RUN_EVAL"), + ("ISL", "128", "ISL"), ("RUN_EVAL", "yes", "RUN_EVAL"), ("PP_SIZE", "2", "PP_SIZE"), ("RESULT_FILENAME", "", "Missing runtime input"), ("EP_SIZE", "2", "expert-parallel-size"), ("SPEC_DECODING", "mtp", "SPEC_DECODING"), ]) @@ -78,8 +78,22 @@ def test_mismatched_point_fails_before_submission(point, field, value, message): runtime_arguments(f"{path}:base", {**env, field: value}) -def test_multi_variant_selection_is_rejected(point): +def test_native_variants_select_only_the_matching_matrix_point(point): path, _, env = point + config, recipe = select_recipe(str(path), {**env, "CONC": "4"}) + assert config == f"{path}:zip_override_conc[1]" + assert recipe["benchmark"]["env"]["CONC"] == "4" + argv = runtime_arguments(config, {**env, "CONC": "4"}) + assert plan_commands(config, "sglang", ["--json", *argv], env) == [[ + "srtctl", "apply", "--json", *argv, "--file", f"{path}:zip_override_conc[1]", + ]] + with pytest.raises(ValueError, match="exactly one"): + select_recipe(str(path), {**env, "CONC": "8"}) + + +def test_ambiguous_native_variants_are_rejected(point): + path, recipe, env = point + path.write_text(yaml.safe_dump({"base": recipe, "override_first": {}, "override_second": {}})) with pytest.raises(ValueError, match="exactly one"): runtime_arguments(str(path), env) @@ -121,6 +135,41 @@ def test_concurrency_selector_keeps_graph_capture_coupled_to_client(point): runtime_arguments(f"{path}:zip_override_conc[1]", env) +def test_eval_binding_changes_context_without_changing_selected_concurrency(point): + path, recipe, env = point + recipe["roles"]["agg"]["args"]["context-length"] = 512 + path.write_text(yaml.safe_dump({"base": recipe, "zip_override_conc": { + "benchmark": {"env": {"CONC": ["2", "4"]}}, + }})) + env = {**env, "EVAL_ONLY": "true", "RUN_EVAL": "true", "CONC": "4", "MAX_MODEL_LEN": "1024"} + config, _ = select_recipe(str(path), env) + argv = runtime_arguments(config, env) + raw = yaml.safe_load(path.read_text()) + apply_overrides_to_recipe(raw, parse_overrides(argv[1::2], [])) + actual = selected_recipes(raw, "zip_override_conc[1]")[0][1] + assert actual["roles"]["agg"]["args"]["context-length"] == 1024 + assert actual["benchmark"]["env"]["CONC"] == "4" + assert len(plan_commands(config, "sglang", ["--json", *argv], env)) == 1 + + +def test_dp_attention_is_validated_without_replacing_recipe_topology(point): + path, recipe, env = point + recipe["roles"]["agg"]["args"].update({ + "data-parallel-size": 4, "expert-parallel-size": 4, "enable-dp-attention": True, + }) + path.write_text(yaml.safe_dump({"base": recipe})) + env = {**env, "DP_ATTENTION": "true", "EP_SIZE": "4"} + actual = copy.deepcopy(recipe) + argv = runtime_arguments(f"{path}:base", env) + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual["roles"]["agg"]["args"] == { + "tensor-parallel-size": 4, "data-parallel-size": 4, + "max-running-requests": 32, "expert-parallel-size": 4, "enable-dp-attention": True, + } + with pytest.raises(ValueError, match="data-parallel-size|DP_ATTENTION"): + runtime_arguments(f"{path}:base", {**env, "DP_ATTENTION": "false"}) + + @pytest.mark.parametrize("record,expected", [ ({"status": "submitted", "slurm_job_id": "42", "output_dir": "/shared/42"}, ("42", "/shared/42")), ({"status": "error"}, None), @@ -137,8 +186,14 @@ def test_submission_manifest(tmp_path, record, expected): assert submission_fields(path) == expected -@pytest.mark.parametrize("failure", ["none", "allocation", "submission", "bootstrap"]) -def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, failure): +@pytest.mark.parametrize("pool,failure", [ + ("h200-dgxc-slurm", "none"), ("h200-dgxc-slurm", "allocation"), + ("h200-dgxc-slurm", "submission"), ("h200-dgxc-slurm", "bootstrap"), + ("h200-cw", "none"), ("h100-cw", "none"), ("h100-dgxc-slurm", "none"), + ("b200-cw", "none"), ("b200-nb", "none"), ("b200-nscale-slurm", "none"), + ("b300-dsxe", "none"), +]) +def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, pool, failure): path, _, point_env = point binaries = tmp_path / "bin" binaries.mkdir() @@ -186,13 +241,16 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "IS_MULTINODE": "false", "REQUIRE_POWER": "1", "SALLOC_TIME_LIMIT": "10", "HF_HUB_CACHE_MOUNT": str(tmp_path), "AIPERF_MMAP_CACHE_HOST_PATH": str(tmp_path), "HF_HUB_CACHE": "/hf", "SRT_MODEL_PATH": str(model), "MODEL_PREFIX": "dsr1", + "SLURM_ACCOUNT": "fixture", "SLURM_PARTITION": "fixture", + "B200_SQUASH_DIR": str(tmp_path), "B300_HF_CACHE_HOST_DIR": str(tmp_path), + "B300_HF_CACHE_CONTAINER_DIR": "/hf", "ENROOT_IMPORT_TIME_LIMIT": "10", "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), } env.pop("AIPERF_DRAIN_TIMEOUT_SECONDS", None) env.pop("AIPERF_DRAIN_POLL_SECONDS", None) result = subprocess.run( - ["bash", str(ROOT / "runners/launch_h200-dgxc-slurm.sh")], cwd=tmp_path, + ["bash", str(ROOT / f"runners/launch_{pool}.sh")], cwd=tmp_path, env=env, capture_output=True, text=True, timeout=30, ) assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13}[failure], result.stderr From a67435c0f8d2cb2e2d0c61c829ca4197a1368e64 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:56:25 -0500 Subject: [PATCH 11/29] feat: migrate fixed-sequence TRT recipes to native SRT MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将八个定长 TRT 配方迁移至原生 SRT,保留 63 个测试点的引擎参数、客户端和 eval token 预算。 --- .../dsr1/trtllm/b200-fp4-mtp/8k1k.yaml | 154 ++++++++++++++ .../dsr1/trtllm/b200-fp4/8k1k.yaml | 137 ++++++++++++ .../dsr1/trtllm/b200-fp8-mtp/8k1k.yaml | 95 +++++++++ .../dsr1/trtllm/b200-fp8/8k1k.yaml | 107 ++++++++++ .../dsr1/trtllm/h200-fp8-mtp/8k1k.yaml | 91 ++++++++ .../dsr1/trtllm/h200-fp8/8k1k.yaml | 80 +++++++ .../qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml | 157 ++++++++++++++ .../qwen3.5/trtllm/b200-fp4/8k1k.yaml | 172 +++++++++++++++ benchmarks/single_node/srt_fixed_sequence.sh | 9 +- configs/nvidia-master.yaml | 199 +++++++++++++++--- infx/srt_slurm/single_node.py | 67 ++++-- perf-changelog.yaml | 16 ++ utils/test_srt_fixed_sequence.py | 14 +- utils/test_srt_single_node.py | 32 +++ 14 files changed, 1275 insertions(+), 55 deletions(-) create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..6956d4eac4 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml @@ -0,0 +1,154 @@ +# Serving settings from dsr1_fp4_b200_trt_mtp.sh. +base: + schema: 2 + name: dsr1-fp4-b200-trt-mtp-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: nvidia/DeepSeek-R1-0528-FP4-V2 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + cuda_graph_config: + enable_padding: true + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.8 + enable_block_reuse: false + stream_interval: 10 + speculative_config: + decoding_type: MTP + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + # The pinned native launcher needs this CLI flag as well as engine metadata. + extra_args: [--served_model_name, nvidia/DeepSeek-R1-0528-FP4-V2] + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [4, 8, 16] + enable_attention_dp: false + moe_config: + backend: TRTLLM + speculative_config: + num_nextn_predict_layers: 3 + max_batch_size: [4, 8, 16] + max_num_tokens: 8320 + tensor_parallel_size: 4 + moe_expert_parallel_size: 1 + benchmark: + env: + CONC: ['4', '8', '16'] +zip_override_tp4_ep4: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 32 + enable_attention_dp: false + moe_config: + backend: TRTLLM + speculative_config: + num_nextn_predict_layers: 3 + max_batch_size: 32 + max_num_tokens: 8384 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + benchmark: + env: + CONC: ['32'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 64 + enable_attention_dp: true + moe_config: + backend: CUTLASS + speculative_config: + num_nextn_predict_layers: 1 + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_batch_size: 64 + max_num_tokens: 8384 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + benchmark: + env: + CONC: ['256'] +zip_override_tp8_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 4 + enable_attention_dp: false + moe_config: + backend: TRTLLM + speculative_config: + num_nextn_predict_layers: 3 + max_batch_size: 4 + max_num_tokens: 8320 + tensor_parallel_size: 8 + moe_expert_parallel_size: 1 + gpus: 8 + benchmark: + env: + CONC: ['4'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [16, 32, 64] + enable_attention_dp: true + moe_config: + backend: CUTLASS + speculative_config: + num_nextn_predict_layers: 1 + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_batch_size: [16, 32, 64] + max_num_tokens: [8320, 8320, 8384] + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml new file mode 100644 index 0000000000..b96ce513a1 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml @@ -0,0 +1,137 @@ +# Serving settings from dsr1_fp4_b200_trt.sh. +base: + schema: 2 + name: dsr1-fp4-b200-trt-8k1k + model: + path: hf:nvidia/DeepSeek-R1-0528-FP4-V2 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: nvidia/DeepSeek-R1-0528-FP4-V2 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + cuda_graph_config: + enable_padding: true + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.8 + enable_block_reuse: false + stream_interval: 10 + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + # The pinned native launcher needs this CLI flag as well as engine metadata. + extra_args: [--served_model_name, nvidia/DeepSeek-R1-0528-FP4-V2] + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/DeepSeek-R1-0528-FP4-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [4, 8, 16, 32] + enable_attention_dp: false + moe_config: + backend: TRTLLM + max_num_tokens: 8320 + tensor_parallel_size: 4 + moe_expert_parallel_size: 1 + benchmark: + env: + CONC: ['4', '8', '16', '32'] +zip_override_tp4_ep4: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 32 + enable_attention_dp: false + moe_config: + backend: TRTLLM + max_num_tokens: 8320 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + benchmark: + env: + CONC: ['32'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 64 + enable_attention_dp: true + moe_config: + backend: CUTLASS + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_num_tokens: 8512 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + benchmark: + env: + CONC: ['256'] +zip_override_tp8_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: 4 + enable_attention_dp: false + moe_config: + backend: TRTLLM + max_num_tokens: 8320 + tensor_parallel_size: 8 + moe_expert_parallel_size: 1 + gpus: 8 + benchmark: + env: + CONC: ['4'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [32, 64] + enable_attention_dp: true + moe_config: + backend: CUTLASS + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_num_tokens: [8384, 8512] + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..6ac275035c --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml @@ -0,0 +1,95 @@ +# Serving settings from dsr1_fp8_b200_trt_mtp.sh. +base: + schema: 2 + name: dsr1-fp8-b200-trt-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: deepseek-ai/DeepSeek-R1-0528 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + enable_padding: true + enable_attention_dp: false + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.8 + enable_block_reuse: false + stream_interval: 10 + moe_config: + backend: TRTLLM + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + tensor_parallel_size: 8 + moe_expert_parallel_size: 1 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + # The pinned native launcher needs this CLI flag as well as engine metadata. + extra_args: [--served_model_name, deepseek-ai/DeepSeek-R1-0528] + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp8_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [4, 8, 16] + max_batch_size: [4, 8, 16] + max_num_tokens: 8320 + benchmark: + env: + CONC: ['4', '8', '16'] +zip_override_tp8_ep1_graphs: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [32, 64, 128, 256] + torch_compile_config: + capture_num_tokens: [[1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, 2048, + 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, 5888, + 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, 8192, 8384], [1, 2, 4, 8, 16, 32, 64, 128, + 256, 512, 768, 1024, 1280, 1536, 1792, 2048, 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, + 4352, 4608, 4864, 5120, 5376, 5632, 5888, 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, + 8192, 8448, 8512], [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, + 2048, 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, + 5888, 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, 8192, 8448, 8704, 8768], [1, 2, 4, + 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, 2048, 2304, 2560, 2816, 3072, + 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, 5888, 6144, 6400, 6656, 6912, + 7168, 7424, 7680, 7936, 8192, 8448, 8704, 8960, 9216, 9280]] + enable_piecewise_cuda_graph: true + max_batch_size: [32, 64, 128, 256] + max_num_tokens: [8384, 8512, 8768, 9280] + benchmark: + env: + CONC: ['32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml new file mode 100644 index 0000000000..76dc46eae9 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml @@ -0,0 +1,107 @@ +# Serving settings from dsr1_fp8_b200_trt.sh. +base: + schema: 2 + name: dsr1-fp8-b200-trt-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp8 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: deepseek-ai/DeepSeek-R1-0528 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + enable_padding: true + enable_attention_dp: false + print_iter_log: true + kv_cache_config: + dtype: fp8 + enable_block_reuse: false + stream_interval: 10 + moe_config: + backend: TRTLLM + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + moe_expert_parallel_size: 1 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + env: + TLLM_OVERRIDE_LAYER_NUM: '61' + # The pinned native launcher needs this CLI flag as well as engine metadata. + extra_args: [--served_model_name, deepseek-ai/DeepSeek-R1-0528] + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep1_graphs: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [64, 128, 256] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + torch_compile_config: + capture_num_tokens: [[1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, 2048, + 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, 5888, + 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, 8192, 8320], [1, 2, 4, 8, 16, 32, 64, 128, + 256, 512, 768, 1024, 1280, 1536, 1792, 2048, 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, + 4352, 4608, 4864, 5120, 5376, 5632, 5888, 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, + 8192, 8384], [1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 768, 1024, 1280, 1536, 1792, 2048, + 2304, 2560, 2816, 3072, 3328, 3584, 3840, 4096, 4352, 4608, 4864, 5120, 5376, 5632, 5888, + 6144, 6400, 6656, 6912, 7168, 7424, 7680, 7936, 8192, 8448, 8512]] + enable_piecewise_cuda_graph: true + max_num_tokens: [8320, 8384, 8512] + tensor_parallel_size: 8 + benchmark: + env: + CONC: ['64', '128', '256'] +zip_override_tp4_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [8, 16, 32] + kv_cache_config: + free_gpu_memory_fraction: 0.75 + max_num_tokens: 8320 + tensor_parallel_size: 4 + gpus: 4 + benchmark: + env: + CONC: ['8', '16', '32'] +zip_override_tp8_ep1: + roles: + agg: + args: + cuda_graph_config: + max_batch_size: [4, 8] + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_num_tokens: 8320 + tensor_parallel_size: 8 + benchmark: + env: + CONC: ['4', '8'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..751e74537e --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml @@ -0,0 +1,91 @@ +# Serving settings from dsr1_fp8_h200_trt_mtp.sh. +base: + schema: 2 + name: dsr1-fp8-h200-trt-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: deepseek-ai/DeepSeek-R1-0528 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.75 + enable_block_reuse: false + stream_interval: 10 + moe_config: + backend: CUTLASS + speculative_config: + decoding_type: MTP + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + env: + PYTHONNOUSERSITE: '1' + # The pinned native launcher needs this CLI flag as well as engine metadata. + extra_args: [--served_model_name, deepseek-ai/DeepSeek-R1-0528] + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp8_ep8: + roles: + agg: + args: + enable_attention_dp: false + speculative_config: + num_nextn_predict_layers: 3 + max_batch_size: [4, 8, 16, 32] + max_num_tokens: [8320, 8320, 8320, 8384] + benchmark: + env: + CONC: ['4', '8', '16', '32'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + enable_attention_dp: true + speculative_config: + num_nextn_predict_layers: 1 + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + max_batch_size: [8, 16, 32] + max_num_tokens: 8320 + env: + PYTORCH_CUDA_ALLOC_CONF: max_split_size_mb:8192 + benchmark: + env: + CONC: ['64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml new file mode 100644 index 0000000000..1a98b08736 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml @@ -0,0 +1,80 @@ +# Serving settings from dsr1_fp8_h200_trt.sh. +base: + schema: 2 + name: dsr1-fp8-h200-trt-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 + precision: fp8 + resources: + gpu_type: h200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: deepseek-ai/DeepSeek-R1-0528 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + enable_padding: true + max_batch_size: 128 + print_iter_log: true + kv_cache_config: + dtype: fp8 + free_gpu_memory_fraction: 0.75 + enable_block_reuse: false + stream_interval: 10 + moe_config: + backend: CUTLASS + trust_remote_code: true + backend: pytorch + max_seq_len: 9472 + max_num_tokens: 8320 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + pipeline_parallel_size: 1 + enable_iter_perf_stats: false + return_perf_metrics: false + env: + PYTHONNOUSERSITE: '1' + # The pinned native launcher needs this CLI flag as well as engine metadata. + extra_args: [--served_model_name, deepseek-ai/DeepSeek-R1-0528] + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp8_ep8: + roles: + agg: + args: + enable_attention_dp: false + benchmark: + env: + CONC: ['4', '8', '16', '32'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + enable_attention_dp: true + attention_dp_config: + batching_wait_iters: 0 + enable_balance: true + timeout_iters: 60 + benchmark: + env: + CONC: ['64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..19abfd8efb --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml @@ -0,0 +1,157 @@ +# Serving settings from qwen3.5_fp4_b200_trt_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp4-b200-trt-mtp-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc18 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: nvidia/Qwen3.5-397B-A17B-NVFP4 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + backend: pytorch + print_iter_log: true + enable_layerwise_nvtx_marker: false + disable_overlap_scheduler: false + enable_iter_perf_stats: true + enable_chunked_prefill: false + stream_interval: 20 + num_postprocess_workers: 4 + scheduler_config: + capacity_scheduler_policy: MAX_UTILIZATION + context_chunking_policy: FIRST_COME_FIRST_SERVED + kv_cache_config: + enable_block_reuse: false + dtype: fp8 + cuda_graph_config: + enable_padding: true + batch_sizes: [1, 2, 4, 8, 16, 32, 64, 128] + moe_config: + use_low_precision_moe_combine: true + speculative_config: + decoding_type: MTP + num_nextn_predict_layers: 3 + trust_remote_code: true + max_seq_len: 9472 + max_num_tokens: 32768 + pipeline_parallel_size: 1 + return_perf_metrics: false + env: + TLLM_USE_FLASHINFER_GDN_PREFILL: '0' + # The pinned native launcher needs this CLI flag as well as engine metadata. + extra_args: [--served_model_name, nvidia/Qwen3.5-397B-A17B-NVFP4] + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp2_ep1: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: 0.7 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 4 + tensor_parallel_size: 2 + moe_expert_parallel_size: 1 + enable_attention_dp: false + benchmark: + env: + CONC: ['4'] +zip_override_tp2_ep2: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: 0.6 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: [8, 16] + tensor_parallel_size: 2 + moe_expert_parallel_size: 2 + enable_attention_dp: false + benchmark: + env: + CONC: ['8', '16'] +zip_override_tp4_ep4: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: 0.75 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 4 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + enable_attention_dp: false + gpus: 4 + benchmark: + env: + CONC: ['4'] +zip_override_tp8_ep8: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: 0.8 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 4 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + enable_attention_dp: false + gpus: 8 + benchmark: + env: + CONC: ['4'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + kv_cache_config: + free_gpu_memory_fraction: [0.9, 0.9, 0.8] + moe_config: + backend: CUTEDSL + enable_attention_dp: true + attention_dp_config: + enable_balance: true + batching_wait_iters: 10 + timeout_iters: 500 + max_batch_size: [16, 32, 128] + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['128', '256', '1024'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml new file mode 100644 index 0000000000..1e1762b73e --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml @@ -0,0 +1,172 @@ +# Serving settings from qwen3.5_fp4_b200_trt.sh. +base: + schema: 2 + name: qwen3.5-fp4-b200-trt-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4 + container: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc18 + precision: fp4 + resources: + gpu_type: b200 + gpus_per_node: 8 + frontend: + type: trtllm_serve + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: + type: trtllm + served_model_name: nvidia/Qwen3.5-397B-A17B-NVFP4 + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + backend: pytorch + print_iter_log: true + enable_layerwise_nvtx_marker: false + disable_overlap_scheduler: false + enable_iter_perf_stats: true + enable_chunked_prefill: false + stream_interval: 20 + num_postprocess_workers: 4 + scheduler_config: + capacity_scheduler_policy: MAX_UTILIZATION + context_chunking_policy: FIRST_COME_FIRST_SERVED + kv_cache_config: + free_gpu_memory_fraction: 0.9 + enable_block_reuse: false + dtype: fp8 + cuda_graph_config: + enable_padding: true + moe_config: + use_low_precision_moe_combine: true + trust_remote_code: true + max_seq_len: 9472 + max_num_tokens: 32768 + pipeline_parallel_size: 1 + return_perf_metrics: false + # The pinned native launcher needs this CLI flag as well as engine metadata. + extra_args: [--served_model_name, nvidia/Qwen3.5-397B-A17B-NVFP4] + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp2_ep1: + roles: + agg: + args: + enable_attention_dp: false + cuda_graph_config: + max_batch_size: 256 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 256 + tensor_parallel_size: 2 + moe_expert_parallel_size: 1 + benchmark: + env: + CONC: ['4', '16'] +zip_override_tp4_ep1: + roles: + agg: + args: + enable_attention_dp: false + cuda_graph_config: + max_batch_size: 512 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 512 + tensor_parallel_size: 4 + moe_expert_parallel_size: 1 + gpus: 4 + benchmark: + env: + CONC: ['4'] +zip_override_tp2_ep2: + roles: + agg: + args: + enable_attention_dp: false + cuda_graph_config: + max_batch_size: [256, 32] + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: [256, 32] + tensor_parallel_size: 2 + moe_expert_parallel_size: 2 + benchmark: + env: + CONC: ['8', '32'] +zip_override_tp8_ep8: + roles: + agg: + args: + enable_attention_dp: false + cuda_graph_config: + max_batch_size: 512 + moe_config: + backend: TRTLLM + batch_wait_timeout_iters: 50 + batch_wait_max_tokens_ratio: 0.45 + max_batch_size: 512 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['4'] +zip_override_tp4_ep4_dpa: + roles: + agg: + args: + enable_attention_dp: true + cuda_graph_config: + max_batch_size: 256 + moe_config: + backend: CUTEDSL + attention_dp_config: + enable_balance: true + batching_wait_iters: 10 + timeout_iters: 500 + max_batch_size: 256 + tensor_parallel_size: 4 + moe_expert_parallel_size: 4 + gpus: 4 + benchmark: + env: + CONC: ['1024'] +zip_override_tp8_ep8_dpa: + roles: + agg: + args: + enable_attention_dp: true + cuda_graph_config: + max_batch_size: 128 + moe_config: + backend: CUTEDSL + attention_dp_config: + enable_balance: true + batching_wait_iters: 10 + timeout_iters: 500 + max_batch_size: 128 + tensor_parallel_size: 8 + moe_expert_parallel_size: 8 + gpus: 8 + benchmark: + env: + CONC: ['256', '512', '1024'] diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index 3f2f081d74..452a0b7c3d 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -4,13 +4,18 @@ set -eo pipefail source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only check_env_vars MODEL CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME RESULT_DIR \ - SRT_FRONTEND_HOST SRT_FRONTEND_PORT RUN_EVAL EVAL_ONLY GPU_MONITOR_INTERVAL USE_CHAT_TEMPLATE + SRT_FRONTEND_HOST SRT_FRONTEND_PORT RUN_EVAL EVAL_ONLY GPU_MONITOR_INTERVAL USE_CHAT_TEMPLATE FRAMEWORK for name in RUN_EVAL EVAL_ONLY; do if [[ "${!name}" != true && "${!name}" != false ]]; then echo "ERROR: $name must be true or false" >&2 exit 1 fi done +case "$FRAMEWORK" in + sglang) CLIENT_BACKEND=vllm ;; + trt) CLIENT_BACKEND=openai ;; + *) echo "ERROR: unsupported fixed-sequence FRAMEWORK: $FRAMEWORK" >&2; exit 1 ;; +esac SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" CLIENT_ARGS=() case "$USE_CHAT_TEMPLATE" in @@ -42,7 +47,7 @@ run_benchmark_serving \ --model "$MODEL" \ --port "$SRT_FRONTEND_PORT" \ --base-url "http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" \ - --backend vllm \ + --backend "$CLIENT_BACKEND" \ --input-len "$ISL" \ --output-len "$OSL" \ --random-range-ratio "$RANDOM_RANGE_RATIO" \ diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 19e90681ce..75baf30ff1 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -995,11 +995,31 @@ dsr1-fp4-b200-trt: # low concurrency cases use TP only # concurrency 32 uses TP & EP # high concurrency cases use TP & EP & DP-ATTN - - { tp: 4, conc-start: 4, conc-end: 32 } - - { tp: 4, ep: 4, conc-start: 32, conc-end: 32 } - - { tp: 4, ep: 4, dp-attn: true, conc-start: 256, conc-end: 256 } - - { tp: 8, conc-start: 4, conc-end: 4 } - - { tp: 8, ep: 8, dp-attn: true, conc-start: 128, conc-end: 256 } + - tp: 4 + conc-start: 4 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml + - tp: 4 + ep: 4 + conc-start: 32 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-start: 256 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml + - tp: 8 + conc-start: 4 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-start: 128 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml dsr1-fp4-b200-trt-mtp: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -1015,12 +1035,37 @@ dsr1-fp4-b200-trt-mtp: osl: 1024 search-space: # TP=4 configurations - - { tp: 4, conc-start: 4, conc-end: 16, spec-decoding: mtp } - - { tp: 4, ep: 4, conc-start: 32, conc-end: 32, spec-decoding: mtp } - - { tp: 4, ep: 4, dp-attn: true, conc-start: 256, conc-end: 256, spec-decoding: mtp } + - tp: 4 + conc-start: 4 + conc-end: 16 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 4 + ep: 4 + conc-start: 32 + conc-end: 32 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-start: 256 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml # TP=8 configurations - - { tp: 8, conc-start: 4, conc-end: 4, spec-decoding: mtp } - - { tp: 8, ep: 8, dp-attn: true, conc-start: 64, conc-end: 256, spec-decoding: mtp } + - tp: 8 + conc-start: 4 + conc-end: 4 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-start: 64 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml dsr1-fp8-b200-sglang: image: lmsysorg/sglang:v0.5.12-cu130 @@ -1535,9 +1580,21 @@ dsr1-fp8-b200-trt: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 64, conc-end: 256 } - - { tp: 4, ep: 1, conc-start: 8, conc-end: 32 } - - { tp: 8, ep: 1, conc-start: 4, conc-end: 8 } + - tp: 8 + ep: 1 + conc-start: 64 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 8 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 8 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml dsr1-fp8-b200-trt-mtp: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -1553,7 +1610,12 @@ dsr1-fp8-b200-trt-mtp: osl: 1024 search-space: # TP8 for all points - - { tp: 8, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml dsr1-fp8-h200-sglang: image: lmsysorg/sglang:v0.5.12-cu130 @@ -1698,8 +1760,17 @@ dsr1-fp8-h200-trt: osl: 1024 # If CONC > 32, then DP_ATTN=true search-space: - - { tp: 8, ep: 8, conc-start: 4, conc-end: 32 } - - { tp: 8, ep: 8, dp-attn: true, conc-start: 64, conc-end: 64 } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-start: 64 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml dsr1-fp8-h200-trt-mtp: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc14 @@ -1716,8 +1787,19 @@ dsr1-fp8-h200-trt-mtp: osl: 1024 search-space: # If CONC >= 64, then DP_ATTN=true, MTP=1 - - { tp: 8, ep: 8, conc-start: 4, conc-end: 32, spec-decoding: mtp } - - { tp: 8, ep: 8, dp-attn: true, conc-start: 64, conc-end: 256, spec-decoding: mtp } + - tp: 8 + ep: 8 + conc-start: 4 + conc-end: 32 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-start: 64 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml dsr1-fp8-h200-dynamo-trt: image: nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:0.8.1.post1 @@ -5277,12 +5359,42 @@ qwen3.5-fp4-b200-trt: - isl: 8192 osl: 1024 search-space: - - { tp: 2, ep: 1, conc-list: [4, 16] } - - { tp: 4, ep: 1, conc-list: [4] } - - { tp: 2, ep: 2, conc-list: [8, 32] } - - { tp: 8, ep: 8, conc-list: [4] } - - { tp: 4, ep: 4, dp-attn: true, conc-list: [1024] } - - { tp: 8, ep: 8, dp-attn: true, conc-list: [256, 512, 1024] } + - tp: 2 + ep: 1 + conc-list: + - 4 + - 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 4 + ep: 1 + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 2 + ep: 2 + conc-list: + - 8 + - 32 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 8 + ep: 8 + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 4 + ep: 4 + dp-attn: true + conc-list: + - 1024 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + conc-list: + - 256 + - 512 + - 1024 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml qwen3.5-fp4-b200-trt-mtp: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc18 @@ -5297,11 +5409,40 @@ qwen3.5-fp4-b200-trt-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 2, ep: 1, spec-decoding: "mtp", conc-list: [4] } - - { tp: 2, ep: 2, spec-decoding: "mtp", conc-list: [8, 16] } - - { tp: 4, ep: 4, spec-decoding: "mtp", conc-list: [4] } - - { tp: 8, ep: 8, spec-decoding: "mtp", conc-list: [4] } - - { tp: 8, ep: 8, dp-attn: true, spec-decoding: "mtp", conc-list: [128, 256, 1024] } + - tp: 2 + ep: 1 + spec-decoding: mtp + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 2 + ep: 2 + spec-decoding: mtp + conc-list: + - 8 + - 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 4 + ep: 4 + spec-decoding: mtp + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 8 + ep: 8 + spec-decoding: mtp + conc-list: + - 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml + - tp: 8 + ep: 8 + dp-attn: true + spec-decoding: mtp + conc-list: + - 128 + - 256 + - 1024 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml minimaxm3-fp8-h100-vllm-agentic-mtp: image: vllm/vllm-openai:v0.27.1 diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index 2fc1233383..5441ee3fff 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -11,7 +11,30 @@ import yaml -from infx.srt_slurm.synthetic_acceptance import selected_recipes, spec_parameters +from infx.srt_slurm.synthetic_acceptance import ENGINES, selected_recipes, spec_parameters + + +def parallelism_constraints( + engine: str, args: Mapping[str, Any], environment: Mapping[str, str] +) -> dict[str, tuple[Any, Any]]: + """Read each engine's native topology fields without translating the recipe.""" + tp, ep = int(environment["TP"]), int(environment["EP_SIZE"]) + dp_attention = environment["DP_ATTENTION"] == "true" + if engine == "sglang": + return { + "tensor-parallel-size": (args["tensor-parallel-size"], tp), + "data-parallel-size": (args.get("data-parallel-size", 1), tp if dp_attention else 1), + "expert-parallel-size": (args.get("expert-parallel-size", args.get("ep-size", 1)), ep), + "DP_ATTENTION": (args.get("enable-dp-attention", False), dp_attention), + } + if engine == "trtllm": + return { + "tensor_parallel_size": (args["tensor_parallel_size"], tp), + "moe_expert_parallel_size": (args["moe_expert_parallel_size"], ep), + "pipeline_parallel_size": (args.get("pipeline_parallel_size", 1), 1), + "DP_ATTENTION": (args.get("enable_attention_dp", False), dp_attention), + } + raise ValueError(f"Unsupported single-node SRT engine: {engine!r}") def select_recipe(config: str, environment: Mapping[str, str]) -> tuple[str, dict[str, Any]]: @@ -42,28 +65,20 @@ def validate_recipe(recipe: dict[str, Any], environment: Mapping[str, str]) -> N args = role["args"] benchmark = recipe["benchmark"] workload = benchmark["env"] - spec = spec_parameters(role, "sglang") - if spec and spec["method"] not in {"eagle", "nextn"}: + engine_config = recipe["engine"] + engine = engine_config["type"] if isinstance(engine_config, dict) else engine_config + if environment["FRAMEWORK"] not in {"sglang", "trt"}: + raise ValueError(f"Unsupported single-node framework: {environment['FRAMEWORK']!r}") + spec = spec_parameters(role, engine) + if spec and spec["method"] not in {"eagle", "nextn", "mtp"}: raise ValueError("Single-node SRT supports only native MTP or no speculation") speculation = "mtp" if spec else "none" expected = { - "engine": (recipe["engine"], environment["FRAMEWORK"]), + "engine": (engine, ENGINES[environment["FRAMEWORK"]]), "model": (recipe["model"]["path"], f"hf:{environment['MODEL']}"), "image": (recipe["model"]["container"], environment["IMAGE"]), "precision": (recipe["model"]["precision"], environment["PRECISION"]), - "tensor-parallel-size": (args["tensor-parallel-size"], int(environment["TP"])), - "data-parallel-size": ( - args.get("data-parallel-size", 1), - int(environment["TP"]) if environment["DP_ATTENTION"] == "true" else 1, - ), - "DP_ATTENTION": ( - args.get("enable-dp-attention", False), - environment["DP_ATTENTION"] == "true", - ), - "expert-parallel-size": ( - args.get("expert-parallel-size", args.get("ep-size", 1)), - int(environment["EP_SIZE"]), - ), + **parallelism_constraints(engine, args, environment), "gpus": (role["gpus"], int(environment["GPU_COUNT"])), "nodes": (role["nodes"], 1), "workers": (role["workers"], 1), @@ -79,7 +94,6 @@ def validate_recipe(recipe: dict[str, Any], environment: Mapping[str, str]) -> N expected[name] = (str(workload[name]), environment[name]) # Multi-node and AgentX workloads use their existing connector. for name, value in { - "FRAMEWORK": "sglang", "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", @@ -98,7 +112,14 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: if environment[name] not in {"true", "false"}: raise ValueError(f"{name} must be true or false") overrides = [] - for name in ("CONC", "RESULT_FILENAME", "GPU_MONITOR_INTERVAL", "RUN_EVAL", "EVAL_ONLY"): + for name in ( + "CONC", + "RESULT_FILENAME", + "GPU_MONITOR_INTERVAL", + "RUN_EVAL", + "EVAL_ONLY", + "FRAMEWORK", + ): value = environment[name] if not value: raise ValueError(f"Missing runtime input: {name}") @@ -111,7 +132,13 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: context = int(environment["MAX_MODEL_LEN"]) if context <= 0: raise ValueError("MAX_MODEL_LEN must be positive") - overrides += ["--set", f"roles.agg.args.context-length={context}"] + context_keys = ( + ("context-length",) + if environment["FRAMEWORK"] == "sglang" + else ("max_seq_len", "max_num_tokens") + ) + for key in context_keys: + overrides += ["--set", f"roles.agg.args.{key}={context}"] return [*overrides, "--set", 'benchmark.env.RESULT_DIR="/logs"'] diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 8dc1cacdc0..dba875407e 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8639,3 +8639,19 @@ - "Cut over 21 active H100/H200/B200/B300 SGLang fixed-sequence recipes to native SRT-Slurm across their Slurm pools, retaining all 181 configured points, per-point serving settings, real MTP verification, shared eval dispatch, and existing result/power artifacts" - "将 21 个活跃的 H100/H200/B200/B300 SGLang 定长配方切换到各自 Slurm 池的原生 SRT-Slurm 路径,保留全部 181 个配置点、逐点服务参数、真实 MTP 验证、共享 eval 调度及现有结果和功耗产物" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-b200-trt + - dsr1-fp4-b200-trt-mtp + - dsr1-fp8-b200-trt + - dsr1-fp8-b200-trt-mtp + - dsr1-fp8-h200-trt + - dsr1-fp8-h200-trt-mtp + - qwen3.5-fp4-b200-trt + - qwen3.5-fp4-b200-trt-mtp + scenario-type: + - fixed-seq-len + description: + - "Cut over eight H200/B200 TRT-LLM fixed-sequence recipes to native SRT-Slurm, preserving all 63 points, engine configuration, real MTP verification, and the OpenAI benchmark client; retain explicit legacy metrics settings and eval token budgets" + - "将八个 H200/B200 TRT-LLM 定长配方切换至原生 SRT-Slurm,保留全部 63 个测试点、引擎配置、真实 MTP 验证及 OpenAI 基准客户端;显式保留原有指标设置和 eval token 预算" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/utils/test_srt_fixed_sequence.py b/utils/test_srt_fixed_sequence.py index 352db5b8e2..5fb9432813 100644 --- a/utils/test_srt_fixed_sequence.py +++ b/utils/test_srt_fixed_sequence.py @@ -48,6 +48,7 @@ def client_environment(tmp_path): "EVAL_ONLY": "false", "GPU_MONITOR_INTERVAL": "2", "USE_CHAT_TEMPLATE": "false", + "FRAMEWORK": "sglang", "IS_AGENTIC": "0", "SCENARIO_TYPE": "fixed-seq-len", "CLIENT_EXIT": "0", @@ -58,11 +59,15 @@ def client_environment(tmp_path): return env -@pytest.mark.parametrize("exit_code,chat_template", [(0, "false"), (7, "false"), (0, "true")]) +@pytest.mark.parametrize("exit_code,chat_template,framework,backend", [ + (0, "false", "sglang", "vllm"), (7, "false", "sglang", "vllm"), + (0, "true", "trt", "openai"), +]) def test_native_endpoint_preserves_client_settings_and_failure( - client_environment, exit_code, chat_template + client_environment, exit_code, chat_template, framework, backend ): - env = {**client_environment, "CLIENT_EXIT": str(exit_code), "USE_CHAT_TEMPLATE": chat_template} + env = {**client_environment, "CLIENT_EXIT": str(exit_code), "USE_CHAT_TEMPLATE": chat_template, + "FRAMEWORK": framework} result = subprocess.run( ["bash", str(CLIENT)], env=env, capture_output=True, text=True ) @@ -74,7 +79,7 @@ def test_native_endpoint_preserves_client_settings_and_failure( "--model", "test/model", "--backend", - "vllm", + backend, "--base-url", "http://10.2.3.4:9444", "--dataset-name", @@ -118,6 +123,7 @@ def test_native_endpoint_preserves_client_settings_and_failure( ("CONC", "0", "CONC must be a positive integer"), ("RUN_EVAL", "yes", "RUN_EVAL must be true or false"), ("EVAL_ONLY", "yes", "EVAL_ONLY must be true or false"), + ("FRAMEWORK", "unknown", "unsupported fixed-sequence FRAMEWORK"), ], ) def test_invalid_runtime_inputs_fail_before_the_client( diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index bcd15efb5d..c89c07ecd1 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -58,6 +58,7 @@ def test_native_binding_submits_one_point_and_keeps_server_settings(point): "USE_CHAT_TEMPLATE": "false", "CONC": "2", "RESULT_FILENAME": "point-identity", "GPU_MONITOR_INTERVAL": "3", "RUN_EVAL": "false", "EVAL_ONLY": "false", "RESULT_DIR": "/logs", + "FRAMEWORK": "sglang", } assert actual["roles"]["agg"]["args"] == { "tensor-parallel-size": 4, "data-parallel-size": 1, "max-running-requests": 32, @@ -170,6 +171,37 @@ def test_dp_attention_is_validated_without_replacing_recipe_topology(point): runtime_arguments(f"{path}:base", {**env, "DP_ATTENTION": "false"}) +def test_trt_binding_keeps_engine_options_and_sets_eval_token_budget(point): + path, recipe, env = point + recipe["engine"] = {"type": "trtllm", "served_model_name": "test/model"} + recipe["roles"]["agg"]["args"] = { + "tensor_parallel_size": 4, "moe_expert_parallel_size": 4, + "enable_attention_dp": True, "max_seq_len": 512, "max_num_tokens": 256, + "speculative_config": {"decoding_type": "MTP", "num_nextn_predict_layers": 3}, + "cuda_graph_config": {"batch_sizes": [1, 2, 4]}, + } + recipe["roles"]["agg"]["env"] = {"TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS": "3"} + recipe["benchmark"]["env"]["USE_CHAT_TEMPLATE"] = "true" + path.write_text(yaml.safe_dump({"base": recipe})) + env = {**env, "FRAMEWORK": "trt", "EP_SIZE": "4", "DP_ATTENTION": "true", + "SPEC_DECODING": "mtp", "EVAL_ONLY": "true", "MAX_MODEL_LEN": "1024"} + argv = runtime_arguments(f"{path}:base", env) + actual = copy.deepcopy(recipe) + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual["roles"]["agg"]["args"] == { + "tensor_parallel_size": 4, "moe_expert_parallel_size": 4, + "enable_attention_dp": True, "max_seq_len": 1024, "max_num_tokens": 1024, + "speculative_config": {"decoding_type": "MTP", "num_nextn_predict_layers": 3}, + "cuda_graph_config": {"batch_sizes": [1, 2, 4]}, + } + assert plan_commands(f"{path}:base", "trt", ["--json", *argv], env) == [[ + "srtctl", "apply", "--json", *argv, "--file", f"{path}:base", + "--unset", "roles.agg.env.TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS", + ]] + with pytest.raises(ValueError, match="moe_expert_parallel_size"): + runtime_arguments(f"{path}:base", {**env, "EP_SIZE": "1"}) + + @pytest.mark.parametrize("record,expected", [ ({"status": "submitted", "slurm_job_id": "42", "output_dir": "/shared/42"}, ("42", "/shared/42")), ({"status": "error"}, None), From 83eaebe9d6a31ddad3be1a7afb85760bba9d3e5e Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:16:31 -0500 Subject: [PATCH 12/29] feat: convert remaining fixed-sequence recipes to native SRT MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将剩余 AMD SGLang/ATOM 与 RTX 定长配方切换到原生 SRT 配置,保留测试点及服务参数;固定直接 ATOM 服务草稿依赖并扩展行为验证。 --- .gitmodules | 2 +- .../dsr1/atom/mi355x-fp4-mtp/8k1k.yaml | 44 +++++ .../dsr1/atom/mi355x-fp4/8k1k.yaml | 50 +++++ .../dsr1/atom/mi355x-fp8-mtp/8k1k.yaml | 44 +++++ .../dsr1/atom/mi355x-fp8/8k1k.yaml | 43 +++++ .../dsr1/sglang/mi300x-fp8/8k1k.yaml | 55 ++++++ .../dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml | 60 ++++++ .../dsr1/sglang/mi325x-fp8/8k1k.yaml | 55 ++++++ .../dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml | 66 +++++++ .../dsr1/sglang/mi355x-fp4/8k1k.yaml | 71 +++++++ .../dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml | 68 +++++++ .../dsr1/sglang/mi355x-fp8/8k1k.yaml | 70 +++++++ .../qwen3.5/atom/mi355x-fp4/8k1k.yaml | 51 +++++ .../qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml | 53 ++++++ .../qwen3.5/atom/mi355x-fp8/8k1k.yaml | 58 ++++++ .../qwen3.5/sglang/mi300x-fp8/8k1k.yaml | 57 ++++++ .../qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml | 61 ++++++ .../qwen3.5/sglang/mi325x-fp8/8k1k.yaml | 57 ++++++ .../qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml | 76 ++++++++ .../qwen3.5/sglang/mi355x-fp4/8k1k.yaml | 72 ++++++++ .../qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml | 68 +++++++ .../qwen3.5/sglang/mi355x-fp8/8k1k.yaml | 64 +++++++ .../sglang/rtx6000pro-fp4-mtp/8k1k.yaml | 88 +++++++++ .../qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml | 82 +++++++++ benchmarks/single_node/srt_docker.sh | 33 ++++ benchmarks/single_node/srt_fixed_sequence.sh | 8 +- configs/amd-master.yaml | 174 +++++++++++++++--- configs/nvidia-master.yaml | 20 +- infx/srt_slurm/docker.py | 117 ++++++++++++ infx/srt_slurm/single_node.py | 26 ++- infx/srt_slurm/synthetic_acceptance.py | 10 +- perf-changelog.yaml | 91 +++++++++ runners/launch_mi300x-amd.sh | 18 ++ runners/launch_mi325x-amds.sh | 18 ++ runners/launch_mi355x-amds.sh | 18 ++ runners/launch_rtx6000pro-lat.sh | 27 ++- runners/slurm_utils.sh | 13 +- runners/srt-slurm/mi300x-amd.yaml | 21 +++ runners/srt-slurm/mi325x-amds.yaml | 17 ++ runners/srt-slurm/mi355x-amds.yaml | 14 ++ utils/srt-slurm | 2 +- utils/test_srt_docker.py | 101 ++++++++++ utils/test_srt_fixed_sequence.py | 13 +- utils/test_srt_single_node.py | 36 +++- 44 files changed, 2137 insertions(+), 55 deletions(-) create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml create mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml create mode 100644 benchmarks/single_node/srt_docker.sh create mode 100644 infx/srt_slurm/docker.py create mode 100644 runners/srt-slurm/mi300x-amd.yaml create mode 100644 runners/srt-slurm/mi325x-amds.yaml create mode 100644 runners/srt-slurm/mi355x-amds.yaml create mode 100644 utils/test_srt_docker.py diff --git a/.gitmodules b/.gitmodules index f7635a307b..c2f2f853f3 100644 --- a/.gitmodules +++ b/.gitmodules @@ -3,4 +3,4 @@ url = https://github.com/SemiAnalysisAI/aiperf.git [submodule "utils/srt-slurm"] path = utils/srt-slurm - url = https://github.com/NVIDIA/srt-slurm.git + url = https://github.com/SemiAnalysisAI/srt-slurm.git diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..d09b7c8ed5 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml @@ -0,0 +1,44 @@ +# Serving settings from dsr1_fp4_mi355x_atom_mtp.sh. +base: + schema: 2 + name: dsr1-fp4-mi355x-atom-mtp-8k1k + model: + path: hf:amd/DeepSeek-R1-0528-MXFP4 + container: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atom + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + method: mtp + env: + OMP_NUM_THREADS: '1' + AMDGCN_USE_BUFFER_OPS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/DeepSeek-R1-0528-MXFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml new file mode 100644 index 0000000000..24d3cad417 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml @@ -0,0 +1,50 @@ +# Serving settings from dsr1_fp4_mi355x_atom.sh. +base: + schema: 2 + name: dsr1-fp4-mi355x-atom-8k1k + model: + path: hf:amd/DeepSeek-R1-0528-MXFP4-Preview + container: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atom + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + block-size: 16 + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/DeepSeek-R1-0528-MXFP4-Preview + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + benchmark: + env: + CONC: ['4'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..ac91c9ac34 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml @@ -0,0 +1,44 @@ +# Serving settings from dsr1_fp8_mi355x_atom_mtp.sh. +base: + schema: 2 + name: dsr1-fp8-mi355x-atom-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: rocm/atom:rocm7.2.4_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.3 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atom + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + method: mtp + num-speculative-tokens: 3 + kv_cache_dtype: fp8 + no-enable_prefix_caching: true + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml new file mode 100644 index 0000000000..68efd10f89 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml @@ -0,0 +1,43 @@ +# Serving settings from dsr1_fp8_mi355x_atom.sh. +base: + schema: 2 + name: dsr1-fp8-mi355x-atom-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atom + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + block-size: 16 + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml new file mode 100644 index 0000000000..b60ad7cf81 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml @@ -0,0 +1,55 @@ +# Serving settings from dsr1_fp8_mi300x.sh. +base: + schema: 2 + name: dsr1-fp8-mi300x-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-rocm700-mi30x + precision: fp8 + resources: + gpu_type: mi300x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 128 + chunked-prefill-size: 131072 + num-continuous-decode-steps: 4 + max-prefill-tokens: 131072 + kv-cache-dtype: fp8_e4m3 + attention-backend: aiter + disable-radix-cache: true + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..1cdc40c0df --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml @@ -0,0 +1,60 @@ +# Serving settings from dsr1_fp8_mi325x_mtp.sh. +base: + schema: 2 + name: dsr1-fp8-mi325x-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-rocm700-mi30x + precision: fp8 + resources: + gpu_type: mi325x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + expert-parallel-size: 1 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 128 + chunked-prefill-size: 131072 + num-continuous-decode-steps: 4 + max-prefill-tokens: 131072 + kv-cache-dtype: fp8_e4m3 + attention-backend: aiter + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + disable-radix-cache: true + data-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + SGLANG_ENABLE_SPEC_V2: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml new file mode 100644 index 0000000000..1d8209510c --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml @@ -0,0 +1,55 @@ +# Serving settings from dsr1_fp8_mi325x.sh. +base: + schema: 2 + name: dsr1-fp8-mi325x-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.19-rocm700-mi30x + precision: fp8 + resources: + gpu_type: mi325x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + mem-fraction-static: 0.8 + cuda-graph-max-bs: 128 + chunked-prefill-size: 131072 + num-continuous-decode-steps: 4 + max-prefill-tokens: 131072 + kv-cache-dtype: fp8_e4m3 + attention-backend: aiter + disable-radix-cache: true + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..af530edec5 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml @@ -0,0 +1,66 @@ +# Serving settings from dsr1_fp4_mi355x_mtp.sh. +base: + schema: 2 + name: dsr1-fp4-mi355x-sglang-mtp-8k1k + model: + path: hf:amd/DeepSeek-R1-0528-MXFP4 + container: lmsysorg/sglang:v0.5.12-rocm700-mi35x + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + trust-remote-code: true + tensor-parallel-size: 8 + expert-parallel-size: 1 + chunked-prefill-size: 196608 + mem-fraction-static: 0.8 + disable-radix-cache: true + num-continuous-decode-steps: 4 + max-prefill-tokens: 196608 + cuda-graph-max-bs: 128 + attention-backend: aiter + kv-cache-dtype: fp8_e4m3 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + data-parallel-size: 1 + served-model-name: amd/DeepSeek-R1-0528-MXFP4 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + SGLANG_ENABLE_SPEC_V2: '1' + ROCM_QUICK_REDUCE_QUANTIZATION: INT4 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/DeepSeek-R1-0528-MXFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + chunked-prefill-size: [196608, 196608, 196608, 196608, 32768] + max-prefill-tokens: [196608, 196608, 196608, 196608, 32768] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml new file mode 100644 index 0000000000..e4ff3977c4 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml @@ -0,0 +1,71 @@ +# Serving settings from dsr1_fp4_mi355x.sh. +base: + schema: 2 + name: dsr1-fp4-mi355x-sglang-8k1k + model: + path: hf:amd/DeepSeek-R1-0528-MXFP4-Preview + container: lmsysorg/sglang:v0.5.12-rocm700-mi35x + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + trust-remote-code: true + tensor-parallel-size: 4 + chunked-prefill-size: 196608 + mem-fraction-static: 0.8 + disable-radix-cache: true + num-continuous-decode-steps: 4 + max-prefill-tokens: 196608 + cuda-graph-max-bs: 128 + attention-backend: aiter + kv-cache-dtype: fp8_e4m3 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: amd/DeepSeek-R1-0528-MXFP4-Preview + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + ROCM_QUICK_REDUCE_QUANTIZATION: INT4 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/DeepSeek-R1-0528-MXFP4-Preview + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + chunked-prefill-size: [196608, 196608, 196608, 196608, 32768] + max-prefill-tokens: [196608, 196608, 196608, 196608, 32768] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + args: + chunked-prefill-size: [196608, 196608, 196608, 196608, 32768] + max-prefill-tokens: [196608, 196608, 196608, 196608, 32768] + tensor-parallel-size: 8 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..2c0d689c9d --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml @@ -0,0 +1,68 @@ +# Serving settings from dsr1_fp8_mi355x_mtp.sh. +base: + schema: 2 + name: dsr1-fp8-mi355x-sglang-mtp-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-rocm700-mi35x + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + attention-backend: aiter + tensor-parallel-size: 8 + expert-parallel-size: 1 + trust-remote-code: true + chunked-prefill-size: 196608 + mem-fraction-static: 0.8 + disable-radix-cache: true + num-continuous-decode-steps: 8 + max-prefill-tokens: 196608 + kv-cache-dtype: fp8_e4m3 + cuda-graph-max-bs: 4 + max-running-requests: 4 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + data-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_AITER_MLA_PERSIST: '1' + SGLANG_ENABLE_SPEC_V2: '1' + RCCL_MSCCL_ENABLE: '0' + ROCM_QUICK_REDUCE_QUANTIZATION: INT4 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + max-running-requests: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml new file mode 100644 index 0000000000..c9f958349d --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml @@ -0,0 +1,70 @@ +# Serving settings from dsr1_fp8_mi355x.sh. +base: + schema: 2 + name: dsr1-fp8-mi355x-sglang-8k1k + model: + path: hf:deepseek-ai/DeepSeek-R1-0528 + container: lmsysorg/sglang:v0.5.12-rocm700-mi35x + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + attention-backend: aiter + tensor-parallel-size: 4 + trust-remote-code: true + chunked-prefill-size: 196608 + mem-fraction-static: 0.8 + disable-radix-cache: true + num-continuous-decode-steps: 8 + max-prefill-tokens: 196608 + kv-cache-dtype: fp8_e4m3 + cuda-graph-max-bs: 32 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: deepseek-ai/DeepSeek-R1-0528 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + RCCL_MSCCL_ENABLE: '0' + ROCM_QUICK_REDUCE_QUANTIZATION: INT4 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: deepseek-ai/DeepSeek-R1-0528 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda-graph-max-bs: [32, 64] + benchmark: + env: + CONC: ['32', '64'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + tensor-parallel-size: 8 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml new file mode 100644 index 0000000000..2ef1824ec9 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml @@ -0,0 +1,51 @@ +# Serving settings from qwen3.5_fp4_mi355x_atom.sh. +base: + schema: 2 + name: qwen3.5-fp4-mi355x-atom-8k1k + model: + path: hf:amd/Qwen3.5-397B-A17B-MXFP4 + container: rocm/atom:rocm7.2.2_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.2.post + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atom + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + gpu-memory-utilization: 0.9 + trust-remote-code: true + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code + env: + MODEL: amd/Qwen3.5-397B-A17B-MXFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp2_ep1: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + benchmark: + env: + CONC: ['4', '8', '16'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..468285d526 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml @@ -0,0 +1,53 @@ +# Serving settings from qwen3.5_fp8_mi355x_atom_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp8-mi355x-atom-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: rocm/atom:rocm7.2.2_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.2.post + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atom + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + gpu-memory-utilization: 0.9 + method: mtp + num-speculative-tokens: 3 + trust-remote-code: true + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml new file mode 100644 index 0000000000..3a1bbaa81f --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml @@ -0,0 +1,58 @@ +# Serving settings from qwen3.5_fp8_mi355x_atom.sh. +base: + schema: 2 + name: qwen3.5-fp8-mi355x-atom-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: atom + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: atom + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + kv_cache_dtype: fp8 + max-model-len: 10240 + gpu-memory-utilization: 0.9 + trust-remote-code: true + env: + OMP_NUM_THREADS: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp2_ep1: + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp8_ep1: + roles: + agg: + gpus: 8 + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml new file mode 100644 index 0000000000..bc00e95f39 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml @@ -0,0 +1,57 @@ +# Serving settings from qwen3.5_fp8_mi300x.sh. +base: + schema: 2 + name: qwen3.5-fp8-mi300x-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-rocm720-mi30x + precision: fp8 + resources: + gpu_type: mi300x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + attention-backend: aiter + tensor-parallel-size: 8 + data-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + cuda-graph-max-bs: 4 + disable-radix-cache: true + max-prefill-tokens: 32768 + scheduler-recv-interval: 30 + mem-fraction-static: 0.75 + context-length: 9236 + expert-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..0f1d40a983 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml @@ -0,0 +1,61 @@ +# Serving settings from qwen3.5_fp8_mi325x_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp8-mi325x-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-rocm720-mi30x + precision: fp8 + resources: + gpu_type: mi325x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + attention-backend: aiter + tensor-parallel-size: 8 + expert-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + cuda-graph-max-bs: 4 + disable-radix-cache: true + max-prefill-tokens: 32768 + scheduler-recv-interval: 30 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + mem-fraction-static: 0.75 + context-length: 9236 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml new file mode 100644 index 0000000000..a17a9e2085 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml @@ -0,0 +1,57 @@ +# Serving settings from qwen3.5_fp8_mi325x.sh. +base: + schema: 2 + name: qwen3.5-fp8-mi325x-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang:v0.5.12-rocm720-mi30x + precision: fp8 + resources: + gpu_type: mi325x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + args: + attention-backend: aiter + tensor-parallel-size: 8 + data-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + cuda-graph-max-bs: 4 + disable-radix-cache: true + max-prefill-tokens: 32768 + scheduler-recv-interval: 30 + mem-fraction-static: 0.75 + context-length: 9236 + expert-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..58b7ce27c0 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml @@ -0,0 +1,76 @@ +# Serving settings from qwen3.5_fp4_mi355x_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp4-mi355x-sglang-mtp-8k1k + model: + path: hf:amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + container: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913 + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + trust-remote-code: true + tensor-parallel-size: 2 + attention-backend: aiter + mem-fraction-static: 0.8 + model-loader-extra-config: '{"enable_multithread_load": true}' + watchdog-timeout: 1200 + disable-radix-cache: true + max-running-requests: 4 + page-size: 16 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + kv-cache-dtype: fp8_e4m3 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_USE_AITER_UNIFIED_ATTN: '1' + AITER_FLYDSL_FORCE: '1' + SGLANG_MAMBA_SSM_DTYPE: bfloat16 + ROCM_QUICK_REDUCE_QUANTIZATION: INT8 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp2_ep1: + roles: + agg: + args: + max-running-requests: [4, 8, 16, 32, 64, 128] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + max-running-requests: [4, 8, 16] + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['4', '8', '16'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml new file mode 100644 index 0000000000..9488267696 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml @@ -0,0 +1,72 @@ +# Serving settings from qwen3.5_fp4_mi355x.sh. +base: + schema: 2 + name: qwen3.5-fp4-mi355x-sglang-8k1k + model: + path: hf:amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + container: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913 + precision: fp4 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + args: + trust-remote-code: true + tensor-parallel-size: 2 + attention-backend: aiter + mem-fraction-static: 0.8 + model-loader-extra-config: '{"enable_multithread_load": true}' + watchdog-timeout: 1200 + disable-radix-cache: true + max-running-requests: 4 + page-size: 16 + kv-cache-dtype: fp8_e4m3 + data-parallel-size: 1 + expert-parallel-size: 1 + served-model-name: amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + enable-metrics: false + env: + SGLANG_USE_AITER: '1' + SGLANG_USE_AITER_UNIFIED_ATTN: '1' + AITER_FLYDSL_FORCE: '1' + SGLANG_MAMBA_SSM_DTYPE: bfloat16 + ROCM_QUICK_REDUCE_QUANTIZATION: INT8 + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: amd/Qwen3.5-397B-A17B-MXFP4-AttnFP8-V2 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp2_ep1: + roles: + agg: + args: + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] +zip_override_tp4_ep1: + roles: + agg: + gpus: 4 + args: + max-running-requests: [4, 8, 16] + tensor-parallel-size: 4 + benchmark: + env: + CONC: ['4', '8', '16'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml new file mode 100644 index 0000000000..9e5e80cad8 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml @@ -0,0 +1,68 @@ +# Serving settings from qwen3.5_fp8_mi355x_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp8-mi355x-sglang-mtp-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + attention-backend: aiter + tensor-parallel-size: 4 + expert-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + max-running-requests: 4 + cuda-graph-max-bs: 4 + disable-radix-cache: true + chunked-prefill-size: 32768 + scheduler-recv-interval: 30 + mem-fraction-static: 0.8 + model-loader-extra-config: '{"enable_multithread_load": true}' + page-size: 16 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + context-length: 9236 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_USE_AITER_UNIFIED_ATTN: '1' + SGLANG_USE_AITER: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml new file mode 100644 index 0000000000..12757766b2 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml @@ -0,0 +1,64 @@ +# Serving settings from qwen3.5_fp8_mi355x.sh. +base: + schema: 2 + name: qwen3.5-fp8-mi355x-sglang-8k1k + model: + path: hf:Qwen/Qwen3.5-397B-A17B-FP8 + container: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 + precision: fp8 + resources: + gpu_type: mi355x + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + attention-backend: aiter + tensor-parallel-size: 4 + expert-parallel-size: 1 + trust-remote-code: true + tokenizer-worker-num: 6 + enable-aiter-allreduce-fusion: true + max-running-requests: 4 + cuda-graph-max-bs: 4 + disable-radix-cache: true + chunked-prefill-size: 32768 + scheduler-recv-interval: 30 + mem-fraction-static: 0.8 + model-loader-extra-config: '{"enable_multithread_load": true}' + page-size: 16 + context-length: 9236 + data-parallel-size: 1 + served-model-name: Qwen/Qwen3.5-397B-A17B-FP8 + enable-metrics: false + env: + SGLANG_USE_AITER_UNIFIED_ATTN: '1' + SGLANG_USE_AITER: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh + env: + MODEL: Qwen/Qwen3.5-397B-A17B-FP8 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_concurrency: + roles: + agg: + args: + cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + max-running-requests: [4, 8, 16, 32, 64, 128, 256] + benchmark: + env: + CONC: ['4', '8', '16', '32', '64', '128', '256'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml new file mode 100644 index 0000000000..9385e888f1 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml @@ -0,0 +1,88 @@ +# Serving settings from qwen3.5_fp4_rtx6000pro_sglang_mtp.sh. +base: + schema: 2 + name: qwen3.5-fp4-rtx6000pro-sglang-mtp-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4 + container: lmsysorg/sglang:v0.5.16-cu130 + precision: fp4 + resources: + gpu_type: rtx6000pro + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + quantization: modelopt_fp4 + fp4-gemm-backend: flashinfer_cutlass + moe-runner-backend: flashinfer_cutlass + attention-backend: flashinfer + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-scheduler-strategy: no_buffer + disable-custom-all-reduce: true + disable-radix-cache: true + mem-fraction-static: 0.8 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 9236 + cuda-graph-max-bs-decode: 1 + max-running-requests: 1 + scheduler-recv-interval: 10 + stream-interval: 20 + speculative-algorithm: EAGLE + speculative-num-steps: 3 + speculative-eagle-topk: 1 + speculative-num-draft-tokens: 4 + enable-metrics: false + env: + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + PYTHONUNBUFFERED: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'true' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda-graph-max-bs-decode: [1, 4, 16, 64] + max-running-requests: [1, 4, 16, 64] + scheduler-recv-interval: [10, 10, 30, 30] + benchmark: + env: + CONC: ['1', '4', '16', '64'] +zip_override_tp4_ep4: + roles: + agg: + args: + cuda-graph-max-bs-decode: [1, 4, 16, 64] + expert-parallel-size: 4 + max-running-requests: [1, 4, 16, 64] + scheduler-recv-interval: [10, 10, 30, 30] + benchmark: + env: + CONC: ['1', '4', '16', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml new file mode 100644 index 0000000000..ac4e793474 --- /dev/null +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml @@ -0,0 +1,82 @@ +# Serving settings from qwen3.5_fp4_rtx6000pro_sglang.sh. +base: + schema: 2 + name: qwen3.5-fp4-rtx6000pro-sglang-8k1k + model: + path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4 + container: lmsysorg/sglang:v0.5.16-cu130 + precision: fp4 + resources: + gpu_type: rtx6000pro + gpus_per_node: 8 + frontend: + type: sglang + enable_multiple_frontends: false + observability: + enabled: false + tachometer: + enabled: false + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 4 + args: + served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 + trust-remote-code: true + tensor-parallel-size: 4 + data-parallel-size: 1 + expert-parallel-size: 1 + reasoning-parser: qwen3 + tool-call-parser: qwen3_coder + quantization: modelopt_fp4 + fp4-gemm-backend: flashinfer_cutlass + moe-runner-backend: flashinfer_cutlass + attention-backend: flashinfer + kv-cache-dtype: fp8_e4m3 + mamba-ssm-dtype: bfloat16 + mamba-scheduler-strategy: no_buffer + disable-custom-all-reduce: true + disable-radix-cache: true + mem-fraction-static: 0.7 + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + context-length: 9236 + cuda-graph-max-bs-decode: 1 + max-running-requests: 128 + scheduler-recv-interval: 10 + stream-interval: 20 + enable-metrics: false + env: + SGLANG_ENABLE_JIT_DEEPGEMM: 'false' + PYTHONUNBUFFERED: '1' + PYTHONNOUSERSITE: '1' + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code + env: + MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4 + ISL: '8192' + OSL: '1024' + RANDOM_RANGE_RATIO: '0.8' + USE_CHAT_TEMPLATE: 'false' +zip_override_tp4_ep1: + roles: + agg: + args: + cuda-graph-max-bs-decode: [1, 4, 16, 64] + scheduler-recv-interval: [10, 10, 30, 30] + benchmark: + env: + CONC: ['1', '4', '16', '64'] +zip_override_tp4_ep4: + roles: + agg: + args: + cuda-graph-max-bs-decode: [1, 4, 16, 64] + expert-parallel-size: 4 + scheduler-recv-interval: [10, 10, 30, 30] + benchmark: + env: + CONC: ['1', '4', '16', '64'] diff --git a/benchmarks/single_node/srt_docker.sh b/benchmarks/single_node/srt_docker.sh new file mode 100644 index 0000000000..47bc95a7f4 --- /dev/null +++ b/benchmarks/single_node/srt_docker.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash + +# Keep the pool's existing Docker lifecycle; commands come from native SRT YAML. +set -eo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only +check_env_vars MODEL PORT RUN_EVAL EVAL_ONLY INFMAX_CONTAINER_WORKSPACE +for flag in RUN_EVAL EVAL_ONLY; do + if [[ "${!flag}" != true && "${!flag}" != false ]]; then + echo "$flag must be true or false" >&2 + exit 1 + fi +done +SERVER_LOG="$INFMAX_CONTAINER_WORKSPACE/server.log" +if [[ -n "${MODEL_PATH:-}" ]]; then + if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then + hf download "$MODEL" --local-dir "$MODEL_PATH" + fi +else + hf download "$MODEL" +fi +bash "$INFMAX_CONTAINER_WORKSPACE/srt-docker-server.sh" > "$SERVER_LOG" 2>&1 & +SERVER_PID=$! +trap 'rc=$?; kill "$SERVER_PID" 2>/dev/null || true; exit "$rc"' EXIT +source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" +export INFERENCEX_SERVER_PID INFERENCEX_SERVER_STATE +if [[ "$EVAL_ONLY" != true ]]; then + bash "$INFMAX_CONTAINER_WORKSPACE/srt-docker-client.sh" +fi +if [[ "$RUN_EVAL" == true || "$EVAL_ONLY" == true ]]; then + bash "$INFMAX_CONTAINER_WORKSPACE/benchmarks/single_node/srt_eval.sh" \ + "http://127.0.0.1:$PORT" "$INFMAX_CONTAINER_WORKSPACE/infx-eval-exit-code" +fi diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index 452a0b7c3d..4ec1578606 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -12,12 +12,18 @@ for name in RUN_EVAL EVAL_ONLY; do fi done case "$FRAMEWORK" in - sglang) CLIENT_BACKEND=vllm ;; + sglang|atom) CLIENT_BACKEND=vllm ;; trt) CLIENT_BACKEND=openai ;; *) echo "ERROR: unsupported fixed-sequence FRAMEWORK: $FRAMEWORK" >&2; exit 1 ;; esac SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" CLIENT_ARGS=() +for argument in "$@"; do + case "$argument" in + --trust-remote-code) CLIENT_ARGS+=("$argument") ;; + *) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;; + esac +done case "$USE_CHAT_TEMPLATE" in true) CLIENT_ARGS+=(--use-chat-template) ;; false) ;; diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index c7f5d254b0..ec13e1a3e3 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -11,8 +11,14 @@ dsr1-fp4-mi355x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, conc-start: 4, conc-end: 64 } - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 4 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml dsr1-fp4-mi355x-sglang-mtp: image: lmsysorg/sglang:v0.5.12-rocm700-mi35x @@ -27,7 +33,12 @@ dsr1-fp4-mi355x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml dsr1-fp4-mi355x-atom: image: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 @@ -42,8 +53,16 @@ dsr1-fp4-mi355x-atom: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256 } - - { tp: 8, ep: 1, conc-start: 4, conc-end: 4 } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 4 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml dsr1-fp4-mi355x-atom-mtp: image: rocm/atom:rocm7.2.3_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom20260511 @@ -60,7 +79,11 @@ dsr1-fp4-mi355x-atom-mtp: osl: 1024 search-space: #- { tp: 4, conc-start: 32, conc-end: 256, spec-decoding: mtp } - - { tp: 8, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 8 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml dsr1-fp8-mi300x-sglang: image: lmsysorg/sglang:v0.5.12-rocm700-mi30x @@ -75,7 +98,10 @@ dsr1-fp8-mi300x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml dsr1-fp8-mi325x-sglang: image: lmsysorg/sglang:v0.5.19-rocm700-mi30x @@ -90,7 +116,10 @@ dsr1-fp8-mi325x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml dsr1-fp8-mi355x-sglang: image: lmsysorg/sglang:v0.5.12-rocm700-mi35x @@ -105,8 +134,14 @@ dsr1-fp8-mi355x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, conc-start: 32, conc-end: 64 } - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 4 + conc-start: 32 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml dsr1-fp8-mi355x-sglang-mtp: image: lmsysorg/sglang:v0.5.12-rocm700-mi35x @@ -121,7 +156,12 @@ dsr1-fp8-mi355x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml qwen3.5-fp8-mi325x-sglang: image: lmsysorg/sglang:v0.5.12-rocm720-mi30x @@ -136,7 +176,10 @@ qwen3.5-fp8-mi325x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml qwen3.5-fp8-mi355x-sglang: image: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 @@ -151,7 +194,11 @@ qwen3.5-fp8-mi355x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256 } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml qwen3.5-fp8-mi355x-sglang-mtp: image: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 @@ -166,7 +213,12 @@ qwen3.5-fp8-mi355x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml # MI325X official matrix selected from the complete 62-point fast sweep. TP2 # peaks at c4, TP4/TEP4 at c40, and TP8/TEP8 at c64; the next point beyond each @@ -202,9 +254,21 @@ qwen3.5-fp8-mi355x-atom: - isl: 8192 osl: 1024 search-space: - - { tp: 2, ep: 1, conc-start: 4, conc-end: 256 } - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256 } - - { tp: 8, ep: 1, conc-start: 4, conc-end: 256 } + - tp: 2 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml qwen3.5-fp8-mi355x-atom-mtp: image: rocm/atom:rocm7.2.2_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.2.post @@ -219,8 +283,18 @@ qwen3.5-fp8-mi355x-atom-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } - - { tp: 8, ep: 1, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 4 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml qwen3.5-fp8-mi355x-sglang-disagg: image: lmsysorg/sglang:v0.5.16-rocm720-mi35x @@ -278,8 +352,14 @@ qwen3.5-fp4-mi355x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 2, conc-start: 4, conc-end: 256 } - - { tp: 4, conc-start: 4, conc-end: 16 } + - tp: 2 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml + - tp: 4 + conc-start: 4 + conc-end: 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml qwen3.5-fp4-mi355x-atom: image: rocm/atom:rocm7.2.2_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.2.post @@ -294,8 +374,14 @@ qwen3.5-fp4-mi355x-atom: - isl: 8192 osl: 1024 search-space: - - { tp: 2, conc-start: 4, conc-end: 256 } - - { tp: 4, conc-start: 4, conc-end: 16 } + - tp: 2 + conc-start: 4 + conc-end: 256 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml + - tp: 4 + conc-start: 4 + conc-end: 16 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml qwen3.5-fp4-mi355x-sglang-mtp: image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260913 @@ -310,8 +396,16 @@ qwen3.5-fp4-mi355x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 2, conc-start: 4, conc-end: 128, spec-decoding: mtp } - - { tp: 4, conc-start: 4, conc-end: 16, spec-decoding: mtp } + - tp: 2 + conc-start: 4 + conc-end: 128 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml + - tp: 4 + conc-start: 4 + conc-end: 16 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml qwen3.5-fp4-mi355x-sglang-agentic-mtp: image: lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260915 @@ -376,7 +470,10 @@ qwen3.5-fp8-mi300x-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 64 } + - tp: 8 + conc-start: 4 + conc-end: 64 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml # AgentX Pareto sweep for Qwen3.5 FP8 on MI300X. The fast discovery run peaks # at c24; c32 records the post-knee throughput and latency cliff. @@ -409,7 +506,10 @@ dsr1-fp8-mi355x-atom: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 128 } + - tp: 8 + conc-start: 4 + conc-end: 128 + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml dsr1-fp8-mi355x-atom-mtp: image: rocm/atom:rocm7.2.4_ubuntu24.04_py3.12_pytorch_release_2.10.0_atom0.1.3 @@ -424,7 +524,11 @@ dsr1-fp8-mi355x-atom-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, conc-start: 4, conc-end: 256, spec-decoding: mtp } + - tp: 8 + conc-start: 4 + conc-end: 256 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml dsr1-fp8-mi355x-sglang-disagg: image: rocm/sgl-dev:sglang-0.5.9-rocm720-mi35x-mori-0227-2 @@ -960,7 +1064,12 @@ dsr1-fp8-mi325x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml qwen3.5-fp8-mi325x-sglang-mtp: image: lmsysorg/sglang:v0.5.12-rocm720-mi30x @@ -975,7 +1084,12 @@ qwen3.5-fp8-mi325x-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + - tp: 8 + ep: 1 + conc-start: 4 + conc-end: 64 + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml dsv4-fp4-mi355x-vllm-agentic-mtp: image: vllm/vllm-openai-rocm:nightly-eed1f3d0c6043bd494424a22443ee198dd56f657 diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index e79903bb1c..2767414982 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -1355,8 +1355,13 @@ qwen3.5-fp4-rtx6000pro-sglang: - isl: 8192 osl: 1024 search-space: - - { tp: 4, conc-list: [1, 4, 16, 64] } - - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64] } + - tp: 4 + conc-list: [1, 4, 16, 64] + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml + - tp: 4 + ep: 4 + conc-list: [1, 4, 16, 64] + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml # Same sweep with the built-in MTP draft head driven through SGLang's EAGLE # speculative path. @@ -1373,8 +1378,15 @@ qwen3.5-fp4-rtx6000pro-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - { tp: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } - - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } + - tp: 4 + conc-list: [1, 4, 16, 64] + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml + - tp: 4 + ep: 4 + conc-list: [1, 4, 16, 64] + spec-decoding: mtp + srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml qwen3.5-fp4-b300-sglang-mtp: image: lmsysorg/sglang:v0.5.19-cu130 diff --git a/infx/srt_slurm/docker.py b/infx/srt_slurm/docker.py new file mode 100644 index 0000000000..dcdb8f8839 --- /dev/null +++ b/infx/srt_slurm/docker.py @@ -0,0 +1,117 @@ +"""Prepare native SRT server/client commands for the existing Docker runner.""" + +from __future__ import annotations + +import argparse +import os +import re +import shlex +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +from infx.srt_slurm.single_node import runtime_arguments, select_recipe +from infx.srt_slurm.synthetic_acceptance import build_overrides + + +def shell_command(command: list[str], environment: Mapping[str, str]) -> str: + """Quote argv and literal recipe environment without embedding caller secrets.""" + lines = ["#!/usr/bin/env bash", "set -eo pipefail"] + for key, value in environment.items(): + if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", key): + raise ValueError(f"Invalid environment key: {key!r}") + lines.append(f"export {key}={shlex.quote(value)}") + lines.append(f"exec {shlex.join(command)}") + return "\n".join(lines) + "\n" + + +def prepare(config: str, environment: Mapping[str, str]) -> tuple[str, str]: + """Use the pinned SRT schema and backend builder; Docker stays pool-owned.""" + from srtctl.core.config import expand_engine_config_defaults, resolve_config_with_defaults + from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides + from srtctl.core.runtime import Nodes, RuntimeContext + from srtctl.core.schema import SrtConfig + from srtctl.core.topology import Process + + selected, recipe = select_recipe(config, environment) + if environment["FRAMEWORK"] != "sglang": + raise ValueError("The Docker runner currently supports native SGLang recipes only") + local_model = environment.get("MODEL_PATH") + port = int(environment["PORT"]) + if not 1 <= port <= 65535: + raise ValueError("PORT must be between 1 and 65535") + overrides = runtime_arguments(selected, environment) + apply_overrides_to_recipe(recipe, parse_overrides(overrides[1::2], [])) + golden = build_overrides(recipe, environment["FRAMEWORK"], environment) + # Fixed-sequence jobs remove any stale simulation flags, as native apply does. + parser = argparse.ArgumentParser(add_help=False) + parser.add_argument("--set", action="append", default=[]) + parser.add_argument("--unset", action="append", default=[]) + parsed = parser.parse_args(golden) + apply_overrides_to_recipe(recipe, parse_overrides(parsed.set, parsed.unset)) + resolved = resolve_config_with_defaults(recipe, {}) + expand_engine_config_defaults(resolved) + native = SrtConfig.Schema().load(resolved) + if ( + native.frontend.type != "sglang" + or native.services + or native.setup_script + or native.host_setup.enabled + or native.dynamo.sidecar + or native.extra_mount + or native.container_mounts + ): + raise ValueError( + "Docker fixed-sequence recipes require one direct server without services or setup" + ) + runtime = RuntimeContext( + job_id="docker", + run_name=native.name, + nodes=Nodes(head="127.0.0.1", bench="127.0.0.1", infra="127.0.0.1", worker=("127.0.0.1",)), + head_node_ip="127.0.0.1", + infra_node_ip="127.0.0.1", + log_dir=Path("/logs"), + model_path=Path(local_model or environment["MODEL"]), + container_image=Path(environment["IMAGE"]), + gpus_per_node=native.resources.gpus_per_node, + network_interface=None, + # Docker already exposes MODEL_PATH through its existing mounts. Use + # the native builder's literal path mode, without Slurm's /model mount. + is_hf_model=True, + frontend_port=port, + ) + process = Process( + node="127.0.0.1", + gpu_indices=frozenset(range(int(environment["GPU_COUNT"]))), + sys_port=port, + http_port=port, + endpoint_mode="agg", + endpoint_index=0, + ) + server = native.backend.build_worker_command( + process, [process], runtime, frontend_type="sglang" + ) + server_env = {**native.backend.get_environment_for_mode("agg"), **native.environment} + benchmark: dict[str, Any] = recipe["benchmark"] + client_env = { + **benchmark["env"], + "SRT_FRONTEND_HOST": "127.0.0.1", + "SRT_FRONTEND_PORT": str(port), + } + return shell_command(server, server_env), shell_command( + shlex.split(benchmark["command"]), client_env + ) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("recipe") + parser.add_argument("output", type=Path) + args = parser.parse_args() + server, client = prepare(args.recipe, os.environ) + (args.output / "srt-docker-server.sh").write_text(server) + (args.output / "srt-docker-client.sh").write_text(client) + + +if __name__ == "__main__": + main() diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index 5441ee3fff..26f47229d6 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -13,6 +13,8 @@ from infx.srt_slurm.synthetic_acceptance import ENGINES, selected_recipes, spec_parameters +SINGLE_NODE_ENGINES = {**ENGINES, "atom": "atom"} + def parallelism_constraints( engine: str, args: Mapping[str, Any], environment: Mapping[str, str] @@ -34,6 +36,13 @@ def parallelism_constraints( "pipeline_parallel_size": (args.get("pipeline_parallel_size", 1), 1), "DP_ATTENTION": (args.get("enable_attention_dp", False), dp_attention), } + if engine == "atom": + if ep not in {1, tp}: + raise ValueError("ATOM expert parallelism must be 1 or TP") + return { + "enable-expert-parallel": (args.get("enable-expert-parallel", False), ep > 1), + "DP_ATTENTION": (args.get("enable-dp-attention", False), dp_attention), + } raise ValueError(f"Unsupported single-node SRT engine: {engine!r}") @@ -67,14 +76,14 @@ def validate_recipe(recipe: dict[str, Any], environment: Mapping[str, str]) -> N workload = benchmark["env"] engine_config = recipe["engine"] engine = engine_config["type"] if isinstance(engine_config, dict) else engine_config - if environment["FRAMEWORK"] not in {"sglang", "trt"}: + if environment["FRAMEWORK"] not in {"sglang", "trt", "atom"}: raise ValueError(f"Unsupported single-node framework: {environment['FRAMEWORK']!r}") spec = spec_parameters(role, engine) if spec and spec["method"] not in {"eagle", "nextn", "mtp"}: raise ValueError("Single-node SRT supports only native MTP or no speculation") speculation = "mtp" if spec else "none" expected = { - "engine": (engine, ENGINES[environment["FRAMEWORK"]]), + "engine": (engine, SINGLE_NODE_ENGINES[environment["FRAMEWORK"]]), "model": (recipe["model"]["path"], f"hf:{environment['MODEL']}"), "image": (recipe["model"]["container"], environment["IMAGE"]), "precision": (recipe["model"]["precision"], environment["PRECISION"]), @@ -90,6 +99,9 @@ def validate_recipe(recipe: dict[str, Any], environment: Mapping[str, str]) -> N } if "CONC" in workload: expected["CONC"] = (str(workload["CONC"]), environment["CONC"]) + if engine == "atom": + # Native ATOM derives -tp from the aggregate worker's GPU allocation. + expected["ATOM TP"] = (role["gpus"], int(environment["TP"])) for name in ("ISL", "OSL", "RANDOM_RANGE_RATIO"): expected[name] = (str(workload[name]), environment[name]) # Multi-node and AgentX workloads use their existing connector. @@ -132,11 +144,11 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: context = int(environment["MAX_MODEL_LEN"]) if context <= 0: raise ValueError("MAX_MODEL_LEN must be positive") - context_keys = ( - ("context-length",) - if environment["FRAMEWORK"] == "sglang" - else ("max_seq_len", "max_num_tokens") - ) + context_keys = { + "sglang": ("context-length",), + "trt": ("max_seq_len", "max_num_tokens"), + "atom": ("max-model-len",), + }[environment["FRAMEWORK"]] for key in context_keys: overrides += ["--set", f"roles.agg.args.{key}={context}"] return [*overrides, "--set", 'benchmark.env.RESULT_DIR="/logs"'] diff --git a/infx/srt_slurm/synthetic_acceptance.py b/infx/srt_slurm/synthetic_acceptance.py index edd6990a51..0a9aa05403 100644 --- a/infx/srt_slurm/synthetic_acceptance.py +++ b/infx/srt_slurm/synthetic_acceptance.py @@ -36,6 +36,14 @@ def spec_parameters(role: Mapping[str, Any], engine: str) -> dict[str, Any]: args = role.get("args", {}) + if engine == "atom": + method = args.get("method") + if not method: + return {} + return { + "method": str(method).lower(), + "num_speculative_tokens": args.get("num-speculative-tokens"), + } if engine == "vllm": raw = args.get("speculative-config") if raw is None: @@ -227,7 +235,7 @@ def plan_commands( """Build native arguments for every selected variant before submitting jobs.""" command = ["srtctl", "apply", *arguments] if framework not in ENGINES: - return [command] + return [[*command, "--file", config]] from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides path, _, selector = config.partition(":") diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 766a40b777..954584cf49 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8690,3 +8690,94 @@ - "Cut over eight H200/B200 TRT-LLM fixed-sequence recipes to native SRT-Slurm, preserving all 63 points, engine configuration, real MTP verification, and the OpenAI benchmark client; retain explicit legacy metrics settings and eval token budgets" - "将八个 H200/B200 TRT-LLM 定长配方切换至原生 SRT-Slurm,保留全部 63 个测试点、引擎配置、真实 MTP 验证及 OpenAI 基准客户端;显式保留原有指标设置和 eval token 预算" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-mi355x-sglang + - dsr1-fp4-mi355x-sglang-mtp + - dsr1-fp4-mi355x-atom + - dsr1-fp4-mi355x-atom-mtp + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - dsr1-fp8-mi355x-sglang + - dsr1-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi355x-atom + - qwen3.5-fp8-mi355x-atom-mtp + - qwen3.5-fp4-mi355x-sglang + - qwen3.5-fp4-mi355x-atom + - qwen3.5-fp4-mi355x-sglang-mtp + - qwen3.5-fp8-mi300x-sglang + - dsr1-fp8-mi355x-atom + - dsr1-fp8-mi355x-atom-mtp + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + - qwen3.5-fp4-rtx6000pro-sglang + - qwen3.5-fp4-rtx6000pro-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - Cut over the remaining 23 AMD SGLang/ATOM and RTX PRO 6000 fixed-sequence recipes to native SRT configuration, retaining all 179 points and legacy server/client settings; use direct ATOM serving and native command generation within the existing Docker pool lifecycle + - 将剩余 23 个 AMD SGLang/ATOM 和 RTX PRO 6000 定长配方切换至原生 SRT 配置,保留全部 179 个测试点及原有服务端和客户端参数;ATOM 使用直接服务模式,Docker 池在现有容器生命周期内使用原生命令生成 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-b200-sglang + - dsr1-fp4-b200-sglang-mtp + - dsr1-fp4-b300-sglang + - dsr1-fp4-b200-trt + - dsr1-fp4-b200-trt-mtp + - dsr1-fp8-b200-sglang + - dsr1-fp8-b300-sglang + - qwen3.5-fp8-b200-sglang + - qwen3.5-fp4-b200-sglang + - qwen3.5-fp4-b200-sglang-mtp + - qwen3.5-fp8-b200-sglang-mtp + - qwen3.5-fp8-b300-sglang-mtp + - qwen3.5-fp8-b300-sglang + - qwen3.5-fp4-b300-sglang + - qwen3.5-fp4-rtx6000pro-sglang + - qwen3.5-fp4-rtx6000pro-sglang-mtp + - qwen3.5-fp4-b300-sglang-mtp + - dsr1-fp8-b200-sglang-mtp + - dsr1-fp8-b300-sglang-mtp + - dsr1-fp8-b200-trt + - dsr1-fp8-b200-trt-mtp + - dsr1-fp8-h200-sglang + - dsr1-fp8-h200-sglang-mtp + - qwen3.5-fp8-h200-sglang + - qwen3.5-fp8-h200-sglang-mtp + - dsr1-fp8-h200-trt + - dsr1-fp8-h200-trt-mtp + - qwen3.5-fp8-h100-sglang + - qwen3.5-fp8-h100-sglang-mtp + - qwen3.5-fp4-b200-trt + - qwen3.5-fp4-b200-trt-mtp + - dsr1-fp4-mi355x-sglang + - dsr1-fp4-mi355x-sglang-mtp + - dsr1-fp4-mi355x-atom + - dsr1-fp4-mi355x-atom-mtp + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - dsr1-fp8-mi355x-sglang + - dsr1-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi355x-atom + - qwen3.5-fp8-mi355x-atom-mtp + - qwen3.5-fp4-mi355x-sglang + - qwen3.5-fp4-mi355x-atom + - qwen3.5-fp4-mi355x-sglang-mtp + - qwen3.5-fp8-mi300x-sglang + - dsr1-fp8-mi355x-atom + - dsr1-fp8-mi355x-atom-mtp + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - Pin the native SRT runtime to the reviewed-in-draft AMD/ATOM integration stack plus direct aggregate ATOM support; keep runner hardware settings in cluster profiles and exclude the held served-model-name PR + - 将原生 SRT 运行时固定到待评审的 AMD/ATOM 集成依赖链及 ATOM 聚合直接服务支持;硬件设置保留在集群配置中,不包含暂缓的 served-model-name PR + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/runners/launch_mi300x-amd.sh b/runners/launch_mi300x-amd.sh index 6d88f743cd..d74cf01df1 100644 --- a/runners/launch_mi300x-amd.sh +++ b/runners/launch_mi300x-amd.sh @@ -4,6 +4,24 @@ source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validat check_env_vars IS_MULTINODE set -eo pipefail +# Select native fixed-sequence execution before the retained AgentX/multi-node paths. +EXECUTION_PATH=legacy +if [[ "$IS_MULTINODE" == false && -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + check_env_vars GITHUB_WORKSPACE MODEL IMAGE + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + export HF_HUB_CACHE_MOUNT=/raid/inferencex/models/hub + export SRT_MODEL_PATH="hf:$MODEL" + export SALLOC_TIME_LIMIT=180 + export SRT_SCRATCH_ROOT="$(dirname "$GITHUB_WORKSPACE")" + export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' + SRT_SQUASH_FILE="/raid/inferencex/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node mi300x-amd + exit $? +fi + export HF_HUB_CACHE_MOUNT="/raid/inferencex/models/hub" export AIPERF_MMAP_CACHE_MOUNT="/raid/inferencex/aiperf-mmap-cache" export AIPERF_DATASET_MMAP_CACHE_DIR="/aiperf_mmap_cache" diff --git a/runners/launch_mi325x-amds.sh b/runners/launch_mi325x-amds.sh index e2cd6501a3..c0411cdc68 100644 --- a/runners/launch_mi325x-amds.sh +++ b/runners/launch_mi325x-amds.sh @@ -4,6 +4,24 @@ source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validat check_env_vars IS_MULTINODE set -eo pipefail +# Select native fixed-sequence execution before the retained AgentX/multi-node paths. +EXECUTION_PATH=legacy +if [[ "$IS_MULTINODE" == false && -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + check_env_vars GITHUB_WORKSPACE MODEL IMAGE + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + export HF_HUB_CACHE_MOUNT=/raid/hf-hub-cache/ + export SRT_MODEL_PATH="hf:$MODEL" + export SALLOC_TIME_LIMIT=480 + export SRT_SCRATCH_ROOT="$(dirname "$GITHUB_WORKSPACE")" + export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' + SRT_SQUASH_FILE="/raid/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node mi325x-amds + exit $? +fi + export HF_HUB_CACHE_MOUNT="/raid/hf-hub-cache/" PARTITION="compute" diff --git a/runners/launch_mi355x-amds.sh b/runners/launch_mi355x-amds.sh index e27c0178d0..41f56f4455 100644 --- a/runners/launch_mi355x-amds.sh +++ b/runners/launch_mi355x-amds.sh @@ -3,6 +3,24 @@ source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 check_env_vars EVAL_ONLY IS_AGENTIC IS_MULTINODE KEEP_LOGS RUN_EVAL +# Select native fixed-sequence execution before the retained AgentX/multi-node paths. +EXECUTION_PATH=legacy +if [[ "$IS_MULTINODE" == false && -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + check_env_vars GITHUB_WORKSPACE MODEL IMAGE + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + export HF_HUB_CACHE_MOUNT=/var/lib/hf-hub-cache/ + export SRT_MODEL_PATH="hf:$MODEL" + export SALLOC_TIME_LIMIT=500 + export SRT_SCRATCH_ROOT="$(dirname "$GITHUB_WORKSPACE")" + export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' + SRT_SQUASH_FILE="/var/lib/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + launch_srt_single_node mi355x-amds + exit $? +fi + scancel_sync() { local jobid=$1 local timeout=${2:-600} diff --git a/runners/launch_rtx6000pro-lat.sh b/runners/launch_rtx6000pro-lat.sh index 32e244b88c..d372a0f4d5 100755 --- a/runners/launch_rtx6000pro-lat.sh +++ b/runners/launch_rtx6000pro-lat.sh @@ -20,6 +20,28 @@ check_env_vars NCCL_IB_DISABLE : "${EXP_NAME:?EXP_NAME must be set}" : "${PRECISION:?PRECISION must be set}" +EXECUTION_PATH=legacy +if [[ -n "${SRT_RECIPE:-}" ]]; then + EXECUTION_PATH=native-single-node +fi +NATIVE_MOUNTS=() +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 + SRTCTL_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-docker.XXXXXX") + setup_srt_slurm "$SRTCTL_ROOT" "$FRAMEWORK" 0 + if ! command -v uv >/dev/null; then + curl -LsSf https://astral.sh/uv/install.sh | sh + source "$HOME/.local/bin/env" + fi + uv venv .venv + source .venv/bin/activate + uv pip install -e . + PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.srt_slurm.docker "$GITHUB_WORKSPACE/$SRT_RECIPE" "$GITHUB_WORKSPACE" + NATIVE_MOUNTS=(--volume "$GITHUB_WORKSPACE:/infmax-workspace" --volume "$GITHUB_WORKSPACE:/logs") + cd "$GITHUB_WORKSPACE" +fi + mkdir -p "$HF_HUB_CACHE_MOUNT" check_env_vars GPU_COUNT @@ -47,7 +69,9 @@ SCENARIO_SUBDIR="${SCENARIO_SUBDIR%/}/" # untagged name for scripts not yet retagged. BENCH_BASE="benchmarks/single_node/${SCENARIO_SUBDIR}${EXP_NAME%%_*}_${PRECISION}_rtx6000pro" BENCH_SCRIPT="${BENCH_BASE}_${FRAMEWORK:-}${SPEC_SUFFIX}.sh" -if [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then +if [[ "$EXECUTION_PATH" == native-single-node ]]; then + BENCH_SCRIPT=benchmarks/single_node/srt_docker.sh +elif [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then BENCH_SCRIPT="${BENCH_BASE}${SPEC_SUFFIX}.sh" fi @@ -77,6 +101,7 @@ done docker run \ "${RUNTIME_ENV_ARGS[@]}" \ + "${NATIVE_MOUNTS[@]}" \ --env IS_MULTINODE \ --env REQUIRE_POWER \ --env INFMAX_CONTAINER_WORKSPACE \ diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 3478e37fad..87939435ff 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -58,7 +58,7 @@ PYENV # native recipe environment and benchmark.env retain their override priority. local source="$INFERENCEX_SLURM_UTILS_DIR/../utils/srt-slurm" if [[ "$framework" == "tilert" ]]; then - # Sole fork exception until NVIDIA supports the TileRT backend and router. + # TileRT still needs its legacy runtime until the native backend and router land. SRT_SLURM_COMMIT=6bc3f306bdafa1edfb5dded2fcda8f1ccede1bde git init "$destination" || return 1 git -C "$destination" remote add origin https://github.com/SemiAnalysisAI/srt-slurm.git || return 1 @@ -137,7 +137,12 @@ launch_srt_single_node() { TP PP_SIZE DCP_SIZE PCP_SIZE EP_SIZE DP_ATTENTION GPU_COUNT IS_AGENTIC SPEC_DECODING \ CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME GPU_MONITOR_INTERVAL SRT_MODEL_PATH \ HF_HUB_CACHE_MOUNT HF_HUB_CACHE SALLOC_TIME_LIMIT - SRT_SINGLE_NODE_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-single.XXXXXX") + local scratch_root="$GITHUB_WORKSPACE" + if [[ -n "${SRT_SCRATCH_ROOT:-}" ]]; then + check_env_vars SRT_SCRATCH_ROOT + scratch_root="$SRT_SCRATCH_ROOT" + fi + SRT_SINGLE_NODE_ROOT=$(mktemp -d "$scratch_root/srt-single.XXXXXX") SRTCTL_ROOT="$SRT_SINGLE_NODE_ROOT/checkout" export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" setup_srt_slurm "$SRTCTL_ROOT" "$FRAMEWORK" 0 @@ -154,6 +159,10 @@ launch_srt_single_node() { mapfile -d '' -t SRT_RUNTIME_ARGS < "$SRT_SINGLE_NODE_ROOT/arguments" SRT_SELECTED_RECIPE="${SRT_RUNTIME_ARGS[0]}" SRT_RUNTIME_ARGS=("${SRT_RUNTIME_ARGS[@]:1}") + if [[ -n "${SRT_SRUN_OPTIONS:-}" ]]; then + check_env_vars SRT_SRUN_OPTIONS + SRT_RUNTIME_ARGS+=(--set "srun_options=$SRT_SRUN_OPTIONS") + fi SRT_RUNTIME_ARGS+=( --set 'post_eval.command=["bash", "{infmax_workspace}/benchmarks/single_node/srt_eval.sh", "{endpoint}", "/logs/infx-eval-exit-code"]' --set "post_eval.passthrough_env=$SRT_EVAL_PASSTHROUGH" diff --git a/runners/srt-slurm/mi300x-amd.yaml b/runners/srt-slurm/mi300x-amd.yaml new file mode 100644 index 0000000000..0bf432ba3e --- /dev/null +++ b/runners/srt-slurm/mi300x-amd.yaml @@ -0,0 +1,21 @@ +default_partition: compute-0 +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: '' +visible_devices_env: ROCR_VISIBLE_DEVICES +srtctl_root: ${SRTCTL_ROOT} +default_mounts: + /dev/kfd: /dev/kfd + /dev/dri: /dev/dri +default_sbatch_directives: + cpus-per-task: '128' +use_gpus_per_node_directive: true +use_segment_sbatch_directive: false +use_exclusive_sbatch_directive: true +default_bash_preamble: | + if [[ "$$MODEL_PREFIX" == dsr1 && "$$FRAMEWORK" == sglang ]]; then + firmware=$$(rocm-smi --showfw | awk '/MEC/ {print $$NF; exit}') + if [[ -z "$$firmware" || "$$firmware" -lt 177 ]]; then + export HSA_NO_SCRATCH_RECLAIM=1 + fi + fi diff --git a/runners/srt-slurm/mi325x-amds.yaml b/runners/srt-slurm/mi325x-amds.yaml new file mode 100644 index 0000000000..c87092f5dd --- /dev/null +++ b/runners/srt-slurm/mi325x-amds.yaml @@ -0,0 +1,17 @@ +default_partition: compute +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: '' +visible_devices_env: ROCR_VISIBLE_DEVICES +srtctl_root: ${SRTCTL_ROOT} +default_mounts: + /dev/kfd: /dev/kfd + /dev/dri: /dev/dri +default_sbatch_directives: + cpus-per-task: '256' +use_gpus_per_node_directive: true +use_segment_sbatch_directive: false +use_exclusive_sbatch_directive: true +default_bash_preamble: | + export XDG_CACHE_HOME="/tmp/xdg-cache-$$SLURM_JOB_ID" + export TRITON_CACHE_DIR="/tmp/triton-cache-$$SLURM_JOB_ID" diff --git a/runners/srt-slurm/mi355x-amds.yaml b/runners/srt-slurm/mi355x-amds.yaml new file mode 100644 index 0000000000..f6a0df40e8 --- /dev/null +++ b/runners/srt-slurm/mi355x-amds.yaml @@ -0,0 +1,14 @@ +default_partition: compute +default_time_limit: ${SRT_DEFAULT_TIME_LIMIT} +gpus_per_node: 8 +network_interface: '' +visible_devices_env: ROCR_VISIBLE_DEVICES +srtctl_root: ${SRTCTL_ROOT} +default_mounts: + /dev/kfd: /dev/kfd + /dev/dri: /dev/dri +default_sbatch_directives: + cpus-per-task: '128' +use_gpus_per_node_directive: true +use_segment_sbatch_directive: false +use_exclusive_sbatch_directive: true diff --git a/utils/srt-slurm b/utils/srt-slurm index 2ac4eb1367..c29ef7c5d0 160000 --- a/utils/srt-slurm +++ b/utils/srt-slurm @@ -1 +1 @@ -Subproject commit 2ac4eb1367dd2a78f597a72ca91afe4211d76b38 +Subproject commit c29ef7c5d0732fbf7fc93aa4b7929b48f0bfa76a diff --git a/utils/test_srt_docker.py b/utils/test_srt_docker.py new file mode 100644 index 0000000000..9e65fb9d2a --- /dev/null +++ b/utils/test_srt_docker.py @@ -0,0 +1,101 @@ +"""Execute native command generation and the Docker server/client lifecycle.""" + +import json +import os +import shutil +import subprocess +import sys +import time +from pathlib import Path + +import pytest +import yaml + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) +from infx.srt_slurm.docker import prepare + + +@pytest.mark.parametrize("local_model,eval_only,context", [("", "false", "256"), ("/cache/model with space", "true", "1024")]) +def test_native_docker_commands_preserve_model_flags_and_literal_environment(tmp_path, local_model, eval_only, context): + recipe = { + "schema": 2, "name": "test", "engine": "sglang", + "model": {"path": "hf:test/model", "container": "test:tag", "precision": "fp8"}, + "resources": {"gpu_type": "h200", "gpus_per_node": 8}, + "frontend": {"type": "sglang", "enable_multiple_frontends": False}, + "roles": {"agg": {"nodes": 1, "workers": 1, "gpus": 2, + "args": {"tensor-parallel-size": 2, "context-length": 256, "served-model-name": "test/model"}, + "env": {"SGLANG_LITERAL": "$(touch injected); 'literal'", "SGLANG_SIMULATE_ACC_LEN": "2.5"}}}, + "benchmark": {"type": "custom", "command": "python3 capture-client", + "env": {"MODEL": "test/model", "ISL": "128", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "USE_CHAT_TEMPLATE": "false"}}, + } + path = tmp_path / "recipe.yaml" + path.write_text(yaml.safe_dump(recipe)) + env = { + "FRAMEWORK": "sglang", "MODEL": "test/model", "MODEL_PREFIX": "test", "MODEL_PATH": local_model, + "IMAGE": "test:tag", "PRECISION": "fp8", "TP": "2", "GPU_COUNT": "2", + "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", "EP_SIZE": "1", "DP_ATTENTION": "false", + "SPEC_DECODING": "none", "IS_AGENTIC": "0", "RUN_EVAL": "false", "EVAL_ONLY": eval_only, + "MAX_MODEL_LEN": "1024", "ISL": "128", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "CONC": "3", + "RESULT_FILENAME": "controlled", "GPU_MONITOR_INTERVAL": "1", "PORT": "9019", + } + server, client = prepare(str(path), env) + stub = tmp_path / "python3" + stub.write_text(f"#!{sys.executable}\nimport json, os, sys\nprint(json.dumps({{'argv': sys.argv[1:], 'env': dict(os.environ)}}))\n") + stub.chmod(0o755) + runtime_env = {**os.environ, "PATH": f"{tmp_path}:{os.environ['PATH']}"} + run = subprocess.run(["bash", "-c", server], cwd=tmp_path, env=runtime_env, capture_output=True, text=True, check=True) + observed = json.loads(run.stdout) + argv = observed["argv"] + assert argv[:2] == ["-m", "sglang.launch_server"] + assert argv[argv.index("--model-path") + 1] == (local_model or "test/model") + assert argv[argv.index("--port") + 1] == "9019" + assert argv[argv.index("--tensor-parallel-size") + 1] == "2" + assert argv[argv.index("--context-length") + 1] == context + assert observed["env"]["SGLANG_LITERAL"] == "$(touch injected); 'literal'" + assert "SGLANG_SIMULATE_ACC_LEN" not in observed["env"] + assert not (tmp_path / "injected").exists() + run = subprocess.run(["bash", "-c", client], cwd=tmp_path, env=runtime_env, capture_output=True, text=True, check=True) + observed = json.loads(run.stdout) + assert observed["argv"] == ["capture-client"] + assert {key: observed["env"][key] for key in ("CONC", "RESULT_FILENAME", "SRT_FRONTEND_HOST", "SRT_FRONTEND_PORT")} == { + "CONC": "3", "RESULT_FILENAME": "controlled", "SRT_FRONTEND_HOST": "127.0.0.1", "SRT_FRONTEND_PORT": "9019", + } + + +@pytest.mark.parametrize("client_exit,ready", [(0, True), (7, True), (0, False)]) +def test_docker_client_failure_and_readiness_clean_up_owned_server(tmp_path, client_exit, ready): + workspace = tmp_path / "repo" + scripts = workspace / "benchmarks/single_node" + scripts.mkdir(parents=True) + shutil.copyfile(ROOT / "benchmarks/benchmark_lib.sh", scripts.parent / "benchmark_lib.sh") + shutil.copyfile(ROOT / "benchmarks/single_node/srt_docker.sh", scripts / "srt_docker.sh") + (workspace / "infx").symlink_to(ROOT / "infx", target_is_directory=True) + (workspace / "srt-docker-server.sh").write_text('echo $$ > "$INFMAX_CONTAINER_WORKSPACE/server.pid"\nexec sleep 120\n' if ready else 'exit 12\n') + (workspace / "srt-docker-client.sh").write_text(f'touch "$INFMAX_CONTAINER_WORKSPACE/client-ran"\nexit {client_exit}\n') + binaries = tmp_path / "bin" + binaries.mkdir() + for name, body in {"hf": 'printf "%s\\n" "$@" > "$INFMAX_CONTAINER_WORKSPACE/download"', "curl": f"exit {0 if ready else 1}", "sleep": "exit 0"}.items(): + # The server uses the real sleep; only readiness retry sleeps are accelerated. + binary = binaries / name + binary.write_text(f"#!/bin/bash\n{body}\n") + binary.chmod(0o755) + if ready: + (workspace / "srt-docker-server.sh").write_text('echo $$ > "$INFMAX_CONTAINER_WORKSPACE/server.pid"\nexec /bin/sleep 120\n') + env = {**os.environ, "PATH": f"{binaries}:{Path(sys.executable).parent}:{os.environ['PATH']}", + "INFMAX_CONTAINER_WORKSPACE": str(workspace), "MODEL": "test/model", "PORT": "9019", + "RUN_EVAL": "false", "EVAL_ONLY": "false", "MODEL_PATH": ""} + result = subprocess.run(["bash", str(scripts / "srt_docker.sh")], env=env, capture_output=True, text=True, timeout=15) + assert result.returncode == (client_exit if ready else 1), result.stdout + result.stderr + assert (workspace / "client-ran").exists() is ready + assert (workspace / "download").read_text() == "download\ntest/model\n" + if ready: + pid = int((workspace / "server.pid").read_text()) + for _ in range(100): + try: + os.kill(pid, 0) + except ProcessLookupError: + break + time.sleep(0.01) + else: + pytest.fail("Docker wrapper left its server process alive") diff --git a/utils/test_srt_fixed_sequence.py b/utils/test_srt_fixed_sequence.py index 5fb9432813..84575b1db2 100644 --- a/utils/test_srt_fixed_sequence.py +++ b/utils/test_srt_fixed_sequence.py @@ -59,17 +59,18 @@ def client_environment(tmp_path): return env -@pytest.mark.parametrize("exit_code,chat_template,framework,backend", [ - (0, "false", "sglang", "vllm"), (7, "false", "sglang", "vllm"), - (0, "true", "trt", "openai"), +@pytest.mark.parametrize("exit_code,chat_template,framework,backend,extra", [ + (0, "false", "sglang", "vllm", []), (7, "false", "sglang", "vllm", []), + (0, "true", "trt", "openai", []), + (0, "true", "atom", "vllm", ["--trust-remote-code"]), ]) def test_native_endpoint_preserves_client_settings_and_failure( - client_environment, exit_code, chat_template, framework, backend + client_environment, exit_code, chat_template, framework, backend, extra ): env = {**client_environment, "CLIENT_EXIT": str(exit_code), "USE_CHAT_TEMPLATE": chat_template, "FRAMEWORK": framework} result = subprocess.run( - ["bash", str(CLIENT)], env=env, capture_output=True, text=True + ["bash", str(CLIENT), *extra], env=env, capture_output=True, text=True ) assert result.returncode == exit_code, result.stderr argv = json.loads(Path(env["CAPTURE"]).read_text()) @@ -106,7 +107,7 @@ def test_native_endpoint_preserves_client_settings_and_failure( env["RESULT_DIR"], "--result-filename", "test-result.json", - ] + (["--use-chat-template"] if chat_template == "true" else []) + ] + (["--use-chat-template"] if chat_template == "true" else []) + extra assert ( (Path(env["RESULT_DIR"]) / "gpu_metrics.csv") .read_text() diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index c89c07ecd1..16e55bf60c 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -202,6 +202,37 @@ def test_trt_binding_keeps_engine_options_and_sets_eval_token_budget(point): runtime_arguments(f"{path}:base", {**env, "EP_SIZE": "1"}) +def test_atom_binding_uses_allocation_tp_and_native_mtp_arguments(point): + path, recipe, env = point + recipe["engine"] = "atom" + recipe["roles"]["agg"]["args"] = { + "method": "mtp", "num-speculative-tokens": 3, "kv_cache_dtype": "fp8", + "enable-expert-parallel": True, "enable-dp-attention": True, + } + recipe["benchmark"]["env"]["USE_CHAT_TEMPLATE"] = "true" + path.write_text(yaml.safe_dump({"base": recipe})) + env = {**env, "FRAMEWORK": "atom", "EP_SIZE": "4", "DP_ATTENTION": "true", + "SPEC_DECODING": "mtp", "EVAL_ONLY": "true", "MAX_MODEL_LEN": "2048"} + argv = runtime_arguments(f"{path}:base", env) + actual = copy.deepcopy(recipe) + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual["roles"]["agg"]["args"] == { + "method": "mtp", "num-speculative-tokens": 3, "kv_cache_dtype": "fp8", + "enable-expert-parallel": True, "enable-dp-attention": True, "max-model-len": 2048, + } + assert plan_commands(f"{path}:base", "atom", ["--json", *argv], env) == [[ + "srtctl", "apply", "--json", *argv, "--file", f"{path}:base", + ]] + for changes, error in [ + ({"EP_SIZE": "2"}, "expert parallelism"), + ({"EP_SIZE": "1"}, "enable-expert-parallel"), + ({"TP": "8", "EP_SIZE": "8"}, "ATOM TP"), + ({"DP_ATTENTION": "false"}, "DP_ATTENTION"), + ]: + with pytest.raises(ValueError, match=error): + runtime_arguments(f"{path}:base", {**env, **changes}) + + @pytest.mark.parametrize("record,expected", [ ({"status": "submitted", "slurm_job_id": "42", "output_dir": "/shared/42"}, ("42", "/shared/42")), ({"status": "error"}, None), @@ -224,6 +255,7 @@ def test_submission_manifest(tmp_path, record, expected): ("h200-cw", "none"), ("h100-cw", "none"), ("h100-dgxc-slurm", "none"), ("b200-cw", "none"), ("b200-nb", "none"), ("b200-nscale-slurm", "none"), ("b300-dsxe", "none"), + ("mi300x-amd", "none"), ("mi325x-amds", "none"), ("mi355x-amds", "none"), ]) def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, pool, failure): path, _, point_env = point @@ -278,6 +310,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "B300_HF_CACHE_CONTAINER_DIR": "/hf", "ENROOT_IMPORT_TIME_LIMIT": "10", "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), + "KEEP_LOGS": "0", } env.pop("AIPERF_DRAIN_TIMEOUT_SECONDS", None) env.pop("AIPERF_DRAIN_POLL_SECONDS", None) @@ -294,7 +327,8 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, assert (tmp_path / "gpu_metrics.csv").read_text() == "gpu,power\n0,300\n" assert json.loads((tmp_path / "gpu_metrics_context.json").read_text()) == {"device_count": 4} assert (tmp_path / "srt-single-node-logs.tar.gz").stat().st_size > 0 - cluster_config = yaml.safe_load(next(tmp_path.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) + scratch = tmp_path.parent if pool.startswith("mi") else tmp_path + cluster_config = yaml.safe_load(next(scratch.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) assert cluster_config["containers"]["test:tag"] == "test:tag" assert cluster_config["use_exclusive_sbatch_directive"] is True assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") From c40b0577d49a563bca9e4e183a11f120a3d1081c Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:23:32 -0500 Subject: [PATCH 13/29] fix: forward native Docker eval model identity MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 RTX 原生 Docker 路径显式传递 eval 模型名称,并验证实际启动器的命令、失败传播及容器清理。 --- benchmarks/single_node/srt_docker.sh | 3 ++ perf-changelog.yaml | 10 +++++ runners/launch_rtx6000pro-lat.sh | 7 ++-- utils/test_srt_docker.py | 59 ++++++++++++++++++++++++++++ 4 files changed, 76 insertions(+), 3 deletions(-) diff --git a/benchmarks/single_node/srt_docker.sh b/benchmarks/single_node/srt_docker.sh index 47bc95a7f4..705e367b52 100644 --- a/benchmarks/single_node/srt_docker.sh +++ b/benchmarks/single_node/srt_docker.sh @@ -10,6 +10,9 @@ for flag in RUN_EVAL EVAL_ONLY; do exit 1 fi done +if [[ "$RUN_EVAL" == true || "$EVAL_ONLY" == true ]]; then + check_env_vars MODEL_NAME +fi SERVER_LOG="$INFMAX_CONTAINER_WORKSPACE/server.log" if [[ -n "${MODEL_PATH:-}" ]]; then if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 3f9f02ea05..67676a910d 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8792,3 +8792,13 @@ - Pin the native SRT runtime to the reviewed-in-draft AMD/ATOM integration stack plus direct aggregate ATOM support; keep runner hardware settings in cluster profiles and exclude the held served-model-name PR - 将原生 SRT 运行时固定到待评审的 AMD/ATOM 集成依赖链及 ATOM 聚合直接服务支持;硬件设置保留在集群配置中,不包含暂缓的 served-model-name PR pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - qwen3.5-fp4-rtx6000pro-sglang + - qwen3.5-fp4-rtx6000pro-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - "Forward the native Docker server model identity into the shared eval path and reject missing eval metadata before startup" + - "将原生 Docker 服务的模型名称传入共享 eval 路径,并在启动前拒绝缺失的 eval 元数据" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/runners/launch_rtx6000pro-lat.sh b/runners/launch_rtx6000pro-lat.sh index d372a0f4d5..4bf36a9134 100755 --- a/runners/launch_rtx6000pro-lat.sh +++ b/runners/launch_rtx6000pro-lat.sh @@ -24,7 +24,7 @@ EXECUTION_PATH=legacy if [[ -n "${SRT_RECIPE:-}" ]]; then EXECUTION_PATH=native-single-node fi -NATIVE_MOUNTS=() +NATIVE_ARGS=() if [[ "$EXECUTION_PATH" == native-single-node ]]; then source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 SRTCTL_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-docker.XXXXXX") @@ -38,7 +38,8 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then uv pip install -e . PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" \ python3 -m infx.srt_slurm.docker "$GITHUB_WORKSPACE/$SRT_RECIPE" "$GITHUB_WORKSPACE" - NATIVE_MOUNTS=(--volume "$GITHUB_WORKSPACE:/infmax-workspace" --volume "$GITHUB_WORKSPACE:/logs") + NATIVE_ARGS=(--volume "$GITHUB_WORKSPACE:/infmax-workspace" --volume "$GITHUB_WORKSPACE:/logs" + --env "MODEL_NAME=$MODEL") cd "$GITHUB_WORKSPACE" fi @@ -101,7 +102,7 @@ done docker run \ "${RUNTIME_ENV_ARGS[@]}" \ - "${NATIVE_MOUNTS[@]}" \ + "${NATIVE_ARGS[@]}" \ --env IS_MULTINODE \ --env REQUIRE_POWER \ --env INFMAX_CONTAINER_WORKSPACE \ diff --git a/utils/test_srt_docker.py b/utils/test_srt_docker.py index 9e65fb9d2a..01701f687c 100644 --- a/utils/test_srt_docker.py +++ b/utils/test_srt_docker.py @@ -99,3 +99,62 @@ def test_docker_client_failure_and_readiness_clean_up_owned_server(tmp_path, cli time.sleep(0.01) else: pytest.fail("Docker wrapper left its server process alive") + + +def test_rtx_launcher_binds_eval_model_and_preserves_container_failure(tmp_path): + binaries = tmp_path / "bin" + binaries.mkdir() + (tmp_path / "benchmarks").symlink_to(ROOT / "benchmarks", target_is_directory=True) + path = tmp_path / "recipe.yaml" + path.write_text(yaml.safe_dump({ + "schema": 2, "name": "fixture", "engine": "sglang", + "model": {"path": "hf:test/model", "container": "test:tag", "precision": "fp4"}, + "resources": {"gpu_type": "rtx6000pro", "gpus_per_node": 8}, + "frontend": {"type": "sglang", "enable_multiple_frontends": False}, + "roles": {"agg": {"nodes": 1, "workers": 1, "gpus": 4, + "args": {"tensor-parallel-size": 4, "served-model-name": "test/model"}}}, + "benchmark": {"type": "custom", "command": "bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh", + "env": {"MODEL": "test/model", "ISL": "128", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "USE_CHAT_TEMPLATE": "false"}}, + })) + scripts = { + "git": 'if [[ "$1" == clone ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', + "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', + } + for name, body in scripts.items(): + binary = binaries / name + binary.write_text(f"#!/usr/bin/env bash\n{body}\n") + binary.chmod(0o755) + docker = binaries / "docker" + docker.write_text(f"#!{sys.executable}\n" + + "import json, os, pathlib, sys\n" + "with pathlib.Path(os.environ['CAPTURE']).open('a') as f: f.write(json.dumps(sys.argv[1:])+'\\n')\n" + "sys.exit(7 if sys.argv[1] == 'run' else 0)\n") + docker.chmod(0o755) + env = {**os.environ, "PATH": f"{binaries}:{Path(sys.executable).parent}:{os.environ['PATH']}", + "PYTHONPATH": f"{ROOT}:{ROOT / 'utils/srt-slurm/src'}", "GITHUB_WORKSPACE": str(tmp_path), + "SRT_RECIPE": path.name, "FRAMEWORK": "sglang", "MODEL": "test/model", "MODEL_PREFIX": "test", + "IMAGE": "test:tag", "PRECISION": "fp4", "TP": "4", "GPU_COUNT": "4", "PP_SIZE": "1", + "DCP_SIZE": "1", "PCP_SIZE": "1", "EP_SIZE": "1", "DP_ATTENTION": "false", "SPEC_DECODING": "none", + "IS_AGENTIC": "0", "RUN_EVAL": "true", "EVAL_ONLY": "true", "MAX_MODEL_LEN": "1024", + "ISL": "128", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "CONC": "3", "RESULT_FILENAME": "point", + "GPU_MONITOR_INTERVAL": "1", "PORT": "9019", "IS_MULTINODE": "false", "HF_HUB_CACHE_MOUNT": str(tmp_path / 'cache'), + "HF_HUB_CACHE": "/hf", "NCCL_IB_DISABLE": "1", "EXP_NAME": "test_8k1k", "SCENARIO_SUBDIR": "fixed_seq_len/", + "RUNNER_NAME": "fixture_00", "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "REQUIRE_POWER": "1", + "CAPTURE": str(tmp_path / "docker.jsonl"), "MODEL_PATH": ""} + result = subprocess.run(["bash", str(ROOT / 'runners/launch_rtx6000pro-lat.sh')], cwd=tmp_path, + env=env, capture_output=True, text=True, timeout=30) + assert result.returncode == 7, result.stdout + result.stderr + calls = [json.loads(line) for line in Path(env["CAPTURE"]).read_text().splitlines()] + run = next(call for call in calls if call[0] == 'run') + assert "MODEL_NAME=test/model" in [run[i+1] for i,v in enumerate(run[:-1]) if v == '--env'] + assert run[-2:] == ["test:tag", "benchmarks/single_node/srt_docker.sh"] + assert calls[-1] == ["rm", "-f", "bmk-server-fixture_00"] + # Run the emitted script against an external Python stub to verify the + # launcher's actual CLI path emitted the eval context and requested model. + stub = binaries / "python3" + stub.write_text(f"#!{sys.executable}\nimport json,sys\nprint(json.dumps(sys.argv[1:]))\n") + stub.chmod(0o755) + observed = subprocess.run(["bash", str(tmp_path / "srt-docker-server.sh")], env=env, capture_output=True, text=True, check=True) + argv = json.loads(observed.stdout) + assert argv[argv.index('--context-length')+1] == '1024' + assert argv[argv.index('--served-model-name')+1] == 'test/model' From 571fa51b3cc87866cff696248148e77bf2fe28a1 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:28:40 -0500 Subject: [PATCH 14/29] fix: apply AMD container options as native leaf overrides MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 按原生 SRT 语义逐项传递容器选项,增加映射行为回归测试并输出提交失败原因。 --- infx/srt_slurm/single_node.py | 12 ++++++++++++ perf-changelog.yaml | 29 +++++++++++++++++++++++++++++ runners/slurm_utils.sh | 11 ++++++----- utils/test_srt_single_node.py | 11 +++++++++++ 4 files changed, 58 insertions(+), 5 deletions(-) diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index 26f47229d6..280819f52a 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -5,6 +5,7 @@ import argparse import json import os +import re from collections.abc import Mapping from pathlib import Path from typing import Any @@ -124,6 +125,17 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: if environment[name] not in {"true", "false"}: raise ValueError(f"{name} must be true or false") overrides = [] + if environment.get("SRT_SRUN_OPTIONS"): + options = json.loads(environment["SRT_SRUN_OPTIONS"]) + if not isinstance(options, dict) or any( + not re.fullmatch(r"[a-z][a-z0-9-]*", key) or not isinstance(value, str) + for key, value in options.items() + ): + raise ValueError("SRT_SRUN_OPTIONS must map option names to string values") + # Native --set preserves whole mappings as JSON strings for engine + # flags. Runtime option mappings therefore need individual leaf sets. + for key, value in options.items(): + overrides += ["--set", f"srun_options.{key}={json.dumps(value)}"] for name in ( "CONC", "RESULT_FILENAME", diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 67676a910d..60cf8af973 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8802,3 +8802,32 @@ - "Forward the native Docker server model identity into the shared eval path and reject missing eval metadata before startup" - "将原生 Docker 服务的模型名称传入共享 eval 路径,并在启动前拒绝缺失的 eval 元数据" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-mi355x-sglang + - dsr1-fp4-mi355x-sglang-mtp + - dsr1-fp4-mi355x-atom + - dsr1-fp4-mi355x-atom-mtp + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - dsr1-fp8-mi355x-sglang + - dsr1-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi355x-atom + - qwen3.5-fp8-mi355x-atom-mtp + - qwen3.5-fp4-mi355x-sglang + - qwen3.5-fp4-mi355x-atom + - qwen3.5-fp4-mi355x-sglang-mtp + - qwen3.5-fp8-mi300x-sglang + - dsr1-fp8-mi355x-atom + - dsr1-fp8-mi355x-atom-mtp + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - Pass AMD container runtime options as native dotted leaf overrides so SRT validates them as a mapping before submission; surface native submission errors in workflow logs + - 将 AMD 容器运行时选项作为原生点路径叶值传递,确保 SRT 在提交前将其校验为映射,并在工作流日志中显示原生提交错误 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 87939435ff..477e70c775 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -159,10 +159,6 @@ launch_srt_single_node() { mapfile -d '' -t SRT_RUNTIME_ARGS < "$SRT_SINGLE_NODE_ROOT/arguments" SRT_SELECTED_RECIPE="${SRT_RUNTIME_ARGS[0]}" SRT_RUNTIME_ARGS=("${SRT_RUNTIME_ARGS[@]:1}") - if [[ -n "${SRT_SRUN_OPTIONS:-}" ]]; then - check_env_vars SRT_SRUN_OPTIONS - SRT_RUNTIME_ARGS+=(--set "srun_options=$SRT_SRUN_OPTIONS") - fi SRT_RUNTIME_ARGS+=( --set 'post_eval.command=["bash", "{infmax_workspace}/benchmarks/single_node/srt_eval.sh", "{endpoint}", "/logs/infx-eval-exit-code"]' --set "post_eval.passthrough_env=$SRT_EVAL_PASSTHROUGH" @@ -209,9 +205,14 @@ launch_srt_single_node() { trap finish_native_single_node EXIT trap 'exit 130' INT trap 'exit 143' TERM + local submission_rc=0 apply_srt_recipe "$SRT_SELECTED_RECIPE" "$FRAMEWORK" \ --json --yes --output "$SRT_SINGLE_NODE_ROOT/outputs" "${SRT_RUNTIME_ARGS[@]}" \ - > "$GITHUB_WORKSPACE/srt-single-node-submission.json" + > "$GITHUB_WORKSPACE/srt-single-node-submission.json" || submission_rc=$? + if (( submission_rc != 0 )); then + cat "$GITHUB_WORKSPACE/srt-single-node-submission.json" >&2 + return "$submission_rc" + fi python3 -m infx.srt_slurm.single_node submission "$GITHUB_WORKSPACE/srt-single-node-submission.json" \ > "$SRT_SINGLE_NODE_ROOT/submission-fields" mapfile -t SRT_SUBMISSION < "$SRT_SINGLE_NODE_ROOT/submission-fields" diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 16e55bf60c..37a8b721f0 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -332,3 +332,14 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, assert cluster_config["containers"]["test:tag"] == "test:tag" assert cluster_config["use_exclusive_sbatch_directive"] is True assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") + + +def test_runtime_container_options_remain_native_mapping(point): + path, recipe, env = point + env = {**env, "SRT_SRUN_OPTIONS": '{"container-remap-root":"", "container-writable":""}'} + argv = runtime_arguments(f"{path}:base", env) + actual = copy.deepcopy(recipe) + apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) + assert actual['srun_options'] == {'container-remap-root': '', 'container-writable': ''} + with pytest.raises(ValueError, match='must map option names to string values'): + runtime_arguments(f"{path}:base", {**env, 'SRT_SRUN_OPTIONS': '{"container-remap-root": true}'}) From 295d7e07bd5e942b3aa07b2659425801d7abb658 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:31:39 -0500 Subject: [PATCH 15/29] refactor: reuse native job workspace for AMD scratch files MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 复用作业检出目录存放 AMD 原生运行时临时文件,由现有 runner 清理流程回收,移除额外 scratch-root 配置。 --- runners/launch_mi300x-amd.sh | 1 - runners/launch_mi325x-amds.sh | 1 - runners/launch_mi355x-amds.sh | 1 - runners/slurm_utils.sh | 7 +------ utils/test_srt_single_node.py | 3 +-- 5 files changed, 2 insertions(+), 11 deletions(-) diff --git a/runners/launch_mi300x-amd.sh b/runners/launch_mi300x-amd.sh index d74cf01df1..d1faa67c50 100644 --- a/runners/launch_mi300x-amd.sh +++ b/runners/launch_mi300x-amd.sh @@ -15,7 +15,6 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then export HF_HUB_CACHE_MOUNT=/raid/inferencex/models/hub export SRT_MODEL_PATH="hf:$MODEL" export SALLOC_TIME_LIMIT=180 - export SRT_SCRATCH_ROOT="$(dirname "$GITHUB_WORKSPACE")" export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' SRT_SQUASH_FILE="/raid/inferencex/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" launch_srt_single_node mi300x-amd diff --git a/runners/launch_mi325x-amds.sh b/runners/launch_mi325x-amds.sh index c0411cdc68..f41e3278a9 100644 --- a/runners/launch_mi325x-amds.sh +++ b/runners/launch_mi325x-amds.sh @@ -15,7 +15,6 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then export HF_HUB_CACHE_MOUNT=/raid/hf-hub-cache/ export SRT_MODEL_PATH="hf:$MODEL" export SALLOC_TIME_LIMIT=480 - export SRT_SCRATCH_ROOT="$(dirname "$GITHUB_WORKSPACE")" export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' SRT_SQUASH_FILE="/raid/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" launch_srt_single_node mi325x-amds diff --git a/runners/launch_mi355x-amds.sh b/runners/launch_mi355x-amds.sh index 41f56f4455..135db385bc 100644 --- a/runners/launch_mi355x-amds.sh +++ b/runners/launch_mi355x-amds.sh @@ -14,7 +14,6 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then export HF_HUB_CACHE_MOUNT=/var/lib/hf-hub-cache/ export SRT_MODEL_PATH="hf:$MODEL" export SALLOC_TIME_LIMIT=500 - export SRT_SCRATCH_ROOT="$(dirname "$GITHUB_WORKSPACE")" export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' SRT_SQUASH_FILE="/var/lib/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" launch_srt_single_node mi355x-amds diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 477e70c775..4643f9c7a1 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -137,12 +137,7 @@ launch_srt_single_node() { TP PP_SIZE DCP_SIZE PCP_SIZE EP_SIZE DP_ATTENTION GPU_COUNT IS_AGENTIC SPEC_DECODING \ CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME GPU_MONITOR_INTERVAL SRT_MODEL_PATH \ HF_HUB_CACHE_MOUNT HF_HUB_CACHE SALLOC_TIME_LIMIT - local scratch_root="$GITHUB_WORKSPACE" - if [[ -n "${SRT_SCRATCH_ROOT:-}" ]]; then - check_env_vars SRT_SCRATCH_ROOT - scratch_root="$SRT_SCRATCH_ROOT" - fi - SRT_SINGLE_NODE_ROOT=$(mktemp -d "$scratch_root/srt-single.XXXXXX") + SRT_SINGLE_NODE_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-single.XXXXXX") SRTCTL_ROOT="$SRT_SINGLE_NODE_ROOT/checkout" export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" setup_srt_slurm "$SRTCTL_ROOT" "$FRAMEWORK" 0 diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 37a8b721f0..b8dba3893c 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -327,8 +327,7 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, assert (tmp_path / "gpu_metrics.csv").read_text() == "gpu,power\n0,300\n" assert json.loads((tmp_path / "gpu_metrics_context.json").read_text()) == {"device_count": 4} assert (tmp_path / "srt-single-node-logs.tar.gz").stat().st_size > 0 - scratch = tmp_path.parent if pool.startswith("mi") else tmp_path - cluster_config = yaml.safe_load(next(scratch.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) + cluster_config = yaml.safe_load(next(tmp_path.glob("srt-single.*/checkout/srtslurm.yaml")).read_text()) assert cluster_config["containers"]["test:tag"] == "test:tag" assert cluster_config["use_exclusive_sbatch_directive"] is True assert (capture.read_text() if capture.exists() else "") == ("42\n" if failure == "submission" else "") From 0bd3bf5f7fd3f0133f9fd2fb227a9dbe80033028 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:46:35 -0500 Subject: [PATCH 16/29] fix: complete AMD native launch and terminal status handling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 修复 AMD 原生启动命令的尾部换行,在 accounting 不可用时通过 Slurm 控制器验证终态,并等待启动失败的日志进程退出。 --- benchmarks/benchmark_lib.sh | 2 ++ perf-changelog.yaml | 16 ++++++++++++++++ runners/slurm_utils.sh | 21 ++++++++++++++++++++- runners/srt-slurm/mi300x-amd.yaml | 2 +- runners/srt-slurm/mi325x-amds.yaml | 2 +- utils/test_srt_docker.py | 21 +++++++++++++++++++++ utils/test_srt_single_node.py | 28 ++++++++++++++++++++++++++++ 7 files changed, 89 insertions(+), 3 deletions(-) diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index 214d7ca7cb..ec5d5e2670 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -718,11 +718,13 @@ wait_for_ready() { if ! kill -0 "$process_pid" 2>/dev/null; then echo "Process died before $endpoint became ready." >&2 kill "$tail_pid" 2>/dev/null || true + wait "$tail_pid" 2>/dev/null || true exit 1 fi if [[ "$deadline" -gt 0 && "$SECONDS" -ge "$deadline" ]]; then echo "Timed out waiting for $endpoint." >&2 kill "$tail_pid" 2>/dev/null || true + wait "$tail_pid" 2>/dev/null || true exit 1 fi sleep "$sleep_interval" diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 60cf8af973..17ebc87e77 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8831,3 +8831,19 @@ - Pass AMD container runtime options as native dotted leaf overrides so SRT validates them as a mapping before submission; surface native submission errors in workflow logs - 将 AMD 容器运行时选项作为原生点路径叶值传递,确保 SRT 在提交前将其校验为映射,并在工作流日志中显示原生提交错误 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - Strip the trailing newline from AMD native shell preambles and verify completed + allocations through the Slurm controller when accounting is unavailable; reap + failed-startup log followers before exit + - 去除 AMD 原生 shell 前置命令的尾部换行,并在 accounting 不可用时通过 Slurm 控制器核验已完成的资源分配;启动失败时等待日志跟随进程退出 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 4643f9c7a1..03c68c7447 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -261,10 +261,29 @@ verify_slurm_job_status() { local job_id="$1" # Disappearing from squeue means terminal, not successful. Accounting can # lag briefly; inspect only the allocation, never successful service steps. - local attempt accounting state exit_code + local attempt accounting state exit_code controller field controller_job_id + local -a controller_fields for attempt in {1..10}; do accounting=$(sacct -X -n -P -j "$job_id" --format=State,ExitCode 2>/dev/null) || accounting="" IFS='|' read -r state exit_code <<< "$accounting" + if [[ -z "$state" ]]; then + # Some pools do not expose slurmdbd. The controller retains recent + # terminal allocations; require its state and exit code, too. + controller=$(scontrol show job -o "$job_id" 2>/dev/null) || controller="" + controller_job_id="" + read -r -a controller_fields <<< "$controller" + for field in "${controller_fields[@]}"; do + case "$field" in + JobId=*) controller_job_id="${field#JobId=}" ;; + JobState=*) state="${field#JobState=}" ;; + ExitCode=*) exit_code="${field#ExitCode=}" ;; + esac + done + if [[ "$controller_job_id" != "$job_id" ]]; then + state="" + exit_code="" + fi + fi case "$state" in COMPLETED) if [[ "$exit_code" == "0:0" ]]; then diff --git a/runners/srt-slurm/mi300x-amd.yaml b/runners/srt-slurm/mi300x-amd.yaml index 0bf432ba3e..7890772181 100644 --- a/runners/srt-slurm/mi300x-amd.yaml +++ b/runners/srt-slurm/mi300x-amd.yaml @@ -12,7 +12,7 @@ default_sbatch_directives: use_gpus_per_node_directive: true use_segment_sbatch_directive: false use_exclusive_sbatch_directive: true -default_bash_preamble: | +default_bash_preamble: |- if [[ "$$MODEL_PREFIX" == dsr1 && "$$FRAMEWORK" == sglang ]]; then firmware=$$(rocm-smi --showfw | awk '/MEC/ {print $$NF; exit}') if [[ -z "$$firmware" || "$$firmware" -lt 177 ]]; then diff --git a/runners/srt-slurm/mi325x-amds.yaml b/runners/srt-slurm/mi325x-amds.yaml index c87092f5dd..4c8681a2d7 100644 --- a/runners/srt-slurm/mi325x-amds.yaml +++ b/runners/srt-slurm/mi325x-amds.yaml @@ -12,6 +12,6 @@ default_sbatch_directives: use_gpus_per_node_directive: true use_segment_sbatch_directive: false use_exclusive_sbatch_directive: true -default_bash_preamble: | +default_bash_preamble: |- export XDG_CACHE_HOME="/tmp/xdg-cache-$$SLURM_JOB_ID" export TRITON_CACHE_DIR="/tmp/triton-cache-$$SLURM_JOB_ID" diff --git a/utils/test_srt_docker.py b/utils/test_srt_docker.py index 01701f687c..ffaf4aec77 100644 --- a/utils/test_srt_docker.py +++ b/utils/test_srt_docker.py @@ -82,6 +82,25 @@ def test_docker_client_failure_and_readiness_clean_up_owned_server(tmp_path, cli binary.chmod(0o755) if ready: (workspace / "srt-docker-server.sh").write_text('echo $$ > "$INFMAX_CONTAINER_WORKSPACE/server.pid"\nexec /bin/sleep 120\n') + else: + # A log follower can take time to stop. Detach its output so EOF alone + # cannot hide a missing wait in the wrapper's failed-readiness path. + follower = binaries / "tail" + follower.write_text(f"#!{sys.executable}\n" + + "import os, pathlib, signal, time\n" + "workspace = pathlib.Path(os.environ['INFMAX_CONTAINER_WORKSPACE'])\n" + "null = os.open(os.devnull, os.O_WRONLY)\n" + "os.dup2(null, 1); os.dup2(null, 2); os.close(null)\n" + "def stop(*_):\n" + " time.sleep(0.1)\n" + " (workspace / 'follower-stopped').touch()\n" + " raise SystemExit(0)\n" + "signal.signal(signal.SIGTERM, stop)\n" + "(workspace / 'follower-ready').touch()\n" + "signal.pause()\n") + follower.chmod(0o755) + (binaries / "curl").write_text( + '#!/bin/bash\nwhile [[ ! -e "$INFMAX_CONTAINER_WORKSPACE/follower-ready" ]]; do /bin/sleep 0.01; done\nexit 1\n') env = {**os.environ, "PATH": f"{binaries}:{Path(sys.executable).parent}:{os.environ['PATH']}", "INFMAX_CONTAINER_WORKSPACE": str(workspace), "MODEL": "test/model", "PORT": "9019", "RUN_EVAL": "false", "EVAL_ONLY": "false", "MODEL_PATH": ""} @@ -89,6 +108,8 @@ def test_docker_client_failure_and_readiness_clean_up_owned_server(tmp_path, cli assert result.returncode == (client_exit if ready else 1), result.stdout + result.stderr assert (workspace / "client-ran").exists() is ready assert (workspace / "download").read_text() == "download\ntest/model\n" + if not ready: + assert (workspace / "follower-stopped").exists() if ready: pid = int((workspace / "server.pid").read_text()) for _ in range(100): diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index b8dba3893c..23f2436095 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -342,3 +342,31 @@ def test_runtime_container_options_remain_native_mapping(point): assert actual['srun_options'] == {'container-remap-root': '', 'container-writable': ''} with pytest.raises(ValueError, match='must map option names to string values'): runtime_arguments(f"{path}:base", {**env, 'SRT_SRUN_OPTIONS': '{"container-remap-root": true}'}) + + +@pytest.mark.parametrize("controller,expected", [ + ("JobId=42 JobState=COMPLETED ExitCode=0:0", 0), + ("JobId=42 JobState=FAILED ExitCode=1:0", 1), + ("JobId=42 JobState=COMPLETED ExitCode=0:9", 1), + ("JobId=43 JobState=COMPLETED ExitCode=0:0", 1), + ("", 1), +]) +def test_terminal_allocation_without_accounting(tmp_path, controller, expected): + binaries = tmp_path / "bin" + binaries.mkdir() + for name, body in { + "sacct": "exit 1", + "scontrol": 'printf "%s\\n" "$CONTROLLER_RECORD"', + "sleep": "exit 0", + }.items(): + binary = binaries / name + binary.write_text(f"#!/usr/bin/env bash\n{body}\n") + binary.chmod(0o755) + result = subprocess.run( + ["bash", "-c", 'source "$1"; verify_slurm_job_status 42', "bash", str(ROOT / "runners/slurm_utils.sh")], + env={**os.environ, "PATH": f"{binaries}:{os.environ['PATH']}", "CONTROLLER_RECORD": controller}, + capture_output=True, text=True, timeout=5, + ) + assert result.returncode == expected, result.stdout + result.stderr + if expected: + assert "ERROR:" in result.stderr From 8e279188ecd33558dfe1742329bb01091d6eb7d1 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:59:25 -0500 Subject: [PATCH 17/29] refactor: require native recipes for the Slurm fixed-sequence cutover MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 移除 Docker 迁移支持和 50 个已被原生配方替代的定长 Bash 实现。Slurm 定长作业必须提供 SRT 配方,保留 AgentX、多节点和显式采集器的独立执行路径。 --- benchmarks/benchmark_lib.sh | 2 - .../fixed_seq_len/dsr1_fp4_b200.sh | 102 ---------- .../fixed_seq_len/dsr1_fp4_b200_mtp.sh | 114 ----------- .../fixed_seq_len/dsr1_fp4_b200_trt.sh | 127 ------------ .../fixed_seq_len/dsr1_fp4_b200_trt_mtp.sh | 136 ------------- .../fixed_seq_len/dsr1_fp4_b300.sh | 81 -------- .../fixed_seq_len/dsr1_fp4_mi355x.sh | 74 ------- .../fixed_seq_len/dsr1_fp4_mi355x_atom.sh | 77 -------- .../fixed_seq_len/dsr1_fp4_mi355x_atom_mtp.sh | 80 -------- .../fixed_seq_len/dsr1_fp4_mi355x_mtp.sh | 84 -------- .../fixed_seq_len/dsr1_fp8_b200.sh | 101 ---------- .../fixed_seq_len/dsr1_fp8_b200_mtp.sh | 114 ----------- .../fixed_seq_len/dsr1_fp8_b200_trt.sh | 145 -------------- .../fixed_seq_len/dsr1_fp8_b200_trt_mtp.sh | 145 -------------- .../fixed_seq_len/dsr1_fp8_b300.sh | 111 ----------- .../fixed_seq_len/dsr1_fp8_b300_mtp.sh | 124 ------------ .../fixed_seq_len/dsr1_fp8_h200.sh | 76 -------- .../fixed_seq_len/dsr1_fp8_h200_mtp.sh | 95 --------- .../fixed_seq_len/dsr1_fp8_h200_trt.sh | 102 ---------- .../fixed_seq_len/dsr1_fp8_h200_trt_mtp.sh | 121 ------------ .../fixed_seq_len/dsr1_fp8_mi300x.sh | 76 -------- .../fixed_seq_len/dsr1_fp8_mi325x.sh | 71 ------- .../fixed_seq_len/dsr1_fp8_mi325x_mtp.sh | 77 -------- .../fixed_seq_len/dsr1_fp8_mi355x.sh | 71 ------- .../fixed_seq_len/dsr1_fp8_mi355x_atom.sh | 77 -------- .../fixed_seq_len/dsr1_fp8_mi355x_atom_mtp.sh | 80 -------- .../fixed_seq_len/dsr1_fp8_mi355x_mtp.sh | 86 --------- .../fixed_seq_len/qwen3.5_fp4_b200.sh | 78 -------- .../fixed_seq_len/qwen3.5_fp4_b200_mtp.sh | 98 ---------- .../fixed_seq_len/qwen3.5_fp4_b200_trt.sh | 142 -------------- .../fixed_seq_len/qwen3.5_fp4_b200_trt_mtp.sh | 159 --------------- .../fixed_seq_len/qwen3.5_fp4_b300.sh | 108 ----------- .../fixed_seq_len/qwen3.5_fp4_b300_mtp.sh | 114 ----------- .../fixed_seq_len/qwen3.5_fp4_mi355x.sh | 71 ------- .../fixed_seq_len/qwen3.5_fp4_mi355x_atom.sh | 76 -------- .../fixed_seq_len/qwen3.5_fp4_mi355x_mtp.sh | 76 -------- .../fixed_seq_len/qwen3.5_fp8_b200.sh | 79 -------- .../fixed_seq_len/qwen3.5_fp8_b200_mtp.sh | 86 --------- .../fixed_seq_len/qwen3.5_fp8_b300.sh | 88 --------- .../fixed_seq_len/qwen3.5_fp8_b300_mtp.sh | 92 --------- .../fixed_seq_len/qwen3.5_fp8_h100.sh | 123 ------------ .../fixed_seq_len/qwen3.5_fp8_h100_mtp.sh | 91 --------- .../fixed_seq_len/qwen3.5_fp8_h200.sh | 84 -------- .../fixed_seq_len/qwen3.5_fp8_h200_mtp.sh | 89 --------- .../fixed_seq_len/qwen3.5_fp8_mi300x.sh | 71 ------- .../fixed_seq_len/qwen3.5_fp8_mi325x.sh | 71 ------- .../fixed_seq_len/qwen3.5_fp8_mi325x_mtp.sh | 78 -------- .../fixed_seq_len/qwen3.5_fp8_mi355x.sh | 76 -------- .../fixed_seq_len/qwen3.5_fp8_mi355x_atom.sh | 76 -------- .../qwen3.5_fp8_mi355x_atom_mtp.sh | 79 -------- .../fixed_seq_len/qwen3.5_fp8_mi355x_mtp.sh | 82 -------- .../dsr1/atom/mi355x-fp4-mtp/8k1k.yaml | 1 - .../dsr1/atom/mi355x-fp4/8k1k.yaml | 1 - .../dsr1/atom/mi355x-fp8-mtp/8k1k.yaml | 1 - .../dsr1/atom/mi355x-fp8/8k1k.yaml | 1 - .../dsr1/sglang/b200-fp4-mtp/8k1k.yaml | 1 - .../dsr1/sglang/b200-fp4/8k1k.yaml | 1 - .../dsr1/sglang/b200-fp8-mtp/8k1k.yaml | 1 - .../dsr1/sglang/b200-fp8/8k1k.yaml | 1 - .../dsr1/sglang/b300-fp4/8k1k.yaml | 1 - .../dsr1/sglang/b300-fp8-mtp/8k1k.yaml | 1 - .../dsr1/sglang/b300-fp8/8k1k.yaml | 1 - .../dsr1/sglang/h200-fp8-mtp/8k1k.yaml | 1 - .../dsr1/sglang/h200-fp8/8k1k.yaml | 1 - .../dsr1/sglang/mi300x-fp8/8k1k.yaml | 1 - .../dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml | 1 - .../dsr1/sglang/mi325x-fp8/8k1k.yaml | 1 - .../dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml | 1 - .../dsr1/sglang/mi355x-fp4/8k1k.yaml | 1 - .../dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml | 1 - .../dsr1/sglang/mi355x-fp8/8k1k.yaml | 1 - .../dsr1/trtllm/b200-fp4-mtp/8k1k.yaml | 1 - .../dsr1/trtllm/b200-fp4/8k1k.yaml | 1 - .../dsr1/trtllm/b200-fp8-mtp/8k1k.yaml | 1 - .../dsr1/trtllm/b200-fp8/8k1k.yaml | 1 - .../dsr1/trtllm/h200-fp8-mtp/8k1k.yaml | 1 - .../dsr1/trtllm/h200-fp8/8k1k.yaml | 1 - .../qwen3.5/atom/mi355x-fp4/8k1k.yaml | 1 - .../qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml | 1 - .../qwen3.5/atom/mi355x-fp8/8k1k.yaml | 1 - .../qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml | 1 - .../qwen3.5/sglang/b200-fp4/8k1k.yaml | 1 - .../qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml | 1 - .../qwen3.5/sglang/b200-fp8/8k1k.yaml | 1 - .../qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml | 1 - .../qwen3.5/sglang/b300-fp4/8k1k.yaml | 1 - .../qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml | 1 - .../qwen3.5/sglang/b300-fp8/8k1k.yaml | 1 - .../qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml | 1 - .../qwen3.5/sglang/h100-fp8/8k1k.yaml | 1 - .../qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml | 1 - .../qwen3.5/sglang/h200-fp8/8k1k.yaml | 1 - .../qwen3.5/sglang/mi300x-fp8/8k1k.yaml | 1 - .../qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml | 1 - .../qwen3.5/sglang/mi325x-fp8/8k1k.yaml | 1 - .../qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml | 1 - .../qwen3.5/sglang/mi355x-fp4/8k1k.yaml | 1 - .../qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml | 1 - .../qwen3.5/sglang/mi355x-fp8/8k1k.yaml | 1 - .../sglang/rtx6000pro-fp4-mtp/8k1k.yaml | 88 --------- .../qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml | 82 -------- .../qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml | 1 - .../qwen3.5/trtllm/b200-fp4/8k1k.yaml | 1 - benchmarks/single_node/srt_docker.sh | 36 ---- configs/nvidia-master.yaml | 20 +- infx/srt_slurm/docker.py | 117 ----------- perf-changelog.yaml | 73 +++++++ runners/launch_b200-cw.sh | 9 +- runners/launch_b200-nb.sh | 9 +- runners/launch_b200-nscale-slurm.sh | 16 +- runners/launch_b300-dsxe.sh | 10 +- runners/launch_h100-cw.sh | 9 +- runners/launch_h100-dgxc-slurm.sh | 7 +- runners/launch_h200-cw.sh | 9 +- runners/launch_h200-dgxc-slurm.sh | 7 +- runners/launch_mi300x-amd.sh | 9 +- runners/launch_mi325x-amds.sh | 9 +- runners/launch_mi355x-amds.sh | 7 +- runners/launch_rtx6000pro-lat.sh | 28 +-- utils/test_srt_docker.py | 181 ------------------ utils/test_srt_single_node.py | 80 +++++++- 121 files changed, 220 insertions(+), 5372 deletions(-) delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp4_b300.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom_mtp.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi300x.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom_mtp.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_mtp.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt_mtp.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_atom.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_mtp.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300_mtp.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200_mtp.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi300x.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x_mtp.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom.sh delete mode 100644 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom_mtp.sh delete mode 100755 benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_mtp.sh delete mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml delete mode 100644 benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml delete mode 100644 benchmarks/single_node/srt_docker.sh delete mode 100644 infx/srt_slurm/docker.py delete mode 100644 utils/test_srt_docker.py diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index ec5d5e2670..214d7ca7cb 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -718,13 +718,11 @@ wait_for_ready() { if ! kill -0 "$process_pid" 2>/dev/null; then echo "Process died before $endpoint became ready." >&2 kill "$tail_pid" 2>/dev/null || true - wait "$tail_pid" 2>/dev/null || true exit 1 fi if [[ "$deadline" -gt 0 && "$SECONDS" -ge "$deadline" ]]; then echo "Timed out waiting for $endpoint." >&2 kill "$tail_pid" 2>/dev/null || true - wait "$tail_pid" 2>/dev/null || true exit 1 fi sleep "$sleep_interval" diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200.sh deleted file mode 100644 index 429d6872e4..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200.sh +++ /dev/null @@ -1,102 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ "$DP_ATTENTION" != "true" && "$DP_ATTENTION" != "false" ]]; then - echo "DP_ATTENTION must be true or false; got '$DP_ATTENTION'" >&2 - exit 1 -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -CHUNKED_PREFILL_SIZE=16384 -SGLANG_PARALLEL_ARGS=( - --tensor-parallel-size="$TP" - --data-parallel-size=1 -) -SGLANG_DPA_ARGS=() - -if [[ "$DP_ATTENTION" == "true" ]]; then - SCHEDULER_RECV_INTERVAL=1 - CHUNKED_PREFILL_SIZE=32768 - SGLANG_PARALLEL_ARGS=( - --tensor-parallel-size="$TP" - --data-parallel-size="$TP" - --enable-dp-attention - --enable-dp-attention-local-control-broadcast - --enable-dp-lm-head - ) - SGLANG_DPA_ARGS=( - --schedule-conservativeness 3.33 - --enable-prefill-delayer - ) -fi - -echo "TP: $TP, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION, CONC: $CONC, ISL: $ISL, OSL: $OSL" -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CHUNKED_PREFILL_SIZE: $CHUNKED_PREFILL_SIZE" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -SGLANG_RADIX_FORCE_MISS=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL --host 0.0.0.0 --port $PORT --trust-remote-code \ -"${SGLANG_PARALLEL_ARGS[@]}" \ ---cuda-graph-max-bs 256 --max-running-requests 256 --mem-fraction-static 0.85 --kv-cache-dtype fp8_e4m3 \ ---chunked-prefill-size "$CHUNKED_PREFILL_SIZE" \ ---ep-size $EP_SIZE --quantization modelopt_fp4 --enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---enable-symm-mem --disable-piecewise-cuda-graph --attention-backend trtllm_mla --moe-runner-backend flashinfer_trtllm --stream-interval 10 "${SGLANG_DPA_ARGS[@]}" $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $((CONC * 10)) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_mtp.sh deleted file mode 100755 index 2a68ebbce2..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_mtp.sh +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ "$DP_ATTENTION" != "true" && "$DP_ATTENTION" != "false" ]]; then - echo "DP_ATTENTION must be true or false; got '$DP_ATTENTION'" >&2 - exit 1 -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -CHUNKED_PREFILL_SIZE=16384 -SGLANG_PARALLEL_ARGS=( - --tensor-parallel-size="$TP" - --data-parallel-size=1 -) -SGLANG_DPA_ARGS=() - -if [[ "$DP_ATTENTION" == "true" ]]; then - SCHEDULER_RECV_INTERVAL=1 - CHUNKED_PREFILL_SIZE=32768 - SGLANG_PARALLEL_ARGS=( - --tensor-parallel-size="$TP" - --data-parallel-size="$TP" - --enable-dp-attention - --enable-dp-attention-local-control-broadcast - --enable-dp-lm-head - ) - SGLANG_DPA_ARGS=( - --schedule-conservativeness 3.33 - --enable-prefill-delayer - ) -fi - -echo "TP: $TP, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION, CONC: $CONC, ISL: $ISL, OSL: $OSL" -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CHUNKED_PREFILL_SIZE: $CHUNKED_PREFILL_SIZE" - -SPECULATIVE_NUM_STEPS=2 -SPECULATIVE_DRAFT_TOKENS=3 -SPECULATIVE_EAGLE_TOPK=1 - -export SGLANG_ENABLE_SPEC_V2=1 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -SGLANG_RADIX_FORCE_MISS=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL --host 0.0.0.0 --port $PORT --trust-remote-code \ -"${SGLANG_PARALLEL_ARGS[@]}" \ ---cuda-graph-max-bs 256 --max-running-requests 256 --mem-fraction-static 0.85 --kv-cache-dtype fp8_e4m3 \ ---chunked-prefill-size "$CHUNKED_PREFILL_SIZE" \ ---ep-size $EP_SIZE --quantization modelopt_fp4 --enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---disable-piecewise-cuda-graph --attention-backend trtllm_mla --moe-runner-backend flashinfer_trtllm --stream-interval 10 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps $SPECULATIVE_NUM_STEPS \ ---speculative-num-draft-tokens $SPECULATIVE_DRAFT_TOKENS \ ---speculative-eagle-topk $SPECULATIVE_EAGLE_TOPK \ -"${SGLANG_DPA_ARGS[@]}" $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $((CONC * 10)) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt.sh deleted file mode 100644 index 346b21b56d..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt.sh +++ /dev/null @@ -1,127 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -CUDA_GRAPH_MAX_BATCH_SIZE=$CONC -MOE_BACKEND="TRTLLM" -PIECEWISE_CUDA_GRAPHS="false" - -if [[ "$ISL" == "1024" && "$OSL" == "1024" ]]; then - if [[ "$TP" == "8" && "$EP_SIZE" == "8" ]]; then - PIECEWISE_CUDA_GRAPHS="true" - fi -fi - -if [[ "$DP_ATTENTION" == "true" ]]; then - MOE_BACKEND="CUTLASS" - CUDA_GRAPH_MAX_BATCH_SIZE=$(( CONC < 4 ? CONC : CONC / 4 )) -fi - -echo "MOE_BACKEND set to '$MOE_BACKEND'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp4.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: $CUDA_GRAPH_MAX_BATCH_SIZE -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: 0.8 - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -start_gpu_monitor - -set -x - -MAX_NUM_TOKENS=$(( ($CONC+$ISL+64+63)/64*64 )) -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) -MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -if [[ "$PIECEWISE_CUDA_GRAPHS" == "true" ]]; then - capture_tokens=(1 2 4 8 16 32 64 128) - capture_tokens+=( $(seq 256 256 $MAX_NUM_TOKENS)) - CAPTURE_TOKENS_LIST=$(printf "%s, " "${capture_tokens[@]}") - - cat << EOF >> $EXTRA_CONFIG_FILE -torch_compile_config: - capture_num_tokens: [${CAPTURE_TOKENS_LIST%, }] - enable_piecewise_cuda_graph: true -EOF -fi - -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt_mtp.sh deleted file mode 100644 index 5a34591884..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b200_trt_mtp.sh +++ /dev/null @@ -1,136 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -MOE_BACKEND="TRTLLM" -PIECEWISE_CUDA_GRAPHS="false" -MAX_BATCH_SIZE=$CONC -MTP=3 - -if [[ "$DP_ATTENTION" == "true" ]]; then - MAX_BATCH_SIZE=$(( CONC < 4 ? CONC : CONC / 4 )) - MOE_BACKEND="CUTLASS" - MTP=1 -fi - -echo "MOE_BACKEND='$MOE_BACKEND', MTP='$MTP'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp4-mtp.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: $MAX_BATCH_SIZE -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: 0.8 - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -speculative_config: - decoding_type: MTP - num_nextn_predict_layers: ${MTP} -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -MAX_NUM_TOKENS=$(( ((MTP+1)*MAX_BATCH_SIZE+ISL+64+63)/64*64 )) - -if [[ "$ISL" == "1024" && "$OSL" == "1024" ]]; then - if [[ $CONC == 32 || $CONC == 64 ]]; then - PIECEWISE_CUDA_GRAPHS="true" - elif [[ $CONC == 128 && $DP_ATTENTION == "false" ]]; then - PIECEWISE_CUDA_GRAPHS="true" - fi -fi - -if [[ "$PIECEWISE_CUDA_GRAPHS" == "true" ]]; then - capture_tokens=(1 2 4 8 16 32 64 128) - capture_tokens+=( $(seq 256 256 $MAX_NUM_TOKENS)) - if [ $((MAX_NUM_TOKENS%256)) -ne 0 ]; then - capture_tokens+=($MAX_NUM_TOKENS) - fi - CAPTURE_TOKENS_LIST=$(printf "%s, " "${capture_tokens[@]}") - - cat << EOF >> $EXTRA_CONFIG_FILE -torch_compile_config: - capture_num_tokens: [${CAPTURE_TOKENS_LIST%, }] - enable_piecewise_cuda_graph: true -EOF -fi # end of set of configs using piecewise_cuda_graphs - -start_gpu_monitor - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -set -x -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size=$MAX_BATCH_SIZE \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b300.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b300.sh deleted file mode 100644 index 6e5b5a6f0b..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_b300.sh +++ /dev/null @@ -1,81 +0,0 @@ -#!/usr/bin/env bash - -# https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 has no B300-specific recipe; this reuses the B200 SGLang tuning. - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT --trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 \ ---cuda-graph-max-bs 256 --max-running-requests 256 --mem-fraction-static 0.85 --kv-cache-dtype fp8_e4m3 \ ---chunked-prefill-size 16384 \ ---ep-size $EP_SIZE --quantization modelopt_fp4 --enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---enable-symm-mem --disable-radix-cache --attention-backend trtllm_mla --moe-runner-backend flashinfer_trtllm --stream-interval 10 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $((CONC * 10)) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x.sh deleted file mode 100644 index 0e77d3069f..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x.sh +++ /dev/null @@ -1,74 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER=1 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT4 - -PREFILL_SIZE=196608 -if [[ "$ISL" == "8192" && "$OSL" == "1024" ]]; then - if [[ "$CONC" -gt "32" ]]; then - PREFILL_SIZE=32768 - fi -fi - -SERVER_LOG=/workspace/server.log - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -python3 -m sglang.launch_server --model-path=$MODEL --trust-remote-code \ ---host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP \ ---chunked-prefill-size=$PREFILL_SIZE \ ---mem-fraction-static=0.8 \ ---disable-radix-cache \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=$PREFILL_SIZE \ ---cuda-graph-max-bs=128 \ ---attention-backend aiter \ ---kv-cache-dtype fp8_e4m3 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom.sh deleted file mode 100644 index 50427bd858..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom.sh +++ /dev/null @@ -1,77 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CALCULATED_MAX_MODEL_LEN=" --max-model-len $EVAL_MAX_MODEL_LEN " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor - -set -x - -BLOCK_SIZE=16 -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --block-size $BLOCK_SIZE > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom_mtp.sh deleted file mode 100644 index 83a95def55..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_atom_mtp.sh +++ /dev/null @@ -1,80 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CALCULATED_MAX_MODEL_LEN=" --max-model-len $EVAL_MAX_MODEL_LEN " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor - -set -x - -export AMDGCN_USE_BUFFER_OPS=1 - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --method mtp \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_mtp.sh deleted file mode 100755 index a72351c59e..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp4_mi355x_mtp.sh +++ /dev/null @@ -1,84 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 -export SGLANG_ENABLE_SPEC_V2=1 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT4 - -PREFILL_SIZE=196608 -if [[ "$ISL" == "8192" && "$OSL" == "1024" ]]; then - if [[ "$CONC" -gt "32" ]]; then - PREFILL_SIZE=32768 - fi -fi - -SERVER_LOG=/workspace/server.log - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -python3 -m sglang.launch_server --model-path=$MODEL --trust-remote-code \ ---host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP \ ---ep-size $EP_SIZE \ ---chunked-prefill-size=$PREFILL_SIZE \ ---mem-fraction-static=0.8 \ ---disable-radix-cache \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=$PREFILL_SIZE \ ---cuda-graph-max-bs=128 \ ---attention-backend aiter \ ---kv-cache-dtype fp8_e4m3 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ -$EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200.sh deleted file mode 100644 index d9a2dc7f61..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200.sh +++ /dev/null @@ -1,101 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGL_ENABLE_JIT_DEEPGEMM=false -export SGLANG_ENABLE_FLASHINFER_GEMM=true -SERVER_LOG=/workspace/server.log - -if [[ $TP -eq 8 ]]; then - if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 - else - SCHEDULER_RECV_INTERVAL=10 - fi - - # Capped so KV memory is not reserved for requests that never run. - MAX_RUNNING_REQUESTS=128 - CUDA_GRAPH_MAX_BATCH_SIZE=128 - - MEM_FRAC_STATIC=0.82 - CHUNKED_PREFILL_SIZE=32768 - MAX_PREFILL_TOKENS=32768 -elif [[ $TP -eq 4 ]]; then - if [[ $ISL -ne 8192 ]] || [[ $OSL -ne 1024 ]]; then - echo "TP=4 not yet supported for ISL=$ISL OSL=$OSL!" - exit 1 - fi - - # Capped so KV memory is not reserved for requests that never run. - MAX_RUNNING_REQUESTS=32 - CUDA_GRAPH_MAX_BATCH_SIZE=32 - - MEM_FRAC_STATIC=0.95 - CHUNKED_PREFILL_SIZE=8192 - MAX_PREFILL_TOKENS=8192 - - SCHEDULER_RECV_INTERVAL=10 -else - echo "Unrecognized TP size $TP!" - exit 1 -fi -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP --data-parallel-size=1 \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE --max-running-requests $MAX_RUNNING_REQUESTS \ ---mem-fraction-static $MEM_FRAC_STATIC --kv-cache-dtype fp8_e4m3 --chunked-prefill-size $CHUNKED_PREFILL_SIZE --max-prefill-tokens $MAX_PREFILL_TOKENS \ ---enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL --disable-radix-cache \ ---attention-backend trtllm_mla --stream-interval 30 --ep-size $EP_SIZE --moe-runner-backend flashinfer_trtllm --quantization fp8 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_mtp.sh deleted file mode 100755 index 39a3e61659..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_mtp.sh +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_ENABLE_JIT_DEEPGEMM=false - -SERVER_LOG=/workspace/server.log - -if [[ $TP -ne 8 ]]; then - echo "MTP only supports TP=8, got TP=$TP!" - exit 1 -fi - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -# Capped so KV memory is not reserved for requests that never run. -MAX_RUNNING_REQUESTS=512 -CUDA_GRAPH_MAX_BATCH_SIZE=512 - -MEM_FRAC_STATIC=0.82 -CHUNKED_PREFILL_SIZE=16384 -MAX_PREFILL_TOKENS=16384 - -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -SPECULATIVE_NUM_STEPS=2 -SPECULATIVE_DRAFT_TOKENS=3 -SPECULATIVE_EAGLE_TOPK=1 - -SGLANG_ENABLE_SPEC_V2=1 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server \ - --model-path=$MODEL \ - --host=0.0.0.0 \ - --port=$PORT \ - --tensor-parallel-size=$TP \ - --data-parallel-size=1 \ - --cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE \ - --max-running-requests $MAX_RUNNING_REQUESTS \ - --mem-fraction-static $MEM_FRAC_STATIC \ - --kv-cache-dtype fp8_e4m3 \ - --chunked-prefill-size $CHUNKED_PREFILL_SIZE \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --enable-flashinfer-allreduce-fusion \ - --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ - --disable-radix-cache \ - --fp8-gemm-backend=flashinfer_trtllm \ - --attention-backend trtllm_mla \ - --stream-interval 30 \ - --ep-size $EP_SIZE \ - --moe-runner-backend flashinfer_trtllm \ - --quantization fp8 \ - --speculative-algorithm EAGLE \ - --speculative-num-steps $SPECULATIVE_NUM_STEPS \ - --speculative-num-draft-tokens $SPECULATIVE_DRAFT_TOKENS \ - --speculative-eagle-topk $SPECULATIVE_EAGLE_TOPK \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt.sh deleted file mode 100644 index d4b66e8cc3..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt.sh +++ /dev/null @@ -1,145 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Guards against OOM. -export TLLM_OVERRIDE_LAYER_NUM=61 - -CUDA_GRAPH_MAX_BATCH_SIZE=$CONC -MOE_BACKEND="TRTLLM" -PIECEWISE_CUDA_GRAPHS="false" -DELAY_BATCHING="false" -KV_CACHE_FREE_MEM_FRACTION=0.8 - -if [[ "$ISL" == "1024" && "$OSL" == "1024" ]]; then - if [[ $CONC -ge 64 ]]; then - PIECEWISE_CUDA_GRAPHS="true" - DELAY_BATCHING="true" - fi -elif [[ "$ISL" == "8192" && "$OSL" == "1024" ]]; then - if [[ $CONC -ge 64 ]]; then - PIECEWISE_CUDA_GRAPHS="true" - fi - if [[ "$TP" == "4" ]]; then - KV_CACHE_FREE_MEM_FRACTION=0.75 - fi -fi - -echo "MOE_BACKEND set to '$MOE_BACKEND'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp8.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: $CUDA_GRAPH_MAX_BATCH_SIZE -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: $KV_CACHE_FREE_MEM_FRACTION - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -if [[ "$DELAY_BATCHING" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -batch_wait_timeout_iters: 40 -batch_wait_max_tokens_ratio: 0.8 -EOF -fi - -start_gpu_monitor - -set -x - -MAX_NUM_TOKENS=$(( ($CONC+$ISL+64+63)/64*64 )) -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) -MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -if [[ "$PIECEWISE_CUDA_GRAPHS" == "true" ]]; then - capture_tokens=(1 2 4 8 16 32 64 128) - capture_tokens+=( $(seq 256 256 $MAX_NUM_TOKENS)) - if [ $((MAX_NUM_TOKENS%256)) -ne 0 ]; then - capture_tokens+=($MAX_NUM_TOKENS) - fi - CAPTURE_TOKENS_LIST=$(printf "%s, " "${capture_tokens[@]}") - - cat << EOF >> $EXTRA_CONFIG_FILE -torch_compile_config: - capture_num_tokens: [${CAPTURE_TOKENS_LIST%, }] - enable_piecewise_cuda_graph: true -EOF -fi - -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt_mtp.sh deleted file mode 100644 index b945a0a6a5..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b200_trt_mtp.sh +++ /dev/null @@ -1,145 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -MOE_BACKEND="TRTLLM" -PIECEWISE_CUDA_GRAPHS="true" -MAX_BATCH_SIZE=$CONC -KV_CACHE_FREE_MEM_FRACTION=0.8 -MTP=3 - -if [[ "$DP_ATTENTION" == "true" ]]; then - MOE_BACKEND="DEEPGEMM" - PIECEWISE_CUDA_GRAPHS="false" - MAX_BATCH_SIZE=$(( CONC < 8 ? CONC : CONC / 8 )) - KV_CACHE_FREE_MEM_FRACTION=0.7 - # Configurable MoE backend has better comms under attention DP. - export ENABLE_CONFIGURABLE_MOE=1 - MTP=1 -fi - -# Low-CONC cases do not benefit from piecewise CUDA graphs. -if [[ "$ISL" == "1024" && "$OSL" == "1024" ]]; then - if [[ $CONC -le 4 ]]; then - PIECEWISE_CUDA_GRAPHS="false" - fi -elif [[ "$ISL" == "8192" && "$OSL" == "1024" ]]; then - if [[ $CONC -le 16 ]]; then - PIECEWISE_CUDA_GRAPHS="false" - fi -fi - - -echo "MOE_BACKEND='$MOE_BACKEND', MTP='$MTP'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp8-mtp.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: $MAX_BATCH_SIZE -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: $KV_CACHE_FREE_MEM_FRACTION - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -speculative_config: - decoding_type: MTP - num_nextn_predict_layers: ${MTP} -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" -fi - -MAX_NUM_TOKENS=$(( ((MTP+1)*MAX_BATCH_SIZE+ISL+64+63)/64*64 )) -if [ "${EVAL_ONLY}" = "true" ]; then - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -if [[ "$PIECEWISE_CUDA_GRAPHS" == "true" ]]; then - capture_tokens=(1 2 4 8 16 32 64 128) - capture_tokens+=( $(seq 256 256 $MAX_NUM_TOKENS)) - if [ $((MAX_NUM_TOKENS%256)) -ne 0 ]; then - capture_tokens+=($MAX_NUM_TOKENS) - fi - CAPTURE_TOKENS_LIST=$(printf "%s, " "${capture_tokens[@]}") - - cat << EOF >> $EXTRA_CONFIG_FILE -torch_compile_config: - capture_num_tokens: [${CAPTURE_TOKENS_LIST%, }] - enable_piecewise_cuda_graph: true -EOF -fi -start_gpu_monitor - -set -x -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size=$MAX_BATCH_SIZE \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300.sh deleted file mode 100644 index b29aa07310..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300.sh +++ /dev/null @@ -1,111 +0,0 @@ -#!/usr/bin/env bash - -# https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 has no B300-specific recipe; this reuses the B200 SGLang tuning. - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - - -export SGL_ENABLE_JIT_DEEPGEMM=false -export SGLANG_ENABLE_FLASHINFER_GEMM=true -SERVER_LOG=/workspace/server.log - -if [[ $TP -eq 8 ]]; then - if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 - else - SCHEDULER_RECV_INTERVAL=10 - fi - - # Capped so KV memory is not reserved for requests that never run. - MAX_RUNNING_REQUESTS=128 - CUDA_GRAPH_MAX_BATCH_SIZE=128 - - MEM_FRAC_STATIC=0.82 - CHUNKED_PREFILL_SIZE=32768 - MAX_PREFILL_TOKENS=32768 -elif [[ $TP -eq 4 ]]; then - if [[ $ISL -ne 8192 ]] || [[ $OSL -ne 1024 ]]; then - echo "TP=4 not yet supported for ISL=$ISL OSL=$OSL!" - exit 1 - fi - - # Capped so KV memory is not reserved for requests that never run. - MAX_RUNNING_REQUESTS=32 - CUDA_GRAPH_MAX_BATCH_SIZE=32 - - MEM_FRAC_STATIC=0.95 - CHUNKED_PREFILL_SIZE=8192 - MAX_PREFILL_TOKENS=8192 - - SCHEDULER_RECV_INTERVAL=10 -else - echo "Unrecognized TP size $TP!" - exit 1 -fi -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---tensor-parallel-size $TP --data-parallel-size 1 \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE --max-running-requests $MAX_RUNNING_REQUESTS \ ---mem-fraction-static $MEM_FRAC_STATIC --kv-cache-dtype fp8_e4m3 --chunked-prefill-size $CHUNKED_PREFILL_SIZE --max-prefill-tokens $MAX_PREFILL_TOKENS \ ---enable-flashinfer-allreduce-fusion --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL --disable-radix-cache \ ---attention-backend trtllm_mla --stream-interval 30 --ep-size $EP_SIZE --moe-runner-backend flashinfer_trtllm --quantization fp8 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300_mtp.sh deleted file mode 100755 index 4915d17224..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_b300_mtp.sh +++ /dev/null @@ -1,124 +0,0 @@ -#!/usr/bin/env bash - -# https://cookbook.sglang.io/autoregressive/DeepSeek/DeepSeek-R1 has no B300-specific recipe; this reuses the B200 SGLang tuning. - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - - -export SGLANG_ENABLE_JIT_DEEPGEMM=false - -SERVER_LOG=/workspace/server.log - -if [[ $TP -ne 8 ]]; then - echo "MTP only supports TP=8, got TP=$TP!" - exit 1 -fi - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -# Capped so KV memory is not reserved for requests that never run. -MAX_RUNNING_REQUESTS=512 -CUDA_GRAPH_MAX_BATCH_SIZE=512 - -MEM_FRAC_STATIC=0.82 -CHUNKED_PREFILL_SIZE=16384 -MAX_PREFILL_TOKENS=16384 - -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -SPECULATIVE_NUM_STEPS=2 -SPECULATIVE_DRAFT_TOKENS=3 -SPECULATIVE_EAGLE_TOPK=1 - -SGLANG_ENABLE_SPEC_V2=1 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server \ - --model-path $MODEL_PATH --served-model-name $MODEL \ - --host 0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --data-parallel-size 1 \ - --cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE \ - --max-running-requests $MAX_RUNNING_REQUESTS \ - --mem-fraction-static $MEM_FRAC_STATIC \ - --kv-cache-dtype fp8_e4m3 \ - --chunked-prefill-size $CHUNKED_PREFILL_SIZE \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --enable-flashinfer-allreduce-fusion \ - --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ - --disable-radix-cache \ - --fp8-gemm-backend flashinfer_trtllm \ - --attention-backend trtllm_mla \ - --stream-interval 30 \ - --ep-size $EP_SIZE \ - --moe-runner-backend flashinfer_trtllm \ - --quantization fp8 \ - --speculative-algorithm EAGLE \ - --speculative-num-steps $SPECULATIVE_NUM_STEPS \ - --speculative-num-draft-tokens $SPECULATIVE_DRAFT_TOKENS \ - --speculative-eagle-topk $SPECULATIVE_EAGLE_TOPK \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh deleted file mode 100644 index 0bec9b79e6..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -pip3 install --user --break-system-packages sentencepiece - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi -SERVER_LOG=/workspace/server.log - -start_gpu_monitor - -export TORCH_CUDA_ARCH_LIST="9.0" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi - -set -x -if [[ $ISL -eq 1024 && $OSL -eq 1024 ]]; then - PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL \ - --host 0.0.0.0 --port $PORT --trust-remote-code \ - --tensor-parallel-size=$TP --data-parallel-size=1 \ - --disable-radix-cache --max-running-requests 512 --cuda-graph-max-bs 512 \ - --chunked-prefill-size 32768 --max-prefill-tokens 32768 --mem-fraction-static 0.82 \ - --attention-backend flashinfer --stream-interval 10 \ - --decode-log-interval 1 \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & -else - PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL \ - --host 0.0.0.0 --port $PORT --trust-remote-code \ - --tensor-parallel-size=$TP --data-parallel-size=1 \ - --disable-radix-cache --max-running-requests 256 --cuda-graph-max-bs 256 \ - --chunked-prefill-size 32768 --max-prefill-tokens 32768 --mem-fraction-static 0.82 \ - --attention-backend flashinfer --stream-interval 10 \ - --decode-log-interval 1 \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & -fi - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_mtp.sh deleted file mode 100755 index f073983d5e..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_mtp.sh +++ /dev/null @@ -1,95 +0,0 @@ -#!/usr/bin/env bash - -# No trtllm_mla attention path on H200 in this image, so attention stays on flashinfer. - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -pip3 install --user --break-system-packages sentencepiece - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -if [[ $TP -ne 8 ]]; then - echo "MTP only supports TP=8, got TP=$TP!" - exit 1 -fi - -SERVER_LOG=/workspace/server.log - -SPECULATIVE_NUM_STEPS=2 -SPECULATIVE_DRAFT_TOKENS=3 -SPECULATIVE_EAGLE_TOPK=1 - -export SGLANG_ENABLE_SPEC_V2=1 -export TORCH_CUDA_ARCH_LIST="9.0" - -start_gpu_monitor - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi - -set -x -if [[ $ISL -eq 1024 && $OSL -eq 1024 ]]; then - MAX_RUNNING_REQUESTS=512 - CUDA_GRAPH_MAX_BS=512 -else - MAX_RUNNING_REQUESTS=256 - CUDA_GRAPH_MAX_BS=256 -fi - -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL \ ---host 0.0.0.0 --port $PORT --trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 \ ---ep-size $EP_SIZE \ ---disable-radix-cache \ ---max-running-requests $MAX_RUNNING_REQUESTS \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BS \ ---chunked-prefill-size 32768 --max-prefill-tokens 32768 --mem-fraction-static 0.82 \ ---attention-backend flashinfer --stream-interval 10 \ ---decode-log-interval 1 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps $SPECULATIVE_NUM_STEPS \ ---speculative-num-draft-tokens $SPECULATIVE_DRAFT_TOKENS \ ---speculative-eagle-topk $SPECULATIVE_EAGLE_TOPK \ -$EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt.sh deleted file mode 100644 index 4e1095bb79..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt.sh +++ /dev/null @@ -1,102 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -MOE_BACKEND="CUTLASS" - -echo "MOE_BACKEND set to '$MOE_BACKEND'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp8.yml" - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: 128 -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: 0.75 - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -start_gpu_monitor - -set -x - -MAX_NUM_TOKENS=$(( (CONC + ISL + 64 + 63) / 64 * 64 )) -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) -MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -PYTHONNOUSERSITE=1 mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt_mtp.sh deleted file mode 100644 index 339468da0f..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_h200_trt_mtp.sh +++ /dev/null @@ -1,121 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -MOE_BACKEND="CUTLASS" - -if [[ "$DP_ATTENTION" == "true" ]]; then - MTP=1 -else - MTP=3 -fi - -echo "MOE_BACKEND='$MOE_BACKEND', MTP='$MTP'" - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="dsr1-fp8-mtp.yml" - -if [[ "$ISL" == "8192" && "$DP_ATTENTION" == "true" ]]; then - export PYTORCH_CUDA_ALLOC_CONF="max_split_size_mb:8192" -fi - -cat > $EXTRA_CONFIG_FILE << EOF -cuda_graph_config: - enable_padding: true - max_batch_size: 128 -enable_attention_dp: $DP_ATTENTION -print_iter_log: true -kv_cache_config: - dtype: fp8 - free_gpu_memory_fraction: 0.75 - enable_block_reuse: false -stream_interval: 10 -moe_config: - backend: $MOE_BACKEND -speculative_config: - decoding_type: MTP - num_nextn_predict_layers: ${MTP} -EOF - -if [[ "$DP_ATTENTION" == "true" ]]; then - cat << EOF >> $EXTRA_CONFIG_FILE -attention_dp_config: - batching_wait_iters: 0 - enable_balance: true - timeout_iters: 60 -EOF -fi - -if [[ "$DP_ATTENTION" == "true" ]]; then - MAX_BATCH_SIZE=$((CONC/TP)) - if [[ $MAX_BATCH_SIZE -lt 1 ]]; then - MAX_BATCH_SIZE=1 - fi -else - MAX_BATCH_SIZE=$CONC -fi - -MAX_NUM_TOKENS=$(( ((MTP+1)*MAX_BATCH_SIZE+ISL+64+63)/64*64 )) - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve $MODEL --port=$PORT \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size=$MAX_BATCH_SIZE \ - --max_seq_len=$MAX_MODEL_LEN \ - --max_num_tokens=$MAX_NUM_TOKENS \ - --tp_size=$TP --ep_size=$EP_SIZE \ - --extra_llm_api_options=$EXTRA_CONFIG_FILE \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi300x.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi300x.sh deleted file mode 100644 index 4b2950ade1..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi300x.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Reference: https://rocm.docs.amd.com/en/docs-7.0-rc1/preview/benchmark-docker/inference-sglang-deepseek-r1-fp8.html#run-the-inference-benchmark - -# On MEC firmware older than 177 RCCL cannot reclaim scratch memory and crashes; disable it. -# See https://rocm.docs.amd.com/en/docs-6.4.3/about/release-notes.html#amdgpu-driver-updates -version=`rocm-smi --showfw | grep MEC | head -n 1 | awk '{print $NF}'` -if [[ "$version" == "" || $version -lt 177 ]]; then - export HSA_NO_SCRATCH_RECLAIM=1 -fi - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 - -SERVER_LOG=/workspace/server.log - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -set -x -python3 -m sglang.launch_server \ ---model-path=$MODEL --host=0.0.0.0 --port=$PORT --trust-remote-code \ ---tensor-parallel-size=$TP \ ---mem-fraction-static=0.8 \ ---cuda-graph-max-bs=128 \ ---chunked-prefill-size=131072 \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=131072 \ ---kv-cache-dtype fp8_e4m3 \ ---attention-backend aiter \ ---disable-radix-cache $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh deleted file mode 100644 index 0f6a801136..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -SERVER_LOG=/workspace/server.log -PORT=8888 -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Reference: https://rocm.docs.amd.com/en/docs-7.0-rc1/preview/benchmark-docker/inference-sglang-deepseek-r1-fp8.html#run-the-inference-benchmark - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 - -start_gpu_monitor - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi - -set -x -python3 -m sglang.launch_server \ ---model-path=$MODEL --host=0.0.0.0 --port=$PORT --trust-remote-code \ ---tensor-parallel-size=$TP \ ---mem-fraction-static=0.8 \ ---cuda-graph-max-bs=128 \ ---chunked-prefill-size=131072 \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=131072 \ ---kv-cache-dtype fp8_e4m3 \ ---attention-backend aiter \ ---disable-radix-cache \ -$EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x_mtp.sh deleted file mode 100755 index df72ca869e..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x_mtp.sh +++ /dev/null @@ -1,77 +0,0 @@ -#!/usr/bin/bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -SERVER_LOG=/workspace/server.log -PORT=8888 -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 -export SGLANG_ENABLE_SPEC_V2=1 - -start_gpu_monitor - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi - -set -x -python3 -m sglang.launch_server \ ---model-path=$MODEL --host=0.0.0.0 --port=$PORT --trust-remote-code \ ---tensor-parallel-size=$TP \ ---ep-size $EP_SIZE \ ---mem-fraction-static=0.8 \ ---cuda-graph-max-bs=128 \ ---chunked-prefill-size=131072 \ ---num-continuous-decode-steps=4 \ ---max-prefill-tokens=131072 \ ---kv-cache-dtype fp8_e4m3 \ ---attention-backend aiter \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---disable-radix-cache \ -$EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts $(( $CONC * 10 )) \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x.sh deleted file mode 100644 index 979ab0d0dd..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Reference: https://rocm.docs.amd.com/en/docs-7.0-docker/benchmark-docker/inference-sglang-deepseek-r1-fp8.html - -export SGLANG_USE_AITER=1 -export RCCL_MSCCL_ENABLE=0 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT4 - -SERVER_LOG=/workspace/server.log - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --trust-remote-code \ - --chunked-prefill-size 196608 \ - --mem-fraction-static 0.8 --disable-radix-cache \ - --num-continuous-decode-steps 8 \ - --max-prefill-tokens 196608 \ - --kv-cache-dtype fp8_e4m3 \ - --cuda-graph-max-bs "$CONC" $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom.sh deleted file mode 100644 index 50427bd858..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom.sh +++ /dev/null @@ -1,77 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CALCULATED_MAX_MODEL_LEN=" --max-model-len $EVAL_MAX_MODEL_LEN " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor - -set -x - -BLOCK_SIZE=16 -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --block-size $BLOCK_SIZE > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x \ No newline at end of file diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom_mtp.sh deleted file mode 100644 index f58b7bd534..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_atom_mtp.sh +++ /dev/null @@ -1,80 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -CALCULATED_MAX_MODEL_LEN="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CALCULATED_MAX_MODEL_LEN=" --max-model-len $EVAL_MAX_MODEL_LEN " -fi - -PARALLEL_ARGS=(-tp "$TP") #TP -if [ "$DP_ATTENTION" = "true" ]; then - if [ "$EP_SIZE" -gt 1 ]; then #DP+EP - PARALLEL_ARGS=(-tp "$TP" --enable-expert-parallel --enable-dp-attention ) - else #DP+TP - PARALLEL_ARGS=(-tp "$TP" --enable-dp-attention ) - fi -fi - -SPEC_ARGS=(--method mtp --num-speculative-tokens 3 ) - -start_gpu_monitor - -set -x - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - "${PARALLEL_ARGS[@]}" \ - "${SPEC_ARGS[@]}" \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN \ - --no-enable_prefix_caching \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -export PYTHONDONTWRITEBYTECODE=1 -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_mtp.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_mtp.sh deleted file mode 100755 index feee07896b..0000000000 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi355x_mtp.sh +++ /dev/null @@ -1,86 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -# Reference: https://rocm.docs.amd.com/en/docs-7.0-docker/benchmark-docker/inference-sglang-deepseek-r1-fp8.html - -export SGLANG_USE_AITER=1 -export SGLANG_AITER_MLA_PERSIST=1 -export SGLANG_ENABLE_SPEC_V2=1 -export RCCL_MSCCL_ENABLE=0 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT4 - -SERVER_LOG=/workspace/server.log - -# Keep server-side speculative decoding capacity aligned with the matrix row. -MAX_RUNNING_REQUESTS="$CONC" -CUDA_GRAPH_MAX_BATCH_SIZE="$CONC" - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -fi -start_gpu_monitor - -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --ep-size $EP_SIZE \ - --trust-remote-code \ - --chunked-prefill-size 196608 \ - --mem-fraction-static 0.8 --disable-radix-cache \ - --num-continuous-decode-steps 8 \ - --max-prefill-tokens 196608 \ - --kv-cache-dtype fp8_e4m3 \ - --cuda-graph-max-bs "$CUDA_GRAPH_MAX_BATCH_SIZE" \ - --max-running-requests "$MAX_RUNNING_REQUESTS" \ - --speculative-algorithm EAGLE \ - --speculative-num-steps 3 \ - --speculative-eagle-topk 1 \ - --speculative-num-draft-tokens 4 \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200.sh deleted file mode 100755 index 904e1eca8e..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200.sh +++ /dev/null @@ -1,78 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 --expert-parallel-size=$EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization modelopt_fp4 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---attention-backend trtllm_mha \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs $CONC \ ---max-prefill-tokens 16384 \ ---chunked-prefill-size 16384 \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval $( [[ $CONC -gt 4 ]] && echo 30 || echo 10 ) \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_mtp.sh deleted file mode 100755 index 53a33b5473..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_mtp.sh +++ /dev/null @@ -1,98 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -LINEAR_ATTN_ARGS=() -if [[ "$TP" == "2" && "$EP_SIZE" == "2" ]]; then - case "$CONC" in - 16|32|64) - LINEAR_ATTN_ARGS=( - --linear-attn-backend triton - --linear-attn-decode-backend flashinfer - --linear-attn-prefill-backend flashinfer - ) - ;; - esac -fi - -set -x -SGLANG_ENABLE_SPEC_V2=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 --expert-parallel-size=$EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization modelopt_fp4 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ -"${LINEAR_ATTN_ARGS[@]}" \ ---attention-backend trtllm_mha \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs $CONC \ ---max-running-requests $CONC \ ---max-prefill-tokens 16384 \ ---chunked-prefill-size 16384 \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval $( [[ $CONC -gt 4 ]] && echo 30 || echo 10 ) \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt.sh deleted file mode 100644 index 7fc3599434..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt.sh +++ /dev/null @@ -1,142 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="qwen3.5-fp4-trt.yml" - -if [[ "$DP_ATTENTION" == "true" ]]; then - case "$TP" in - 4) MAX_BATCH_SIZE=256 ;; # tp4 / ep4 with attention DP - 8) MAX_BATCH_SIZE=128 ;; # tp8 / ep8 with attention DP - *) MAX_BATCH_SIZE=$(( CONC > 16 ? CONC : 16 )) ;; - esac -elif [[ "$TP" == "2" ]]; then - if [[ "$EP_SIZE" == "2" && "$CONC" -ge 32 ]]; then - MAX_BATCH_SIZE=32 # tp2 / ep2 at high concurrency - else - MAX_BATCH_SIZE=256 # tp2 / ep1, or tp2 / ep2 at low concurrency - fi -elif [[ "$TP" -ge 4 ]]; then - MAX_BATCH_SIZE=512 # tp>=4 without attention DP -else - MAX_BATCH_SIZE=$(( CONC > 16 ? CONC : 16 )) -fi - -if [[ "$DP_ATTENTION" == "true" ]]; then - MOE_BACKEND="CUTEDSL" - MODE_CONFIG="attention_dp_config: - enable_balance: true - batching_wait_iters: 10 - timeout_iters: 500" -else - MOE_BACKEND="TRTLLM" - MODE_CONFIG="batch_wait_timeout_iters: 50 -batch_wait_max_tokens_ratio: 0.45" -fi - -cat > "$EXTRA_CONFIG_FILE" << EOF -backend: pytorch -print_iter_log: true -enable_layerwise_nvtx_marker: false -disable_overlap_scheduler: false -enable_iter_perf_stats: true -enable_chunked_prefill: false -stream_interval: 20 -num_postprocess_workers: 4 -enable_attention_dp: $DP_ATTENTION -scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - context_chunking_policy: FIRST_COME_FIRST_SERVED -kv_cache_config: - free_gpu_memory_fraction: 0.9 - enable_block_reuse: false - dtype: fp8 -cuda_graph_config: - enable_padding: true - max_batch_size: $MAX_BATCH_SIZE -moe_config: - backend: $MOE_BACKEND - use_low_precision_moe_combine: true -$MODE_CONFIG -EOF - -echo "Generated config file contents:" -cat "$EXTRA_CONFIG_FILE" - -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) - -case "${ISL}_${OSL}" in - 8192_1024) MAX_NUM_TOKENS=32768 ;; - 1024_1024) MAX_NUM_TOKENS=16384 ;; - *) - MAX_NUM_TOKENS=$(( ISL + OSL + 256 )) - MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - ;; -esac - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve "$MODEL" --port="$PORT" \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size "$MAX_BATCH_SIZE" \ - --max_seq_len="$MAX_MODEL_LEN" \ - --max_num_tokens="$MAX_NUM_TOKENS" \ - --tp_size="$TP" --ep_size="$EP_SIZE" \ - --extra_llm_api_options="$EXTRA_CONFIG_FILE" \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$(( CONC * 10 ))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt_mtp.sh deleted file mode 100644 index 2300dd1b35..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b200_trt_mtp.sh +++ /dev/null @@ -1,159 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - MAX_MODEL_LEN \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - DP_ATTENTION \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -# MTP speculative decode requires the FlashInfer GDN prefill path to be disabled. -export TLLM_USE_FLASHINFER_GDN_PREFILL="0" - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log -EXTRA_CONFIG_FILE="qwen3.5-fp4-trt-mtp.yml" -NUM_NEXTN_PREDICT_LAYERS=3 - -# KV-cache memory fractions below are tuned empirically per layout. -if [[ "$DP_ATTENTION" == "true" ]]; then - MAX_BATCH_SIZE=$(( CONC / 8 )) - MOE_BACKEND="CUTEDSL" - if (( CONC >= 1024 )); then KV_MEMORY_FRACTION=0.8; else KV_MEMORY_FRACTION=0.9; fi - MODE_CONFIG="enable_attention_dp: true -attention_dp_config: - enable_balance: true - batching_wait_iters: 10 - timeout_iters: 500" -else - MAX_BATCH_SIZE="$CONC" - MOE_BACKEND="TRTLLM" - case "${ISL}_tp${TP}_ep${EP_SIZE}" in - 1024_tp2_ep1) KV_MEMORY_FRACTION=0.6 ;; - 1024_tp2_ep2) KV_MEMORY_FRACTION=0.75 ;; - 1024_tp8_ep8) KV_MEMORY_FRACTION=0.8 ;; - 8192_tp2_ep1) KV_MEMORY_FRACTION=0.7 ;; - 8192_tp2_ep2) KV_MEMORY_FRACTION=0.6 ;; - 8192_tp4_ep4) KV_MEMORY_FRACTION=0.75 ;; - 8192_tp8_ep8) KV_MEMORY_FRACTION=0.8 ;; - *) KV_MEMORY_FRACTION=0.8 ;; - esac - # Short-context runs hold less in flight, so they use a tighter token ratio before flushing a batch. - case "$ISL" in - 1024) BATCH_WAIT_MAX_TOKENS_RATIO=0.0625 ;; - *) BATCH_WAIT_MAX_TOKENS_RATIO=0.45 ;; - esac - MODE_CONFIG="batch_wait_timeout_iters: 50 -batch_wait_max_tokens_ratio: $BATCH_WAIT_MAX_TOKENS_RATIO" -fi - -cat > "$EXTRA_CONFIG_FILE" << EOF -backend: pytorch -print_iter_log: true -enable_layerwise_nvtx_marker: false -disable_overlap_scheduler: false -enable_iter_perf_stats: true -enable_chunked_prefill: false -stream_interval: 20 -num_postprocess_workers: 4 -scheduler_config: - capacity_scheduler_policy: MAX_UTILIZATION - context_chunking_policy: FIRST_COME_FIRST_SERVED -kv_cache_config: - free_gpu_memory_fraction: $KV_MEMORY_FRACTION - enable_block_reuse: false - dtype: fp8 -cuda_graph_config: - enable_padding: true - batch_sizes: - - 1 - - 2 - - 4 - - 8 - - 16 - - 32 - - 64 - - 128 -moe_config: - backend: $MOE_BACKEND - use_low_precision_moe_combine: true -speculative_config: - decoding_type: MTP - num_nextn_predict_layers: $NUM_NEXTN_PREDICT_LAYERS -$MODE_CONFIG -EOF - -echo "Generated config file contents:" -cat "$EXTRA_CONFIG_FILE" - -MAX_MODEL_LEN=$(( MAX_MODEL_LEN > 8192 ? MAX_MODEL_LEN : 8192 )) - -case "${ISL}_${OSL}" in - 8192_1024) MAX_NUM_TOKENS=32768 ;; - 1024_1024) MAX_NUM_TOKENS=16384 ;; - *) - MAX_NUM_TOKENS=$(( ISL + OSL + 256 )) - MAX_NUM_TOKENS=$(( MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192 )) - ;; -esac - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" - MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -mpirun -n 1 --oversubscribe --allow-run-as-root \ - trtllm-serve "$MODEL" --port="$PORT" \ - --trust_remote_code \ - --backend=pytorch \ - --max_batch_size "$MAX_BATCH_SIZE" \ - --max_seq_len="$MAX_MODEL_LEN" \ - --max_num_tokens="$MAX_NUM_TOKENS" \ - --tp_size="$TP" --ep_size="$EP_SIZE" \ - --extra_llm_api_options="$EXTRA_CONFIG_FILE" \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend openai \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$(( CONC * 10 ))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300.sh deleted file mode 100755 index 0f04f7f6e2..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300.sh +++ /dev/null @@ -1,108 +0,0 @@ -#!/usr/bin/env bash - -# Follows https://cookbook.sglang.io/autoregressive/Qwen/Qwen3.5 - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - - -export NCCL_NVLS_ENABLE=1 -export SGL_ENABLE_JIT_DEEPGEMM=false -export SGLANG_ENABLE_FLASHINFER_GEMM=true -export PYTHONUNBUFFERED=1 - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -MEM_FRAC_STATIC=0.8 -CHUNKED_PREFILL_SIZE=32768 -MAX_PREFILL_TOKENS=32768 -CUDA_GRAPH_MAX_BATCH_SIZE=$CONC -MAX_RUNNING_REQUESTS=128 -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -if [[ $TP -eq 8 ]]; then - EXTRA_ARGS="--enable-flashinfer-allreduce-fusion" -else - EXTRA_ARGS="" -fi - -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 --ep-size $EP_SIZE \ ---reasoning-parser qwen3 \ ---tool-call-parser qwen3_coder \ ---mamba-scheduler-strategy no_buffer \ ---quantization modelopt_fp4 --fp4-gemm-backend flashinfer_cutlass \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE --max-running-requests $MAX_RUNNING_REQUESTS \ ---mem-fraction-static $MEM_FRAC_STATIC --chunked-prefill-size $CHUNKED_PREFILL_SIZE --max-prefill-tokens $MAX_PREFILL_TOKENS \ ---context-length $CONTEXT_LENGTH --disable-radix-cache \ ---attention-backend trtllm_mha --mm-attention-backend triton_attn --moe-runner-backend flashinfer_trtllm \ -$EXTRA_ARGS --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---tokenizer-worker-num 6 --stream-interval 30 > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300_mtp.sh deleted file mode 100755 index 0c0befe0dc..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_b300_mtp.sh +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env bash - -# Follows https://cookbook.sglang.io/autoregressive/Qwen/Qwen3.5 - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - - -export NCCL_NVLS_ENABLE=1 -export SGL_ENABLE_JIT_DEEPGEMM=false -export SGLANG_ENABLE_FLASHINFER_GEMM=true -export PYTHONUNBUFFERED=1 - -SERVER_LOG=/workspace/server.log - -if [[ $CONC -ge 16 ]]; then - SCHEDULER_RECV_INTERVAL=30 -else - SCHEDULER_RECV_INTERVAL=10 -fi - -MEM_FRAC_STATIC=0.8 -CHUNKED_PREFILL_SIZE=32768 -MAX_PREFILL_TOKENS=32768 -CUDA_GRAPH_MAX_BATCH_SIZE=$CONC -MAX_RUNNING_REQUESTS=128 -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -if [[ $TP -eq 8 ]]; then - EXTRA_ARGS="--enable-flashinfer-allreduce-fusion" -else - EXTRA_ARGS="" -fi - -echo "SCHEDULER_RECV_INTERVAL: $SCHEDULER_RECV_INTERVAL, CONC: $CONC, ISL: $ISL, OSL: $OSL" - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 --ep-size $EP_SIZE \ ---reasoning-parser qwen3 \ ---tool-call-parser qwen3_coder \ ---mamba-scheduler-strategy no_buffer \ ---quantization modelopt_fp4 --fp4-gemm-backend flashinfer_cutlass \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---cuda-graph-max-bs $CUDA_GRAPH_MAX_BATCH_SIZE --max-running-requests $MAX_RUNNING_REQUESTS \ ---mem-fraction-static $MEM_FRAC_STATIC --chunked-prefill-size $CHUNKED_PREFILL_SIZE --max-prefill-tokens $MAX_PREFILL_TOKENS \ ---context-length $CONTEXT_LENGTH --disable-radix-cache \ ---attention-backend trtllm_mha --mm-attention-backend triton_attn --moe-runner-backend flashinfer_trtllm \ -$EXTRA_ARGS --scheduler-recv-interval $SCHEDULER_RECV_INTERVAL \ ---tokenizer-worker-num 6 --stream-interval 30 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ -> $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x.sh deleted file mode 100644 index 9351dd3d7a..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER=1 -export SGLANG_USE_AITER_UNIFIED_ATTN=1 -export AITER_FLYDSL_FORCE=1 -export SGLANG_MAMBA_SSM_DTYPE=bfloat16 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT8 - -SERVER_LOG=/workspace/server.log -MEM_FRAC_STATIC=0.8 - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context -fi - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server --model-path=$MODEL --trust-remote-code \ ---host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP \ ---attention-backend aiter \ ---mem-fraction-static $MEM_FRAC_STATIC \ ---model-loader-extra-config '{"enable_multithread_load": true}' \ ---watchdog-timeout 1200 \ ---disable-radix-cache \ ---max-running-requests $CONC \ ---page-size 16 \ ---kv-cache-dtype fp8_e4m3 \ -> $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" --sleep-interval 60 - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_atom.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_atom.sh deleted file mode 100644 index 98f2416dcc..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_atom.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor -MEM_FRAC_STATIC=0.9 - -set -x - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --gpu-memory-utilization $MEM_FRAC_STATIC \ - --trust-remote-code \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -export PYTHONDONTWRITEBYTECODE=1 -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --trust-remote-code - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_mtp.sh deleted file mode 100755 index 649eb121ab..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_mi355x_mtp.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -hf download "$MODEL" - -export SGLANG_USE_AITER=1 -export SGLANG_USE_AITER_UNIFIED_ATTN=1 -export AITER_FLYDSL_FORCE=1 -export SGLANG_MAMBA_SSM_DTYPE=bfloat16 -export ROCM_QUICK_REDUCE_QUANTIZATION=INT8 - -SERVER_LOG=/workspace/server.log -MEM_FRAC_STATIC=0.8 - -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context -fi - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server --model-path=$MODEL --trust-remote-code \ ---host=0.0.0.0 --port=$PORT \ ---tensor-parallel-size=$TP \ ---attention-backend aiter \ ---mem-fraction-static $MEM_FRAC_STATIC \ ---model-loader-extra-config '{"enable_multithread_load": true}' \ ---watchdog-timeout 1200 \ ---disable-radix-cache \ ---max-running-requests $CONC \ ---page-size 16 \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---kv-cache-dtype fp8_e4m3 \ -> $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" --sleep-interval 60 - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200.sh deleted file mode 100755 index 8e8771369b..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200.sh +++ /dev/null @@ -1,79 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 --expert-parallel-size=$EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization fp8 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---mamba-full-memory-ratio 0.37 \ ---linear-attn-prefill-backend flashinfer \ ---attention-backend trtllm_mha \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs-decode $CONC \ ---max-prefill-tokens 32768 \ ---chunked-prefill-size 32768 \ ---mem-fraction-static 0.86 \ ---stream-interval 50 \ ---scheduler-recv-interval $( [[ $CONC -gt 4 ]] && echo 30 || echo 10 ) \ ---tokenizer-worker-num 6 \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200_mtp.sh deleted file mode 100755 index 5f30b28545..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b200_mtp.sh +++ /dev/null @@ -1,86 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -SGLANG_ENABLE_SPEC_V2=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path=$MODEL --host=0.0.0.0 --port=$PORT \ ---trust-remote-code \ ---tensor-parallel-size=$TP --data-parallel-size=1 --expert-parallel-size=$EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization fp8 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---attention-backend trtllm_mha \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs-decode $CONC \ ---max-running-requests $CONC \ ---max-prefill-tokens 32768 \ ---chunked-prefill-size 32768 \ ---mamba-full-memory-ratio 0.37 \ ---linear-attn-prefill-backend flashinfer \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval $( [[ $CONC -gt 4 ]] && echo 30 || echo 10 ) \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300.sh deleted file mode 100644 index ce0ef07dcd..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300.sh +++ /dev/null @@ -1,88 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 --expert-parallel-size $EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization fp8 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---attention-backend trtllm_mha \ ---mm-attention-backend triton_attn \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs $CONC \ ---max-running-requests $CONC \ ---max-prefill-tokens 16384 \ ---chunked-prefill-size 16384 \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval 10 \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL_PATH \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300_mtp.sh deleted file mode 100644 index ffe9ea2c0c..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_b300_mtp.sh +++ /dev/null @@ -1,92 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" - export MODEL_PATH="$MODEL" -fi - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -SERVER_LOG=/workspace/server.log - -CONTEXT_LENGTH=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - CONTEXT_LENGTH="$EVAL_MAX_MODEL_LEN" -fi - -start_gpu_monitor - -set -x -SGLANG_ENABLE_SPEC_V2=1 PYTHONNOUSERSITE=1 python3 -m sglang.launch_server --model-path $MODEL_PATH --served-model-name $MODEL --host 0.0.0.0 --port $PORT \ ---trust-remote-code \ ---tensor-parallel-size $TP --data-parallel-size 1 --expert-parallel-size $EP_SIZE \ ---enable-symm-mem \ ---disable-radix-cache \ ---quantization fp8 \ ---kv-cache-dtype fp8_e4m3 \ ---mamba-ssm-dtype bfloat16 \ ---attention-backend trtllm_mha \ ---mm-attention-backend triton_attn \ ---moe-runner-backend flashinfer_trtllm \ ---cuda-graph-max-bs $CONC \ ---max-running-requests $CONC \ ---max-prefill-tokens 16384 \ ---chunked-prefill-size 16384 \ ---mem-fraction-static 0.8 \ ---stream-interval 50 \ ---scheduler-recv-interval 10 \ ---tokenizer-worker-num 6 \ ---tokenizer-path $MODEL_PATH \ ---speculative-algorithm EAGLE \ ---speculative-num-steps 3 \ ---speculative-eagle-topk 1 \ ---speculative-num-draft-tokens 4 \ ---context-length $CONTEXT_LENGTH > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100.sh deleted file mode 100755 index c94fbd5793..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100.sh +++ /dev/null @@ -1,123 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -MAX_SEQ_LEN=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_SEQ_LEN="$EVAL_MAX_MODEL_LEN" -fi - -PARALLEL_ARGS=(--tp "$TP") -if [ "${EP_SIZE}" -gt 1 ]; then - PARALLEL_ARGS+=(--expert-parallel-size "$EP_SIZE") -fi - -SCHEDULER_RECV_INTERVAL= -case "$CONC" in - 1|2|4) - SCHEDULER_RECV_INTERVAL=2 - ;; - 8) - SCHEDULER_RECV_INTERVAL=60 - ;; - 16) - SCHEDULER_RECV_INTERVAL=30 - ;; - 32) - SCHEDULER_RECV_INTERVAL=1200 - ;; - 64) - SCHEDULER_RECV_INTERVAL=600 - ;; - 128|256) - SCHEDULER_RECV_INTERVAL=1920 - ;; - *) - echo "Unsupported CONC=$CONC for qwen3.5 FP8 H100 SGLang recipe" >&2 - exit 1 - ;; -esac - -SCHEDULER_ARGS=() -if [ -n "$SCHEDULER_RECV_INTERVAL" ]; then - SCHEDULER_ARGS=(--scheduler-recv-interval "$SCHEDULER_RECV_INTERVAL") -fi - -echo "TP: $TP, EP_SIZE: $EP_SIZE, CONC: $CONC, ISL: $ISL, OSL: $OSL, MAX_SEQ_LEN: $MAX_SEQ_LEN" -echo "SCHEDULER_RECV_INTERVAL: ${SCHEDULER_RECV_INTERVAL:-none}" -echo "SCHEDULER_ARGS: ${SCHEDULER_ARGS[*]}" - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server \ - --model "$MODEL" \ - --host 0.0.0.0 \ - --port "$PORT" \ - "${PARALLEL_ARGS[@]}" \ - --reasoning-parser qwen3 \ - --tool-call-parser qwen3_coder \ - --enable-flashinfer-allreduce-fusion \ - --max-running-requests 256 \ - --chunked-prefill-size 16384 \ - --decode-log-interval 1 \ - --mem-fraction-static 0.8 \ - --cuda-graph-max-bs "$CONC" \ - --context-length "$MAX_SEQ_LEN" \ - --kv-cache-dtype fp8_e4m3 \ - --quantization fp8 \ - --attention-backend flashinfer \ - --stream-interval 50 \ - --tokenizer-worker-num 6 \ - --mamba-ssm-dtype bfloat16 \ - --disable-radix-cache \ - --enable-symm-mem \ - --trust-remote-code \ - "${SCHEDULER_ARGS[@]}" \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100_mtp.sh deleted file mode 100755 index 5860d601cb..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h100_mtp.sh +++ /dev/null @@ -1,91 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_ENABLE_SPEC_V2=1 - -SERVER_LOG=/workspace/server.log -MAX_SEQ_LEN=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_SEQ_LEN="$EVAL_MAX_MODEL_LEN" -fi - -echo "CONC: $CONC, ISL: $ISL, OSL: $OSL, MAX_SEQ_LEN: $MAX_SEQ_LEN" - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server \ - --model "$MODEL" \ - --host 0.0.0.0 \ - --port "$PORT" \ - --tp "$TP" \ - --expert-parallel-size "$EP_SIZE" \ - --reasoning-parser qwen3 \ - --tool-call-parser qwen3_coder \ - --enable-flashinfer-allreduce-fusion \ - --max-running-requests 64 \ - --chunked-prefill-size 8192 \ - --decode-log-interval 1 \ - --mem-fraction-static 0.75 \ - --cuda-graph-max-bs "$CONC" \ - --context-length "$MAX_SEQ_LEN" \ - --kv-cache-dtype fp8_e4m3 \ - --quantization fp8 \ - --attention-backend flashinfer \ - --stream-interval 50 \ - --tokenizer-worker-num 6 \ - --mamba-ssm-dtype bfloat16 \ - --disable-radix-cache \ - --trust-remote-code \ - --speculative-algorithm EAGLE \ - --speculative-num-steps 3 \ - --speculative-eagle-topk 1 \ - --speculative-num-draft-tokens 4 \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200.sh deleted file mode 100644 index 0e1a4b581d..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200.sh +++ /dev/null @@ -1,84 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -MAX_SEQ_LEN=$((ISL + OSL + 20)) -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - MAX_SEQ_LEN="$EVAL_MAX_MODEL_LEN" -fi - -echo "CONC: $CONC, ISL: $ISL, OSL: $OSL, MAX_SEQ_LEN: $MAX_SEQ_LEN" - -start_gpu_monitor - -set -x -python3 -m sglang.launch_server \ - --model "$MODEL" \ - --host 0.0.0.0 \ - --port "$PORT" \ - --tp "$TP" \ - --expert-parallel-size "$EP_SIZE" \ - --reasoning-parser qwen3 \ - --tool-call-parser qwen3_coder \ - --enable-flashinfer-allreduce-fusion \ - --max-running-requests 128 \ - --chunked-prefill-size 16384 \ - --decode-log-interval 1 \ - --mem-fraction-static 0.8 \ - --cuda-graph-max-bs "$CONC" \ - --context-length "$MAX_SEQ_LEN" \ - --kv-cache-dtype fp8_e4m3 \ - --quantization fp8 \ - --attention-backend flashinfer \ - --stream-interval 50 \ - --tokenizer-worker-num 6 \ - --mamba-ssm-dtype bfloat16 \ - --disable-radix-cache \ - --trust-remote-code \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200_mtp.sh deleted file mode 100644 index 46e04ebe84..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_h200_mtp.sh +++ /dev/null @@ -1,89 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - MAX_MODEL_LEN - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -nvidia-smi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log - -SPECULATIVE_NUM_STEPS=3 -SPECULATIVE_DRAFT_TOKENS=4 -SPECULATIVE_EAGLE_TOPK=1 - -echo "CONC: $CONC, ISL: $ISL, OSL: $OSL, MAX_MODEL_LEN: $MAX_MODEL_LEN" - -start_gpu_monitor - -set -x -SGLANG_ENABLE_SPEC_V2=1 python3 -m sglang.launch_server \ - --model "$MODEL" \ - --host 0.0.0.0 \ - --port "$PORT" \ - --tp "$TP" \ - --expert-parallel-size "$EP_SIZE" \ - --reasoning-parser qwen3 \ - --tool-call-parser qwen3_coder \ - --enable-flashinfer-allreduce-fusion \ - --max-running-requests 128 \ - --chunked-prefill-size 16384 \ - --mem-fraction-static 0.8 \ - --cuda-graph-max-bs "$CONC" \ - --context-length "$MAX_MODEL_LEN" \ - --kv-cache-dtype fp8_e4m3 \ - --quantization fp8 \ - --attention-backend flashinfer \ - --stream-interval 50 \ - --tokenizer-worker-num 6 \ - --mamba-ssm-dtype bfloat16 \ - --disable-radix-cache \ - --trust-remote-code \ - --speculative-algorithm EAGLE \ - --speculative-num-steps "$SPECULATIVE_NUM_STEPS" \ - --speculative-num-draft-tokens "$SPECULATIVE_DRAFT_TOKENS" \ - --speculative-eagle-topk "$SPECULATIVE_EAGLE_TOPK" \ - > "$SERVER_LOG" 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -pip install -q datasets pandas - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --use-chat-template \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - export EVAL_CONCURRENT_REQUESTS="$CONC" - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi300x.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi300x.sh deleted file mode 100755 index 1ca367d51c..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi300x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) -MAX_PREFILL_TOKENS=32768 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -# Recipe source: https://www.linkedin.com/feed/update/urn:li:activity:7429203734389280768/ -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --data-parallel-size 1 \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --scheduler-recv-interval 30 \ - --mem-fraction-static 0.75 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x.sh deleted file mode 100755 index 1ca367d51c..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) -MAX_PREFILL_TOKENS=32768 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -# Recipe source: https://www.linkedin.com/feed/update/urn:li:activity:7429203734389280768/ -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --data-parallel-size 1 \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --scheduler-recv-interval 30 \ - --mem-fraction-static 0.75 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x_mtp.sh deleted file mode 100755 index edd44a705d..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi325x_mtp.sh +++ /dev/null @@ -1,78 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME - EP_SIZE \ - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) -MAX_PREFILL_TOKENS=32768 - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -# Recipe source: https://www.linkedin.com/feed/update/urn:li:activity:7429203734389280768/ -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --ep-size $EP_SIZE \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --max-prefill-tokens $MAX_PREFILL_TOKENS \ - --scheduler-recv-interval 30 \ - --speculative-algorithm EAGLE \ - --speculative-num-steps 3 \ - --speculative-eagle-topk 1 \ - --speculative-num-draft-tokens 4 \ - --mem-fraction-static 0.75 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - EP_SIZE \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x.sh deleted file mode 100644 index 1c4e1dceb8..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER_UNIFIED_ATTN=1 -export SGLANG_USE_AITER=1 - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --ep-size $EP_SIZE \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --max-running-requests $CONC \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --chunked-prefill-size 32768 \ - --scheduler-recv-interval 30 \ - --mem-fraction-static 0.8 \ - --model-loader-extra-config '{"enable_multithread_load": true}' \ - --page-size 16 $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom.sh deleted file mode 100644 index 98f2416dcc..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor -MEM_FRAC_STATIC=0.9 - -set -x - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --gpu-memory-utilization $MEM_FRAC_STATIC \ - --trust-remote-code \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -export PYTHONDONTWRITEBYTECODE=1 -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --trust-remote-code - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom_mtp.sh deleted file mode 100644 index 7948322313..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_atom_mtp.sh +++ /dev/null @@ -1,79 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE \ - DP_ATTENTION - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL, EP_SIZE: $EP_SIZE, DP_ATTENTION: $DP_ATTENTION" - -SERVER_LOG=/workspace/server.log - -export OMP_NUM_THREADS=1 - -if [ "$ISL" = "1024" ] && [ "$OSL" = "1024" ]; then - CALCULATED_MAX_MODEL_LEN="" -else - CALCULATED_MAX_MODEL_LEN=" --max-model-len 10240 " -fi - -if [ "$EP_SIZE" -gt 1 ]; then - EP=" --enable-expert-parallel" -else - EP=" " -fi - -start_gpu_monitor -MEM_FRAC_STATIC=0.9 - -set -x - -python3 -m atom.entrypoints.openai_server \ - --model $MODEL \ - --server-port $PORT \ - -tp $TP \ - --kv_cache_dtype fp8 $CALCULATED_MAX_MODEL_LEN $EP \ - --gpu-memory-utilization $MEM_FRAC_STATIC \ - --method mtp \ - --num-speculative-tokens 3 \ - --trust-remote-code \ - > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -export PYTHONDONTWRITEBYTECODE=1 -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --trust-remote-code \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_mtp.sh b/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_mtp.sh deleted file mode 100755 index 2a20db1868..0000000000 --- a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp8_mi355x_mtp.sh +++ /dev/null @@ -1,82 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars \ - MODEL \ - TP \ - CONC \ - ISL \ - OSL \ - RANDOM_RANGE_RATIO \ - RESULT_FILENAME \ - EP_SIZE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi - -export SGLANG_USE_AITER_UNIFIED_ATTN=1 -export SGLANG_USE_AITER=1 - -SERVER_LOG=/workspace/server.log -CONTEXT_LENGTH=$((ISL + OSL + 20)) - -EVAL_CONTEXT_ARGS="" -if [ "${EVAL_ONLY}" = "true" ]; then - setup_eval_context - EVAL_CONTEXT_ARGS="--context-length $EVAL_MAX_MODEL_LEN" -else EVAL_CONTEXT_ARGS="--context-length $CONTEXT_LENGTH" -fi -start_gpu_monitor - -python3 -m sglang.launch_server \ - --attention-backend aiter \ - --model-path $MODEL \ - --host=0.0.0.0 \ - --port $PORT \ - --tensor-parallel-size $TP \ - --ep-size $EP_SIZE \ - --trust-remote-code \ - --tokenizer-worker-num 6 \ - --enable-aiter-allreduce-fusion \ - --max-running-requests $CONC \ - --cuda-graph-max-bs $CONC \ - --disable-radix-cache \ - --chunked-prefill-size 32768 \ - --scheduler-recv-interval 30 \ - --mem-fraction-static 0.8 \ - --model-loader-extra-config '{"enable_multithread_load": true}' \ - --page-size 16 \ - --speculative-algorithm EAGLE \ - --speculative-num-steps 3 \ - --speculative-eagle-topk 1 \ - --speculative-num-draft-tokens 4 \ - $EVAL_CONTEXT_ARGS > $SERVER_LOG 2>&1 & - -SERVER_PID=$! - -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$PORT" \ - --backend vllm \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir /workspace/ \ - --use-chat-template - -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary -fi - -stop_gpu_monitor -set +x diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml index d09b7c8ed5..03c0d172b3 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp4_mi355x_atom_mtp.sh. base: schema: 2 name: dsr1-fp4-mi355x-atom-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml index 24d3cad417..2a258efb17 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp4_mi355x_atom.sh. base: schema: 2 name: dsr1-fp4-mi355x-atom-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml index ac91c9ac34..db5f89aee7 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_mi355x_atom_mtp.sh. base: schema: 2 name: dsr1-fp8-mi355x-atom-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml index 68efd10f89..d7b82c5ed7 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_mi355x_atom.sh. base: schema: 2 name: dsr1-fp8-mi355x-atom-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml index d006f22922..88ae02628c 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp4_b200_mtp.sh. base: schema: 2 name: dsr1-fp4-b200-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml index 2fcd1bb81e..98888e3f91 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp4_b200.sh. base: schema: 2 name: dsr1-fp4-b200-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml index aa2eb02bf7..54d62161ee 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_b200_mtp.sh. base: schema: 2 name: dsr1-fp8-b200-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml index 06af0e4743..301b150f54 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b200-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_b200.sh. base: schema: 2 name: dsr1-fp8-b200-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml index 935a533ac0..4399686cde 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp4_b300.sh. base: schema: 2 name: dsr1-fp4-b300-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml index cb8887141b..952ec213c9 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_b300_mtp.sh. base: schema: 2 name: dsr1-fp8-b300-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml index 4fdbff56bb..0fbaf5bb8a 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/b300-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_b300.sh. base: schema: 2 name: dsr1-fp8-b300-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml index ab5b935067..7818ad85df 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_h200_mtp.sh. base: schema: 2 name: dsr1-fp8-h200-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml index bee0edbea2..23bbfa1d27 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/h200-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_h200.sh. base: schema: 2 name: dsr1-fp8-h200-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml index b60ad7cf81..78e9a937a0 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi300x-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_mi300x.sh. base: schema: 2 name: dsr1-fp8-mi300x-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml index 1cdc40c0df..449ba41109 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_mi325x_mtp.sh. base: schema: 2 name: dsr1-fp8-mi325x-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml index 1d8209510c..06b99ea82f 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi325x-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_mi325x.sh. base: schema: 2 name: dsr1-fp8-mi325x-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml index af530edec5..18cf9bfd4f 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp4_mi355x_mtp.sh. base: schema: 2 name: dsr1-fp4-mi355x-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml index e4ff3977c4..75d0a62f57 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp4_mi355x.sh. base: schema: 2 name: dsr1-fp4-mi355x-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml index 2c0d689c9d..ae4ba64142 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_mi355x_mtp.sh. base: schema: 2 name: dsr1-fp8-mi355x-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml index c9f958349d..07cb2a1783 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/sglang/mi355x-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_mi355x.sh. base: schema: 2 name: dsr1-fp8-mi355x-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml index 6956d4eac4..0533fa0d82 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp4_b200_trt_mtp.sh. base: schema: 2 name: dsr1-fp4-b200-trt-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml index b96ce513a1..2910db349e 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp4_b200_trt.sh. base: schema: 2 name: dsr1-fp4-b200-trt-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml index 6ac275035c..6b31d7e6fa 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_b200_trt_mtp.sh. base: schema: 2 name: dsr1-fp8-b200-trt-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml index 76dc46eae9..6e818ae104 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_b200_trt.sh. base: schema: 2 name: dsr1-fp8-b200-trt-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml index 751e74537e..8d446aef24 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_h200_trt_mtp.sh. base: schema: 2 name: dsr1-fp8-h200-trt-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml index 1a98b08736..cfe32cd8b9 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from dsr1_fp8_h200_trt.sh. base: schema: 2 name: dsr1-fp8-h200-trt-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml index 2ef1824ec9..b748775c8a 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp4_mi355x_atom.sh. base: schema: 2 name: qwen3.5-fp4-mi355x-atom-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml index 468285d526..ac1435ee2e 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_mi355x_atom_mtp.sh. base: schema: 2 name: qwen3.5-fp8-mi355x-atom-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml index 3a1bbaa81f..4091212fb7 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_mi355x_atom.sh. base: schema: 2 name: qwen3.5-fp8-mi355x-atom-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml index cb84cd7218..8604043522 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp4_b200_mtp.sh. base: schema: 2 name: qwen3.5-fp4-b200-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml index 3589bf278a..39801bfa8e 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp4_b200.sh. base: schema: 2 name: qwen3.5-fp4-b200-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml index bf219a8c68..8ea8fbae21 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_b200_mtp.sh. base: schema: 2 name: qwen3.5-fp8-b200-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml index 1236c91d90..68df2a0c0a 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b200-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_b200.sh. base: schema: 2 name: qwen3.5-fp8-b200-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml index 21096c3047..3297e6b7df 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp4_b300_mtp.sh. base: schema: 2 name: qwen3.5-fp4-b300-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml index 7484259bfa..00af900818 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp4_b300.sh. base: schema: 2 name: qwen3.5-fp4-b300-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml index 5489fe7106..aa6af8a4b7 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_b300_mtp.sh. base: schema: 2 name: qwen3.5-fp8-b300-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml index 789199c51d..98f877110c 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_b300.sh. base: schema: 2 name: qwen3.5-fp8-b300-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml index 0bd0858956..ff59b74291 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_h100_mtp.sh. base: schema: 2 name: qwen3.5-fp8-h100-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml index b1b117d767..9f5dac04cc 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h100-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_h100.sh. base: schema: 2 name: qwen3.5-fp8-h100-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml index fd74c5af14..e5bf9adca9 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_h200_mtp.sh. base: schema: 2 name: qwen3.5-fp8-h200-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml index 41d490f7cb..a7c1df61c2 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/h200-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_h200.sh. base: schema: 2 name: qwen3.5-fp8-h200-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml index bc00e95f39..6691e5ee92 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi300x-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_mi300x.sh. base: schema: 2 name: qwen3.5-fp8-mi300x-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml index 0f1d40a983..bd0dcd6623 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_mi325x_mtp.sh. base: schema: 2 name: qwen3.5-fp8-mi325x-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml index a17a9e2085..7d4ef623dc 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_mi325x.sh. base: schema: 2 name: qwen3.5-fp8-mi325x-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml index 58b7ce27c0..244b8d64dc 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp4_mi355x_mtp.sh. base: schema: 2 name: qwen3.5-fp4-mi355x-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml index 9488267696..7ea8244e0d 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp4_mi355x.sh. base: schema: 2 name: qwen3.5-fp4-mi355x-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml index 9e5e80cad8..488994e3bf 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_mi355x_mtp.sh. base: schema: 2 name: qwen3.5-fp8-mi355x-sglang-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml index 12757766b2..c753c2b465 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp8_mi355x.sh. base: schema: 2 name: qwen3.5-fp8-mi355x-sglang-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml deleted file mode 100644 index 9385e888f1..0000000000 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml +++ /dev/null @@ -1,88 +0,0 @@ -# Serving settings from qwen3.5_fp4_rtx6000pro_sglang_mtp.sh. -base: - schema: 2 - name: qwen3.5-fp4-rtx6000pro-sglang-mtp-8k1k - model: - path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4 - container: lmsysorg/sglang:v0.5.16-cu130 - precision: fp4 - resources: - gpu_type: rtx6000pro - gpus_per_node: 8 - frontend: - type: sglang - enable_multiple_frontends: false - observability: - enabled: false - tachometer: - enabled: false - engine: sglang - roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - quantization: modelopt_fp4 - fp4-gemm-backend: flashinfer_cutlass - moe-runner-backend: flashinfer_cutlass - attention-backend: flashinfer - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: no_buffer - disable-custom-all-reduce: true - disable-radix-cache: true - mem-fraction-static: 0.8 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 9236 - cuda-graph-max-bs-decode: 1 - max-running-requests: 1 - scheduler-recv-interval: 10 - stream-interval: 20 - speculative-algorithm: EAGLE - speculative-num-steps: 3 - speculative-eagle-topk: 1 - speculative-num-draft-tokens: 4 - enable-metrics: false - env: - SGLANG_ENABLE_JIT_DEEPGEMM: 'false' - PYTHONUNBUFFERED: '1' - PYTHONNOUSERSITE: '1' - benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code - env: - MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4 - ISL: '8192' - OSL: '1024' - RANDOM_RANGE_RATIO: '0.8' - USE_CHAT_TEMPLATE: 'true' -zip_override_tp4_ep1: - roles: - agg: - args: - cuda-graph-max-bs-decode: [1, 4, 16, 64] - max-running-requests: [1, 4, 16, 64] - scheduler-recv-interval: [10, 10, 30, 30] - benchmark: - env: - CONC: ['1', '4', '16', '64'] -zip_override_tp4_ep4: - roles: - agg: - args: - cuda-graph-max-bs-decode: [1, 4, 16, 64] - expert-parallel-size: 4 - max-running-requests: [1, 4, 16, 64] - scheduler-recv-interval: [10, 10, 30, 30] - benchmark: - env: - CONC: ['1', '4', '16', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml deleted file mode 100644 index ac4e793474..0000000000 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml +++ /dev/null @@ -1,82 +0,0 @@ -# Serving settings from qwen3.5_fp4_rtx6000pro_sglang.sh. -base: - schema: 2 - name: qwen3.5-fp4-rtx6000pro-sglang-8k1k - model: - path: hf:nvidia/Qwen3.5-397B-A17B-NVFP4 - container: lmsysorg/sglang:v0.5.16-cu130 - precision: fp4 - resources: - gpu_type: rtx6000pro - gpus_per_node: 8 - frontend: - type: sglang - enable_multiple_frontends: false - observability: - enabled: false - tachometer: - enabled: false - engine: sglang - roles: - agg: - nodes: 1 - workers: 1 - gpus: 4 - args: - served-model-name: nvidia/Qwen3.5-397B-A17B-NVFP4 - trust-remote-code: true - tensor-parallel-size: 4 - data-parallel-size: 1 - expert-parallel-size: 1 - reasoning-parser: qwen3 - tool-call-parser: qwen3_coder - quantization: modelopt_fp4 - fp4-gemm-backend: flashinfer_cutlass - moe-runner-backend: flashinfer_cutlass - attention-backend: flashinfer - kv-cache-dtype: fp8_e4m3 - mamba-ssm-dtype: bfloat16 - mamba-scheduler-strategy: no_buffer - disable-custom-all-reduce: true - disable-radix-cache: true - mem-fraction-static: 0.7 - chunked-prefill-size: 16384 - max-prefill-tokens: 16384 - context-length: 9236 - cuda-graph-max-bs-decode: 1 - max-running-requests: 128 - scheduler-recv-interval: 10 - stream-interval: 20 - enable-metrics: false - env: - SGLANG_ENABLE_JIT_DEEPGEMM: 'false' - PYTHONUNBUFFERED: '1' - PYTHONNOUSERSITE: '1' - benchmark: - type: custom - command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh --trust-remote-code - env: - MODEL: nvidia/Qwen3.5-397B-A17B-NVFP4 - ISL: '8192' - OSL: '1024' - RANDOM_RANGE_RATIO: '0.8' - USE_CHAT_TEMPLATE: 'false' -zip_override_tp4_ep1: - roles: - agg: - args: - cuda-graph-max-bs-decode: [1, 4, 16, 64] - scheduler-recv-interval: [10, 10, 30, 30] - benchmark: - env: - CONC: ['1', '4', '16', '64'] -zip_override_tp4_ep4: - roles: - agg: - args: - cuda-graph-max-bs-decode: [1, 4, 16, 64] - expert-parallel-size: 4 - scheduler-recv-interval: [10, 10, 30, 30] - benchmark: - env: - CONC: ['1', '4', '16', '64'] diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml index 19abfd8efb..8ea47be0f0 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp4_b200_trt_mtp.sh. base: schema: 2 name: qwen3.5-fp4-b200-trt-mtp-8k1k diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml index 1e1762b73e..0a6ca11dc8 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml @@ -1,4 +1,3 @@ -# Serving settings from qwen3.5_fp4_b200_trt.sh. base: schema: 2 name: qwen3.5-fp4-b200-trt-8k1k diff --git a/benchmarks/single_node/srt_docker.sh b/benchmarks/single_node/srt_docker.sh deleted file mode 100644 index 705e367b52..0000000000 --- a/benchmarks/single_node/srt_docker.sh +++ /dev/null @@ -1,36 +0,0 @@ -#!/usr/bin/env bash - -# Keep the pool's existing Docker lifecycle; commands come from native SRT YAML. -set -eo pipefail -source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only -check_env_vars MODEL PORT RUN_EVAL EVAL_ONLY INFMAX_CONTAINER_WORKSPACE -for flag in RUN_EVAL EVAL_ONLY; do - if [[ "${!flag}" != true && "${!flag}" != false ]]; then - echo "$flag must be true or false" >&2 - exit 1 - fi -done -if [[ "$RUN_EVAL" == true || "$EVAL_ONLY" == true ]]; then - check_env_vars MODEL_NAME -fi -SERVER_LOG="$INFMAX_CONTAINER_WORKSPACE/server.log" -if [[ -n "${MODEL_PATH:-}" ]]; then - if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then - hf download "$MODEL" --local-dir "$MODEL_PATH" - fi -else - hf download "$MODEL" -fi -bash "$INFMAX_CONTAINER_WORKSPACE/srt-docker-server.sh" > "$SERVER_LOG" 2>&1 & -SERVER_PID=$! -trap 'rc=$?; kill "$SERVER_PID" 2>/dev/null || true; exit "$rc"' EXIT -source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" -wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" -export INFERENCEX_SERVER_PID INFERENCEX_SERVER_STATE -if [[ "$EVAL_ONLY" != true ]]; then - bash "$INFMAX_CONTAINER_WORKSPACE/srt-docker-client.sh" -fi -if [[ "$RUN_EVAL" == true || "$EVAL_ONLY" == true ]]; then - bash "$INFMAX_CONTAINER_WORKSPACE/benchmarks/single_node/srt_eval.sh" \ - "http://127.0.0.1:$PORT" "$INFMAX_CONTAINER_WORKSPACE/infx-eval-exit-code" -fi diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 7584527843..4cd00ec44a 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -1355,13 +1355,8 @@ qwen3.5-fp4-rtx6000pro-sglang: - isl: 8192 osl: 1024 search-space: - - tp: 4 - conc-list: [1, 4, 16, 64] - srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml - - tp: 4 - ep: 4 - conc-list: [1, 4, 16, 64] - srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4/8k1k.yaml + - { tp: 4, conc-list: [1, 4, 16, 64] } + - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64] } # Same sweep with the built-in MTP draft head driven through SGLang's EAGLE # speculative path. @@ -1378,15 +1373,8 @@ qwen3.5-fp4-rtx6000pro-sglang-mtp: - isl: 8192 osl: 1024 search-space: - - tp: 4 - conc-list: [1, 4, 16, 64] - spec-decoding: mtp - srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml - - tp: 4 - ep: 4 - conc-list: [1, 4, 16, 64] - spec-decoding: mtp - srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/rtx6000pro-fp4-mtp/8k1k.yaml + - { tp: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } + - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } qwen3.5-fp4-b300-sglang-mtp: image: lmsysorg/sglang:v0.5.19-cu130 diff --git a/infx/srt_slurm/docker.py b/infx/srt_slurm/docker.py deleted file mode 100644 index dcdb8f8839..0000000000 --- a/infx/srt_slurm/docker.py +++ /dev/null @@ -1,117 +0,0 @@ -"""Prepare native SRT server/client commands for the existing Docker runner.""" - -from __future__ import annotations - -import argparse -import os -import re -import shlex -from collections.abc import Mapping -from pathlib import Path -from typing import Any - -from infx.srt_slurm.single_node import runtime_arguments, select_recipe -from infx.srt_slurm.synthetic_acceptance import build_overrides - - -def shell_command(command: list[str], environment: Mapping[str, str]) -> str: - """Quote argv and literal recipe environment without embedding caller secrets.""" - lines = ["#!/usr/bin/env bash", "set -eo pipefail"] - for key, value in environment.items(): - if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", key): - raise ValueError(f"Invalid environment key: {key!r}") - lines.append(f"export {key}={shlex.quote(value)}") - lines.append(f"exec {shlex.join(command)}") - return "\n".join(lines) + "\n" - - -def prepare(config: str, environment: Mapping[str, str]) -> tuple[str, str]: - """Use the pinned SRT schema and backend builder; Docker stays pool-owned.""" - from srtctl.core.config import expand_engine_config_defaults, resolve_config_with_defaults - from srtctl.core.overrides import apply_overrides_to_recipe, parse_overrides - from srtctl.core.runtime import Nodes, RuntimeContext - from srtctl.core.schema import SrtConfig - from srtctl.core.topology import Process - - selected, recipe = select_recipe(config, environment) - if environment["FRAMEWORK"] != "sglang": - raise ValueError("The Docker runner currently supports native SGLang recipes only") - local_model = environment.get("MODEL_PATH") - port = int(environment["PORT"]) - if not 1 <= port <= 65535: - raise ValueError("PORT must be between 1 and 65535") - overrides = runtime_arguments(selected, environment) - apply_overrides_to_recipe(recipe, parse_overrides(overrides[1::2], [])) - golden = build_overrides(recipe, environment["FRAMEWORK"], environment) - # Fixed-sequence jobs remove any stale simulation flags, as native apply does. - parser = argparse.ArgumentParser(add_help=False) - parser.add_argument("--set", action="append", default=[]) - parser.add_argument("--unset", action="append", default=[]) - parsed = parser.parse_args(golden) - apply_overrides_to_recipe(recipe, parse_overrides(parsed.set, parsed.unset)) - resolved = resolve_config_with_defaults(recipe, {}) - expand_engine_config_defaults(resolved) - native = SrtConfig.Schema().load(resolved) - if ( - native.frontend.type != "sglang" - or native.services - or native.setup_script - or native.host_setup.enabled - or native.dynamo.sidecar - or native.extra_mount - or native.container_mounts - ): - raise ValueError( - "Docker fixed-sequence recipes require one direct server without services or setup" - ) - runtime = RuntimeContext( - job_id="docker", - run_name=native.name, - nodes=Nodes(head="127.0.0.1", bench="127.0.0.1", infra="127.0.0.1", worker=("127.0.0.1",)), - head_node_ip="127.0.0.1", - infra_node_ip="127.0.0.1", - log_dir=Path("/logs"), - model_path=Path(local_model or environment["MODEL"]), - container_image=Path(environment["IMAGE"]), - gpus_per_node=native.resources.gpus_per_node, - network_interface=None, - # Docker already exposes MODEL_PATH through its existing mounts. Use - # the native builder's literal path mode, without Slurm's /model mount. - is_hf_model=True, - frontend_port=port, - ) - process = Process( - node="127.0.0.1", - gpu_indices=frozenset(range(int(environment["GPU_COUNT"]))), - sys_port=port, - http_port=port, - endpoint_mode="agg", - endpoint_index=0, - ) - server = native.backend.build_worker_command( - process, [process], runtime, frontend_type="sglang" - ) - server_env = {**native.backend.get_environment_for_mode("agg"), **native.environment} - benchmark: dict[str, Any] = recipe["benchmark"] - client_env = { - **benchmark["env"], - "SRT_FRONTEND_HOST": "127.0.0.1", - "SRT_FRONTEND_PORT": str(port), - } - return shell_command(server, server_env), shell_command( - shlex.split(benchmark["command"]), client_env - ) - - -def main() -> None: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("recipe") - parser.add_argument("output", type=Path) - args = parser.parse_args() - server, client = prepare(args.recipe, os.environ) - (args.output / "srt-docker-server.sh").write_text(server) - (args.output / "srt-docker-client.sh").write_text(client) - - -if __name__ == "__main__": - main() diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 17ebc87e77..076d5275af 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8847,3 +8847,76 @@ failed-startup log followers before exit - 去除 AMD 原生 shell 前置命令的尾部换行,并在 accounting 不可用时通过 Slurm 控制器核验已完成的资源分配;启动失败时等待日志跟随进程退出 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - qwen3.5-fp4-rtx6000pro-sglang + - qwen3.5-fp4-rtx6000pro-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - Exclude the Docker-only RTX PRO 6000 configs from the native SRT migration and + retain their existing Docker execution; remove the proposed Docker adapter and + recipes from this PR + - 将仅使用 Docker 的 RTX PRO 6000 配置排除在原生 SRT 迁移之外,保留现有 Docker 执行路径;从本 PR 中移除拟新增的 Docker + 适配器与配方 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-mi355x-sglang + - dsr1-fp4-mi355x-sglang-mtp + - dsr1-fp4-mi355x-atom + - dsr1-fp4-mi355x-atom-mtp + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - dsr1-fp8-mi355x-sglang + - dsr1-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi355x-atom + - qwen3.5-fp8-mi355x-atom-mtp + - qwen3.5-fp4-mi355x-sglang + - qwen3.5-fp4-mi355x-atom + - qwen3.5-fp4-mi355x-sglang-mtp + - qwen3.5-fp8-mi300x-sglang + - dsr1-fp8-mi355x-atom + - dsr1-fp8-mi355x-atom-mtp + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + - dsr1-fp4-b200-sglang + - dsr1-fp4-b200-sglang-mtp + - dsr1-fp4-b300-sglang + - dsr1-fp4-b200-trt + - dsr1-fp4-b200-trt-mtp + - dsr1-fp8-b200-sglang + - dsr1-fp8-b300-sglang + - qwen3.5-fp8-b200-sglang + - qwen3.5-fp4-b200-sglang + - qwen3.5-fp4-b200-sglang-mtp + - qwen3.5-fp8-b200-sglang-mtp + - qwen3.5-fp8-b300-sglang-mtp + - qwen3.5-fp8-b300-sglang + - qwen3.5-fp4-b300-sglang + - qwen3.5-fp4-b300-sglang-mtp + - dsr1-fp8-b200-sglang-mtp + - dsr1-fp8-b300-sglang-mtp + - dsr1-fp8-b200-trt + - dsr1-fp8-b200-trt-mtp + - dsr1-fp8-h200-sglang + - dsr1-fp8-h200-sglang-mtp + - qwen3.5-fp8-h200-sglang + - qwen3.5-fp8-h200-sglang-mtp + - dsr1-fp8-h200-trt + - dsr1-fp8-h200-trt-mtp + - qwen3.5-fp8-h100-sglang + - qwen3.5-fp8-h100-sglang-mtp + - qwen3.5-fp4-b200-trt + - qwen3.5-fp4-b200-trt-mtp + scenario-type: + - fixed-seq-len + description: + - Require native SRT recipes for active Slurm single-node fixed-sequence jobs and + remove their superseded Bash implementations; keep AgentX, multi-node, and explicitly + selected SPEED-Bench collector paths separate + - 活跃的 Slurm 单节点定长作业必须提供原生 SRT 配方,并移除已替代的 Bash 实现;AgentX、多节点和显式选择的 SPEED-Bench 采集器保留独立执行路径 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/runners/launch_b200-cw.sh b/runners/launch_b200-cw.sh index 576f2ef026..655599e755 100644 --- a/runners/launch_b200-cw.sh +++ b/runners/launch_b200-cw.sh @@ -1,10 +1,13 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC -EXECUTION_PATH=legacy-single-node -if [[ "$IS_MULTINODE" != true && -n "${SRT_RECIPE:-}" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi if [[ "$EXECUTION_PATH" == native-single-node ]]; then diff --git a/runners/launch_b200-nb.sh b/runners/launch_b200-nb.sh index 92f1e62e69..419af005de 100644 --- a/runners/launch_b200-nb.sh +++ b/runners/launch_b200-nb.sh @@ -1,10 +1,13 @@ #!/usr/bin/bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC -EXECUTION_PATH=legacy-single-node -if [[ "$IS_MULTINODE" != true && -n "${SRT_RECIPE:-}" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi if [[ "$EXECUTION_PATH" == native-single-node ]]; then diff --git a/runners/launch_b200-nscale-slurm.sh b/runners/launch_b200-nscale-slurm.sh index ddbb491e66..f010f76765 100755 --- a/runners/launch_b200-nscale-slurm.sh +++ b/runners/launch_b200-nscale-slurm.sh @@ -4,13 +4,14 @@ # # The reusable workflows run runners/launch_${RUNNER_NAME%%_*}.sh, so every # b200-nscale-slurm_* runner enters here and this is the pool's only launcher. -# Three execution paths share the file and are selected once, below: +# Execution paths share the file and are selected once, below: # native-srt multi-node lanes whose srt-slurm recipes are maintained # against this cluster (DSV4 / Kimi K3 / GLM-5.2 # FP4 and GLM-5.1 FP8 TileRT) # multinode-srt every other multi-node job, through srt-slurm with the # cluster-wide model table -# single-node salloc + srun of the benchmarks/single_node script +# native-single-node fixed-sequence jobs, which require an SRT recipe +# agentic salloc + srun of the existing AgentX script source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 check_env_vars EVAL_ONLY IS_AGENTIC IS_MULTINODE RUN_EVAL # Exported for this pool by runners/runtime_settings.sh. @@ -49,10 +50,11 @@ if uses_native_srt_lane; then LAUNCH_PATH="native-srt" elif [[ "$IS_MULTINODE" == "true" ]]; then LAUNCH_PATH="multinode-srt" -elif [[ -n "${SRT_RECIPE:-}" ]]; then +elif [[ "$IS_AGENTIC" == "0" ]]; then + check_env_vars SRT_RECIPE LAUNCH_PATH="native-single-node" else - LAUNCH_PATH="single-node" + LAUNCH_PATH="agentic" fi echo "B200 Nscale launch path: $LAUNCH_PATH" @@ -765,10 +767,10 @@ run_multinode_srt() { } # --------------------------------------------------------------------------- -# single-node: salloc + srun of the benchmarks/single_node script +# agentic: salloc + srun of the existing AgentX script # --------------------------------------------------------------------------- -run_single_node() { +run_agentic() { # The runner lease reserves the Slurm nodes before this single-node job is # submitted to the Nscale batch_1 partition. check_env_vars SALLOC_TIME_LIMIT GPU_COUNT @@ -845,5 +847,5 @@ run_single_node() { case "$LAUNCH_PATH" in native-srt) run_native_srt_lane ;; multinode-srt) run_multinode_srt ;; - single-node) run_single_node ;; + agentic) run_agentic ;; esac diff --git a/runners/launch_b300-dsxe.sh b/runners/launch_b300-dsxe.sh index a841fa2fcb..e5bf987eb7 100755 --- a/runners/launch_b300-dsxe.sh +++ b/runners/launch_b300-dsxe.sh @@ -8,7 +8,7 @@ source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 # B300 DSXE Slurm cluster (dsxe-sa-b300-prd0); runners run as sa-gha-runner. # Cluster-specific facts live in this block. Multi-node jobs go through -# srt-slurm/srtctl, single-node jobs through salloc + pyxis. +# srt-slurm/srtctl; AgentX and explicit collector scripts retain salloc + pyxis. SLURM_PARTITION="batch_1" SLURM_ACCOUNT="benchmark" @@ -84,10 +84,14 @@ import_squash_image() { test -r "$sqsh" || { echo "Error: squash file not readable: $sqsh" >&2; exit 1; } } -EXECUTION_PATH=legacy-single-node +EXECUTION_PATH=agentic if [[ "$IS_MULTINODE" == true ]]; then EXECUTION_PATH=multinode -elif [[ -n "${SRT_RECIPE:-}" ]]; then +elif [[ -n "${BENCH_SCRIPT_OVERRIDE:-}" ]]; then + # SPEED-Bench collectors explicitly supply their script outside this migration. + EXECUTION_PATH=script +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi diff --git a/runners/launch_h100-cw.sh b/runners/launch_h100-cw.sh index 7252f11362..693211a505 100644 --- a/runners/launch_h100-cw.sh +++ b/runners/launch_h100-cw.sh @@ -1,10 +1,13 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC -EXECUTION_PATH=legacy-single-node -if [[ "$IS_MULTINODE" != true && -n "${SRT_RECIPE:-}" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi if [[ "$EXECUTION_PATH" == native-single-node ]]; then diff --git a/runners/launch_h100-dgxc-slurm.sh b/runners/launch_h100-dgxc-slurm.sh index 8040a9ea0e..6571df5224 100644 --- a/runners/launch_h100-dgxc-slurm.sh +++ b/runners/launch_h100-dgxc-slurm.sh @@ -1,7 +1,7 @@ #!/usr/bin/bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars EVAL_ONLY IS_MULTINODE RUN_EVAL SALLOC_TIME_LIMIT +check_env_vars EVAL_ONLY IS_MULTINODE RUN_EVAL SALLOC_TIME_LIMIT IS_AGENTIC set -e # shellcheck source=runners/slurm_utils.sh @@ -14,10 +14,11 @@ SPEC_SUFFIX=$([[ "$SPEC_DECODING" == "mtp" ]] && printf '_mtp' || printf '') set -x -EXECUTION_PATH=legacy-single-node +EXECUTION_PATH=agentic if [[ "$IS_MULTINODE" == true ]]; then EXECUTION_PATH=multinode -elif [[ -n "${SRT_RECIPE:-}" ]]; then +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi diff --git a/runners/launch_h200-cw.sh b/runners/launch_h200-cw.sh index ff91570102..d101457309 100644 --- a/runners/launch_h200-cw.sh +++ b/runners/launch_h200-cw.sh @@ -1,10 +1,13 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC -EXECUTION_PATH=legacy-single-node -if [[ "$IS_MULTINODE" != true && -n "${SRT_RECIPE:-}" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi if [[ "$EXECUTION_PATH" == native-single-node ]]; then diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index c7f6127711..2d287115a7 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -1,7 +1,7 @@ #!/usr/bin/bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars EVAL_ONLY IS_MULTINODE REQUIRE_POWER RUN_EVAL SALLOC_TIME_LIMIT +check_env_vars EVAL_ONLY IS_MULTINODE REQUIRE_POWER RUN_EVAL SALLOC_TIME_LIMIT IS_AGENTIC set -eo pipefail SLURM_PARTITION="main" @@ -15,10 +15,11 @@ set -x source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 -EXECUTION_PATH=legacy-single-node +EXECUTION_PATH=agentic if [[ "$IS_MULTINODE" == true ]]; then EXECUTION_PATH=multinode -elif [[ -n "${SRT_RECIPE:-}" ]]; then +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi diff --git a/runners/launch_mi300x-amd.sh b/runners/launch_mi300x-amd.sh index d1faa67c50..d07899d6fb 100644 --- a/runners/launch_mi300x-amd.sh +++ b/runners/launch_mi300x-amd.sh @@ -1,12 +1,15 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC set -eo pipefail # Select native fixed-sequence execution before the retained AgentX/multi-node paths. -EXECUTION_PATH=legacy -if [[ "$IS_MULTINODE" == false && -n "${SRT_RECIPE:-}" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi if [[ "$EXECUTION_PATH" == native-single-node ]]; then diff --git a/runners/launch_mi325x-amds.sh b/runners/launch_mi325x-amds.sh index f41e3278a9..8063915ea5 100644 --- a/runners/launch_mi325x-amds.sh +++ b/runners/launch_mi325x-amds.sh @@ -1,12 +1,15 @@ #!/usr/bin/env bash source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE +check_env_vars IS_MULTINODE IS_AGENTIC set -eo pipefail # Select native fixed-sequence execution before the retained AgentX/multi-node paths. -EXECUTION_PATH=legacy -if [[ "$IS_MULTINODE" == false && -n "${SRT_RECIPE:-}" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi if [[ "$EXECUTION_PATH" == native-single-node ]]; then diff --git a/runners/launch_mi355x-amds.sh b/runners/launch_mi355x-amds.sh index 135db385bc..b476f28994 100644 --- a/runners/launch_mi355x-amds.sh +++ b/runners/launch_mi355x-amds.sh @@ -4,8 +4,11 @@ source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validat check_env_vars EVAL_ONLY IS_AGENTIC IS_MULTINODE KEEP_LOGS RUN_EVAL # Select native fixed-sequence execution before the retained AgentX/multi-node paths. -EXECUTION_PATH=legacy -if [[ "$IS_MULTINODE" == false && -n "${SRT_RECIPE:-}" ]]; then +EXECUTION_PATH=agentic +if [[ "$IS_MULTINODE" == true ]]; then + EXECUTION_PATH=multinode +elif [[ "$IS_AGENTIC" == 0 ]]; then + check_env_vars SRT_RECIPE EXECUTION_PATH=native-single-node fi if [[ "$EXECUTION_PATH" == native-single-node ]]; then diff --git a/runners/launch_rtx6000pro-lat.sh b/runners/launch_rtx6000pro-lat.sh index 4bf36a9134..32e244b88c 100755 --- a/runners/launch_rtx6000pro-lat.sh +++ b/runners/launch_rtx6000pro-lat.sh @@ -20,29 +20,6 @@ check_env_vars NCCL_IB_DISABLE : "${EXP_NAME:?EXP_NAME must be set}" : "${PRECISION:?PRECISION must be set}" -EXECUTION_PATH=legacy -if [[ -n "${SRT_RECIPE:-}" ]]; then - EXECUTION_PATH=native-single-node -fi -NATIVE_ARGS=() -if [[ "$EXECUTION_PATH" == native-single-node ]]; then - source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh" || exit 1 - SRTCTL_ROOT=$(mktemp -d "$GITHUB_WORKSPACE/srt-docker.XXXXXX") - setup_srt_slurm "$SRTCTL_ROOT" "$FRAMEWORK" 0 - if ! command -v uv >/dev/null; then - curl -LsSf https://astral.sh/uv/install.sh | sh - source "$HOME/.local/bin/env" - fi - uv venv .venv - source .venv/bin/activate - uv pip install -e . - PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" \ - python3 -m infx.srt_slurm.docker "$GITHUB_WORKSPACE/$SRT_RECIPE" "$GITHUB_WORKSPACE" - NATIVE_ARGS=(--volume "$GITHUB_WORKSPACE:/infmax-workspace" --volume "$GITHUB_WORKSPACE:/logs" - --env "MODEL_NAME=$MODEL") - cd "$GITHUB_WORKSPACE" -fi - mkdir -p "$HF_HUB_CACHE_MOUNT" check_env_vars GPU_COUNT @@ -70,9 +47,7 @@ SCENARIO_SUBDIR="${SCENARIO_SUBDIR%/}/" # untagged name for scripts not yet retagged. BENCH_BASE="benchmarks/single_node/${SCENARIO_SUBDIR}${EXP_NAME%%_*}_${PRECISION}_rtx6000pro" BENCH_SCRIPT="${BENCH_BASE}_${FRAMEWORK:-}${SPEC_SUFFIX}.sh" -if [[ "$EXECUTION_PATH" == native-single-node ]]; then - BENCH_SCRIPT=benchmarks/single_node/srt_docker.sh -elif [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then +if [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then BENCH_SCRIPT="${BENCH_BASE}${SPEC_SUFFIX}.sh" fi @@ -102,7 +77,6 @@ done docker run \ "${RUNTIME_ENV_ARGS[@]}" \ - "${NATIVE_ARGS[@]}" \ --env IS_MULTINODE \ --env REQUIRE_POWER \ --env INFMAX_CONTAINER_WORKSPACE \ diff --git a/utils/test_srt_docker.py b/utils/test_srt_docker.py deleted file mode 100644 index ffaf4aec77..0000000000 --- a/utils/test_srt_docker.py +++ /dev/null @@ -1,181 +0,0 @@ -"""Execute native command generation and the Docker server/client lifecycle.""" - -import json -import os -import shutil -import subprocess -import sys -import time -from pathlib import Path - -import pytest -import yaml - -ROOT = Path(__file__).resolve().parents[1] -sys.path.insert(0, str(ROOT / "utils/srt-slurm/src")) -from infx.srt_slurm.docker import prepare - - -@pytest.mark.parametrize("local_model,eval_only,context", [("", "false", "256"), ("/cache/model with space", "true", "1024")]) -def test_native_docker_commands_preserve_model_flags_and_literal_environment(tmp_path, local_model, eval_only, context): - recipe = { - "schema": 2, "name": "test", "engine": "sglang", - "model": {"path": "hf:test/model", "container": "test:tag", "precision": "fp8"}, - "resources": {"gpu_type": "h200", "gpus_per_node": 8}, - "frontend": {"type": "sglang", "enable_multiple_frontends": False}, - "roles": {"agg": {"nodes": 1, "workers": 1, "gpus": 2, - "args": {"tensor-parallel-size": 2, "context-length": 256, "served-model-name": "test/model"}, - "env": {"SGLANG_LITERAL": "$(touch injected); 'literal'", "SGLANG_SIMULATE_ACC_LEN": "2.5"}}}, - "benchmark": {"type": "custom", "command": "python3 capture-client", - "env": {"MODEL": "test/model", "ISL": "128", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "USE_CHAT_TEMPLATE": "false"}}, - } - path = tmp_path / "recipe.yaml" - path.write_text(yaml.safe_dump(recipe)) - env = { - "FRAMEWORK": "sglang", "MODEL": "test/model", "MODEL_PREFIX": "test", "MODEL_PATH": local_model, - "IMAGE": "test:tag", "PRECISION": "fp8", "TP": "2", "GPU_COUNT": "2", - "PP_SIZE": "1", "DCP_SIZE": "1", "PCP_SIZE": "1", "EP_SIZE": "1", "DP_ATTENTION": "false", - "SPEC_DECODING": "none", "IS_AGENTIC": "0", "RUN_EVAL": "false", "EVAL_ONLY": eval_only, - "MAX_MODEL_LEN": "1024", "ISL": "128", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "CONC": "3", - "RESULT_FILENAME": "controlled", "GPU_MONITOR_INTERVAL": "1", "PORT": "9019", - } - server, client = prepare(str(path), env) - stub = tmp_path / "python3" - stub.write_text(f"#!{sys.executable}\nimport json, os, sys\nprint(json.dumps({{'argv': sys.argv[1:], 'env': dict(os.environ)}}))\n") - stub.chmod(0o755) - runtime_env = {**os.environ, "PATH": f"{tmp_path}:{os.environ['PATH']}"} - run = subprocess.run(["bash", "-c", server], cwd=tmp_path, env=runtime_env, capture_output=True, text=True, check=True) - observed = json.loads(run.stdout) - argv = observed["argv"] - assert argv[:2] == ["-m", "sglang.launch_server"] - assert argv[argv.index("--model-path") + 1] == (local_model or "test/model") - assert argv[argv.index("--port") + 1] == "9019" - assert argv[argv.index("--tensor-parallel-size") + 1] == "2" - assert argv[argv.index("--context-length") + 1] == context - assert observed["env"]["SGLANG_LITERAL"] == "$(touch injected); 'literal'" - assert "SGLANG_SIMULATE_ACC_LEN" not in observed["env"] - assert not (tmp_path / "injected").exists() - run = subprocess.run(["bash", "-c", client], cwd=tmp_path, env=runtime_env, capture_output=True, text=True, check=True) - observed = json.loads(run.stdout) - assert observed["argv"] == ["capture-client"] - assert {key: observed["env"][key] for key in ("CONC", "RESULT_FILENAME", "SRT_FRONTEND_HOST", "SRT_FRONTEND_PORT")} == { - "CONC": "3", "RESULT_FILENAME": "controlled", "SRT_FRONTEND_HOST": "127.0.0.1", "SRT_FRONTEND_PORT": "9019", - } - - -@pytest.mark.parametrize("client_exit,ready", [(0, True), (7, True), (0, False)]) -def test_docker_client_failure_and_readiness_clean_up_owned_server(tmp_path, client_exit, ready): - workspace = tmp_path / "repo" - scripts = workspace / "benchmarks/single_node" - scripts.mkdir(parents=True) - shutil.copyfile(ROOT / "benchmarks/benchmark_lib.sh", scripts.parent / "benchmark_lib.sh") - shutil.copyfile(ROOT / "benchmarks/single_node/srt_docker.sh", scripts / "srt_docker.sh") - (workspace / "infx").symlink_to(ROOT / "infx", target_is_directory=True) - (workspace / "srt-docker-server.sh").write_text('echo $$ > "$INFMAX_CONTAINER_WORKSPACE/server.pid"\nexec sleep 120\n' if ready else 'exit 12\n') - (workspace / "srt-docker-client.sh").write_text(f'touch "$INFMAX_CONTAINER_WORKSPACE/client-ran"\nexit {client_exit}\n') - binaries = tmp_path / "bin" - binaries.mkdir() - for name, body in {"hf": 'printf "%s\\n" "$@" > "$INFMAX_CONTAINER_WORKSPACE/download"', "curl": f"exit {0 if ready else 1}", "sleep": "exit 0"}.items(): - # The server uses the real sleep; only readiness retry sleeps are accelerated. - binary = binaries / name - binary.write_text(f"#!/bin/bash\n{body}\n") - binary.chmod(0o755) - if ready: - (workspace / "srt-docker-server.sh").write_text('echo $$ > "$INFMAX_CONTAINER_WORKSPACE/server.pid"\nexec /bin/sleep 120\n') - else: - # A log follower can take time to stop. Detach its output so EOF alone - # cannot hide a missing wait in the wrapper's failed-readiness path. - follower = binaries / "tail" - follower.write_text(f"#!{sys.executable}\n" + - "import os, pathlib, signal, time\n" - "workspace = pathlib.Path(os.environ['INFMAX_CONTAINER_WORKSPACE'])\n" - "null = os.open(os.devnull, os.O_WRONLY)\n" - "os.dup2(null, 1); os.dup2(null, 2); os.close(null)\n" - "def stop(*_):\n" - " time.sleep(0.1)\n" - " (workspace / 'follower-stopped').touch()\n" - " raise SystemExit(0)\n" - "signal.signal(signal.SIGTERM, stop)\n" - "(workspace / 'follower-ready').touch()\n" - "signal.pause()\n") - follower.chmod(0o755) - (binaries / "curl").write_text( - '#!/bin/bash\nwhile [[ ! -e "$INFMAX_CONTAINER_WORKSPACE/follower-ready" ]]; do /bin/sleep 0.01; done\nexit 1\n') - env = {**os.environ, "PATH": f"{binaries}:{Path(sys.executable).parent}:{os.environ['PATH']}", - "INFMAX_CONTAINER_WORKSPACE": str(workspace), "MODEL": "test/model", "PORT": "9019", - "RUN_EVAL": "false", "EVAL_ONLY": "false", "MODEL_PATH": ""} - result = subprocess.run(["bash", str(scripts / "srt_docker.sh")], env=env, capture_output=True, text=True, timeout=15) - assert result.returncode == (client_exit if ready else 1), result.stdout + result.stderr - assert (workspace / "client-ran").exists() is ready - assert (workspace / "download").read_text() == "download\ntest/model\n" - if not ready: - assert (workspace / "follower-stopped").exists() - if ready: - pid = int((workspace / "server.pid").read_text()) - for _ in range(100): - try: - os.kill(pid, 0) - except ProcessLookupError: - break - time.sleep(0.01) - else: - pytest.fail("Docker wrapper left its server process alive") - - -def test_rtx_launcher_binds_eval_model_and_preserves_container_failure(tmp_path): - binaries = tmp_path / "bin" - binaries.mkdir() - (tmp_path / "benchmarks").symlink_to(ROOT / "benchmarks", target_is_directory=True) - path = tmp_path / "recipe.yaml" - path.write_text(yaml.safe_dump({ - "schema": 2, "name": "fixture", "engine": "sglang", - "model": {"path": "hf:test/model", "container": "test:tag", "precision": "fp4"}, - "resources": {"gpu_type": "rtx6000pro", "gpus_per_node": 8}, - "frontend": {"type": "sglang", "enable_multiple_frontends": False}, - "roles": {"agg": {"nodes": 1, "workers": 1, "gpus": 4, - "args": {"tensor-parallel-size": 4, "served-model-name": "test/model"}}}, - "benchmark": {"type": "custom", "command": "bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh", - "env": {"MODEL": "test/model", "ISL": "128", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "USE_CHAT_TEMPLATE": "false"}}, - })) - scripts = { - "git": 'if [[ "$1" == clone ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', - "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', - } - for name, body in scripts.items(): - binary = binaries / name - binary.write_text(f"#!/usr/bin/env bash\n{body}\n") - binary.chmod(0o755) - docker = binaries / "docker" - docker.write_text(f"#!{sys.executable}\n" + - "import json, os, pathlib, sys\n" - "with pathlib.Path(os.environ['CAPTURE']).open('a') as f: f.write(json.dumps(sys.argv[1:])+'\\n')\n" - "sys.exit(7 if sys.argv[1] == 'run' else 0)\n") - docker.chmod(0o755) - env = {**os.environ, "PATH": f"{binaries}:{Path(sys.executable).parent}:{os.environ['PATH']}", - "PYTHONPATH": f"{ROOT}:{ROOT / 'utils/srt-slurm/src'}", "GITHUB_WORKSPACE": str(tmp_path), - "SRT_RECIPE": path.name, "FRAMEWORK": "sglang", "MODEL": "test/model", "MODEL_PREFIX": "test", - "IMAGE": "test:tag", "PRECISION": "fp4", "TP": "4", "GPU_COUNT": "4", "PP_SIZE": "1", - "DCP_SIZE": "1", "PCP_SIZE": "1", "EP_SIZE": "1", "DP_ATTENTION": "false", "SPEC_DECODING": "none", - "IS_AGENTIC": "0", "RUN_EVAL": "true", "EVAL_ONLY": "true", "MAX_MODEL_LEN": "1024", - "ISL": "128", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "CONC": "3", "RESULT_FILENAME": "point", - "GPU_MONITOR_INTERVAL": "1", "PORT": "9019", "IS_MULTINODE": "false", "HF_HUB_CACHE_MOUNT": str(tmp_path / 'cache'), - "HF_HUB_CACHE": "/hf", "NCCL_IB_DISABLE": "1", "EXP_NAME": "test_8k1k", "SCENARIO_SUBDIR": "fixed_seq_len/", - "RUNNER_NAME": "fixture_00", "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "REQUIRE_POWER": "1", - "CAPTURE": str(tmp_path / "docker.jsonl"), "MODEL_PATH": ""} - result = subprocess.run(["bash", str(ROOT / 'runners/launch_rtx6000pro-lat.sh')], cwd=tmp_path, - env=env, capture_output=True, text=True, timeout=30) - assert result.returncode == 7, result.stdout + result.stderr - calls = [json.loads(line) for line in Path(env["CAPTURE"]).read_text().splitlines()] - run = next(call for call in calls if call[0] == 'run') - assert "MODEL_NAME=test/model" in [run[i+1] for i,v in enumerate(run[:-1]) if v == '--env'] - assert run[-2:] == ["test:tag", "benchmarks/single_node/srt_docker.sh"] - assert calls[-1] == ["rm", "-f", "bmk-server-fixture_00"] - # Run the emitted script against an external Python stub to verify the - # launcher's actual CLI path emitted the eval context and requested model. - stub = binaries / "python3" - stub.write_text(f"#!{sys.executable}\nimport json,sys\nprint(json.dumps(sys.argv[1:]))\n") - stub.chmod(0o755) - observed = subprocess.run(["bash", str(tmp_path / "srt-docker-server.sh")], env=env, capture_output=True, text=True, check=True) - argv = json.loads(observed.stdout) - assert argv[argv.index('--context-length')+1] == '1024' - assert argv[argv.index('--served-model-name')+1] == 'test/model' diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 23f2436095..c5038d36f0 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -254,9 +254,14 @@ def test_submission_manifest(tmp_path, record, expected): ("h200-dgxc-slurm", "submission"), ("h200-dgxc-slurm", "bootstrap"), ("h200-cw", "none"), ("h100-cw", "none"), ("h100-dgxc-slurm", "none"), ("b200-cw", "none"), ("b200-nb", "none"), ("b200-nscale-slurm", "none"), + ("b200-nscale-slurm", "agentic"), ("b300-dsxe", "none"), ("mi300x-amd", "none"), ("mi325x-amds", "none"), ("mi355x-amds", "none"), -]) +] + [(pool, "missing-recipe") for pool in ( + "b200-cw", "b200-nb", "b200-nscale-slurm", "b300-dsxe", "h100-cw", + "h100-dgxc-slurm", "h200-cw", "h200-dgxc-slurm", "mi300x-amd", + "mi325x-amds", "mi355x-amds", +)]) def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, pool, failure): path, _, point_env = point binaries = tmp_path / "bin" @@ -272,7 +277,8 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "git": 'if [[ "$1" == clone ]]; then mkdir -p "${@: -1}/configs"; else echo test-commit; fi', "uv": 'if [[ "$1" == venv ]]; then mkdir -p .venv/bin; echo ":" > .venv/bin/activate; fi', "make": '[[ "$TEST_FAILURE" == bootstrap ]] && exit 13; mkdir -p bin; touch bin/uv', - "squeue": '[[ "$TEST_FAILURE" == submission ]] && echo "42 RUNNING"; exit 0', + "squeue": '[[ "$TEST_FAILURE" == submission || "$TEST_FAILURE" == agentic ]] && echo "42"; exit 0', + "salloc": 'echo "Granted job allocation 42"', "sacct": 'if [[ "$TEST_FAILURE" == allocation ]]; then echo "FAILED|1:0"; else echo "COMPLETED|0:0"; fi', "scancel": 'printf "%s\\n" "$@" >> "$CANCEL_CAPTURE"', "tail": 'exit 0', @@ -297,6 +303,11 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "sys.exit(7 if os.environ['TEST_FAILURE'] == 'submission' else 0)\n" ) srtctl.chmod(0o755) + srun = binaries / "srun" + srun.write_text(f"#!{sys.executable}\n" + + "import json, os, pathlib, sys\n" + "with pathlib.Path(os.environ['SRUN_CAPTURE']).open('a') as f: f.write(json.dumps(sys.argv[1:])+'\\n')\n") + srun.chmod(0o755) env = { **os.environ, **point_env, "PATH": f"{binaries}:{Path(sys.executable).parent}:{os.environ['PATH']}", @@ -310,15 +321,34 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, "B300_HF_CACHE_CONTAINER_DIR": "/hf", "ENROOT_IMPORT_TIME_LIMIT": "10", "INFERENCEX_RUNTIME_ENV_VARS": "REQUIRE_POWER", "TEST_FAILURE": failure, "CANCEL_CAPTURE": str(capture), + "SRUN_CAPTURE": str(tmp_path / "srun.jsonl"), "KEEP_LOGS": "0", } env.pop("AIPERF_DRAIN_TIMEOUT_SECONDS", None) env.pop("AIPERF_DRAIN_POLL_SECONDS", None) + env.pop("BENCH_SCRIPT_OVERRIDE", None) + if failure == "missing-recipe": + env.pop("SRT_RECIPE") + if failure == "agentic": + env.update(IS_AGENTIC="1", SCENARIO_SUBDIR="agentic/", EXP_NAME="fixture_agentic", + RUNNER_NAME="fixture_00", SRT_RECIPE="unused.yaml") result = subprocess.run( ["bash", str(ROOT / f"runners/launch_{pool}.sh")], cwd=tmp_path, env=env, capture_output=True, text=True, timeout=30, ) - assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13}[failure], result.stderr + assert result.returncode == {"none": 0, "allocation": 1, "submission": 7, "bootstrap": 13, "missing-recipe": 1, "agentic": 0}[failure], result.stderr + if failure == "agentic": + calls = [json.loads(line) for line in Path(env["SRUN_CAPTURE"]).read_text().splitlines()] + assert calls[-1][-2:] == ["bash", "benchmarks/single_node/agentic/fixture_fp8_b200.sh"] + assert "--jobid=42" in calls[-1] + assert not (tmp_path / "srt-single-node-submission.json").exists() + return + if failure == "missing-recipe": + assert "SRT_RECIPE" in result.stdout + assert not (tmp_path / "srt-single-node-submission.json").exists() + assert not (tmp_path / "point-identity.json").exists() + assert not capture.exists() + return if failure == "bootstrap": assert not (tmp_path / "srt-single-node-submission.json").exists() assert not capture.exists() @@ -370,3 +400,47 @@ def test_terminal_allocation_without_accounting(tmp_path, controller, expected): assert result.returncode == expected, result.stdout + result.stderr if expected: assert "ERROR:" in result.stderr + + +@pytest.mark.parametrize("collector", [False, True]) +def test_b300_keeps_agentic_and_explicit_collector_dispatch(tmp_path, collector): + binaries = tmp_path / "bin" + binaries.mkdir() + for name, body in { + # Host image-cache directories and Slurm operations are external here. + "mkdir": "exit 0", + "unsquashfs": "exit 0", + "salloc": 'echo "Granted job allocation 42"', + "scancel": 'printf "%s\\n" "$@" > "$CANCEL_CAPTURE"', + }.items(): + binary = binaries / name + binary.write_text(f"#!/usr/bin/env bash\n{body}\n") + binary.chmod(0o755) + srun = binaries / "srun" + srun.write_text(f"#!{sys.executable}\n" + + "import json, os, pathlib, sys\n" + "with pathlib.Path(os.environ['SRUN_CAPTURE']).open('a') as f: f.write(json.dumps(sys.argv[1:])+'\\n')\n" + "sys.exit(7 if '--container-image' in ' '.join(sys.argv) else 0)\n") + srun.chmod(0o755) + env = {**os.environ, "PATH": f"{binaries}:{os.environ['PATH']}", + "GITHUB_WORKSPACE": str(tmp_path), "B300_HF_CACHE_HOST_DIR": str(tmp_path / "cache"), + "B300_HF_CACHE_CONTAINER_DIR": "/cache", "RUNNER_NAME": "fixture_00", + "ENROOT_IMPORT_TIME_LIMIT": "10", "SALLOC_TIME_LIMIT": "10", "IS_MULTINODE": "false", + "IS_AGENTIC": "0" if collector else "1", "EVAL_ONLY": "false", "RUN_EVAL": "false", + "MODEL": "test/DeepSeek-V4-Pro", "MODEL_PREFIX": "fixture", "PRECISION": "fp4", + "FRAMEWORK": "vllm", "EXP_NAME": "fixture_workload", "IMAGE": "fixture:tag", + "SPEC_DECODING": "none", "GPU_COUNT": "4", + "SCENARIO_SUBDIR": "fixed_seq_len/" if collector else "agentic/", + "SRUN_CAPTURE": str(tmp_path / "srun.jsonl"), "CANCEL_CAPTURE": str(tmp_path / "cancelled")} + env.pop("SRT_RECIPE", None) + env.pop("BENCH_SCRIPT_OVERRIDE", None) + if collector: + env["BENCH_SCRIPT_OVERRIDE"] = "benchmarks/single_node/speedbench/fixture.py" + result = subprocess.run(["bash", str(ROOT / "runners/launch_b300-dsxe.sh")], cwd=tmp_path, + env=env, capture_output=True, text=True, timeout=10) + assert result.returncode == 7, result.stdout + result.stderr + calls = [json.loads(line) for line in Path(env["SRUN_CAPTURE"]).read_text().splitlines()] + expected = "benchmarks/single_node/speedbench/fixture.py" if collector else "benchmarks/single_node/agentic/fixture_fp4_b300.sh" + assert calls[-1][-2:] == ["bash", expected] + assert "--jobid=42" in calls[-1] + assert (tmp_path / "cancelled").read_text() == "42\n" From 8f3f0d5514adbadc7dd712bf96e0af706accdb42 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:28:40 -0500 Subject: [PATCH 18/29] refactor: make single-node fixed-sequence coverage SRT-only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 归档仅支持 Docker 的 RTX 定长配置及脚本,移除无调用方的 runner 路由;保留历史 changelog 条目,但不为已归档的精确配置键生成作业。 --- MODELS.md | 2 + MODELS_zh.md | 2 + .../qwen3.5_fp4_rtx6000pro_sglang.sh | 0 .../qwen3.5_fp4_rtx6000pro_sglang_mtp.sh | 0 configs/deprecated/nvidia-master.yaml | 37 ++++ configs/nvidia-master.yaml | 36 ---- configs/runners.yaml | 9 - docs/ci-procedures.md | 2 + docs/ci-procedures_zh.md | 2 + infx/matrix/plan.py | 31 +++- perf-changelog.yaml | 12 ++ runners/launch_rtx6000pro-lat.sh | 161 ------------------ runners/runtime_settings.sh | 4 - utils/test_process_changelog.py | 33 ++++ 14 files changed, 118 insertions(+), 213 deletions(-) rename benchmarks/single_node/fixed_seq_len/{ => deprecated}/qwen3.5_fp4_rtx6000pro_sglang.sh (100%) rename benchmarks/single_node/fixed_seq_len/{ => deprecated}/qwen3.5_fp4_rtx6000pro_sglang_mtp.sh (100%) delete mode 100755 runners/launch_rtx6000pro-lat.sh diff --git a/MODELS.md b/MODELS.md index f845453d54..fc4aabd9f7 100644 --- a/MODELS.md +++ b/MODELS.md @@ -58,6 +58,8 @@ Rationale: `dsv4` carries the largest single-turn footprint in the repository. 4 **Deprecation parity audit (2026-09-21):** Active master configs and benchmark-script locations match the enacted retirements above and in the support matrix below. GLM-5.1 B200 TileRT remains the documented exception to the earlier GLM-5/5.1 and 1k1k retirements. Conditional A/B baseline retirement remains pending; non-speculative Pareto contributors remain supported. The broader routing audit also removed stale retired-model branches from launchers/runtime settings and a GLM-5-only environment override, and corrected workflow/agent guidance that still recommended retired coverage. SPEED-Bench collectors, historical result readers, and the explicitly retained recipe YAMLs remain available. Deprecated configs are consolidated in [`configs/deprecated/amd-master.yaml`](configs/deprecated/amd-master.yaml) and [`configs/deprecated/nvidia-master.yaml`](configs/deprecated/nvidia-master.yaml). +**Single-node SRT-only cutover (2026-09-22):** Active single-node fixed-sequence recipes now use SRT-Slurm. The two Docker-only Qwen3.5 RTX PRO 6000 FP4 configs (with and without MTP) are retired, with their original settings preserved in `configs/deprecated/nvidia-master.yaml` and their scripts in `benchmarks/single_node/fixed_seq_len/deprecated/`. The unused `rtx6000pro-lat` runner mappings, launcher, and runtime settings are removed. Qwen3.5 remains active on the other supported Slurm pools; AgentX and multi-node coverage are unchanged. + ## Scenarios | Scenario | ISL/OSL | Status | diff --git a/MODELS_zh.md b/MODELS_zh.md index 8215a8de98..ee263f4a4c 100644 --- a/MODELS_zh.md +++ b/MODELS_zh.md @@ -58,6 +58,8 @@ InferenceX-e2e 运行在数量固定且有限的 GPU 资源池上,并由一支 **弃用状态一致性核查(2026-09-21):** 启用的主配置及基准测试脚本位置与上述已执行的退役事项和下方支持矩阵一致。GLM-5.1 B200 TileRT 仍是文档明确保留的例外,不受此前 GLM-5/5.1 和 1k1k 退役范围限制。有条件的 A/B 基线退役仍待执行;对 Pareto 前沿有贡献的非投机解码配置继续受支持。进一步的路由核查还移除了启动器和运行时设置中遗留的退役模型分支及 GLM-5 专用环境覆盖,并修正了仍推荐退役配置的工作流和智能体指南。SPEED-Bench 采集器、历史结果读取逻辑及明确保留的配方 YAML 继续保留。弃用配置现统一归档至 [`configs/deprecated/amd-master.yaml`](configs/deprecated/amd-master.yaml) 和 [`configs/deprecated/nvidia-master.yaml`](configs/deprecated/nvidia-master.yaml)。 +**单节点切换为仅使用 SRT(2026-09-22):** 活跃的单节点定长配方现统一使用 SRT-Slurm。两个仅支持 Docker 的 Qwen3.5 RTX PRO 6000 FP4 配置(启用和关闭 MTP)已退役,原始设置保留在 `configs/deprecated/nvidia-master.yaml`,脚本保留在 `benchmarks/single_node/fixed_seq_len/deprecated/`。已移除不再使用的 `rtx6000pro-lat` runner 映射、启动器及运行时设置。Qwen3.5 在其他受支持的 Slurm 池上继续启用;AgentX 和多节点覆盖保持不变。 + ## 场景 | 场景 | ISL/OSL | 状态 | diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_rtx6000pro_sglang.sh b/benchmarks/single_node/fixed_seq_len/deprecated/qwen3.5_fp4_rtx6000pro_sglang.sh similarity index 100% rename from benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_rtx6000pro_sglang.sh rename to benchmarks/single_node/fixed_seq_len/deprecated/qwen3.5_fp4_rtx6000pro_sglang.sh diff --git a/benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_rtx6000pro_sglang_mtp.sh b/benchmarks/single_node/fixed_seq_len/deprecated/qwen3.5_fp4_rtx6000pro_sglang_mtp.sh similarity index 100% rename from benchmarks/single_node/fixed_seq_len/qwen3.5_fp4_rtx6000pro_sglang_mtp.sh rename to benchmarks/single_node/fixed_seq_len/deprecated/qwen3.5_fp4_rtx6000pro_sglang_mtp.sh diff --git a/configs/deprecated/nvidia-master.yaml b/configs/deprecated/nvidia-master.yaml index a72715dd2f..9029f6235a 100644 --- a/configs/deprecated/nvidia-master.yaml +++ b/configs/deprecated/nvidia-master.yaml @@ -16549,3 +16549,40 @@ qwen3.5-bf16-b300-sglang-mtp: search-space: - { tp: 8, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } - { tp: 4, ep: 1, conc-start: 4, conc-end: 64, spec-decoding: mtp } + +# Retired on 2026-09-22: Docker fixed-sequence coverage is outside the SRT-only cutover. +# Qwen3.5-397B-A17B NVFP4 single-node SGLang sweep using 4 of 8 RTX PRO +# 6000 Blackwell GPUs. Both arms use ordinary NCCL collectives on PCIe. +qwen3.5-fp4-rtx6000pro-sglang: + image: lmsysorg/sglang:v0.5.16-cu130 + model: nvidia/Qwen3.5-397B-A17B-NVFP4 + model-prefix: qwen3.5 + runner: rtx6000pro-lat + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 4, conc-list: [1, 4, 16, 64] } + - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64] } + +# Same sweep with the built-in MTP draft head driven through SGLang's EAGLE +# speculative path. +qwen3.5-fp4-rtx6000pro-sglang-mtp: + image: lmsysorg/sglang:v0.5.16-cu130 + model: nvidia/Qwen3.5-397B-A17B-NVFP4 + model-prefix: qwen3.5 + runner: rtx6000pro-lat + precision: fp4 + framework: sglang + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } + - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 4cd00ec44a..ad28da4f4a 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -1340,42 +1340,6 @@ qwen3.5-fp4-b300-sglang: conc-end: 128 srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/b300-fp4/8k1k.yaml -# Qwen3.5-397B-A17B NVFP4 single-node SGLang sweep using 4 of 8 RTX PRO -# 6000 Blackwell GPUs. Both arms use ordinary NCCL collectives on PCIe. -qwen3.5-fp4-rtx6000pro-sglang: - image: lmsysorg/sglang:v0.5.16-cu130 - model: nvidia/Qwen3.5-397B-A17B-NVFP4 - model-prefix: qwen3.5 - runner: rtx6000pro-lat - precision: fp4 - framework: sglang - multinode: false - scenarios: - fixed-seq-len: - - isl: 8192 - osl: 1024 - search-space: - - { tp: 4, conc-list: [1, 4, 16, 64] } - - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64] } - -# Same sweep with the built-in MTP draft head driven through SGLang's EAGLE -# speculative path. -qwen3.5-fp4-rtx6000pro-sglang-mtp: - image: lmsysorg/sglang:v0.5.16-cu130 - model: nvidia/Qwen3.5-397B-A17B-NVFP4 - model-prefix: qwen3.5 - runner: rtx6000pro-lat - precision: fp4 - framework: sglang - multinode: false - scenarios: - fixed-seq-len: - - isl: 8192 - osl: 1024 - search-space: - - { tp: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } - - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } - qwen3.5-fp4-b300-sglang-mtp: image: lmsysorg/sglang:v0.5.19-cu130 model: nvidia/Qwen3.5-397B-A17B-NVFP4-V2 diff --git a/configs/runners.yaml b/configs/runners.yaml index 2b85b54431..f233788d41 100644 --- a/configs/runners.yaml +++ b/configs/runners.yaml @@ -138,10 +138,6 @@ labels: - gb300-nv_15 - gb300-nv_16 - gb300-nv_17 - rtx6000pro: - - rtx6000pro-lat_00 - rtx6000pro-lat: - - rtx6000pro-lat_00 cluster:h100-cw: - h100-cw_00 - h100-cw_01 @@ -239,8 +235,6 @@ labels: - gb300-nv_15 - gb300-nv_16 - gb300-nv_17 - cluster:rtx6000pro-lat: - - rtx6000pro-lat_00 cluster:mi300x-amd: - mi300x-amd_00 - mi300x-amd_01 @@ -299,9 +293,6 @@ hardware: cluster:gb300-nv: available-cpu-dram-mib: 860_160 gpus-per-node: 4 - cluster:rtx6000pro-lat: - available-cpu-dram-mib: 1_500_000 - gpus-per-node: 8 cluster:mi300x-amd: available-cpu-dram-mib: 1_547_820 gpus-per-node: 8 diff --git a/docs/ci-procedures.md b/docs/ci-procedures.md index 95311bfd20..c16b1025cf 100644 --- a/docs/ci-procedures.md +++ b/docs/ci-procedures.md @@ -180,6 +180,8 @@ uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with p Add `--all-evals` and/or `--evals-only` when those PR modifiers will be active. [`run-sweep.yml`](../.github/workflows/run-sweep.yml) passes the same flags. Never use a formatter to rewrite `perf-changelog.yaml`, and never treat `yaml.safe_load` alone as sufficient changelog validation. +Changelog entries may name configs retired later in the same PR. Exact keys absent from active masters but present in the matching vendor `configs/deprecated/` archive remain in changelog metadata and produce no jobs. Active definitions take precedence over archived versions; wildcards match only active configs, unknown keys still fail, and `append-only` entries cannot retire configs. + ## Manual end-to-end dispatch Use [`e2e-tests.yml`](../.github/workflows/e2e-tests.yml) for a bounded one-off run only after the identical generator command succeeds locally. Make the test name unique. In the common pattern, `--ref main` selects the deployed workflow definition while input `ref` selects the branch or SHA to measure. Setup resolves that ref once and passes its checkout SHA to all eight benchmark/eval routes, covering single-node, multi-node, fixed-sequence, and AgentX jobs. Queued jobs keep that SHA if the branch advances. With no input `ref`, the run uses `github.sha` as before. diff --git a/docs/ci-procedures_zh.md b/docs/ci-procedures_zh.md index fd303217cd..d6ceff99d0 100644 --- a/docs/ci-procedures_zh.md +++ b/docs/ci-procedures_zh.md @@ -172,6 +172,8 @@ uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with p 当 PR 将启用对应修饰标签时,加入 `--all-evals` 和/或 `--evals-only`;[`run-sweep.yml`](../.github/workflows/run-sweep.yml) 会传递相同参数。绝不使用 Formatter 重写 `perf-changelog.yaml`,也绝不能把单独通过 `yaml.safe_load` 当作充分的 Changelog 验证。 +Changelog 条目可能引用同一 PR 后续提交中退役的配置。若精确键名已从活跃主配置移除,但仍存在于对应厂商的 `configs/deprecated/` 归档中,该条目会保留在 Changelog 元数据中,但不生成作业。活跃定义优先于归档版本;通配符仅匹配活跃配置,未知键名仍报错,`append-only` 条目不得退役配置。 + ## 手动端到端派发 仅在完全相同的生成器命令已于本地成功后,才使用 [`e2e-tests.yml`](../.github/workflows/e2e-tests.yml) 执行受限的一次性 Run。测试名称必须唯一。在通用模式中,`--ref main` 选择已部署的 Workflow 定义,输入 `ref` 则选择要测量的 Branch 或 SHA。Setup 只解析一次该 ref,并将 checkout SHA 传给全部八条基准测试和评测路径,覆盖单节点、多节点、固定序列和 AgentX Job。即使分支在排队期间前进,后续 Job 仍使用该 SHA。未提供输入 `ref` 时,仍使用 `github.sha`。 diff --git a/infx/matrix/plan.py b/infx/matrix/plan.py index a964e1984f..66f06e85e5 100644 --- a/infx/matrix/plan.py +++ b/infx/matrix/plan.py @@ -11,7 +11,7 @@ import tempfile import traceback from collections import defaultdict -from collections.abc import Iterator +from collections.abc import Collection, Iterator from contextlib import ExitStack, contextmanager from dataclasses import dataclass from pathlib import Path @@ -90,7 +90,12 @@ def filter_eval_rows_by_prefill_ep(eval_rows: list[dict], min_prefill_ep: int | return kept -def get_config_keys_from_master(config_keys: list[str], master_config: dict) -> list[str]: +def get_config_keys_from_master( + config_keys: list[str], + master_config: dict, + *, + deprecated_config_keys: Collection[str] = (), +) -> list[str]: resolved_keys = {} for key in config_keys: if "*" in key: @@ -103,6 +108,8 @@ def get_config_keys_from_master(config_keys: list[str], master_config: dict) -> for matched_key in matched_keys: resolved_keys.setdefault(matched_key, None) elif key not in master_config: + if key in deprecated_config_keys: + continue raise ValueError(f"Config key '{key}' not found in master configs.") else: resolved_keys.setdefault(key, None) @@ -487,9 +494,27 @@ def generate_current( error, ) from error + # A later commit can retire a config named by an earlier changelog entry. + # Preserve that history without scheduling archives or accepting typos. + # Append-only entries still require every selected config to remain active. + deprecated_keys: set[str] = set() + if not has_append_only and any( + "*" not in key and key not in master_config + for entry in parsed_entries + for key in entry.config_keys + ): + archives = [Path(path).parent / "deprecated" / Path(path).name for path in config_files] + deprecated_keys = set( + load_config_files( + [str(path) for path in archives if path.is_file()], validate=False + ) + ) + resolved_entries = [] for entry in parsed_entries: - all_configs = get_config_keys_from_master(entry.config_keys, master_config) + all_configs = get_config_keys_from_master( + entry.config_keys, master_config, deprecated_config_keys=deprecated_keys + ) resolved_entries.append((entry, all_configs)) base_inputs = None diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 076d5275af..97840db484 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8920,3 +8920,15 @@ selected SPEED-Bench collector paths separate - 活跃的 Slurm 单节点定长作业必须提供原生 SRT 配方,并移除已替代的 Bash 实现;AgentX、多节点和显式选择的 SPEED-Bench 采集器保留独立执行路径 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - qwen3.5-fp4-rtx6000pro-sglang + - qwen3.5-fp4-rtx6000pro-sglang-mtp + scenario-type: + - fixed-seq-len + description: + - Retire the Docker-only RTX PRO 6000 fixed-sequence configs and runner routing + so active single-node fixed-sequence coverage uses SRT-Slurm exclusively; + preserve the original configs and scripts in the existing deprecated archives + - 退役仅支持 Docker 的 RTX PRO 6000 定长配置及 runner 路由,使活跃的单节点定长覆盖仅使用 SRT-Slurm;原始配置和脚本保留在现有弃用归档中 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/runners/launch_rtx6000pro-lat.sh b/runners/launch_rtx6000pro-lat.sh deleted file mode 100755 index 32e244b88c..0000000000 --- a/runners/launch_rtx6000pro-lat.sh +++ /dev/null @@ -1,161 +0,0 @@ -#!/usr/bin/bash - -source "$(dirname "${BASH_SOURCE[0]}")/../benchmarks/benchmark_lib.sh" --validation-only || exit 1 -check_env_vars IS_MULTINODE -set -eo pipefail - -# This runner executes directly on the single RTX PRO 6000 GPU node. Docker -# therefore owns a separate image cache from the node's RKE2/containerd cache. -check_env_vars HF_HUB_CACHE_MOUNT -check_env_vars HF_HUB_CACHE -check_env_vars PORT - -# NCCL 2.28.9 segfaults while probing this node's bnxt_re RDMA devices. -# Disable that RDMA path by default while preserving local CUDA P2P/SHM and -# allowing an explicit caller override. -check_env_vars NCCL_IB_DISABLE - -: "${GITHUB_WORKSPACE:?GITHUB_WORKSPACE must be set}" -: "${IMAGE:?IMAGE must be set}" -: "${EXP_NAME:?EXP_NAME must be set}" -: "${PRECISION:?PRECISION must be set}" - -mkdir -p "$HF_HUB_CACHE_MOUNT" - -check_env_vars GPU_COUNT -if [[ ! "$GPU_COUNT" =~ ^[1-9][0-9]*$ ]]; then - echo "GPU_COUNT must be a positive integer, got: $GPU_COUNT" >&2 - exit 1 -fi - -export CUDA_VISIBLE_DEVICES -CUDA_VISIBLE_DEVICES="$(seq -s, 0 "$((GPU_COUNT - 1))")" - -# Some Slurm/enroot configs spell registry paths as nvcr.io#namespace/image. -# Docker requires the normal slash form. -DOCKER_IMAGE="${IMAGE//#//}" - -SPEC_SUFFIX="" -if [[ "${SPEC_DECODING:-}" == "mtp" ]]; then - SPEC_SUFFIX="_mtp" -fi - -check_env_vars SCENARIO_SUBDIR -SCENARIO_SUBDIR="${SCENARIO_SUBDIR#/}" -SCENARIO_SUBDIR="${SCENARIO_SUBDIR%/}/" -# Prefer a framework-tagged script so engines can coexist; fall back to the -# untagged name for scripts not yet retagged. -BENCH_BASE="benchmarks/single_node/${SCENARIO_SUBDIR}${EXP_NAME%%_*}_${PRECISION}_rtx6000pro" -BENCH_SCRIPT="${BENCH_BASE}_${FRAMEWORK:-}${SPEC_SUFFIX}.sh" -if [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then - BENCH_SCRIPT="${BENCH_BASE}${SPEC_SUFFIX}.sh" -fi - -if [[ ! -f "$GITHUB_WORKSPACE/$BENCH_SCRIPT" ]]; then - echo "Benchmark script not found: $GITHUB_WORKSPACE/$BENCH_SCRIPT" >&2 - exit 1 -fi - -check_env_vars RUNNER_NAME -server_name="bmk-server-${RUNNER_NAME}" -server_name="${server_name//[^a-zA-Z0-9_.-]/-}" - -cleanup() { - docker rm -f "$server_name" >/dev/null 2>&1 || true -} -trap cleanup EXIT - -# Clear a container left behind by a cancelled or interrupted workflow. -cleanup - -check_env_vars INFERENCEX_RUNTIME_ENV_VARS -RUNTIME_ENV_ARGS=() -for runtime_var in $INFERENCEX_RUNTIME_ENV_VARS; do - check_env_vars "$runtime_var" - RUNTIME_ENV_ARGS+=(--env "$runtime_var") -done - -docker run \ - "${RUNTIME_ENV_ARGS[@]}" \ - --env IS_MULTINODE \ - --env REQUIRE_POWER \ - --env INFMAX_CONTAINER_WORKSPACE \ - --env AIPERF_EXPERIMENTAL_FAST \ - --rm \ - --pull=missing \ - --name="$server_name" \ - --runtime=nvidia \ - --gpus="$GPU_COUNT" \ - --network=host \ - --ipc=host \ - --privileged \ - --shm-size=32g \ - --ulimit memlock=-1 \ - --ulimit stack=67108864 \ - --security-opt seccomp=unconfined \ - --cap-add=SYS_PTRACE \ - --volume "$HF_HUB_CACHE_MOUNT:$HF_HUB_CACHE" \ - --volume "$GITHUB_WORKSPACE:/workspace/" \ - --workdir=/workspace/ \ - --env HF_TOKEN \ - --env HF_HUB_CACHE \ - --env MODEL \ - --env MODEL_PREFIX \ - --env MODEL_PATH \ - --env TP \ - --env PP_SIZE \ - --env DCP_SIZE \ - --env PCP_SIZE \ - --env EP_SIZE \ - --env DP_SIZE \ - --env DP_ATTENTION \ - --env GPU_COUNT \ - --env CONC \ - --env MAX_MODEL_LEN \ - --env ISL \ - --env OSL \ - --env FRAMEWORK \ - --env PRECISION \ - --env DISAGG \ - --env SPEC_DECODING \ - --env NUM_SPEC_TOKENS \ - --env RUN_EVAL \ - --env EVAL_ONLY \ - --env EVAL_FRAMEWORK \ - --env EVAL_LIMIT \ - --env EVAL_SUITE \ - --env EVAL_MAX_MODEL_LEN \ - --env RUNNER_TYPE \ - --env RUNNER_NAME \ - --env RESULT_FILENAME \ - --env RESULT_DIR \ - --env RANDOM_RANGE_RATIO \ - --env GPU_MEM_UTIL \ - --env AIPERF_FAILED_REQUEST_THRESHOLD \ - --env KV_OFFLOADING \ - --env KV_OFFLOAD_BACKEND \ - --env KV_OFFLOAD_BACKEND_METADATA \ - --env ROUTER_METADATA \ - --env KV_P2P_TRANSFER \ - --env TOTAL_CPU_DRAM_GB \ - --env DURATION \ - --env SCENARIO_TYPE \ - --env SCENARIO_SUBDIR \ - --env IS_AGENTIC \ - --env SWEBENCH_GEN_MODE \ - --env SWEBENCH_USE_MODAL \ - --env MODAL_TOKEN_ID \ - --env MODAL_TOKEN_SECRET \ - --env PROFILE \ - --env SGLANG_TORCH_PROFILER_DIR \ - --env VLLM_TORCH_PROFILER_DIR \ - --env VLLM_RPC_TIMEOUT \ - --env PYTHONDONTWRITEBYTECODE \ - --env PYTHONPYCACHEPREFIX=/tmp/pycache/ \ - --env PORT="$PORT" \ - --env CUDA_DEVICE_ORDER=PCI_BUS_ID \ - --env CUDA_VISIBLE_DEVICES \ - --env NCCL_IB_DISABLE \ - --entrypoint=/bin/bash \ - "$DOCKER_IMAGE" \ - "$BENCH_SCRIPT" diff --git a/runners/runtime_settings.sh b/runners/runtime_settings.sh index 06d50bdfa7..38e7837bb2 100644 --- a/runners/runtime_settings.sh +++ b/runners/runtime_settings.sh @@ -46,8 +46,4 @@ case "${RUNNER_NAME%%_*}" in check_env_vars GITHUB_WORKSPACE export BENCHMARK_LOGS_DIR="$GITHUB_WORKSPACE/benchmark_logs" ;; - rtx6000pro-lat) - export HF_HUB_CACHE_MOUNT=/var/lib/inferencex/hf-hub-cache - export NCCL_IB_DISABLE=1 - ;; esac diff --git a/utils/test_process_changelog.py b/utils/test_process_changelog.py index febb282ebd..f2d1c1b984 100644 --- a/utils/test_process_changelog.py +++ b/utils/test_process_changelog.py @@ -691,6 +691,39 @@ def test_invalid_key_after_valid_key_rejects_entire_selection(changelog_run, key changelog_run([{"config-keys": keys}]) +@pytest.mark.parametrize("keys,expected_concs", [ + (["retired", "single"], [16, 32, 64]), + (["retired"], []), +]) +def test_archived_changelog_keys_do_not_schedule_retired_configs( + planning_repo, changelog_run, keys, expected_concs, +): + archive = planning_repo[0] / "configs/deprecated" + archive.mkdir() + # Historical settings need not satisfy the current runtime schema. An older + # archived version of an active key must not replace its current definition. + (archive / "nvidia-master.yaml").write_text("retired: {}\nsingle: {}\n") + output = changelog_run([{"config-keys": keys, "scenario-type": ["fixed-seq-len"]}]) + assert [row["conc"] for row in output["single_node"].get("8k1k", [])] == expected_concs + assert [row["conc"] for row in output["evals"]] == ([32, 64] if expected_concs else []) + assert output["multi_node"] == {} + + +@pytest.mark.parametrize("keys,flags,message", [ + (["retired", "missing"], {}, "Config key 'missing' not found"), + (["retired-*"], {}, "No config keys matched"), + (["retired"], {"append-only": True}, "Config key 'retired' not found"), +]) +def test_archives_do_not_relax_unknown_keys_wildcards_or_append_only( + planning_repo, changelog_run, keys, flags, message, +): + archive = planning_repo[0] / "configs/deprecated" + archive.mkdir() + (archive / "nvidia-master.yaml").write_text("retired: {}\nretired-old: {}\n") + with pytest.raises(ValueError, match=message): + changelog_run([{"config-keys": keys, **flags}]) + + @pytest.mark.parametrize("trim", [False, True]) def test_append_only_main_runs_only_added_points_and_skips_evals(planning_repo, changelog_run, trim): root, master, _ = planning_repo From c430090a89e0f5063da8e21bda64b86a61bd6874 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:43:19 -0500 Subject: [PATCH 19/29] fix: allow client dependencies in container virtual environments MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 修复容器虚拟环境中的客户端依赖安装,移除不兼容的用户目录安装参数。 --- benchmarks/single_node/srt_fixed_sequence.sh | 2 +- perf-changelog.yaml | 59 ++++++++++++++++++++ 2 files changed, 60 insertions(+), 1 deletion(-) diff --git a/benchmarks/single_node/srt_fixed_sequence.sh b/benchmarks/single_node/srt_fixed_sequence.sh index 4ec1578606..0bc79b9ab4 100644 --- a/benchmarks/single_node/srt_fixed_sequence.sh +++ b/benchmarks/single_node/srt_fixed_sequence.sh @@ -44,7 +44,7 @@ fi source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" cd "$INFERENCEX_REPO_ROOT" -pip3 install --user --break-system-packages sentencepiece datasets pandas +pip3 install --break-system-packages sentencepiece datasets pandas start_gpu_monitor --output "$RESULT_DIR/gpu_metrics.csv" --interval "$SRT_MONITOR_INTERVAL" trap 'rc=$?; stop_gpu_monitor; exit "$rc"' EXIT diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 97840db484..c192c778b1 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8932,3 +8932,62 @@ preserve the original configs and scripts in the existing deprecated archives - 退役仅支持 Docker 的 RTX PRO 6000 定长配置及 runner 路由,使活跃的单节点定长覆盖仅使用 SRT-Slurm;原始配置和脚本保留在现有弃用归档中 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-mi355x-sglang + - dsr1-fp4-mi355x-sglang-mtp + - dsr1-fp4-mi355x-atom + - dsr1-fp4-mi355x-atom-mtp + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - dsr1-fp8-mi355x-sglang + - dsr1-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi355x-atom + - qwen3.5-fp8-mi355x-atom-mtp + - qwen3.5-fp4-mi355x-sglang + - qwen3.5-fp4-mi355x-atom + - qwen3.5-fp4-mi355x-sglang-mtp + - qwen3.5-fp8-mi300x-sglang + - dsr1-fp8-mi355x-atom + - dsr1-fp8-mi355x-atom-mtp + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + - dsr1-fp4-b200-sglang + - dsr1-fp4-b200-sglang-mtp + - dsr1-fp4-b300-sglang + - dsr1-fp4-b200-trt + - dsr1-fp4-b200-trt-mtp + - dsr1-fp8-b200-sglang + - dsr1-fp8-b300-sglang + - qwen3.5-fp8-b200-sglang + - qwen3.5-fp4-b200-sglang + - qwen3.5-fp4-b200-sglang-mtp + - qwen3.5-fp8-b200-sglang-mtp + - qwen3.5-fp8-b300-sglang-mtp + - qwen3.5-fp8-b300-sglang + - qwen3.5-fp4-b300-sglang + - qwen3.5-fp4-b300-sglang-mtp + - dsr1-fp8-b200-sglang-mtp + - dsr1-fp8-b300-sglang-mtp + - dsr1-fp8-b200-trt + - dsr1-fp8-b200-trt-mtp + - dsr1-fp8-h200-sglang + - dsr1-fp8-h200-sglang-mtp + - qwen3.5-fp8-h200-sglang + - qwen3.5-fp8-h200-sglang-mtp + - dsr1-fp8-h200-trt + - dsr1-fp8-h200-trt-mtp + - qwen3.5-fp8-h100-sglang + - qwen3.5-fp8-h100-sglang-mtp + - qwen3.5-fp4-b200-trt + - qwen3.5-fp4-b200-trt-mtp + scenario-type: + - fixed-seq-len + description: + - Install native fixed-sequence client dependencies in the container Python environment + without forcing user-site packages, which ROCm image virtual environments disable + - 原生定长客户端依赖安装到容器 Python 环境,不再强制使用 ROCm 镜像虚拟环境禁用的用户级软件包目录 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 From 6f60a8bf2695bc5d5b44f47fa5f8660030b0c70e Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:59:56 -0500 Subject: [PATCH 20/29] fix: limit native single-node steps to serving GPUs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 保留节点独占预留,同时将原生单节点服务和客户端步骤限制为配方指定的 GPU 数量,避免采集空闲设备功耗。 --- infx/srt_slurm/single_node.py | 4 ++- perf-changelog.yaml | 60 +++++++++++++++++++++++++++++++++++ utils/test_srt_single_node.py | 6 +++- 3 files changed, 68 insertions(+), 2 deletions(-) diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index 280819f52a..cae4d70c86 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -124,7 +124,9 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: for name in ("RUN_EVAL", "EVAL_ONLY", "DP_ATTENTION"): if environment[name] not in {"true", "false"}: raise ValueError(f"{name} must be true or false") - overrides = [] + # Exclusive nodes include idle GPUs. Restrict each server/client step to + # the serving GPU count so client-side power collection sees the same set. + overrides = ["--set", f"srun_options.gpus-per-node={json.dumps(environment['GPU_COUNT'])}"] if environment.get("SRT_SRUN_OPTIONS"): options = json.loads(environment["SRT_SRUN_OPTIONS"]) if not isinstance(options, dict) or any( diff --git a/perf-changelog.yaml b/perf-changelog.yaml index c192c778b1..713630fac0 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8991,3 +8991,63 @@ without forcing user-site packages, which ROCm image virtual environments disable - 原生定长客户端依赖安装到容器 Python 环境,不再强制使用 ROCm 镜像虚拟环境禁用的用户级软件包目录 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-mi355x-sglang + - dsr1-fp4-mi355x-sglang-mtp + - dsr1-fp4-mi355x-atom + - dsr1-fp4-mi355x-atom-mtp + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - dsr1-fp8-mi355x-sglang + - dsr1-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi355x-atom + - qwen3.5-fp8-mi355x-atom-mtp + - qwen3.5-fp4-mi355x-sglang + - qwen3.5-fp4-mi355x-atom + - qwen3.5-fp4-mi355x-sglang-mtp + - qwen3.5-fp8-mi300x-sglang + - dsr1-fp8-mi355x-atom + - dsr1-fp8-mi355x-atom-mtp + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + - dsr1-fp4-b200-sglang + - dsr1-fp4-b200-sglang-mtp + - dsr1-fp4-b300-sglang + - dsr1-fp4-b200-trt + - dsr1-fp4-b200-trt-mtp + - dsr1-fp8-b200-sglang + - dsr1-fp8-b300-sglang + - qwen3.5-fp8-b200-sglang + - qwen3.5-fp4-b200-sglang + - qwen3.5-fp4-b200-sglang-mtp + - qwen3.5-fp8-b200-sglang-mtp + - qwen3.5-fp8-b300-sglang-mtp + - qwen3.5-fp8-b300-sglang + - qwen3.5-fp4-b300-sglang + - qwen3.5-fp4-b300-sglang-mtp + - dsr1-fp8-b200-sglang-mtp + - dsr1-fp8-b300-sglang-mtp + - dsr1-fp8-b200-trt + - dsr1-fp8-b200-trt-mtp + - dsr1-fp8-h200-sglang + - dsr1-fp8-h200-sglang-mtp + - qwen3.5-fp8-h200-sglang + - qwen3.5-fp8-h200-sglang-mtp + - dsr1-fp8-h200-trt + - dsr1-fp8-h200-trt-mtp + - qwen3.5-fp8-h100-sglang + - qwen3.5-fp8-h100-sglang-mtp + - qwen3.5-fp4-b200-trt + - qwen3.5-fp4-b200-trt-mtp + scenario-type: + - fixed-seq-len + description: + - Bind native single-node Slurm steps to the validated serving GPU count while retaining + exclusive node reservations, so partial-node recipes do not expose idle devices + to benchmark power collection + - 原生单节点 Slurm 步骤使用已验证的推理 GPU 数量并保留节点独占预留,避免部分节点配方将空闲设备暴露给基准功耗采集 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index c5038d36f0..6d7386ad31 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -22,6 +22,7 @@ def point(tmp_path): recipe = { "engine": "sglang", + "resources": {"gpus_per_node": 8}, "model": {"path": "hf:test/model", "container": "test:tag", "precision": "fp8"}, "roles": {"agg": { "nodes": 1, "workers": 1, "gpus": 4, @@ -53,6 +54,7 @@ def test_native_binding_submits_one_point_and_keeps_server_settings(point): overrides = parse_overrides(argv[1::2], []) actual = copy.deepcopy(recipe) apply_overrides_to_recipe(actual, overrides) + assert actual["srun_options"]["gpus-per-node"] == "4" assert actual["benchmark"]["env"] == { "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "USE_CHAT_TEMPLATE": "false", @@ -369,7 +371,9 @@ def test_runtime_container_options_remain_native_mapping(point): argv = runtime_arguments(f"{path}:base", env) actual = copy.deepcopy(recipe) apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) - assert actual['srun_options'] == {'container-remap-root': '', 'container-writable': ''} + assert actual['srun_options'] == { + 'gpus-per-node': '4', 'container-remap-root': '', 'container-writable': '', + } with pytest.raises(ValueError, match='must map option names to string values'): runtime_arguments(f"{path}:base", {**env, 'SRT_SRUN_OPTIONS': '{"container-remap-root": true}'}) From 161aa243adc5fabac72a9742f3ffab644b732d09 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 19:02:42 -0500 Subject: [PATCH 21/29] fix: restore repository workdir for native container steps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 原生容器步骤从已有仓库挂载目录启动,避免 Python 动态模块在根目录下导入失败,并保留调用方运行时覆盖。 --- infx/srt_slurm/single_node.py | 3 ++ perf-changelog.yaml | 60 +++++++++++++++++++++++++++++++++++ utils/test_srt_single_node.py | 9 ++++-- 3 files changed, 70 insertions(+), 2 deletions(-) diff --git a/infx/srt_slurm/single_node.py b/infx/srt_slurm/single_node.py index cae4d70c86..31655f6dc7 100644 --- a/infx/srt_slurm/single_node.py +++ b/infx/srt_slurm/single_node.py @@ -127,6 +127,9 @@ def runtime_arguments(config: str, environment: Mapping[str, str]) -> list[str]: # Exclusive nodes include idle GPUs. Restrict each server/client step to # the serving GPU count so client-side power collection sees the same set. overrides = ["--set", f"srun_options.gpus-per-node={json.dumps(environment['GPU_COUNT'])}"] + # Match the legacy container working directory using the existing repo mount. + # PyTorch's generated module imports fail from / with PYTHONPYCACHEPREFIX set. + overrides += ["--set", 'srun_options.container-workdir="/infmax-workspace"'] if environment.get("SRT_SRUN_OPTIONS"): options = json.loads(environment["SRT_SRUN_OPTIONS"]) if not isinstance(options, dict) or any( diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 7dd48849be..026f0f0f1c 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -9115,3 +9115,63 @@ to benchmark power collection - 原生单节点 Slurm 步骤使用已验证的推理 GPU 数量并保留节点独占预留,避免部分节点配方将空闲设备暴露给基准功耗采集 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-mi355x-sglang + - dsr1-fp4-mi355x-sglang-mtp + - dsr1-fp4-mi355x-atom + - dsr1-fp4-mi355x-atom-mtp + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - dsr1-fp8-mi355x-sglang + - dsr1-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi355x-atom + - qwen3.5-fp8-mi355x-atom-mtp + - qwen3.5-fp4-mi355x-sglang + - qwen3.5-fp4-mi355x-atom + - qwen3.5-fp4-mi355x-sglang-mtp + - qwen3.5-fp8-mi300x-sglang + - dsr1-fp8-mi355x-atom + - dsr1-fp8-mi355x-atom-mtp + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + - dsr1-fp4-b200-sglang + - dsr1-fp4-b200-sglang-mtp + - dsr1-fp4-b300-sglang + - dsr1-fp4-b200-trt + - dsr1-fp4-b200-trt-mtp + - dsr1-fp8-b200-sglang + - dsr1-fp8-b300-sglang + - qwen3.5-fp8-b200-sglang + - qwen3.5-fp4-b200-sglang + - qwen3.5-fp4-b200-sglang-mtp + - qwen3.5-fp8-b200-sglang-mtp + - qwen3.5-fp8-b300-sglang-mtp + - qwen3.5-fp8-b300-sglang + - qwen3.5-fp4-b300-sglang + - qwen3.5-fp4-b300-sglang-mtp + - dsr1-fp8-b200-sglang-mtp + - dsr1-fp8-b300-sglang-mtp + - dsr1-fp8-b200-trt + - dsr1-fp8-b200-trt-mtp + - dsr1-fp8-h200-sglang + - dsr1-fp8-h200-sglang-mtp + - qwen3.5-fp8-h200-sglang + - qwen3.5-fp8-h200-sglang-mtp + - dsr1-fp8-h200-trt + - dsr1-fp8-h200-trt-mtp + - qwen3.5-fp8-h100-sglang + - qwen3.5-fp8-h100-sglang-mtp + - qwen3.5-fp4-b200-trt + - qwen3.5-fp4-b200-trt-mtp + scenario-type: + - fixed-seq-len + description: + - Run native single-node container steps from the existing InferenceX repository + mount, preserving legacy working-directory behavior and avoiding Python generated-module + import failures at the filesystem root + - 原生单节点容器步骤从现有 InferenceX 仓库挂载目录启动,保留旧版工作目录行为,避免在文件系统根目录导入 Python 动态生成模块时失败 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/utils/test_srt_single_node.py b/utils/test_srt_single_node.py index 6d7386ad31..979f286daf 100644 --- a/utils/test_srt_single_node.py +++ b/utils/test_srt_single_node.py @@ -54,7 +54,9 @@ def test_native_binding_submits_one_point_and_keeps_server_settings(point): overrides = parse_overrides(argv[1::2], []) actual = copy.deepcopy(recipe) apply_overrides_to_recipe(actual, overrides) - assert actual["srun_options"]["gpus-per-node"] == "4" + assert actual["srun_options"] == { + "gpus-per-node": "4", "container-workdir": "/infmax-workspace", + } assert actual["benchmark"]["env"] == { "MODEL": "test/model", "ISL": "256", "OSL": "64", "RANDOM_RANGE_RATIO": "0.5", "USE_CHAT_TEMPLATE": "false", @@ -367,12 +369,15 @@ def test_pool_launcher_stages_artifacts_and_propagates_failure(point, tmp_path, def test_runtime_container_options_remain_native_mapping(point): path, recipe, env = point - env = {**env, "SRT_SRUN_OPTIONS": '{"container-remap-root":"", "container-writable":""}'} + env = {**env, "SRT_SRUN_OPTIONS": json.dumps({ + "container-remap-root": "", "container-writable": "", "container-workdir": "/custom", + })} argv = runtime_arguments(f"{path}:base", env) actual = copy.deepcopy(recipe) apply_overrides_to_recipe(actual, parse_overrides(argv[1::2], [])) assert actual['srun_options'] == { 'gpus-per-node': '4', 'container-remap-root': '', 'container-writable': '', + 'container-workdir': '/custom', } with pytest.raises(ValueError, match='must map option names to string values'): runtime_arguments(f"{path}:base", {**env, 'SRT_SRUN_OPTIONS': '{"container-remap-root": true}'}) From dd955705c0a85abd0a012d4bec2053145f4c26d6 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 19:37:19 -0500 Subject: [PATCH 22/29] fix: stream AMD power samples without input buffering MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 逐行写出 AMD 功耗 CSV,避免 awk 输入缓冲在采样器关闭时丢失末尾数据;新增真实流式读取回归测试。 --- benchmarks/benchmark_lib.sh | 23 +++++++++++++++--- perf-changelog.yaml | 47 ++++++++++++++++++++++++++++++++++++ utils/test_process_result.py | 38 +++++++++++++++++++++++++++++ 3 files changed, 105 insertions(+), 3 deletions(-) diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index 214d7ca7cb..b51f6dd3ca 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -425,6 +425,24 @@ GPU_METRICS_CSV="${GPU_METRICS_CSV:-gpu_metrics.csv}" NVIDIA_GPU_MONITOR_QUERY="timestamp,index,power.draw,temperature.gpu,clocks.current.sm,clocks.current.memory,utilization.gpu,utilization.memory" export GPU_METRICS_CSV +# Keep one AMD CSV header and forward each complete row immediately. Some awk +# implementations buffer pipe input even with fflush(), losing the final ticks +# when the monitor stops. +_filter_amd_smi_metrics() { + local line header_seen=false + while IFS= read -r line; do + if [[ "$line" == timestamp,* ]]; then + if [[ "$header_seen" == true ]]; then + continue + fi + header_seen=true + fi + if [[ "$header_seen" == true ]]; then + printf '%s\n' "$line" + fi + done +} + # Background nvidia-smi/amd-smi sampler writing CSV. # Usage: start_gpu_monitor [--output /path/to/output.csv] [--interval 1] start_gpu_monitor() { @@ -457,10 +475,9 @@ start_gpu_monitor() { elif command -v amd-smi &>/dev/null; then GPU_MONITOR_VENDOR="amd" # amd-smi is Python and block-buffers stdout; without PYTHONUNBUFFERED the - # trailing ticks were lost at kill (measured on MI355X). awk keeps the first - # CSV header, drops repeated ones, and flushes every row for the same reason. + # trailing ticks were lost at kill (measured on MI355X). PYTHONUNBUFFERED=1 amd-smi metric -p -c -t -u -w "$interval" --csv 2>/dev/null \ - | awk '/^timestamp,/{if(!h){print;h=1};next} h{print;fflush()}' > "$output" & + | _filter_amd_smi_metrics > "$output" & GPU_MONITOR_PID=$! # Hardware energy-accumulator + identity snapshots; the end-side twin in # stop_gpu_monitor lets auditors cross-check the integrated energy diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 26751663cf..7f4d0f79cc 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -9188,3 +9188,50 @@ import failures at the filesystem root - 原生单节点容器步骤从现有 InferenceX 仓库挂载目录启动,保留旧版工作目录行为,避免在文件系统根目录导入 Python 动态生成模块时失败 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-mi355x-sglang + - dsr1-fp4-mi355x-sglang-mtp + - dsr1-fp4-mi355x-atom + - dsr1-fp4-mi355x-atom-mtp + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - dsr1-fp8-mi355x-sglang + - dsr1-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-agentic-mtp + - qwen3.5-fp8-mi355x-atom + - qwen3.5-fp8-mi355x-atom-mtp + - qwen3.5-fp4-mi355x-sglang + - qwen3.5-fp4-mi355x-atom + - qwen3.5-fp4-mi355x-sglang-mtp + - qwen3.5-fp4-mi355x-sglang-agentic-mtp + - qwen3.5-fp8-mi300x-sglang + - qwen3.5-fp8-mi300x-sglang-agentic-mtp + - dsr1-fp8-mi355x-atom + - dsr1-fp8-mi355x-atom-mtp + - kimik3-fp4-mi355x-vllm-agentic-mtp + - kimik3-fp4-mi355x-atom-agentic-mtp + - minimaxm3-fp4-mi355x-atom-agentic-mtp + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + - dsv4-fp4-mi355x-vllm-agentic-mtp + - dsv4-fp4-mi355x-atom-agentic-mtp + - minimaxm3-fp8-mi300x-vllm-agentic-mtp + - dsv41flash-fp4-mi300x-vllm-agentic-dspark + - glm5.2-fp8-mi325x-sglang-agentic-mtp + - minimaxm3-fp8-mi325x-vllm-agentic-mtp + - dsv41flash-fp4-mi325x-vllm-agentic-dspark + - minimaxm3-fp4-mi355x-vllm-agentic-mtp + - glm5.2-fp4-mi355x-sglang-agentic-mtp + - glm5.2-fp8-mi355x-sglang-agentic-mtp + - glm5.2-fp4-mi355x-atom-agentic-mtp + - dsv4-fp4-mi355x-sglang-agentic-mtp + - dsv41flash-fp4-mi355x-vllm-agentic-dspark + description: + - Stream AMD power CSV rows without awk input buffering so the sampler preserves final benchmark-window + samples during shutdown. Keep benchmark settings and power validation unchanged. + - AMD 功耗 CSV 逐行写出,避免 awk 输入缓冲导致采样器关闭时丢失基准窗口末尾的样本;基准设置和功耗校验保持不变。 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/utils/test_process_result.py b/utils/test_process_result.py index 472ede7e1c..62f173ec59 100644 --- a/utils/test_process_result.py +++ b/utils/test_process_result.py @@ -1,9 +1,11 @@ """Exercise the fixed-sequence module CLI with controlled environment and artifacts.""" import json import os +import select import signal import subprocess import sys +import time from pathlib import Path import pytest @@ -979,6 +981,42 @@ def test_multinode_internal_error_preserves_validation( "message": "forced import failure" if fail_import else "forced aggregation failure", } + def test_amd_csv_filter_streams_complete_rows_before_eof(self): + """A live producer must not leave telemetry buffered until shutdown.""" + benchmark_lib = REPO_ROOT / "benchmarks/benchmark_lib.sh" + expected = b"timestamp,gpu,socket_power\n123,0,400\n124,0,410\n" + with subprocess.Popen( + ["bash", "-c", f"source {str(benchmark_lib)!r}; _filter_amd_smi_metrics"], + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + env={"PATH": os.environ["PATH"], "PYTHONDONTWRITEBYTECODE": "1"}, + ) as process: + try: + process.stdin.write( + b"diagnostic before header\ntimestamp,gpu,socket_power\n123,0,400\n" + b"timestamp,gpu,socket_power\n124,0,410\n125,0,4" + ) + process.stdin.flush() + received = b"" + deadline = time.monotonic() + 5 + while len(received) < len(expected): + ready, _, _ = select.select( + [process.stdout], [], [], max(0, deadline - time.monotonic()) + ) + assert ready, "CSV rows remained buffered while the producer was open" + chunk = os.read(process.stdout.fileno(), 4096) + assert chunk, "filter exited before consuming the live stream" + received += chunk + assert received == expected + process.stdin.close() + assert process.wait(timeout=5) == 0 + assert process.stdout.read() == b"" # Discard the incomplete final row. + finally: + if process.poll() is None: + process.kill() + process.wait(timeout=5) + def test_stop_gpu_monitor_appends_final_nvidia_sample(self, tmp_path): """Stopping between 1 Hz ticks still records one post-benchmark sample.""" fake_bin = tmp_path / "bin" From 77961ed4204cac15be737052cd03afb719e75178 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Tue, 22 Sep 2026 20:35:57 -0500 Subject: [PATCH 23/29] fix: match registered H200 runner labels MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 H200 DGXC runner 清单对齐到实际注册的两位数字标签,恢复精确 smoke 选择和 sweep 调度。 --- configs/runners.yaml | 40 ++++++++++++++++++++-------------------- perf-changelog.yaml | 28 ++++++++++++++++++++++++++++ 2 files changed, 48 insertions(+), 20 deletions(-) diff --git a/configs/runners.yaml b/configs/runners.yaml index f233788d41..b9ede289c5 100644 --- a/configs/runners.yaml +++ b/configs/runners.yaml @@ -25,16 +25,16 @@ labels: h200: - h200-cw_00 - h200-cw_01 - - h200-dgxc-slurm_0 - - h200-dgxc-slurm_1 - - h200-dgxc-slurm_2 - - h200-dgxc-slurm_3 - - h200-dgxc-slurm_4 - - h200-dgxc-slurm_5 - - h200-dgxc-slurm_6 - - h200-dgxc-slurm_7 - - h200-dgxc-slurm_8 - - h200-dgxc-slurm_9 + - h200-dgxc-slurm_00 + - h200-dgxc-slurm_01 + - h200-dgxc-slurm_02 + - h200-dgxc-slurm_03 + - h200-dgxc-slurm_04 + - h200-dgxc-slurm_05 + - h200-dgxc-slurm_06 + - h200-dgxc-slurm_07 + - h200-dgxc-slurm_08 + - h200-dgxc-slurm_09 - h200-dgxc-slurm_10 - h200-dgxc-slurm_11 - h200-dgxc-slurm_12 @@ -166,16 +166,16 @@ labels: - h200-cw_00 - h200-cw_01 cluster:h200-dgxc: - - h200-dgxc-slurm_0 - - h200-dgxc-slurm_1 - - h200-dgxc-slurm_2 - - h200-dgxc-slurm_3 - - h200-dgxc-slurm_4 - - h200-dgxc-slurm_5 - - h200-dgxc-slurm_6 - - h200-dgxc-slurm_7 - - h200-dgxc-slurm_8 - - h200-dgxc-slurm_9 + - h200-dgxc-slurm_00 + - h200-dgxc-slurm_01 + - h200-dgxc-slurm_02 + - h200-dgxc-slurm_03 + - h200-dgxc-slurm_04 + - h200-dgxc-slurm_05 + - h200-dgxc-slurm_06 + - h200-dgxc-slurm_07 + - h200-dgxc-slurm_08 + - h200-dgxc-slurm_09 - h200-dgxc-slurm_10 - h200-dgxc-slurm_11 - h200-dgxc-slurm_12 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 7f4d0f79cc..287e90b1aa 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -9235,3 +9235,31 @@ samples during shutdown. Keep benchmark settings and power validation unchanged. - AMD 功耗 CSV 逐行写出,避免 awk 输入缓冲导致采样器关闭时丢失基准窗口末尾的样本;基准设置和功耗校验保持不变。 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - kimik3-fp4-h200-vllm-agentic-latency + - kimik3-fp4-h200-vllm-agentic-balanced + - kimik3-fp4-h200-vllm-agentic-simple + - dsr1-fp8-h200-sglang + - dsr1-fp8-h200-sglang-mtp + - dsv4-fp8-h200-dynamo-sglang-agentic-agg + - qwen3.5-fp8-h200-sglang + - qwen3.5-fp8-h200-sglang-mtp + - dsr1-fp8-h200-trt + - dsr1-fp8-h200-trt-mtp + - dsr1-fp8-h200-dynamo-trt + - dsr1-fp8-h200-dynamo-sglang + - qwen3.5-fp8-h200-sglang-agentic-mtp + - qwen3.8next-fp8-h200-sglang-agentic-mtp + - qwen3.5-fp8-h200-sglang-agentic-hicache-mtp + - minimaxm3-fp8-h200-vllm-agentic-mtp + - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-2p2d + - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-1p1d-hicache + - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-agg + - dsv41flash-fp4-h200-vllm-agentic-dspark + - dsv41flash-fp4-h200-sglang-agentic-dspark + description: + - Align H200 DGXC runner inventory with the verified two-digit GitHub runner labels so exact point selections + and sweep assignments reach registered workers; benchmark recipes and settings are unchanged. + - H200 DGXC runner 清单与已核实的两位数字 GitHub 标签对齐,使精确测试点选择和 sweep 分配能够匹配已注册的 worker;基准配方和设置保持不变。 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 From a3dc8988b0f62640e5fae8c1778eb5abfc9d1d58 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Wed, 23 Sep 2026 15:43:22 -0500 Subject: [PATCH 24/29] chore: pin srt-slurm submodule to main [skip ci] --- .gitmodules | 2 +- utils/srt-slurm | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/.gitmodules b/.gitmodules index c2f2f853f3..f7635a307b 100644 --- a/.gitmodules +++ b/.gitmodules @@ -3,4 +3,4 @@ url = https://github.com/SemiAnalysisAI/aiperf.git [submodule "utils/srt-slurm"] path = utils/srt-slurm - url = https://github.com/SemiAnalysisAI/srt-slurm.git + url = https://github.com/NVIDIA/srt-slurm.git diff --git a/utils/srt-slurm b/utils/srt-slurm index c29ef7c5d0..2ac4eb1367 160000 --- a/utils/srt-slurm +++ b/utils/srt-slurm @@ -1 +1 @@ -Subproject commit c29ef7c5d0732fbf7fc93aa4b7929b48f0bfa76a +Subproject commit 2ac4eb1367dd2a78f597a72ca91afe4211d76b38 From 082279de5efb2e3ecf5821af3baeb1fb013ddc3c Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:49:04 -0500 Subject: [PATCH 25/29] refactor(srt): return single-node migration to NVIDIA upstream MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将单节点迁移切回 NVIDIA 上游 srt-slurm。ATOM 使用原生 AToMesh 及单独固定的官方路由器镜像;TRT-LLM 使用原生模型名称转发。保留 worker 镜像、服务参数、并发范围与拓扑,不增加 Bash 分支。 --- .gitmodules | 2 +- .../dsr1/atom/mi355x-fp4-mtp/8k1k.yaml | 3 +- .../dsr1/atom/mi355x-fp4/8k1k.yaml | 3 +- .../dsr1/atom/mi355x-fp8-mtp/8k1k.yaml | 3 +- .../dsr1/atom/mi355x-fp8/8k1k.yaml | 3 +- .../dsr1/trtllm/b200-fp4-mtp/8k1k.yaml | 2 - .../dsr1/trtllm/b200-fp4/8k1k.yaml | 2 - .../dsr1/trtllm/b200-fp8-mtp/8k1k.yaml | 2 - .../dsr1/trtllm/b200-fp8/8k1k.yaml | 2 - .../dsr1/trtllm/h200-fp8-mtp/8k1k.yaml | 2 - .../dsr1/trtllm/h200-fp8/8k1k.yaml | 2 - .../qwen3.5/atom/mi355x-fp4/8k1k.yaml | 3 +- .../qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml | 3 +- .../qwen3.5/atom/mi355x-fp8/8k1k.yaml | 3 +- .../qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml | 2 - .../qwen3.5/trtllm/b200-fp4/8k1k.yaml | 2 - docs/configuration-procedures.md | 11 ++ docs/configuration-procedures_zh.md | 8 ++ perf-changelog.yaml | 133 ++++++++++++++++++ utils/srt-slurm | 2 +- 20 files changed, 168 insertions(+), 25 deletions(-) diff --git a/.gitmodules b/.gitmodules index c2f2f853f3..f7635a307b 100644 --- a/.gitmodules +++ b/.gitmodules @@ -3,4 +3,4 @@ url = https://github.com/SemiAnalysisAI/aiperf.git [submodule "utils/srt-slurm"] path = utils/srt-slurm - url = https://github.com/SemiAnalysisAI/srt-slurm.git + url = https://github.com/NVIDIA/srt-slurm.git diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml index 03c0d172b3..2d11e33ab1 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4-mtp/8k1k.yaml @@ -9,7 +9,8 @@ base: gpu_type: mi355x gpus_per_node: 8 frontend: - type: atom + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b enable_multiple_frontends: false observability: enabled: false diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml index 2a258efb17..9769e663d7 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp4/8k1k.yaml @@ -9,7 +9,8 @@ base: gpu_type: mi355x gpus_per_node: 8 frontend: - type: atom + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b enable_multiple_frontends: false observability: enabled: false diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml index db5f89aee7..40bc67ac50 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8-mtp/8k1k.yaml @@ -9,7 +9,8 @@ base: gpu_type: mi355x gpus_per_node: 8 frontend: - type: atom + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b enable_multiple_frontends: false observability: enabled: false diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml index d7b82c5ed7..cdcc9646dc 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/atom/mi355x-fp8/8k1k.yaml @@ -9,7 +9,8 @@ base: gpu_type: mi355x gpus_per_node: 8 frontend: - type: atom + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b enable_multiple_frontends: false observability: enabled: false diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml index 0533fa0d82..d2a484832f 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4-mtp/8k1k.yaml @@ -40,8 +40,6 @@ base: pipeline_parallel_size: 1 enable_iter_perf_stats: false return_perf_metrics: false - # The pinned native launcher needs this CLI flag as well as engine metadata. - extra_args: [--served_model_name, nvidia/DeepSeek-R1-0528-FP4-V2] benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml index 2910db349e..bfee802b20 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp4/8k1k.yaml @@ -38,8 +38,6 @@ base: pipeline_parallel_size: 1 enable_iter_perf_stats: false return_perf_metrics: false - # The pinned native launcher needs this CLI flag as well as engine metadata. - extra_args: [--served_model_name, nvidia/DeepSeek-R1-0528-FP4-V2] benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml index 6b31d7e6fa..70d909c580 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8-mtp/8k1k.yaml @@ -46,8 +46,6 @@ base: pipeline_parallel_size: 1 enable_iter_perf_stats: false return_perf_metrics: false - # The pinned native launcher needs this CLI flag as well as engine metadata. - extra_args: [--served_model_name, deepseek-ai/DeepSeek-R1-0528] benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml index 6e818ae104..9223814440 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/b200-fp8/8k1k.yaml @@ -43,8 +43,6 @@ base: return_perf_metrics: false env: TLLM_OVERRIDE_LAYER_NUM: '61' - # The pinned native launcher needs this CLI flag as well as engine metadata. - extra_args: [--served_model_name, deepseek-ai/DeepSeek-R1-0528] benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml index 8d446aef24..6e6685878d 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8-mtp/8k1k.yaml @@ -47,8 +47,6 @@ base: return_perf_metrics: false env: PYTHONNOUSERSITE: '1' - # The pinned native launcher needs this CLI flag as well as engine metadata. - extra_args: [--served_model_name, deepseek-ai/DeepSeek-R1-0528] benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh diff --git a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml index cfe32cd8b9..f544e6bbbc 100644 --- a/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/dsr1/trtllm/h200-fp8/8k1k.yaml @@ -46,8 +46,6 @@ base: return_perf_metrics: false env: PYTHONNOUSERSITE: '1' - # The pinned native launcher needs this CLI flag as well as engine metadata. - extra_args: [--served_model_name, deepseek-ai/DeepSeek-R1-0528] benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml index b748775c8a..7677e9a68a 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp4/8k1k.yaml @@ -9,7 +9,8 @@ base: gpu_type: mi355x gpus_per_node: 8 frontend: - type: atom + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b enable_multiple_frontends: false observability: enabled: false diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml index ac1435ee2e..6e5b336ee5 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8-mtp/8k1k.yaml @@ -9,7 +9,8 @@ base: gpu_type: mi355x gpus_per_node: 8 frontend: - type: atom + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b enable_multiple_frontends: false observability: enabled: false diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml index 4091212fb7..6c6fb60a3e 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/atom/mi355x-fp8/8k1k.yaml @@ -9,7 +9,8 @@ base: gpu_type: mi355x gpus_per_node: 8 frontend: - type: atom + type: atomesh + container_image: rocm/atom-dev@sha256:f00a9dc588a380038fdad54cd57a956043cb316ddda54ebb1b62a24f8caa343b enable_multiple_frontends: false observability: enabled: false diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml index 8ea47be0f0..9d0c9375cd 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4-mtp/8k1k.yaml @@ -53,8 +53,6 @@ base: return_perf_metrics: false env: TLLM_USE_FLASHINFER_GDN_PREFILL: '0' - # The pinned native launcher needs this CLI flag as well as engine metadata. - extra_args: [--served_model_name, nvidia/Qwen3.5-397B-A17B-NVFP4] benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml index 0a6ca11dc8..6b705c00cb 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/trtllm/b200-fp4/8k1k.yaml @@ -48,8 +48,6 @@ base: max_num_tokens: 32768 pipeline_parallel_size: 1 return_perf_metrics: false - # The pinned native launcher needs this CLI flag as well as engine metadata. - extra_args: [--served_model_name, nvidia/Qwen3.5-397B-A17B-NVFP4] benchmark: type: custom command: bash /infmax-workspace/benchmarks/single_node/srt_fixed_sequence.sh diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index c6941009a7..6f4c12983c 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -35,6 +35,17 @@ git submodule update --init To upgrade, fetch and check out the desired commit inside the relevant submodule, then commit the updated submodule pointer in InferenceX. Benchmark workflows already initialize submodules. Slurm launchers make a local Git clone for each job so recipe staging and runtime writes do not modify the submodule, and record the actual commit for result provenance. NVIDIA setup clones locally; TileRT setup fetches its pinned fork commit over the network. +Single-node fixed-sequence recipes use NVIDIA upstream srt-slurm. ATOM recipes use +the native `atomesh` frontend with one aggregate worker and +`enable_multiple_frontends: false`. The router's pinned official image belongs in +`frontend.container_image`: older benchmark worker images do not include AToMesh. +Keep `model.container` aligned with the master config's worker `image`; changing the +router image does not require changing the worker image. TRT-LLM recipes use native +`engine.served_model_name`, without duplicating that flag in `roles.agg.extra_args`. +The former fork's direct ATOM frontend is not required. Historical direct-serving +smokes do not qualify the new router path; verify startup, requests, power, cleanup, +and performance on the new pin. + ### Cluster profiles Launchers that use srt-slurm keep their cluster configuration in diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 54986db75c..652fc490cd 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -35,6 +35,14 @@ git submodule update --init 升级时,在对应子模块中获取并检出目标提交,再将更新后的子模块指针提交到 InferenceX。基准测试工作流已配置为自动初始化子模块。Slurm 启动器为每个作业创建本地 Git 克隆,避免配方准备和运行时写入修改子模块,并记录实际提交以供结果溯源。NVIDIA 启动器使用本地克隆;TileRT 启动器通过网络获取固定的分支提交。 +单节点固定序列长度配方使用 NVIDIA 上游 srt-slurm。ATOM 配方使用原生 `atomesh` +frontend、一个聚合 worker,并设置 `enable_multiple_frontends: false`。旧版基准 worker +镜像不包含 AToMesh,因此通过 `frontend.container_image` 单独固定路由器的官方镜像。 +`model.container` 必须与主配置中的 worker `image` 一致;更换路由器镜像无需更换 worker +镜像。TRT-LLM 配方使用原生 `engine.served_model_name`,不再通过 `roles.agg.extra_args` +重复传入该参数。不再依赖此前分叉中的 ATOM 直连 frontend。历史直连 smoke 不能作为新路由 +路径的验收证据;需在新固定版本上验证启动、请求、功耗、清理和性能。 + ## 规程索引 1. [准备 worktree](#准备-worktree) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 78b2912d37..5aa58662f5 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -9336,3 +9336,136 @@ and sweep assignments reach registered workers; benchmark recipes and settings are unchanged. - H200 DGXC runner 清单与已核实的两位数字 GitHub 标签对齐,使精确测试点选择和 sweep 分配能够匹配已注册的 worker;基准配方和设置保持不变。 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 + +- config-keys: + - dsr1-fp4-b200-dynamo-trt + - dsr1-fp8-b200-dynamo-trt + - dsr1-fp4-b300-dynamo-trt + - dsr1-fp8-b300-dynamo-trt + - dsr1-fp4-b200-sglang + - dsr1-fp4-b200-sglang-mtp + - dsr1-fp4-b300-sglang + - dsr1-fp4-b200-trt + - dsr1-fp4-b200-trt-mtp + - dsr1-fp8-b200-sglang + - dsr1-fp8-b300-sglang + - qwen3.5-fp8-b200-sglang + - qwen3.5-fp4-b200-sglang + - qwen3.5-fp4-b200-sglang-mtp + - qwen3.5-fp8-b200-sglang-mtp + - qwen3.5-fp8-b300-sglang-mtp + - qwen3.5-fp8-b300-sglang + - qwen3.5-fp4-b300-sglang + - qwen3.5-fp4-b300-sglang-mtp + - kimik3-fp4-h200-vllm-agentic-latency + - kimik3-fp4-h200-vllm-agentic-balanced + - kimik3-fp4-h200-vllm-agentic-simple + - dsr1-fp8-b200-sglang-mtp + - dsr1-fp8-b300-sglang-mtp + - dsr1-fp8-b200-trt + - dsr1-fp8-b200-trt-mtp + - dsr1-fp8-h200-sglang + - dsr1-fp8-h200-sglang-mtp + - dsv4-fp8-h200-dynamo-sglang-agentic-agg + - qwen3.5-fp8-h200-sglang + - qwen3.5-fp8-h200-sglang-mtp + - dsr1-fp8-h200-trt + - dsr1-fp8-h200-trt-mtp + - dsr1-fp8-h200-dynamo-trt + - dsr1-fp8-h100-dynamo-trt + - dsr1-fp8-h100-dynamo-sglang + - dsr1-fp4-gb200-dynamo-trt + - dsr1-fp8-gb200-dynamo-trt + - dsr1-fp8-gb200-dynamo-sglang + - dsr1-fp8-gb300-dynamo-sglang + - dsr1-fp4-gb200-dynamo-sglang + - dsr1-fp4-gb300-dynamo-trt + - dsr1-fp4-gb300-dynamo-sglang + - dsr1-fp8-gb300-dynamo-trt + - dsr1-fp8-h200-dynamo-sglang + - dsr1-fp4-b200-dynamo-sglang + - dsr1-fp8-b200-dynamo-sglang + - dsr1-fp8-b200-dynamo-sglang-mtp + - dsr1-fp4-b200-dynamo-sglang-mtp + - qwen3.5-fp8-gb200-dynamo-sglang + - qwen3.5-fp8-h100-sglang + - qwen3.5-fp8-h100-sglang-mtp + - qwen3.5-fp4-gb300-dynamo-sglang + - qwen3.5-fp4-gb300-dynamo-trt + - qwen3.5-fp4-gb300-dynamo-trt-mtp + - qwen3.5-fp4-gb300-dynamo-trt-agentic-disagg + - qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg + - qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg + - qwen3.5-fp4-gb300-dynamo-sglang-agentic-pp-pareto + - qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg-pareto + - dsv4-fp4-gb300-dynamo-vllm-agentic + - qwen3.5-fp4-b200-trt + - qwen3.5-fp4-b200-trt-mtp + - qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp + - minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg + - minimaxm3-fp4-gb200-dynamo-vllm-agentic-agg-mtp + - minimaxm3-fp4-gb200-dynamo-vllm-agentic-disagg-mtp + - minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-agg + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-disagg + - dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-agg + - dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-agg + - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-disagg + - kimik3-fp4-gb200-dynamo-vllm-agentic + - kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-dcp16-agg + - kimik3-fp4-gb200-dynamo-vllm-agentic-mooncake-dcp16-agg + - dsv4-fp4-gb300-dynamo-trt-agentx + - kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2 + - dsv4-fp4-gb300-dynamo-sglang-agentic-agg + - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg + - qwen3.5-fp8-gb300-dynamo-sglang + - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-2p2d + - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-1p1d-hicache + - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-agg + - glm5.2-fp4-b200-dynamo-sglang-agentic-agg + - glm5.2-fp4-b200-dynamo-sglang-agentic-disagg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-agg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-disagg + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp + - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg + - glm5.2-fp4-gb300-dynamo-sglang-agentic-agg + - glm5.2-fp4-gb300-dynamo-sglang-agentic-disagg + - glm5.2-fp4-gb300-dynamo-trt-agentic-disagg-mtp + - qwen3.5-fp8-gb200-dynamo-sglang-mtp + - kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg + - kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-agg + - kimik3-fp4-b200-dynamo-vllm-agentic-dspark + - qwen3.5-fp8-gb300-dynamo-sglang-mtp + - dsv4-fp4-b200-dynamo-sglang-agentic-agg + - dsv4-fp4-b200-dynamo-sglang-agentic-disagg + - qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp + - qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg + - dsr1-fp4-mi355x-sglang + - dsr1-fp4-mi355x-sglang-mtp + - dsr1-fp4-mi355x-atom + - dsr1-fp4-mi355x-atom-mtp + - dsr1-fp8-mi300x-sglang + - dsr1-fp8-mi325x-sglang + - dsr1-fp8-mi355x-sglang + - dsr1-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + - qwen3.5-fp8-mi355x-atom + - qwen3.5-fp8-mi355x-atom-mtp + - qwen3.5-fp4-mi355x-sglang + - qwen3.5-fp4-mi355x-atom + - qwen3.5-fp4-mi355x-sglang-mtp + - qwen3.5-fp8-mi300x-sglang + - dsr1-fp8-mi355x-atom + - dsr1-fp8-mi355x-atom-mtp + - dsr1-fp8-mi325x-sglang-mtp + - qwen3.5-fp8-mi325x-sglang-mtp + description: + - Return the shared SRT submodule to NVIDIA upstream main 3cbc5dd256af2bfd2fed09b724628c3f5456c85f. + Use native AToMesh with a digest-pinned official router image for single-node ATOM while preserving + worker images and settings; use native TRT-LLM served_model_name forwarding. + - 共享 SRT 子模块切回 NVIDIA 上游 main 的 3cbc5dd256af2bfd2fed09b724628c3f5456c85f。单节点 ATOM 使用原生 AToMesh 和按摘要固定的官方路由器镜像,保留 + worker 镜像及设置;TRT-LLM 使用原生 served_model_name 参数转发。 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/utils/srt-slurm b/utils/srt-slurm index c29ef7c5d0..3cbc5dd256 160000 --- a/utils/srt-slurm +++ b/utils/srt-slurm @@ -1 +1 @@ -Subproject commit c29ef7c5d0732fbf7fc93aa4b7929b48f0bfa76a +Subproject commit 3cbc5dd256af2bfd2fed09b724628c3f5456c85f From b0a9064970cb6eb32f00a6a86dd41cfe43b7d672 Mon Sep 17 00:00:00 2001 From: cquil11 <60715037+cquil11@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:22:32 +0000 Subject: [PATCH 26/29] chore: bump srt-slurm submodule to v2.23.2 Move the shared SRT submodule from v2.22.1 (3cbc5dd) to upstream release v2.23.2 (8dace5f). Upstream delta is three additive changes: vllm-router multi-node hybrid-DP rank offsets, SGLang leader IP honoring the cluster network interface, and nsys/DSight coverage for SGLang and Dynamo. No recipe or connector changes required; utils/ and runners/ tests pass. --- perf-changelog.yaml | 4 ++-- utils/srt-slurm | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 5aa58662f5..f96aca850b 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -9463,9 +9463,9 @@ - dsr1-fp8-mi325x-sglang-mtp - qwen3.5-fp8-mi325x-sglang-mtp description: - - Return the shared SRT submodule to NVIDIA upstream main 3cbc5dd256af2bfd2fed09b724628c3f5456c85f. + - Return the shared SRT submodule to NVIDIA upstream release v2.23.2 (8dace5f9596907a5075bf056251563b2e9563e7d). Use native AToMesh with a digest-pinned official router image for single-node ATOM while preserving worker images and settings; use native TRT-LLM served_model_name forwarding. - - 共享 SRT 子模块切回 NVIDIA 上游 main 的 3cbc5dd256af2bfd2fed09b724628c3f5456c85f。单节点 ATOM 使用原生 AToMesh 和按摘要固定的官方路由器镜像,保留 + - 共享 SRT 子模块切回 NVIDIA 上游发行版 v2.23.2(8dace5f9596907a5075bf056251563b2e9563e7d)。单节点 ATOM 使用原生 AToMesh 和按摘要固定的官方路由器镜像,保留 worker 镜像及设置;TRT-LLM 使用原生 served_model_name 参数转发。 pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/utils/srt-slurm b/utils/srt-slurm index 3cbc5dd256..8dace5f959 160000 --- a/utils/srt-slurm +++ b/utils/srt-slurm @@ -1 +1 @@ -Subproject commit 3cbc5dd256af2bfd2fed09b724628c3f5456c85f +Subproject commit 8dace5f9596907a5075bf056251563b2e9563e7d From 91f040d147c6c06d20ef1647f7fd16b4935d2a59 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 24 Sep 2026 11:51:26 -0500 Subject: [PATCH 27/29] refactor(srt): check MI300X firmware in a setup hook Replace the MI300X container preamble with a host setup hook that fails on MEC firmware older than 177 instead of exporting HSA_NO_SCRATCH_RECLAIM in every container. --- runners/launch_mi300x-amd.sh | 2 +- runners/srt-slurm/hooks/mi300x-amd/setup.sh | 12 ++++++++++++ runners/srt-slurm/mi300x-amd.yaml | 10 +++------- 3 files changed, 16 insertions(+), 8 deletions(-) create mode 100755 runners/srt-slurm/hooks/mi300x-amd/setup.sh diff --git a/runners/launch_mi300x-amd.sh b/runners/launch_mi300x-amd.sh index d07899d6fb..e881f41dc2 100644 --- a/runners/launch_mi300x-amd.sh +++ b/runners/launch_mi300x-amd.sh @@ -20,7 +20,7 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then export SALLOC_TIME_LIMIT=180 export SRT_SRUN_OPTIONS='{"container-remap-root":"", "container-writable":""}' SRT_SQUASH_FILE="/raid/inferencex/squash/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" - launch_srt_single_node mi300x-amd + launch_srt_single_node mi300x-amd --var GITHUB_WORKSPACE "$GITHUB_WORKSPACE" exit $? fi diff --git a/runners/srt-slurm/hooks/mi300x-amd/setup.sh b/runners/srt-slurm/hooks/mi300x-amd/setup.sh new file mode 100755 index 0000000000..1eeb9bfe32 --- /dev/null +++ b/runners/srt-slurm/hooks/mi300x-amd/setup.sh @@ -0,0 +1,12 @@ +#!/usr/bin/env bash +set -eo pipefail + +# RCCL cannot reclaim scratch memory on MEC firmware older than 177 and crashes. +# See https://rocm.docs.amd.com/en/docs-6.4.3/about/release-notes.html#amdgpu-driver-updates +minimum=177 +firmware=$(rocm-smi --showfw | awk '/MEC firmware version/ {print $NF}' | sort -n | head -n 1) +if [[ -z "$firmware" || "$firmware" -lt "$minimum" ]]; then + echo "[$(hostname -s)] MEC firmware ${firmware:-unknown} is older than $minimum" >&2 + exit 1 +fi +echo "[$(hostname -s)] MEC firmware $firmware" diff --git a/runners/srt-slurm/mi300x-amd.yaml b/runners/srt-slurm/mi300x-amd.yaml index 7890772181..9bb06997c8 100644 --- a/runners/srt-slurm/mi300x-amd.yaml +++ b/runners/srt-slurm/mi300x-amd.yaml @@ -12,10 +12,6 @@ default_sbatch_directives: use_gpus_per_node_directive: true use_segment_sbatch_directive: false use_exclusive_sbatch_directive: true -default_bash_preamble: |- - if [[ "$$MODEL_PREFIX" == dsr1 && "$$FRAMEWORK" == sglang ]]; then - firmware=$$(rocm-smi --showfw | awk '/MEC/ {print $$NF; exit}') - if [[ -z "$$firmware" || "$$firmware" -lt 177 ]]; then - export HSA_NO_SCRATCH_RECLAIM=1 - fi - fi +default_host_setup: + commands: + - bash "${GITHUB_WORKSPACE}/runners/srt-slurm/hooks/mi300x-amd/setup.sh" From 5fe80b5bd90c553ebadce7bc95950de91afcff29 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 24 Sep 2026 12:05:56 -0500 Subject: [PATCH 28/29] fix(srt): apply pending upstream srt-slurm patches at setup Post-eval steps dropped the recipe's srun_options, so ATOM evals on MI355X ran in a read-only container. Apply NVIDIA/srt-slurm#504 to the job clone until it merges. --- runners/slurm_utils.sh | 6 +++++ .../patches/504-post-eval-srun-options.patch | 25 +++++++++++++++++++ runners/srt-slurm/patches/README.md | 9 +++++++ 3 files changed, 40 insertions(+) create mode 100644 runners/srt-slurm/patches/504-post-eval-srun-options.patch create mode 100644 runners/srt-slurm/patches/README.md diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 2a77e90cc4..17b0525f86 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -73,6 +73,12 @@ PYENV SRTCTL_EVAL_ARGS+=(--set benchmark.stream_output=true) # A local clone keeps job writes isolated and preserves upstream Git provenance. git clone --no-hardlinks "$source" "$destination" || return 1 + # Temporary fixes awaiting upstream merge; see runners/srt-slurm/patches/README.md. + local patch + for patch in "$GITHUB_WORKSPACE"/runners/srt-slurm/patches/*.patch; do + [[ -e "$patch" ]] || continue + git -C "$destination" apply "$patch" || return 1 + done fi cd "$destination" || return 1 [[ "$(git rev-parse HEAD)" == "$SRT_SLURM_COMMIT" ]] || return 1 diff --git a/runners/srt-slurm/patches/504-post-eval-srun-options.patch b/runners/srt-slurm/patches/504-post-eval-srun-options.patch new file mode 100644 index 0000000000..196295312d --- /dev/null +++ b/runners/srt-slurm/patches/504-post-eval-srun-options.patch @@ -0,0 +1,25 @@ +diff --git a/docs/config-reference.md b/docs/config-reference.md +index 17cfccba8..d6ddf2874 100644 +--- a/docs/config-reference.md ++++ b/docs/config-reference.md +@@ -2014,6 +2014,8 @@ post_eval: + + `MODEL_NAME` (the served model name) and `EVAL_CONC` are always set by srtctl. `srtctl dry-run` prints the effective dispatch. + ++Eval steps inherit the recipe's `srun_options`, including container flags such as `container-writable`. ++ + --- + + ## services +diff --git a/src/srtctl/cli/do_sweep.py b/src/srtctl/cli/do_sweep.py +index 4ff19ade6..ceba6ad6b 100644 +--- a/src/srtctl/cli/do_sweep.py ++++ b/src/srtctl/cli/do_sweep.py +@@ -603,6 +603,7 @@ def _run_post_eval(self, stop_event: threading.Event) -> int: + container_image=str(self.runtime.container_image), + container_mounts=self.runtime.container_mounts, + env_to_set=env_to_set, ++ srun_options=self.runtime.srun_options, + het_group=self.runtime.nodes.het_group_for(self.runtime.nodes.head), + ) + diff --git a/runners/srt-slurm/patches/README.md b/runners/srt-slurm/patches/README.md new file mode 100644 index 0000000000..cb619d0506 --- /dev/null +++ b/runners/srt-slurm/patches/README.md @@ -0,0 +1,9 @@ +# srt-slurm patches + +`setup_srt_slurm()` in [`runners/slurm_utils.sh`](../../slurm_utils.sh) applies every `*.patch` here to the job's srt-slurm clone after checking out the pinned submodule. TileRT jobs use the fork checkout and skip these patches. + +Each patch is a temporary fix for an open upstream PR. When the PR merges and the submodule pin includes it, delete the patch and its row. + +| Patch | Upstream PR | Fix | +|-------|-------------|-----| +| `504-post-eval-srun-options.patch` | [NVIDIA/srt-slurm#504](https://github.com/NVIDIA/srt-slurm/pull/504) | Forward recipe `srun_options` (e.g. `container-writable`) to post-eval steps | From 2016415bef90b1772e97172ebedb5845ed98baf3 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 24 Sep 2026 12:24:05 -0500 Subject: [PATCH 29/29] chore: remove perf changelog and smoke-test notes from the single-node port Drop the pilot, cutover and eval-fix changelog entries, the planner support for changelog keys retired later in the PR that only those entries needed, and the doc note about historical smokes. --- docs/ci-procedures.md | 2 - docs/ci-procedures_zh.md | 2 - docs/configuration-procedures.md | 4 +- docs/configuration-procedures_zh.md | 3 +- infx/matrix/plan.py | 31 +- perf-changelog.yaml | 686 ---------------------------- utils/test_process_changelog.py | 33 -- 7 files changed, 5 insertions(+), 756 deletions(-) diff --git a/docs/ci-procedures.md b/docs/ci-procedures.md index 234ace440d..74d08d3867 100644 --- a/docs/ci-procedures.md +++ b/docs/ci-procedures.md @@ -180,8 +180,6 @@ uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with p Add `--all-evals` and/or `--evals-only` when those PR modifiers will be active. [`run-sweep.yml`](../.github/workflows/run-sweep.yml) passes the same flags. Never use a formatter to rewrite `perf-changelog.yaml`, and never treat `yaml.safe_load` alone as sufficient changelog validation. -Changelog entries may name configs retired later in the same PR. Exact keys absent from active masters but present in the matching vendor `configs/deprecated/` archive remain in changelog metadata and produce no jobs. Active definitions take precedence over archived versions; wildcards match only active configs, unknown keys still fail, and `append-only` entries cannot retire configs. - ## Manual end-to-end dispatch Use [`e2e-tests.yml`](../.github/workflows/e2e-tests.yml) for a bounded one-off run only after the identical generator command succeeds locally. Make the test name unique. In the common pattern, `--ref main` selects the deployed workflow definition while input `ref` selects the branch or SHA to measure. Setup resolves that ref once and passes its checkout SHA to all eight benchmark/eval routes, covering single-node, multi-node, fixed-sequence, and AgentX jobs. Queued jobs keep that SHA if the branch advances. With no input `ref`, the run uses `github.sha` as before. diff --git a/docs/ci-procedures_zh.md b/docs/ci-procedures_zh.md index 5e97e2b75e..fd0f769ac3 100644 --- a/docs/ci-procedures_zh.md +++ b/docs/ci-procedures_zh.md @@ -172,8 +172,6 @@ uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with p 当 PR 将启用对应修饰标签时,加入 `--all-evals` 和/或 `--evals-only`;[`run-sweep.yml`](../.github/workflows/run-sweep.yml) 会传递相同参数。绝不使用 Formatter 重写 `perf-changelog.yaml`,也绝不能把单独通过 `yaml.safe_load` 当作充分的 Changelog 验证。 -Changelog 条目可能引用同一 PR 后续提交中退役的配置。若精确键名已从活跃主配置移除,但仍存在于对应厂商的 `configs/deprecated/` 归档中,该条目会保留在 Changelog 元数据中,但不生成作业。活跃定义优先于归档版本;通配符仅匹配活跃配置,未知键名仍报错,`append-only` 条目不得退役配置。 - ## 手动端到端派发 仅在完全相同的生成器命令已于本地成功后,才使用 [`e2e-tests.yml`](../.github/workflows/e2e-tests.yml) 执行受限的一次性 Run。测试名称必须唯一。在通用模式中,`--ref main` 选择已部署的 Workflow 定义,输入 `ref` 则选择要测量的 Branch 或 SHA。Setup 只解析一次该 ref,并将 checkout SHA 传给全部八条基准测试和评测路径,覆盖单节点、多节点、固定序列和 AgentX Job。即使分支在排队期间前进,后续 Job 仍使用该 SHA。未提供输入 `ref` 时,仍使用 `github.sha`。 diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 6f4c12983c..c8285d40bf 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -42,9 +42,7 @@ the native `atomesh` frontend with one aggregate worker and Keep `model.container` aligned with the master config's worker `image`; changing the router image does not require changing the worker image. TRT-LLM recipes use native `engine.served_model_name`, without duplicating that flag in `roles.agg.extra_args`. -The former fork's direct ATOM frontend is not required. Historical direct-serving -smokes do not qualify the new router path; verify startup, requests, power, cleanup, -and performance on the new pin. +The former fork's direct ATOM frontend is not required. ### Cluster profiles diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 652fc490cd..a91e4415ea 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -40,8 +40,7 @@ frontend、一个聚合 worker,并设置 `enable_multiple_frontends: false`。 镜像不包含 AToMesh,因此通过 `frontend.container_image` 单独固定路由器的官方镜像。 `model.container` 必须与主配置中的 worker `image` 一致;更换路由器镜像无需更换 worker 镜像。TRT-LLM 配方使用原生 `engine.served_model_name`,不再通过 `roles.agg.extra_args` -重复传入该参数。不再依赖此前分叉中的 ATOM 直连 frontend。历史直连 smoke 不能作为新路由 -路径的验收证据;需在新固定版本上验证启动、请求、功耗、清理和性能。 +重复传入该参数。不再依赖此前分叉中的 ATOM 直连 frontend。 ## 规程索引 diff --git a/infx/matrix/plan.py b/infx/matrix/plan.py index 66f06e85e5..a964e1984f 100644 --- a/infx/matrix/plan.py +++ b/infx/matrix/plan.py @@ -11,7 +11,7 @@ import tempfile import traceback from collections import defaultdict -from collections.abc import Collection, Iterator +from collections.abc import Iterator from contextlib import ExitStack, contextmanager from dataclasses import dataclass from pathlib import Path @@ -90,12 +90,7 @@ def filter_eval_rows_by_prefill_ep(eval_rows: list[dict], min_prefill_ep: int | return kept -def get_config_keys_from_master( - config_keys: list[str], - master_config: dict, - *, - deprecated_config_keys: Collection[str] = (), -) -> list[str]: +def get_config_keys_from_master(config_keys: list[str], master_config: dict) -> list[str]: resolved_keys = {} for key in config_keys: if "*" in key: @@ -108,8 +103,6 @@ def get_config_keys_from_master( for matched_key in matched_keys: resolved_keys.setdefault(matched_key, None) elif key not in master_config: - if key in deprecated_config_keys: - continue raise ValueError(f"Config key '{key}' not found in master configs.") else: resolved_keys.setdefault(key, None) @@ -494,27 +487,9 @@ def generate_current( error, ) from error - # A later commit can retire a config named by an earlier changelog entry. - # Preserve that history without scheduling archives or accepting typos. - # Append-only entries still require every selected config to remain active. - deprecated_keys: set[str] = set() - if not has_append_only and any( - "*" not in key and key not in master_config - for entry in parsed_entries - for key in entry.config_keys - ): - archives = [Path(path).parent / "deprecated" / Path(path).name for path in config_files] - deprecated_keys = set( - load_config_files( - [str(path) for path in archives if path.is_file()], validate=False - ) - ) - resolved_entries = [] for entry in parsed_entries: - all_configs = get_config_keys_from_master( - entry.config_keys, master_config, deprecated_config_keys=deprecated_keys - ) + all_configs = get_config_keys_from_master(entry.config_keys, master_config) resolved_entries.append((entry, all_configs)) base_inputs = None diff --git a/perf-changelog.yaml b/perf-changelog.yaml index f96aca850b..3323c6fbec 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8783,689 +8783,3 @@ - "Run a new canonical full sweep with 16 performance and 16 full-evaluation cells. Earlier EP4/EP8 and DP8 measurements remain historical evidence and do not qualify the EP1 grid." - "Extend only H200 SGLang AgentX performance cells at C64/C128 to 24-hour Slurm allocations and 24.5-hour GitHub jobs after the original eight-hour allocation expired during progressing warmup. Preserve normal warmup, the full 3600-second profile, image, model, precision and topology." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3341 - -- config-keys: - - dsr1-fp8-h200-sglang - description: - - "Begin the parallel single-node SRT-Slurm port with native H200 SGLang 8k1k serving settings and the existing InferenceX benchmark client; production routing is unchanged and GPU parity remains unqualified" - - "开始并行构建单节点 SRT-Slurm 迁移:采用原生 H200 SGLang 8k1k 服务设置并复用现有 InferenceX 基准客户端;生产路由未变,GPU 性能一致性尚未验收" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp8-h200-sglang - description: - - "Wire the opt-in H200 native SRT single-node pilot through the matrix, launcher, result and GPU power paths; keep production routing unchanged pending qualification" - - "将显式启用的 H200 原生 SRT 单节点试点接入矩阵、启动器、结果及 GPU 功耗流程;验收完成前保留现有生产路由" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp8-h200-sglang - - dsr1-fp8-h200-sglang-mtp - - qwen3.5-fp8-h200-sglang - description: - - "Extend the opt-in native H200 SRT pilot to DeepSeek-R1 MTP and Qwen3.5 EP8, preserving real verification, chat formatting, and concurrency-dependent graph capture; select 4/16/64 for sequential regression checks without production cutover" - - "将显式启用的原生 H200 SRT 试点扩展到 DeepSeek-R1 MTP 和 Qwen3.5 EP8,保留真实验证、聊天模板及随并发变化的图捕获;选择 4/16/64 逐点检查回归,尚不切换生产路由" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-b200-sglang - - dsr1-fp4-b200-sglang-mtp - - dsr1-fp4-b300-sglang - - dsr1-fp8-b200-sglang - - dsr1-fp8-b300-sglang - - qwen3.5-fp8-b200-sglang - - qwen3.5-fp4-b200-sglang - - qwen3.5-fp4-b200-sglang-mtp - - qwen3.5-fp8-b200-sglang-mtp - - qwen3.5-fp8-b300-sglang-mtp - - qwen3.5-fp8-b300-sglang - - qwen3.5-fp4-b300-sglang - - qwen3.5-fp4-b300-sglang-mtp - - dsr1-fp8-b200-sglang-mtp - - dsr1-fp8-b300-sglang-mtp - - dsr1-fp8-h200-sglang - - dsr1-fp8-h200-sglang-mtp - - qwen3.5-fp8-h200-sglang - - qwen3.5-fp8-h200-sglang-mtp - - qwen3.5-fp8-h100-sglang - - qwen3.5-fp8-h100-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - "Cut over 21 active H100/H200/B200/B300 SGLang fixed-sequence recipes to native SRT-Slurm across their Slurm pools, retaining all 181 configured points, per-point serving settings, real MTP verification, shared eval dispatch, and existing result/power artifacts" - - "将 21 个活跃的 H100/H200/B200/B300 SGLang 定长配方切换到各自 Slurm 池的原生 SRT-Slurm 路径,保留全部 181 个配置点、逐点服务参数、真实 MTP 验证、共享 eval 调度及现有结果和功耗产物" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-b200-trt - - dsr1-fp4-b200-trt-mtp - - dsr1-fp8-b200-trt - - dsr1-fp8-b200-trt-mtp - - dsr1-fp8-h200-trt - - dsr1-fp8-h200-trt-mtp - - qwen3.5-fp4-b200-trt - - qwen3.5-fp4-b200-trt-mtp - scenario-type: - - fixed-seq-len - description: - - "Cut over eight H200/B200 TRT-LLM fixed-sequence recipes to native SRT-Slurm, preserving all 63 points, engine configuration, real MTP verification, and the OpenAI benchmark client; retain explicit legacy metrics settings and eval token budgets" - - "将八个 H200/B200 TRT-LLM 定长配方切换至原生 SRT-Slurm,保留全部 63 个测试点、引擎配置、真实 MTP 验证及 OpenAI 基准客户端;显式保留原有指标设置和 eval token 预算" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-mi355x-sglang - - dsr1-fp4-mi355x-sglang-mtp - - dsr1-fp4-mi355x-atom - - dsr1-fp4-mi355x-atom-mtp - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - dsr1-fp8-mi355x-sglang - - dsr1-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi355x-sglang - - qwen3.5-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi355x-atom - - qwen3.5-fp8-mi355x-atom-mtp - - qwen3.5-fp4-mi355x-sglang - - qwen3.5-fp4-mi355x-atom - - qwen3.5-fp4-mi355x-sglang-mtp - - qwen3.5-fp8-mi300x-sglang - - dsr1-fp8-mi355x-atom - - dsr1-fp8-mi355x-atom-mtp - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - - qwen3.5-fp4-rtx6000pro-sglang - - qwen3.5-fp4-rtx6000pro-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - Cut over the remaining 23 AMD SGLang/ATOM and RTX PRO 6000 fixed-sequence recipes to native SRT configuration, retaining all 179 points and legacy server/client settings; use direct ATOM serving and native command generation within the existing Docker pool lifecycle - - 将剩余 23 个 AMD SGLang/ATOM 和 RTX PRO 6000 定长配方切换至原生 SRT 配置,保留全部 179 个测试点及原有服务端和客户端参数;ATOM 使用直接服务模式,Docker 池在现有容器生命周期内使用原生命令生成 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-b200-sglang - - dsr1-fp4-b200-sglang-mtp - - dsr1-fp4-b300-sglang - - dsr1-fp4-b200-trt - - dsr1-fp4-b200-trt-mtp - - dsr1-fp8-b200-sglang - - dsr1-fp8-b300-sglang - - qwen3.5-fp8-b200-sglang - - qwen3.5-fp4-b200-sglang - - qwen3.5-fp4-b200-sglang-mtp - - qwen3.5-fp8-b200-sglang-mtp - - qwen3.5-fp8-b300-sglang-mtp - - qwen3.5-fp8-b300-sglang - - qwen3.5-fp4-b300-sglang - - qwen3.5-fp4-rtx6000pro-sglang - - qwen3.5-fp4-rtx6000pro-sglang-mtp - - qwen3.5-fp4-b300-sglang-mtp - - dsr1-fp8-b200-sglang-mtp - - dsr1-fp8-b300-sglang-mtp - - dsr1-fp8-b200-trt - - dsr1-fp8-b200-trt-mtp - - dsr1-fp8-h200-sglang - - dsr1-fp8-h200-sglang-mtp - - qwen3.5-fp8-h200-sglang - - qwen3.5-fp8-h200-sglang-mtp - - dsr1-fp8-h200-trt - - dsr1-fp8-h200-trt-mtp - - qwen3.5-fp8-h100-sglang - - qwen3.5-fp8-h100-sglang-mtp - - qwen3.5-fp4-b200-trt - - qwen3.5-fp4-b200-trt-mtp - - dsr1-fp4-mi355x-sglang - - dsr1-fp4-mi355x-sglang-mtp - - dsr1-fp4-mi355x-atom - - dsr1-fp4-mi355x-atom-mtp - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - dsr1-fp8-mi355x-sglang - - dsr1-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi355x-sglang - - qwen3.5-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi355x-atom - - qwen3.5-fp8-mi355x-atom-mtp - - qwen3.5-fp4-mi355x-sglang - - qwen3.5-fp4-mi355x-atom - - qwen3.5-fp4-mi355x-sglang-mtp - - qwen3.5-fp8-mi300x-sglang - - dsr1-fp8-mi355x-atom - - dsr1-fp8-mi355x-atom-mtp - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - Pin the native SRT runtime to the reviewed-in-draft AMD/ATOM integration stack plus direct aggregate ATOM support; keep runner hardware settings in cluster profiles and exclude the held served-model-name PR - - 将原生 SRT 运行时固定到待评审的 AMD/ATOM 集成依赖链及 ATOM 聚合直接服务支持;硬件设置保留在集群配置中,不包含暂缓的 served-model-name PR - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - qwen3.5-fp4-rtx6000pro-sglang - - qwen3.5-fp4-rtx6000pro-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - "Forward the native Docker server model identity into the shared eval path and reject missing eval metadata before startup" - - "将原生 Docker 服务的模型名称传入共享 eval 路径,并在启动前拒绝缺失的 eval 元数据" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-mi355x-sglang - - dsr1-fp4-mi355x-sglang-mtp - - dsr1-fp4-mi355x-atom - - dsr1-fp4-mi355x-atom-mtp - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - dsr1-fp8-mi355x-sglang - - dsr1-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi355x-sglang - - qwen3.5-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi355x-atom - - qwen3.5-fp8-mi355x-atom-mtp - - qwen3.5-fp4-mi355x-sglang - - qwen3.5-fp4-mi355x-atom - - qwen3.5-fp4-mi355x-sglang-mtp - - qwen3.5-fp8-mi300x-sglang - - dsr1-fp8-mi355x-atom - - dsr1-fp8-mi355x-atom-mtp - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - Pass AMD container runtime options as native dotted leaf overrides so SRT validates them as a mapping before submission; surface native submission errors in workflow logs - - 将 AMD 容器运行时选项作为原生点路径叶值传递,确保 SRT 在提交前将其校验为映射,并在工作流日志中显示原生提交错误 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - Strip the trailing newline from AMD native shell preambles and verify completed - allocations through the Slurm controller when accounting is unavailable; reap - failed-startup log followers before exit - - 去除 AMD 原生 shell 前置命令的尾部换行,并在 accounting 不可用时通过 Slurm 控制器核验已完成的资源分配;启动失败时等待日志跟随进程退出 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - qwen3.5-fp4-rtx6000pro-sglang - - qwen3.5-fp4-rtx6000pro-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - Exclude the Docker-only RTX PRO 6000 configs from the native SRT migration and - retain their existing Docker execution; remove the proposed Docker adapter and - recipes from this PR - - 将仅使用 Docker 的 RTX PRO 6000 配置排除在原生 SRT 迁移之外,保留现有 Docker 执行路径;从本 PR 中移除拟新增的 Docker - 适配器与配方 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-mi355x-sglang - - dsr1-fp4-mi355x-sglang-mtp - - dsr1-fp4-mi355x-atom - - dsr1-fp4-mi355x-atom-mtp - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - dsr1-fp8-mi355x-sglang - - dsr1-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi355x-sglang - - qwen3.5-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi355x-atom - - qwen3.5-fp8-mi355x-atom-mtp - - qwen3.5-fp4-mi355x-sglang - - qwen3.5-fp4-mi355x-atom - - qwen3.5-fp4-mi355x-sglang-mtp - - qwen3.5-fp8-mi300x-sglang - - dsr1-fp8-mi355x-atom - - dsr1-fp8-mi355x-atom-mtp - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - - dsr1-fp4-b200-sglang - - dsr1-fp4-b200-sglang-mtp - - dsr1-fp4-b300-sglang - - dsr1-fp4-b200-trt - - dsr1-fp4-b200-trt-mtp - - dsr1-fp8-b200-sglang - - dsr1-fp8-b300-sglang - - qwen3.5-fp8-b200-sglang - - qwen3.5-fp4-b200-sglang - - qwen3.5-fp4-b200-sglang-mtp - - qwen3.5-fp8-b200-sglang-mtp - - qwen3.5-fp8-b300-sglang-mtp - - qwen3.5-fp8-b300-sglang - - qwen3.5-fp4-b300-sglang - - qwen3.5-fp4-b300-sglang-mtp - - dsr1-fp8-b200-sglang-mtp - - dsr1-fp8-b300-sglang-mtp - - dsr1-fp8-b200-trt - - dsr1-fp8-b200-trt-mtp - - dsr1-fp8-h200-sglang - - dsr1-fp8-h200-sglang-mtp - - qwen3.5-fp8-h200-sglang - - qwen3.5-fp8-h200-sglang-mtp - - dsr1-fp8-h200-trt - - dsr1-fp8-h200-trt-mtp - - qwen3.5-fp8-h100-sglang - - qwen3.5-fp8-h100-sglang-mtp - - qwen3.5-fp4-b200-trt - - qwen3.5-fp4-b200-trt-mtp - scenario-type: - - fixed-seq-len - description: - - Require native SRT recipes for active Slurm single-node fixed-sequence jobs and - remove their superseded Bash implementations; keep AgentX, multi-node, and explicitly - selected SPEED-Bench collector paths separate - - 活跃的 Slurm 单节点定长作业必须提供原生 SRT 配方,并移除已替代的 Bash 实现;AgentX、多节点和显式选择的 SPEED-Bench 采集器保留独立执行路径 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - qwen3.5-fp4-rtx6000pro-sglang - - qwen3.5-fp4-rtx6000pro-sglang-mtp - scenario-type: - - fixed-seq-len - description: - - Retire the Docker-only RTX PRO 6000 fixed-sequence configs and runner routing - so active single-node fixed-sequence coverage uses SRT-Slurm exclusively; - preserve the original configs and scripts in the existing deprecated archives - - 退役仅支持 Docker 的 RTX PRO 6000 定长配置及 runner 路由,使活跃的单节点定长覆盖仅使用 SRT-Slurm;原始配置和脚本保留在现有弃用归档中 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-mi355x-sglang - - dsr1-fp4-mi355x-sglang-mtp - - dsr1-fp4-mi355x-atom - - dsr1-fp4-mi355x-atom-mtp - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - dsr1-fp8-mi355x-sglang - - dsr1-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi355x-sglang - - qwen3.5-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi355x-atom - - qwen3.5-fp8-mi355x-atom-mtp - - qwen3.5-fp4-mi355x-sglang - - qwen3.5-fp4-mi355x-atom - - qwen3.5-fp4-mi355x-sglang-mtp - - qwen3.5-fp8-mi300x-sglang - - dsr1-fp8-mi355x-atom - - dsr1-fp8-mi355x-atom-mtp - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - - dsr1-fp4-b200-sglang - - dsr1-fp4-b200-sglang-mtp - - dsr1-fp4-b300-sglang - - dsr1-fp4-b200-trt - - dsr1-fp4-b200-trt-mtp - - dsr1-fp8-b200-sglang - - dsr1-fp8-b300-sglang - - qwen3.5-fp8-b200-sglang - - qwen3.5-fp4-b200-sglang - - qwen3.5-fp4-b200-sglang-mtp - - qwen3.5-fp8-b200-sglang-mtp - - qwen3.5-fp8-b300-sglang-mtp - - qwen3.5-fp8-b300-sglang - - qwen3.5-fp4-b300-sglang - - qwen3.5-fp4-b300-sglang-mtp - - dsr1-fp8-b200-sglang-mtp - - dsr1-fp8-b300-sglang-mtp - - dsr1-fp8-b200-trt - - dsr1-fp8-b200-trt-mtp - - dsr1-fp8-h200-sglang - - dsr1-fp8-h200-sglang-mtp - - qwen3.5-fp8-h200-sglang - - qwen3.5-fp8-h200-sglang-mtp - - dsr1-fp8-h200-trt - - dsr1-fp8-h200-trt-mtp - - qwen3.5-fp8-h100-sglang - - qwen3.5-fp8-h100-sglang-mtp - - qwen3.5-fp4-b200-trt - - qwen3.5-fp4-b200-trt-mtp - scenario-type: - - fixed-seq-len - description: - - Install native fixed-sequence client dependencies in the container Python environment - without forcing user-site packages, which ROCm image virtual environments disable - - 原生定长客户端依赖安装到容器 Python 环境,不再强制使用 ROCm 镜像虚拟环境禁用的用户级软件包目录 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-mi355x-sglang - - dsr1-fp4-mi355x-sglang-mtp - - dsr1-fp4-mi355x-atom - - dsr1-fp4-mi355x-atom-mtp - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - dsr1-fp8-mi355x-sglang - - dsr1-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi355x-sglang - - qwen3.5-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi355x-atom - - qwen3.5-fp8-mi355x-atom-mtp - - qwen3.5-fp4-mi355x-sglang - - qwen3.5-fp4-mi355x-atom - - qwen3.5-fp4-mi355x-sglang-mtp - - qwen3.5-fp8-mi300x-sglang - - dsr1-fp8-mi355x-atom - - dsr1-fp8-mi355x-atom-mtp - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - - dsr1-fp4-b200-sglang - - dsr1-fp4-b200-sglang-mtp - - dsr1-fp4-b300-sglang - - dsr1-fp4-b200-trt - - dsr1-fp4-b200-trt-mtp - - dsr1-fp8-b200-sglang - - dsr1-fp8-b300-sglang - - qwen3.5-fp8-b200-sglang - - qwen3.5-fp4-b200-sglang - - qwen3.5-fp4-b200-sglang-mtp - - qwen3.5-fp8-b200-sglang-mtp - - qwen3.5-fp8-b300-sglang-mtp - - qwen3.5-fp8-b300-sglang - - qwen3.5-fp4-b300-sglang - - qwen3.5-fp4-b300-sglang-mtp - - dsr1-fp8-b200-sglang-mtp - - dsr1-fp8-b300-sglang-mtp - - dsr1-fp8-b200-trt - - dsr1-fp8-b200-trt-mtp - - dsr1-fp8-h200-sglang - - dsr1-fp8-h200-sglang-mtp - - qwen3.5-fp8-h200-sglang - - qwen3.5-fp8-h200-sglang-mtp - - dsr1-fp8-h200-trt - - dsr1-fp8-h200-trt-mtp - - qwen3.5-fp8-h100-sglang - - qwen3.5-fp8-h100-sglang-mtp - - qwen3.5-fp4-b200-trt - - qwen3.5-fp4-b200-trt-mtp - scenario-type: - - fixed-seq-len - description: - - Bind native single-node Slurm steps to the validated serving GPU count while retaining - exclusive node reservations, so partial-node recipes do not expose idle devices - to benchmark power collection - - 原生单节点 Slurm 步骤使用已验证的推理 GPU 数量并保留节点独占预留,避免部分节点配方将空闲设备暴露给基准功耗采集 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-mi355x-sglang - - dsr1-fp4-mi355x-sglang-mtp - - dsr1-fp4-mi355x-atom - - dsr1-fp4-mi355x-atom-mtp - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - dsr1-fp8-mi355x-sglang - - dsr1-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi355x-sglang - - qwen3.5-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi355x-atom - - qwen3.5-fp8-mi355x-atom-mtp - - qwen3.5-fp4-mi355x-sglang - - qwen3.5-fp4-mi355x-atom - - qwen3.5-fp4-mi355x-sglang-mtp - - qwen3.5-fp8-mi300x-sglang - - dsr1-fp8-mi355x-atom - - dsr1-fp8-mi355x-atom-mtp - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - - dsr1-fp4-b200-sglang - - dsr1-fp4-b200-sglang-mtp - - dsr1-fp4-b300-sglang - - dsr1-fp4-b200-trt - - dsr1-fp4-b200-trt-mtp - - dsr1-fp8-b200-sglang - - dsr1-fp8-b300-sglang - - qwen3.5-fp8-b200-sglang - - qwen3.5-fp4-b200-sglang - - qwen3.5-fp4-b200-sglang-mtp - - qwen3.5-fp8-b200-sglang-mtp - - qwen3.5-fp8-b300-sglang-mtp - - qwen3.5-fp8-b300-sglang - - qwen3.5-fp4-b300-sglang - - qwen3.5-fp4-b300-sglang-mtp - - dsr1-fp8-b200-sglang-mtp - - dsr1-fp8-b300-sglang-mtp - - dsr1-fp8-b200-trt - - dsr1-fp8-b200-trt-mtp - - dsr1-fp8-h200-sglang - - dsr1-fp8-h200-sglang-mtp - - qwen3.5-fp8-h200-sglang - - qwen3.5-fp8-h200-sglang-mtp - - dsr1-fp8-h200-trt - - dsr1-fp8-h200-trt-mtp - - qwen3.5-fp8-h100-sglang - - qwen3.5-fp8-h100-sglang-mtp - - qwen3.5-fp4-b200-trt - - qwen3.5-fp4-b200-trt-mtp - scenario-type: - - fixed-seq-len - description: - - Run native single-node container steps from the existing InferenceX repository - mount, preserving legacy working-directory behavior and avoiding Python generated-module - import failures at the filesystem root - - 原生单节点容器步骤从现有 InferenceX 仓库挂载目录启动,保留旧版工作目录行为,避免在文件系统根目录导入 Python 动态生成模块时失败 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-mi355x-sglang - - dsr1-fp4-mi355x-sglang-mtp - - dsr1-fp4-mi355x-atom - - dsr1-fp4-mi355x-atom-mtp - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - dsr1-fp8-mi355x-sglang - - dsr1-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi355x-sglang - - qwen3.5-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-agentic-mtp - - qwen3.5-fp8-mi355x-atom - - qwen3.5-fp8-mi355x-atom-mtp - - qwen3.5-fp4-mi355x-sglang - - qwen3.5-fp4-mi355x-atom - - qwen3.5-fp4-mi355x-sglang-mtp - - qwen3.5-fp4-mi355x-sglang-agentic-mtp - - qwen3.5-fp8-mi300x-sglang - - qwen3.5-fp8-mi300x-sglang-agentic-mtp - - dsr1-fp8-mi355x-atom - - dsr1-fp8-mi355x-atom-mtp - - kimik3-fp4-mi355x-vllm-agentic-mtp - - kimik3-fp4-mi355x-atom-agentic-mtp - - minimaxm3-fp4-mi355x-atom-agentic-mtp - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - - dsv4-fp4-mi355x-vllm-agentic-mtp - - dsv4-fp4-mi355x-atom-agentic-mtp - - minimaxm3-fp8-mi300x-vllm-agentic-mtp - - dsv41flash-fp4-mi300x-vllm-agentic-dspark - - glm5.2-fp8-mi325x-sglang-agentic-mtp - - minimaxm3-fp8-mi325x-vllm-agentic-mtp - - dsv41flash-fp4-mi325x-vllm-agentic-dspark - - minimaxm3-fp4-mi355x-vllm-agentic-mtp - - glm5.2-fp4-mi355x-sglang-agentic-mtp - - glm5.2-fp8-mi355x-sglang-agentic-mtp - - glm5.2-fp4-mi355x-atom-agentic-mtp - - dsv4-fp4-mi355x-sglang-agentic-mtp - - dsv41flash-fp4-mi355x-vllm-agentic-dspark - description: - - Stream AMD power CSV rows without awk input buffering so the sampler preserves final benchmark-window - samples during shutdown. Keep benchmark settings and power validation unchanged. - - AMD 功耗 CSV 逐行写出,避免 awk 输入缓冲导致采样器关闭时丢失基准窗口末尾的样本;基准设置和功耗校验保持不变。 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - kimik3-fp4-h200-vllm-agentic-latency - - kimik3-fp4-h200-vllm-agentic-balanced - - kimik3-fp4-h200-vllm-agentic-simple - - dsr1-fp8-h200-sglang - - dsr1-fp8-h200-sglang-mtp - - dsv4-fp8-h200-dynamo-sglang-agentic-agg - - qwen3.5-fp8-h200-sglang - - qwen3.5-fp8-h200-sglang-mtp - - dsr1-fp8-h200-trt - - dsr1-fp8-h200-trt-mtp - - dsr1-fp8-h200-dynamo-trt - - dsr1-fp8-h200-dynamo-sglang - - qwen3.5-fp8-h200-sglang-agentic-mtp - - qwen3.8next-fp8-h200-sglang-agentic-mtp - - qwen3.5-fp8-h200-sglang-agentic-hicache-mtp - - minimaxm3-fp8-h200-vllm-agentic-mtp - - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-2p2d - - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-1p1d-hicache - - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-agg - - dsv41flash-fp4-h200-vllm-agentic-dspark - - dsv41flash-fp4-h200-sglang-agentic-dspark - description: - - Align H200 DGXC runner inventory with the verified two-digit GitHub runner labels so exact point selections - and sweep assignments reach registered workers; benchmark recipes and settings are unchanged. - - H200 DGXC runner 清单与已核实的两位数字 GitHub 标签对齐,使精确测试点选择和 sweep 分配能够匹配已注册的 worker;基准配方和设置保持不变。 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 - -- config-keys: - - dsr1-fp4-b200-dynamo-trt - - dsr1-fp8-b200-dynamo-trt - - dsr1-fp4-b300-dynamo-trt - - dsr1-fp8-b300-dynamo-trt - - dsr1-fp4-b200-sglang - - dsr1-fp4-b200-sglang-mtp - - dsr1-fp4-b300-sglang - - dsr1-fp4-b200-trt - - dsr1-fp4-b200-trt-mtp - - dsr1-fp8-b200-sglang - - dsr1-fp8-b300-sglang - - qwen3.5-fp8-b200-sglang - - qwen3.5-fp4-b200-sglang - - qwen3.5-fp4-b200-sglang-mtp - - qwen3.5-fp8-b200-sglang-mtp - - qwen3.5-fp8-b300-sglang-mtp - - qwen3.5-fp8-b300-sglang - - qwen3.5-fp4-b300-sglang - - qwen3.5-fp4-b300-sglang-mtp - - kimik3-fp4-h200-vllm-agentic-latency - - kimik3-fp4-h200-vllm-agentic-balanced - - kimik3-fp4-h200-vllm-agentic-simple - - dsr1-fp8-b200-sglang-mtp - - dsr1-fp8-b300-sglang-mtp - - dsr1-fp8-b200-trt - - dsr1-fp8-b200-trt-mtp - - dsr1-fp8-h200-sglang - - dsr1-fp8-h200-sglang-mtp - - dsv4-fp8-h200-dynamo-sglang-agentic-agg - - qwen3.5-fp8-h200-sglang - - qwen3.5-fp8-h200-sglang-mtp - - dsr1-fp8-h200-trt - - dsr1-fp8-h200-trt-mtp - - dsr1-fp8-h200-dynamo-trt - - dsr1-fp8-h100-dynamo-trt - - dsr1-fp8-h100-dynamo-sglang - - dsr1-fp4-gb200-dynamo-trt - - dsr1-fp8-gb200-dynamo-trt - - dsr1-fp8-gb200-dynamo-sglang - - dsr1-fp8-gb300-dynamo-sglang - - dsr1-fp4-gb200-dynamo-sglang - - dsr1-fp4-gb300-dynamo-trt - - dsr1-fp4-gb300-dynamo-sglang - - dsr1-fp8-gb300-dynamo-trt - - dsr1-fp8-h200-dynamo-sglang - - dsr1-fp4-b200-dynamo-sglang - - dsr1-fp8-b200-dynamo-sglang - - dsr1-fp8-b200-dynamo-sglang-mtp - - dsr1-fp4-b200-dynamo-sglang-mtp - - qwen3.5-fp8-gb200-dynamo-sglang - - qwen3.5-fp8-h100-sglang - - qwen3.5-fp8-h100-sglang-mtp - - qwen3.5-fp4-gb300-dynamo-sglang - - qwen3.5-fp4-gb300-dynamo-trt - - qwen3.5-fp4-gb300-dynamo-trt-mtp - - qwen3.5-fp4-gb300-dynamo-trt-agentic-disagg - - qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg - - qwen3.5-fp4-gb300-dynamo-sglang-agentic-disagg - - qwen3.5-fp4-gb300-dynamo-sglang-agentic-pp-pareto - - qwen3.5-fp4-gb300-dynamo-sglang-agentic-agg-pareto - - dsv4-fp4-gb300-dynamo-vllm-agentic - - qwen3.5-fp4-b200-trt - - qwen3.5-fp4-b200-trt-mtp - - qwen3.5-fp4-gb200-dynamo-sglang-agentic-mtp - - minimaxm3-fp4-gb300-dynamo-vllm-agentic-mtp-disagg - - minimaxm3-fp4-gb200-dynamo-vllm-agentic-agg-mtp - - minimaxm3-fp4-gb200-dynamo-vllm-agentic-disagg-mtp - - minimaxm3-fp4-gb200-dynamo-trt-agentic-agg-mtp - - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-agg - - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp-disagg - - dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-agg - - dsv4-fp4-gb300-dynamo-vllm-agentic-mtp-disagg - - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-agg - - dsv4-fp4-gb200-dynamo-vllm-agentic-mtp2-disagg - - kimik3-fp4-gb200-dynamo-vllm-agentic - - kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-dcp16-agg - - kimik3-fp4-gb200-dynamo-vllm-agentic-mooncake-dcp16-agg - - dsv4-fp4-gb300-dynamo-trt-agentx - - kimik3-fp4-gb200-dynamo-vllm-agentic-dspark-mooncake-tp8pp2 - - dsv4-fp4-gb300-dynamo-sglang-agentic-agg - - dsv4-fp4-gb300-dynamo-sglang-agentic-disagg - - qwen3.5-fp8-gb300-dynamo-sglang - - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-2p2d - - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-1p1d-hicache - - glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-agg - - glm5.2-fp4-b200-dynamo-sglang-agentic-agg - - glm5.2-fp4-b200-dynamo-sglang-agentic-disagg - - glm5.2-fp4-gb200-dynamo-sglang-agentic-agg - - glm5.2-fp4-gb200-dynamo-sglang-agentic-disagg - - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp - - glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg - - glm5.2-fp4-gb300-dynamo-sglang-agentic-agg - - glm5.2-fp4-gb300-dynamo-sglang-agentic-disagg - - glm5.2-fp4-gb300-dynamo-trt-agentic-disagg-mtp - - qwen3.5-fp8-gb200-dynamo-sglang-mtp - - kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-disagg - - kimik3-fp4-gb300-dynamo-vllm-agentic-dspark-mooncake-dcp8-agg - - kimik3-fp4-b200-dynamo-vllm-agentic-dspark - - qwen3.5-fp8-gb300-dynamo-sglang-mtp - - dsv4-fp4-b200-dynamo-sglang-agentic-agg - - dsv4-fp4-b200-dynamo-sglang-agentic-disagg - - qwen3.5-fp8-b200-dynamo-sglang-agentic-disagg-mtp - - qwen3.5-fp8-b300-dynamo-sglang-agentic-disagg - - dsr1-fp4-mi355x-sglang - - dsr1-fp4-mi355x-sglang-mtp - - dsr1-fp4-mi355x-atom - - dsr1-fp4-mi355x-atom-mtp - - dsr1-fp8-mi300x-sglang - - dsr1-fp8-mi325x-sglang - - dsr1-fp8-mi355x-sglang - - dsr1-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang - - qwen3.5-fp8-mi355x-sglang - - qwen3.5-fp8-mi355x-sglang-mtp - - qwen3.5-fp8-mi355x-atom - - qwen3.5-fp8-mi355x-atom-mtp - - qwen3.5-fp4-mi355x-sglang - - qwen3.5-fp4-mi355x-atom - - qwen3.5-fp4-mi355x-sglang-mtp - - qwen3.5-fp8-mi300x-sglang - - dsr1-fp8-mi355x-atom - - dsr1-fp8-mi355x-atom-mtp - - dsr1-fp8-mi325x-sglang-mtp - - qwen3.5-fp8-mi325x-sglang-mtp - description: - - Return the shared SRT submodule to NVIDIA upstream release v2.23.2 (8dace5f9596907a5075bf056251563b2e9563e7d). - Use native AToMesh with a digest-pinned official router image for single-node ATOM while preserving - worker images and settings; use native TRT-LLM served_model_name forwarding. - - 共享 SRT 子模块切回 NVIDIA 上游发行版 v2.23.2(8dace5f9596907a5075bf056251563b2e9563e7d)。单节点 ATOM 使用原生 AToMesh 和按摘要固定的官方路由器镜像,保留 - worker 镜像及设置;TRT-LLM 使用原生 served_model_name 参数转发。 - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3352 diff --git a/utils/test_process_changelog.py b/utils/test_process_changelog.py index f2d1c1b984..febb282ebd 100644 --- a/utils/test_process_changelog.py +++ b/utils/test_process_changelog.py @@ -691,39 +691,6 @@ def test_invalid_key_after_valid_key_rejects_entire_selection(changelog_run, key changelog_run([{"config-keys": keys}]) -@pytest.mark.parametrize("keys,expected_concs", [ - (["retired", "single"], [16, 32, 64]), - (["retired"], []), -]) -def test_archived_changelog_keys_do_not_schedule_retired_configs( - planning_repo, changelog_run, keys, expected_concs, -): - archive = planning_repo[0] / "configs/deprecated" - archive.mkdir() - # Historical settings need not satisfy the current runtime schema. An older - # archived version of an active key must not replace its current definition. - (archive / "nvidia-master.yaml").write_text("retired: {}\nsingle: {}\n") - output = changelog_run([{"config-keys": keys, "scenario-type": ["fixed-seq-len"]}]) - assert [row["conc"] for row in output["single_node"].get("8k1k", [])] == expected_concs - assert [row["conc"] for row in output["evals"]] == ([32, 64] if expected_concs else []) - assert output["multi_node"] == {} - - -@pytest.mark.parametrize("keys,flags,message", [ - (["retired", "missing"], {}, "Config key 'missing' not found"), - (["retired-*"], {}, "No config keys matched"), - (["retired"], {"append-only": True}, "Config key 'retired' not found"), -]) -def test_archives_do_not_relax_unknown_keys_wildcards_or_append_only( - planning_repo, changelog_run, keys, flags, message, -): - archive = planning_repo[0] / "configs/deprecated" - archive.mkdir() - (archive / "nvidia-master.yaml").write_text("retired: {}\nretired-old: {}\n") - with pytest.raises(ValueError, match=message): - changelog_run([{"config-keys": keys, **flags}]) - - @pytest.mark.parametrize("trim", [False, True]) def test_append_only_main_runs_only_added_points_and_skips_evals(planning_repo, changelog_run, trim): root, master, _ = planning_repo