Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 14 additions & 0 deletions .github/workflows/benchmark-tmpl.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,10 @@ name: Template - Benchmark
on:
workflow_call:
secrets:
SRT_STATUS_ENDPOINT:
required: false
SRTCTL_STATUS_TOKEN:
required: false
INFERENCEX_OFFICIAL_RO_HF_TOKEN:
required: true
MODAL_TOKEN_ID:
Expand Down Expand Up @@ -278,6 +282,13 @@ jobs:
submodules: true
persist-credentials: false

- name: Download the patched Tachometer build
if: hashFiles('inferencex-e2e/runners/srt-slurm/patches/539.patch') != ''
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
name: srt-pr539-tachometer
path: srt-streaming-build

- name: Checkout result tooling
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
Expand All @@ -299,6 +310,9 @@ jobs:

- name: Launch job script
env:
SRT_STATUS_ENDPOINT: ${{ secrets.SRT_STATUS_ENDPOINT }} # zizmor: ignore[secrets-outside-env]
SRTCTL_STATUS_TOKEN: ${{ secrets.SRTCTL_STATUS_TOKEN }} # zizmor: ignore[secrets-outside-env]
SRT_TACHOMETER_BUILD_DIR: ${{ github.workspace }}/srt-streaming-build
RUNNER_NAME: ${{ runner.name }}
RUNNER_TYPE: ${{ inputs.runner }}
RESULT_FILENAME_BASE: ${{ env.EXP_NAME }}_${{ env.PRECISION }}_${{ env.FRAMEWORK }}_tp${{ env.TP }}-pp${{ env.PP_SIZE }}-dcp${{ env.DCP_SIZE }}-pcp${{ env.PCP_SIZE }}-ep${{ env.EP_SIZE }}-dpa${{ env.DP_ATTENTION }}_disagg-${{ env.DISAGG }}_spec-${{ env.SPEC_DECODING }}_conc${{ env.CONC }}_${{ runner.name }}
Expand Down
58 changes: 52 additions & 6 deletions .github/workflows/run-sweep.yml
Original file line number Diff line number Diff line change
Expand Up @@ -404,6 +404,35 @@ jobs:
['Reused benchmark source', process.env.REUSE_SHA],
]).write();

build-srt-tachometer:
name: Build patched Tachometer
needs: setup
if: ${{ !cancelled() && needs.setup.result == 'success' && needs.setup.outputs.reuse-enabled != 'true' }}
runs-on: ubuntu-24.04
timeout-minutes: 30
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: ${{ github.event.pull_request.head.sha || github.sha }}
submodules: true
persist-credentials: false
- name: Apply PR 539 and build its atomic writer
working-directory: inferencex-e2e/utils/srt-slurm
run: |
git apply ../../runners/srt-slurm/patches/539.patch
cargo build --release --locked --bin tachometer-scraper
mkdir -p "$GITHUB_WORKSPACE/srt-streaming-build"
cp target/release/tachometer-scraper "$GITHUB_WORKSPACE/srt-streaming-build/"
git rev-parse HEAD > "$GITHUB_WORKSPACE/srt-streaming-build/base-commit.txt"
cd "$GITHUB_WORKSPACE/srt-streaming-build"
sha256sum tachometer-scraper > tachometer.sha256
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: srt-pr539-tachometer
path: srt-streaming-build
if-no-files-found: error
retention-days: 7

canary-select:
name: canary-select
needs: setup
Expand Down Expand Up @@ -462,7 +491,7 @@ jobs:
} >> "$GITHUB_OUTPUT"

canary-sweep:
needs: canary-select
needs: [canary-select, build-srt-tachometer]
if: ${{ needs.canary-select.outputs.canary-config != '' && needs.canary-select.outputs.canary-config != '[]' }}
uses: $/.github/workflows/benchmark-tmpl.yml
name: canary /
Expand All @@ -471,6 +500,8 @@ jobs:
matrix:
config: ${{ fromJson(needs.canary-select.outputs.canary-config) }}
secrets:
SRT_STATUS_ENDPOINT: ${{ secrets.SRT_STATUS_ENDPOINT }}
SRTCTL_STATUS_TOKEN: ${{ secrets.SRTCTL_STATUS_TOKEN }}
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
Expand Down Expand Up @@ -575,11 +606,12 @@ jobs:
with: *multi-node-inputs

sweep-single-node-1k1k:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep, build-srt-tachometer]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.build-srt-tachometer.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
Expand All @@ -593,6 +625,8 @@ jobs:
matrix:
config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k'] }}
secrets:
SRT_STATUS_ENDPOINT: ${{ secrets.SRT_STATUS_ENDPOINT }}
SRTCTL_STATUS_TOKEN: ${{ secrets.SRTCTL_STATUS_TOKEN }}
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
Expand All @@ -608,11 +642,12 @@ jobs:
run-eval: ${{ matrix.config.run-eval }}

sweep-single-node-8k1k:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep, build-srt-tachometer]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.build-srt-tachometer.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
Expand All @@ -626,17 +661,20 @@ jobs:
matrix:
config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k'] }}
secrets:
SRT_STATUS_ENDPOINT: ${{ secrets.SRT_STATUS_ENDPOINT }}
SRTCTL_STATUS_TOKEN: ${{ secrets.SRTCTL_STATUS_TOKEN }}
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with: *single-node-inputs

sweep-agentic:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep, build-srt-tachometer]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.build-srt-tachometer.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
Expand All @@ -650,6 +688,8 @@ jobs:
matrix:
config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['agentic'] }}
secrets:
SRT_STATUS_ENDPOINT: ${{ secrets.SRT_STATUS_ENDPOINT }}
SRTCTL_STATUS_TOKEN: ${{ secrets.SRTCTL_STATUS_TOKEN }}
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
Expand Down Expand Up @@ -705,11 +745,12 @@ jobs:
agentx-fast: ${{ contains(github.event.pull_request.labels.*.name, 'agentx-fast') }}

sweep-evals:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep, build-srt-tachometer]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.build-srt-tachometer.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
Expand All @@ -723,6 +764,8 @@ jobs:
matrix:
config: ${{ fromJson(needs.setup.outputs.search-space-config).evals }}
secrets:
SRT_STATUS_ENDPOINT: ${{ secrets.SRT_STATUS_ENDPOINT }}
SRTCTL_STATUS_TOKEN: ${{ secrets.SRTCTL_STATUS_TOKEN }}
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
Expand All @@ -743,11 +786,12 @@ jobs:
# dispatched with sweep-agentic's inputs rather than sweep-evals' fixed-seq-len
# inputs (isl/osl/max-model-len, which agentic rows don't have).
sweep-agentic-evals:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep, build-srt-tachometer]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.build-srt-tachometer.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
Expand All @@ -761,6 +805,8 @@ jobs:
matrix:
config: ${{ fromJson(needs.setup.outputs.search-space-config).agentic_evals }}
secrets:
SRT_STATUS_ENDPOINT: ${{ secrets.SRT_STATUS_ENDPOINT }}
SRTCTL_STATUS_TOKEN: ${{ secrets.SRTCTL_STATUS_TOKEN }}
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,8 @@ base:
observability:
enabled: false
tachometer:
enabled: false
enabled: true
collect_interval_ms: 1000
engine: sglang
roles:
agg:
Expand Down
13 changes: 13 additions & 0 deletions inferencex-e2e/docs/configuration-procedures.md
Original file line number Diff line number Diff line change
Expand Up @@ -691,3 +691,16 @@ runtime directories stay out of `/workspace`. The MI300X launcher also raises it
allocation from 180 to 480 minutes for this checkpoint: the HF cache there is node-local, so
the first arm on each node downloads 511 GB before serving. GPU sweep and eval evidence is
required before calling either arm validated.

### Testing SRT raw streaming on B300

PR #3591 applies NVIDIA/srt-slurm#539 to the pinned submodule in each job's
checkout. The normal sweep builds the patched Tachometer on a GitHub-hosted runner
once, verifies its checksum and base commit, and installs it before `make setup`;
the released binary does not include the required atomic Arrow writer.

The existing `glm5.2-fp8-b300-sglang-agentic-mtp` recipe collects Tachometer
metrics every second and sends raw logs/captures every five seconds. The workflow
forwards `SRT_STATUS_ENDPOINT` and `SRTCTL_STATUS_TOKEN` repository secrets into
the B300 profile. Start its normal sweep with `non-canary-full-sweep-enabled`
after the Dash collector is deployed.
12 changes: 12 additions & 0 deletions inferencex-e2e/docs/configuration-procedures_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -592,3 +592,15 @@ python -m pytest infx/tests/matrix/ -v
检查点将仓库挂载到 `/ix` 并重写 `RESULT_DIR`,使 AgentX 运行目录不落在 `/workspace` 下。MI300X
launcher 还为该检查点将 Slurm 分配时长从 180 分钟提高到 480 分钟:那里的 HF 缓存为节点本地,
每个节点上的首次运行需先下载 511 GB。在获得 GPU sweep 与 eval 证据之前,不得将任一配方视为已验证。

### 在 B300 上测试 SRT 原始数据流

PR #3591 在每个作业的独立检出中,将 NVIDIA/srt-slurm#539 补丁应用到固定的
子模块提交。常规 sweep 在 GitHub 托管 runner 上构建一次补丁版 Tachometer,
验证校验和及基础提交,并在 `make setup` 前安装;已发布的二进制不包含所需的
原子 Arrow 写入实现。

现有 `glm5.2-fp8-b300-sglang-agentic-mtp` 配方每秒采集 Tachometer 指标,
每五秒上传原始日志及采集文件。工作流将仓库 Secret `SRT_STATUS_ENDPOINT` 和
`SRTCTL_STATUS_TOKEN` 传给 B300 配置。部署 Dash 收集 API 后,添加
`non-canary-full-sweep-enabled` 标签即可启动常规 sweep。
8 changes: 8 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9104,3 +9104,11 @@
- "Restore the DCP8 LMCache bands on the native srt-slurm recipe: concurrency 14 and 16 (DSpark 3, ReplaySSM) and 48, 56 and 72 (no draft) run ATOM's in-process lmcache_offload connector through roles.agg.args.extra-kv-connectors (srt-slurm patch 507), with 128 GB/rank up to 48 and 192 GB/rank at 56 and 72. Concurrency 1 and 4 stay GPU-resident."
- "No change to the Inferact/Kimi-K3-DSpark draft's precision: online_quant_config still excludes every draft linear (layers.*, context_proj), so its weights and activations stay BF16, and it keeps the target's FP8 KV cache (kv_cache_dtype fp8). FlyDSL FP8 prefill attention applies only to the target, since the draft runs its block pass as decode attention."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3407

- config-keys:
- glm5.2-fp8-b300-sglang-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Test SRT Slurm PR 539 raw log streaming and Tachometer capture during the existing B300 GLM-5.2 FP8 SGLang AgentX sweep. Enable 1-second metric collection and 5-second HTTP uploads; serving settings and concurrency points are unchanged."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3591
11 changes: 11 additions & 0 deletions inferencex-e2e/runners/slurm_utils.sh
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,7 @@ write_srt_cluster_config() {
--var SLURM_ACCOUNT "$SLURM_ACCOUNT" --var SLURM_PARTITION "$SLURM_PARTITION" \
--var SRTCTL_ROOT "$SRTCTL_ROOT" --var SQUASH_FILE "$SQUASH_FILE" \
--var NGINX_SQUASH_FILE "$NGINX_SQUASH_FILE" --var IMAGE "$IMAGE" \
--var SRT_STATUS_ENDPOINT "${SRT_STATUS_ENDPOINT:-}" \
"$@" "${power_args[@]}"
}

Expand Down Expand Up @@ -87,6 +88,15 @@ PYENV
if [[ "$uses_power" == "1" ]]; then
cp "$GITHUB_WORKSPACE/srt-slurm-sha.txt" "$GITHUB_WORKSPACE/power-producer-sha.txt" || return 1
fi
if [[ -n "${SRT_TACHOMETER_BUILD_DIR:-}" ]]; then
[[ "$(cat "$SRT_TACHOMETER_BUILD_DIR/base-commit.txt")" == "$SRT_SLURM_COMMIT" ]] || return 1
(cd "$SRT_TACHOMETER_BUILD_DIR" && sha256sum --check tachometer.sha256) || return 1
mkdir -p bin || return 1
install -m755 "$SRT_TACHOMETER_BUILD_DIR/tachometer-scraper" bin/tachometer-scraper || return 1
fi
if [[ -n "${SRT_STATUS_ENDPOINT:-}" ]]; then
check_env_vars SRTCTL_STATUS_TOKEN SRT_TACHOMETER_BUILD_DIR
fi
mkdir -p recipes benchmarks/multi_node || return 1
cp -R "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/." recipes/ || return 1
# Both CONFIG_FILE spellings currently occur in master configs.
Expand Down Expand Up @@ -191,6 +201,7 @@ launch_srt_single_node() {
--var SRTCTL_ROOT "$SRTCTL_ROOT" --var SQUASH_FILE "$SRT_CONTAINER" \
--var IMAGE "$IMAGE" --var NGINX_SQUASH_FILE nginx:1.27.4 \
--var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" \
--var SRT_STATUS_ENDPOINT "${SRT_STATUS_ENDPOINT:-}" \
--model "hf:$MODEL" "$SRT_MODEL_PATH" --container "$IMAGE" "$SRT_CONTAINER" \
--mount "$HF_HUB_CACHE_MOUNT" "$HF_HUB_CACHE" --exclusive "$@"
run_srt_setup "ARCH=${SRT_SETUP_ARCH:-x86_64}"
Expand Down
6 changes: 6 additions & 0 deletions inferencex-e2e/runners/srt-slurm/b300-dsxe.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -36,3 +36,9 @@ default_sbatch_directives:
cpus-per-task: '192'
# gpu-16 retains a foreign 1.63 TB tmpfs allocation; rechecked 2026-09-21.
exclude: dsxe-sa-b300-prd0-gpu-16

reporting:
status:
endpoint: ${SRT_STATUS_ENDPOINT}
token_env: SRTCTL_STATUS_TOKEN
logging-stream-interval: 5
Loading
Loading