From 7ababb3a24c7c9fee1e70c7ba9f2d85fd9d8aeb4 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 17:51:19 -0700 Subject: [PATCH 1/6] feat(agentx): add GB300 DSV4.1 Flash Dynamo SGLang curve MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port the nine exercised aggregated and disaggregated AgentX points to the pluggable launcher. Add a named restricted GB300 Slurm route for the alternate partition and shared storage.\n\n将九个已验证的聚合与分离式 AgentX 点迁移到可插拔启动器,并为备用分区和共享存储添加命名的 GB300 restricted Slurm 路由。 --- .../sglang/gb300-fp4/agentx/agg-variants.yaml | 177 ++++++++++++++ .../gb300-fp4/agentx/disagg-variants.yaml | 231 ++++++++++++++++++ inferencex-e2e/configs/CONFIGS.md | 4 + inferencex-e2e/configs/nvidia-master.yaml | 143 +++++++++++ inferencex-e2e/configs/runners.yaml | 9 + .../docs/configuration-procedures.md | 4 + .../docs/configuration-procedures_zh.md | 3 + inferencex-e2e/infx/clusters/slurm.py | 30 +++ .../infx/launch/drivers/srt/__init__.py | 9 +- .../infx/launch/drivers/srt/lanes.py | 20 ++ inferencex-e2e/infx/launch/drivers/srt/run.py | 12 + .../infx/tests/launch/test_srt_driver.py | 13 + inferencex-e2e/perf-changelog.yaml | 18 ++ 13 files changed, 671 insertions(+), 2 deletions(-) create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml new file mode 100644 index 0000000000..9b5232f7fd --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -0,0 +1,177 @@ +# AgentX dsv41flash sglang gb300-fp4 recipes (Dynamo frontend + SGLang, AGGREGATED +# topology): shared settings in base, one override per benchmark point. Select one +# with CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_. +# +# One SGLang worker serves prefill and decode with the checkpoint's bundled DSpark +# draft (block size 5); the Dynamo frontend routes with the KV-aware router and +# session affinity. The aggregated arms cover the two ends of the curve; the +# middle band (c8-c96) comes from the disaggregated recipes in disagg-variants.yaml. +# Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, P90 TTFT): +# override_tp2_pdi16_c64 : 153,542 @ 83.3 tok/s/user, 15.2 s (2 GPUs; GSM8K 0.9719) +# override_tp2_pdi8_c64 : 159,510 @ 62.7 tok/s/user, 6.0 s (2 GPUs; GSM8K 0.9712) +# override_tp4_c1 : lowest-latency arm, the standalone recipe's pure TP4 (EP1) +# c1 settings (../../gb300-fp4-mtp/agentic.yaml) behind Dynamo; +# not measured in the campaign (4 GPUs) +# The harness applies the golden acceptance length (3.51 at K=5) itself. + +schema: 2 + +base: + name: dsv41flash-fp4-gb300-dynamo-sglang-agentx-agg + model: + path: hf:deepseek-ai/DeepSeek-V4.1-Flash + container: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 + precision: fp4 + identity: + model: + repo: deepseek-ai/DeepSeek-V4.1-Flash + dynamo: + install: true + source: + wheel: "1.6.0.dev20260928" + slurm: + time_limit: "4:00:00" + # Cold weight loads from the shared HF cache plus graph capture take 15-20 min. + health_check: + max_attempts: 1440 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + # Dynamo 1.6 uses the TCP request plane; etcd is the only discovery service needed. + services: + - name: etcd + type: etcd + placement: + node: infra + frontend: + type: dynamo + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: kv + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + engine: sglang + roles: + agg: + nodes: 1 + workers: 1 + gpus: 2 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + PYTHONNOUSERSITE: "1" + PYTHONUNBUFFERED: "1" + HF_HUB_CACHE: /hf_hub_cache + # Outlast AIPerf's pooled connections past Uvicorn's keep-alive. + SGLANG_TIMEOUT_KEEP_ALIVE: "900" + # AgentX measures thinking on, the regime of the golden acceptance curve. + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV41_REASONING_EFFORT: high + SGLANG_DSPARK_OPT_MARKOV_W2_BF16: "True" + # TP2 keeps the row-sharded Engram tables in host DRAM. + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1" + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + trust-remote-code: true + tensor-parallel-size: 2 + expert-parallel-size: 2 + mem-fraction-static: 0.8 + # Keep active decode requests progressing while long prefixes are queued. + prefill-decode-interval: 16 + # Admission never exceeds the captured decode graph batch. + cuda-graph-max-bs-decode: 64 + # DSpark is the checkpoint's bundled draft; the block size is its only knob. + speculative-algorithm: DSPARK + speculative-dspark-block-size: 5 + reasoning-parser: auto + tool-call-parser: auto + # Draft passes under long-context load outlast the 1800 s default watchdog. + watchdog-timeout: 3600 + enable-metrics: true + weight-loader-prefetch-checkpoints: true + weight-loader-drop-cache-after-load: false + model-loader-extra-config: '{"enable_multithread_load":true}' + sbatch_directives: + mem: "0" + exclusive: "" + srun_options: + container-remap-root: "" + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "false" + MODEL: deepseek-ai/DeepSeek-V4.1-Flash + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + # Warmup can leave bytes unacknowledged past AIPerf's 30 s default. + AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" + TP: "2" + +# Lowest-latency arm: one pure TP4 (EP1) worker on all four GPUs at c1, the standalone +# recipe's override_tp4_c1 behind the Dynamo frontend: static ragged verify, Engram +# tables in HBM (host-table path off), admission 2x CONC, 4096-token prefill chunks. +override_tp4_c1: + name: agg-gb300-tp4ep1-c1 + roles: + agg: + gpus: 4 + args: + tensor-parallel-size: 4 + expert-parallel-size: 1 + chunked-prefill-size: 4096 + max-running-requests: 2 + env: + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "0" + SGLANG_RAGGED_VERIFY_MODE: static + benchmark: + env: + CONC: "1" + TP: "4" + AGENTIC_WARMUP_GRACE_PERIOD: "1800" + +# c64 with the default prefill-decode-interval 16 (512 prefill tokens per decode step): +# 153,542 tokens/s/GPU @ 83.3 tok/s/user, P90 TTFT 15.2 s. +override_tp2_pdi16_c64: + name: agg-gb300-tp2ep2-pdi16-c64 + roles: + agg: + args: + chunked-prefill-size: 8192 + prefill-decode-interval: 16 + swa-prefix-tails: 4096 + max-running-requests: 64 + benchmark: + env: + CONC: "64" + AGENTIC_WARMUP_GRACE_PERIOD: "3600" + +# c64 with prefill-decode-interval 8 (1024 prefill tokens per decode step): the +# throughput end of the aggregated curve, 159,510 tokens/s/GPU @ 62.7 tok/s/user, +# P90 TTFT 6.0 s (+3.9% tokens/s/GPU and 2.5x lower TTFT than interval 16, at +# -25% interactivity). +override_tp2_pdi8_c64: + name: agg-gb300-tp2ep2-pdi8-c64 + roles: + agg: + args: + chunked-prefill-size: 8192 + prefill-decode-interval: 8 + swa-prefix-tails: 4096 + max-running-requests: 64 + benchmark: + env: + CONC: "64" + AGENTIC_WARMUP_GRACE_PERIOD: "3600" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml new file mode 100644 index 0000000000..4f042083d2 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -0,0 +1,231 @@ +# AgentX dsv41flash sglang gb300-fp4 recipes (Dynamo frontend + SGLang, DISAGGREGATED +# topology with Mooncake KV transfer): shared settings in base, one override per +# benchmark point. Select one with +# CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_. +# +# Prefill and decode are separate TP2/EP2 SGLang workers (two GPUs each) behind the +# Dynamo KV router; with the DSpark draft in PD-disaggregated mode both sides must +# use the same TP, so capacity is added with more TP2 workers: one prefill worker +# feeds one, two or four decode workers (1PxD), which sets the sessions per decode +# worker and with it the interactivity band. Measured on GB300 NVL72 (tokens/s/GPU +# @ P90 interactivity, P90 TTFT): +# override_1p1d_c8 : 15,724 @ 326.1 tok/s/user, 0.6 s ( 4 GPUs; GSM8K gate passed) +# override_1p1d_c16 : 28,810 @ 288.3 tok/s/user, 0.7 s ( 4 GPUs; GSM8K gate passed) +# override_1p1d_c64 : 114,682 @ 147.5 tok/s/user, 1.8 s ( 4 GPUs; GSM8K 0.9742) +# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s ( 4 GPUs; GSM8K gate passed) +# override_1p2d_c64 : 80,214 @ 202.9 tok/s/user, 2.0 s ( 6 GPUs, 2 nodes; GSM8K gate pending) +# override_1p4d_c64 : 49,846 @ 253.9 tok/s/user, 1.9 s (10 GPUs, 3 nodes; GSM8K gate pending) +# Against the published MI355X ATOM curve at matched P90 interactivity: 1P1D c16 1.53x, +# c64 2.30x, c96 2.72x, 1P2D c64 2.15x, 1P4D c64 1.90x tokens/s/GPU; 1P1D c8 (1.41x) +# extends the curve to 326 tok/s/user. The harness applies the golden acceptance +# length (3.51 at K=5) itself. + +schema: 2 + +base: + name: dsv41flash-fp4-gb300-dynamo-sglang-agentx-disagg + model: + path: hf:deepseek-ai/DeepSeek-V4.1-Flash + container: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 + precision: fp4 + identity: + model: + repo: deepseek-ai/DeepSeek-V4.1-Flash + dynamo: + install: true + source: + wheel: "1.6.0.dev20260928" + slurm: + # c96 AgentX runs take ~2.5 h including the 60 min profiling phase and the gate. + time_limit: "6:00:00" + health_check: + max_attempts: 1440 + interval_seconds: 10 + resources: + gpu_type: gb300 + gpus_per_node: 4 + # Dynamo 1.6 uses the TCP request plane; etcd is the only discovery service needed + # and runs with the frontend on the prefill node (two nodes total for 1P1D). + services: + - name: etcd + type: etcd + placement: + node: infra + frontend: + type: dynamo + enable_multiple_frontends: false + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" + args: + router-mode: kv + router-session-affinity-ttl-secs: "3600" + active-decode-blocks-threshold: "None" + active-prefill-tokens-threshold: "None" + active-prefill-tokens-threshold-frac: "None" + engine: sglang + roles: + prefill: + nodes: 1 + workers: 1 + gpus: 2 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + PYTHONNOUSERSITE: "1" + PYTHONUNBUFFERED: "1" + HF_HUB_CACHE: /hf_hub_cache + SGLANG_TIMEOUT_KEEP_ALIVE: "900" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV41_REASONING_EFFORT: high + SGLANG_DSPARK_OPT_MARKOV_W2_BF16: "True" + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1" + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_ENABLE_PREFILL_WAR_READ_DONE: "1" + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" + NCCL_MNNVL_ENABLE: "1" + NCCL_CUMEM_ENABLE: "1" + MC_FORCE_MNNVL: "1" + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + trust-remote-code: true + tensor-parallel-size: 2 + expert-parallel-size: 2 + disaggregation-transfer-backend: mooncake + # Long-context first turns (P90 ISL ~250k tokens): one 16k-token chunk per step. + chunked-prefill-size: 16384 + max-prefill-tokens: 16384 + max-running-requests: 64 + swa-prefix-tails: 4096 + mem-fraction-static: 0.85 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 5 + reasoning-parser: auto + tool-call-parser: auto + watchdog-timeout: 3600 + enable-metrics: true + weight-loader-prefetch-checkpoints: true + weight-loader-drop-cache-after-load: false + model-loader-extra-config: '{"enable_multithread_load":true}' + # KV events for the Dynamo KV router; srtctl allocates the ZMQ ports per worker. + kv_events: true + decode: + nodes: 1 + workers: 1 + gpus: 2 + env: + PIP_BREAK_SYSTEM_PACKAGES: "1" + PYTHONNOUSERSITE: "1" + PYTHONUNBUFFERED: "1" + HF_HUB_CACHE: /hf_hub_cache + SGLANG_TIMEOUT_KEEP_ALIVE: "900" + SGLANG_DEFAULT_THINKING: "1" + SGLANG_DSV41_REASONING_EFFORT: high + SGLANG_DSPARK_OPT_MARKOV_W2_BF16: "True" + SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1" + SGLANG_DSV41_ENGRAM_HOST_TABLE_LAYOUT: per_rank + SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "900" + SGLANG_MOONCAKE_CUSTOM_MEM_POOL: "True" + NCCL_MNNVL_ENABLE: "1" + NCCL_CUMEM_ENABLE: "1" + MC_FORCE_MNNVL: "1" + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + trust-remote-code: true + tensor-parallel-size: 2 + expert-parallel-size: 2 + disaggregation-transfer-backend: mooncake + max-running-requests: 128 + cuda-graph-max-bs-decode: 128 + mem-fraction-static: 0.85 + speculative-algorithm: DSPARK + speculative-dspark-block-size: 5 + reasoning-parser: auto + tool-call-parser: auto + watchdog-timeout: 3600 + enable-metrics: true + weight-loader-prefetch-checkpoints: true + weight-loader-drop-cache-after-load: false + model-loader-extra-config: '{"enable_multithread_load":true}' + kv_events: true + sbatch_directives: + mem: "0" + exclusive: "" + srun_options: + container-remap-root: "" + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + RESULT_DIR: /logs/agentic + PORT: "8000" + IS_MULTINODE: "true" + MODEL: deepseek-ai/DeepSeek-V4.1-Flash + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true" + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "sglang:" + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + HF_HUB_CACHE: /hf_hub_cache + AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" + AGENTIC_WARMUP_GRACE_PERIOD: "3600" + +# 1 prefill TP2/EP2 + 1 decode TP2/EP2 (4 GPUs, 2 nodes). The decode worker alone +# sets the interactivity: 8 sessions give 326 tok/s/user, 16 give 288, 64 give +# 147.5 and 96 give 132; the single prefill worker holds P90 TTFT under 4.1 s up +# to c96 (it saturates between c96 and c128). +override_1p1d_c8: + name: disagg-gb300-1p1d-tp2ep2-c8 + benchmark: + env: + CONC: "8" + +override_1p1d_c16: + name: disagg-gb300-1p1d-tp2ep2-c16 + benchmark: + env: + CONC: "16" + +# 114,682 tokens/s/GPU @ 147.5 tok/s/user with a 1.8 s P90 TTFT. +override_1p1d_c64: + name: disagg-gb300-1p1d-tp2ep2-c64 + benchmark: + env: + CONC: "64" + +# 144,448 tokens/s/GPU @ 132.2 tok/s/user with a 4.1 s P90 TTFT. +override_1p1d_c96: + name: disagg-gb300-1p1d-tp2ep2-c96 + benchmark: + env: + CONC: "96" + +# 1 prefill TP2/EP2 + 2 decode TP2/EP2 on one node (6 GPUs, 2 nodes), 32 sessions per +# decode worker: 80,214 tokens/s/GPU @ 202.9 tok/s/user, P90 TTFT 2.0 s. The KV +# router kept the two decode workers evenly loaded without per-worker caps. +override_1p2d_c64: + name: disagg-gb300-1p2d-tp2ep2-c64 + roles: + decode: + workers: 2 + args: + max-running-requests: 64 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: "64" + +# 1 prefill TP2/EP2 + 4 decode TP2/EP2 on two nodes (10 GPUs, 3 nodes), 16 sessions +# per decode worker: 49,846 tokens/s/GPU @ 253.9 tok/s/user, P90 TTFT 1.9 s. +override_1p4d_c64: + name: disagg-gb300-1p4d-tp2ep2-c64 + roles: + decode: + nodes: 2 + workers: 4 + args: + max-running-requests: 32 + cuda-graph-max-bs-decode: 32 + benchmark: + env: + CONC: "64" diff --git a/inferencex-e2e/configs/CONFIGS.md b/inferencex-e2e/configs/CONFIGS.md index 03a6cce4c1..211fc2ee8c 100644 --- a/inferencex-e2e/configs/CONFIGS.md +++ b/inferencex-e2e/configs/CONFIGS.md @@ -200,6 +200,10 @@ schema; unknown keys fail. - `slurm:` ([`infx/clusters/slurm.py`](../infx/clusters/slurm.py)) holds the partition, account, exclusivity, GRES, excluded nodes and extra `srun`/`salloc` options; its volumes are host `path`s that jobs see at the same place. +- `slurm.routes` declares named alternate partition/account/storage routes for the same + physical runner pool. A route may replace selected volume paths and the squash-cache + directory. Workload matching stays in the applicable named launcher policy table; the + route itself contains only cluster facts. - `slurm.squash` is the Pyxis squash cache: `dir`, `visibility`, `lock-timeout-s`, `key-style` (`underscore`, `plus` or `plus-strip-nvcr`) and `import`: `submit-host` (on the launching host), `compute` (once on one compute node), `all-nodes` (on every diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index bfa6c1499b..8fba7f5782 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8939,3 +8939,146 @@ dsv41flash-fp4-gb300-sglang-agentic-dspark: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 1, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } +dsv41flash-fp4-gb300-dynamo-sglang-agentic-agg: + image: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:gb300-nv + precision: fp4 + framework: dynamo-sglang + router: { name: dynamo-router, version: "1.6.0.dev20260928" } + multinode: true + disagg: false + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - spec-decoding: mtp + conc-list: [1] + num-nodes: 1 + worker: + num-worker: 1 + tp: 4 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4_c1" + - spec-decoding: mtp + conc-list: [64] + num-nodes: 1 + worker: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_pdi16_c64" + - spec-decoding: mtp + conc-list: [64] + num-nodes: 1 + worker: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_pdi8_c64" +dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: + image: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 + model: deepseek-ai/DeepSeek-V4.1-Flash + model-prefix: dsv41flash + runner: cluster:gb300-nv + precision: fp4 + framework: dynamo-sglang + router: { name: dynamo-router, version: "1.6.0.dev20260928" } + kv-p2p-transfer: mooncake + multinode: true + disagg: true + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - spec-decoding: mtp + conc-list: [8] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c8" + decode: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [16] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c16" + decode: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [64] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c64" + decode: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [96] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c96" + decode: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [64] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_c64" + decode: + num-worker: 2 + tp: 2 + ep: 2 + dp-attn: false + - spec-decoding: mtp + conc-list: [64] + prefill: + num-worker: 1 + tp: 2 + ep: 2 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p4d_c64" + decode: + num-worker: 4 + tp: 2 + ep: 2 + dp-attn: false diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index cf67501a63..caaad93b91 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -621,6 +621,15 @@ clusters: Qwen3.5-397B-A17B-NVFP4: {root: scratch, dir: Qwen3.5-397B-A17B-NVFP4-V2} scheduler: slurm slurm: + routes: + restricted: + partition: batch_2 + account: restricted + volumes: + hf-hub-cache: {path: /data/home/slurm-shared/gharunners/hf-hub-cache} + aiperf-cache: {path: /data/home/slurm-shared/gharunners/ai-perf-cache} + dynamo-wheels: {path: /data/home/slurm-shared/gharunners/dynamo-wheels} + squash-dir: /data/home/slurm-shared/gharunners/squash partition: batch_1 account: benchmark exclusive: false diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index d0ed2d739b..a4e3f36d80 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -169,6 +169,10 @@ Setup source: [`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNE ### Repository registration 1. Add the fleet's `clusters.` record to [`configs/runners.yaml`](../configs/runners.yaml) (node shape, workload env, models, and the scheduler sub-record: for Slurm the partition, volumes, squash cache and srt-slurm facts; schema in [`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners)). Put launch rules that depend on model, framework, precision or recipe in [`infx/launch/policy.py`](../infx/launch/policy.py) or beside the one driver that reads them, never in a driver branch on the cluster id. A cluster on a new scheduler needs that scheduler's settings model under [`infx/clusters/`](../infx/clusters) and its backend under [`infx/launch/backends/`](../infx/launch/backends), one registry entry each, and no driver change; it runs only script-driver (`BENCH_SCRIPT_OVERRIDE`) points. + When one physical Slurm pool exposes an alternate partition/account and storage root, + declare those facts as a named `slurm.routes` entry and select it from the applicable + named workload-policy table. Do not duplicate runner ownership or embed the alternate + cluster facts in driver control flow. 2. Add each exact registered runner name under the intended `labels:` key in [`configs/runners.yaml`](../configs/runners.yaml). New names use `_` with zero-padded indices. 3. Add every runner name to exactly one `cluster:` label matching that record. `python -m infx.launch run` resolves the cluster from the runner name, so a runner outside every cluster label fails validation and fails at launch. 4. Master entries whose facts depend on one physical fleet use that exact `cluster:` label. Agentic configs require it. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index a659d68c46..1f74c8a5c5 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -150,6 +150,9 @@ STP(Single Token Prediction,单 Token 预测)是每次前向传播生成 ### 仓库注册 1. 在 [`configs/runners.yaml`](../configs/runners.yaml) 中为 fleet 添加 `clusters.` 记录(节点形状、工作负载环境、模型以及调度器子记录:Slurm 为分区、卷、squash 缓存和 srt-slurm 事实;schema 见 [`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners))。依赖模型、框架、精度或配方的启动规则写进 [`infx/launch/policy.py`](../infx/launch/policy.py) 或唯一读取它的驱动旁边,绝不在驱动中按集群 id 分支。新调度器上的集群需要在 [`infx/clusters/`](../infx/clusters) 下新增该调度器的设置模型、在 [`infx/launch/backends/`](../infx/launch/backends) 下新增其后端,各登记一行,无需修改驱动;这类集群只运行 script 驱动(`BENCH_SCRIPT_OVERRIDE`)的点。 + 当同一个物理 Slurm 池提供备用分区/账号和存储根目录时,请将这些事实声明为命名的 + `slurm.routes` 条目,并在相应的命名工作负载策略表中选择它。不要重复 runner 所有权, + 也不要把备用集群事实嵌入驱动控制流。 2. 在 [`configs/runners.yaml`](../configs/runners.yaml) 预期的 `labels:` key 下添加每个精确的已注册 runner 名称。新名称使用 `_`,索引必须两位补零。 3. 把每个 runner 名称加入且仅加入一个与该记录对应的 `cluster:` 标签。`python -m infx.launch run` 通过 runner 名称解析集群,因此不属于任何集群标签的 runner 会导致校验失败,并在启动时失败。 4. 事实依赖某个物理 fleet 的主条目使用对应的精确 `cluster:` 标签;agentic 配置强制要求该标签。 diff --git a/inferencex-e2e/infx/clusters/slurm.py b/inferencex-e2e/infx/clusters/slurm.py index 2fbc41f75e..492f19e29c 100644 --- a/inferencex-e2e/infx/clusters/slurm.py +++ b/inferencex-e2e/infx/clusters/slurm.py @@ -177,6 +177,15 @@ class SrtSlurmSettings(Record): extra: dict[str, Any] = Field(default_factory=dict) +class SlurmRoute(Record): + """A workload-selected route through another partition and shared-storage root.""" + + partition: str = Field(min_length=1) + account: str = Field(min_length=1) + volumes: dict[str, HostVolume] = Field(default_factory=dict) + squash_dir: HostPath | None = Field(default=None, alias="squash-dir") + + class SlurmSettings(SchedulerSettings): """Scheduler facts shared by every Slurm submission on the cluster.""" @@ -192,6 +201,7 @@ class SlurmSettings(SchedulerSettings): salloc_args: tuple[LongOption, ...] = Field(default=(), alias="salloc-args") squash: SquashCache | None = None srt_slurm: SrtSlurmSettings | None = Field(default=None, alias="srt-slurm") + routes: dict[str, SlurmRoute] = Field(default_factory=dict) @field_validator("gres") @classmethod @@ -239,6 +249,26 @@ def path(self, volume: str) -> Path | None: declared = self.volumes.get(volume) return None if declared is None else declared.path + def routed(self, name: str) -> Self: + """Return these settings with the named route's scheduler and storage facts.""" + try: + route = self.routes[name] + except KeyError: + raise ValueError(f"unknown Slurm route {name!r}") from None + squash = self.squash + if route.squash_dir is not None: + if squash is None: + raise ValueError(f"Slurm route {name!r} sets squash-dir without a squash cache") + squash = squash.model_copy(update={"dir": route.squash_dir}) + return self.model_copy( + update={ + "partition": route.partition, + "account": route.account, + "volumes": {**self.volumes, **route.volumes}, + "squash": squash, + } + ) + def slurm_settings(cluster: Cluster) -> SlurmSettings: """The Slurm sub-record of ``cluster``; a cluster on another scheduler is a caller bug.""" diff --git a/inferencex-e2e/infx/launch/drivers/srt/__init__.py b/inferencex-e2e/infx/launch/drivers/srt/__init__.py index 4a3aa0cd3d..765665a579 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/__init__.py +++ b/inferencex-e2e/infx/launch/drivers/srt/__init__.py @@ -15,7 +15,7 @@ from pathlib import Path from typing import TYPE_CHECKING -from infx.clusters.slurm import SlurmSettings +from infx.clusters.slurm import SlurmSettings, slurm_settings from infx.launch import policy from infx.launch.backends.base import BackendError from infx.launch.backends.slurm import srtctl_job_name @@ -30,7 +30,7 @@ run_setup, ) from infx.launch.drivers.srt.recipe import eval_overrides, prepare_recipe -from infx.launch.drivers.srt.run import SrtRun, require, slurm_backend +from infx.launch.drivers.srt.run import SrtRun, require, routed_launch, slurm_backend from infx.launch.request import BATCH_REENTRY_ENV, RequestError, SingleNodeRequest, SrtRequest if TYPE_CHECKING: @@ -124,6 +124,7 @@ def run_multinode(launch: Launch) -> int: lane = lanes.srt_lane(launch.cluster.id, launch.path) request = SrtRequest.from_env(launch.request.env) lanes.check_request(lane, request) + launch = routed_launch(launch, lanes.scheduler_route(lane, request)) config_file = lanes.config_file(request) decision = power.resolve_power(launch.cluster.id, launch.path, request) model = models.checkpoint(launch.cluster, request) @@ -190,6 +191,10 @@ def volume(where: str, cluster_id: str, name: str) -> None: where = f"SRT_LANES[{cluster_id!r}, {path}]" for mount in lane.mounts: volume(where, cluster_id, mount.volume) + settings = slurm_settings(clusters[cluster_id]) + for _, route in lane.scheduler_routes: + if route not in settings.routes: + problems.append(f"{where}: no Slurm route {route!r}") if lane.shared_run_root and srt.shared_run_root is None: problems.append(f"{where}: no srt-slurm.shared-run-root") if srt.default_time_limit is not None and (lane.time_limit or lane.long_time_limit): diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index e8217eb607..46ed24f3d3 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -41,6 +41,7 @@ class SrtLane: time_limit: str | None = None long_time_limit: str | None = None long_time: Match | None = None + scheduler_routes: tuple[tuple[Match, str], ...] = () _DYNAMO = any_of("dynamo-sglang", "dynamo-trt", "dynamo-vllm") @@ -91,6 +92,17 @@ class SrtLane: long_time=Match( any_of("dsv4"), frameworks=any_of("dynamo-sglang", "dynamo-trt"), agentic=True ), + scheduler_routes=( + ( + Match( + any_of("dsv41flash"), + any_of("fp4"), + frameworks=any_of("dynamo-sglang"), + agentic=True, + ), + "restricted", + ), + ), ), ("h100-dgxc", LaunchPath.SRT_MULTI): SrtLane(frameworks=any_of("dynamo-sglang", "dynamo-trt")), ("h200-dgxc", LaunchPath.SRT_MULTI): SrtLane( @@ -130,6 +142,14 @@ def check_request(lane: SrtLane, request: SrtRequest) -> None: raise LaunchError(f"{message} (FRAMEWORK={framework})") +def scheduler_route(lane: SrtLane, request: SrtRequest) -> str | None: + """The one named scheduler route selected for this request, if any.""" + selected = [name for match, name in lane.scheduler_routes if match(request)] + if len(selected) > 1: + raise LaunchError(f"request selects several scheduler routes: {', '.join(selected)}") + return selected[0] if selected else None + + def config_file(request: SrtRequest) -> str: """CONFIG_FILE, or on an eval-only run its real-verification EVAL_CONFIG_FILE.""" if request.eval_only and request.eval_config_file: diff --git a/inferencex-e2e/infx/launch/drivers/srt/run.py b/inferencex-e2e/infx/launch/drivers/srt/run.py index e1f1e92c4b..f060a6f37e 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/run.py +++ b/inferencex-e2e/infx/launch/drivers/srt/run.py @@ -8,6 +8,7 @@ from pathlib import Path from typing import TYPE_CHECKING +from infx.clusters.slurm import slurm_settings from infx.config import repository_root from infx.launch.backends.slurm import SlurmBackend, cli from infx.launch.context import Launch, LaunchError @@ -69,3 +70,14 @@ def slurm_backend(launch: Launch) -> SlurmBackend: f"{launch.path} needs the Slurm backend, got {type(launch.backend).__name__}" ) return launch.backend + + +def routed_launch(launch: Launch, route: str | None) -> Launch: + """Apply a named cluster-declared Slurm route without changing the physical cluster.""" + if route is None: + return launch + settings = slurm_settings(launch.cluster).routed(route) + cluster = launch.cluster.model_copy(update={"scheduler_settings": settings}) + cluster.bind_id(launch.cluster.id) + backend = SlurmBackend(cluster, launch.request, launch.life) + return Launch(cluster, backend, launch.request, launch.life, launch.path) diff --git a/inferencex-e2e/infx/tests/launch/test_srt_driver.py b/inferencex-e2e/infx/tests/launch/test_srt_driver.py index da513d1dd4..1ae1b0b499 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_driver.py @@ -176,16 +176,19 @@ def test_single_node_failed_allocation_fails_the_launch(harness): lane=SrtLane( setup_scripts={"dynamo-sglang": "setup.sh"}, mounts=(LaneMount(Match(), "cache", "/cache"),), time_limit="2:00:00", + scheduler_routes=((Match(), "restricted"),), ), env=dict(FRAMEWORK="dynamo-sglang"), model="nvme/model", preflight=False, tag="lab,dsr1,fp8,1024x1024,", setup_script="setup.sh", served="served-model", dist_timeout=True, time="2:00:00", mounts=("/cache",), staging="import", + partition="p2", account="restricted", cache="restricted-cache", squash="restricted-squash", ), "lab-b": dict( lane=SrtLane(shared_run_root=(Match(),)), env=dict(FRAMEWORK="dynamo-vllm", IS_AGENTIC="1", ISL="0", OSL="0", FAKE_RESULTS="agentic"), model="models/model", preflight=True, tag=None, setup_script=None, served=None, dist_timeout=False, time="10", mounts=(), staging="registry", shared_checkout=True, + partition="p", account=None, cache=None, squash=None, ), } # fmt: skip @@ -199,6 +202,11 @@ def lab_config(tmp: Path) -> Path: "volumes": {"nvme": {"path": str(tmp / "nvme"), "visibility": "node-local"}, "cache": {"path": str(tmp / "cache")}}, "squash": {"dir": str(tmp / "squash"), "import": "submit-host"}, + "routes": {"restricted": { + "partition": "p2", "account": "restricted", + "volumes": {"cache": {"path": str(tmp / "restricted-cache")}}, + "squash-dir": str(tmp / "restricted-squash"), + }}, "srt-slurm": {"network-interface": "", "job-tag": "lab", "dist-timeout-s": 1800}, }}, "lab-b": {**common, "models": {"entries": {"Model": {"root": "models", "dir": "model"}}}, "slurm": { @@ -275,10 +283,15 @@ def test_multinode_lane_stages_workflow_artifacts(harness, monkeypatch, cluster_ config = srtslurm(checkout) assert config["model_paths"] == {"alias": str(tmp / lab["model"])} assert config["default_time_limit"] == lab["time"] + assert config["default_partition"] == lab["partition"] + assert config.get("default_account") == lab["account"] assert set(lab["mounts"]) <= set(config.get("default_mounts", {}).values()) + if lab["cache"] is not None: + assert config["default_mounts"][str(tmp / lab["cache"])] == "/cache" imported = [line.split()[-1] for line in lines(harness.logs, "enroot")] if lab["staging"] == "import": assert config["containers"][env["IMAGE"]].endswith(".sqsh") and imported == ["docker://test:tag"] + assert Path(config["containers"][env["IMAGE"]]).parent == tmp / lab["squash"] else: assert config["containers"][env["IMAGE"]] == env["IMAGE"] and imported == [] outputs = Path(json.loads((workspace / "srt-submission.json").read_text())["output_dir"]) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index cdd9f07f34..3c6b07793d 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9104,3 +9104,21 @@ - "Restore the DCP8 LMCache bands on the native srt-slurm recipe: concurrency 14 and 16 (DSpark 3, ReplaySSM) and 48, 56 and 72 (no draft) run ATOM's in-process lmcache_offload connector through roles.agg.args.extra-kv-connectors (srt-slurm patch 507), with 128 GB/rank up to 48 and 192 GB/rank at 56 and 72. Concurrency 1 and 4 stay GPU-resident." - "No change to the Inferact/Kimi-K3-DSpark draft's precision: online_quant_config still excludes every draft linear (layers.*, context_proj), so its weights and activations stay BF16, and it keeps the target's FP8 KV cache (kv_cache_dtype fp8). FlyDSL FP8 prefill attention applies only to the target, since the draft runs its block pass as decode attention." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3407 + +- config-keys: + - dsv41flash-fp4-gb300-dynamo-sglang-agentic-agg + - dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg + scenario-type: + - agentic-coding + description: + - "Add DeepSeek-V4.1-Flash FP4 AgentX on GB300 with the Dynamo frontend (KV router, session affinity) and SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928, DSpark block size 5." + - "Enable the GB300 multi-node launcher route for DeepSeek-V4.1-Flash FP4 with Dynamo + SGLang." + - "Use the mounted shared Hugging Face cache for every SGLang server role." + - "Aggregated arms: a pure TP4 (EP1) c1 lowest-latency point (the standalone recipe's TP4 c1 settings, static ragged verify and Engram in HBM, behind Dynamo) and the TP2/EP2 c64 prefill-decode-interval 16 and 8 variants (153,542 tok/s/GPU @ 83.3 tok/s/user P90 and 159,510 @ 62.7 measured on GB300 NVL72)." + - "Disaggregated curve with Mooncake KV transfer, one TP2/EP2 prefill worker feeding one, two or four TP2/EP2 decode workers: 1P1D at c8/c16/c64/c96 (15,724 @ 326.1, 28,810 @ 288.3, 114,682 @ 147.5 and 144,448 @ 132.2 tok/s/GPU @ tok/s/user P90; P90 TTFT 0.6-4.1 s), 1P2D at c64 (80,214 @ 202.9, 2.0 s) and 1P4D at c64 (49,846 @ 253.9, 1.9 s). The complete nine-point sweep passed, including GSM8K evaluation for all configured eval points." + - "新增 GB300 上 DeepSeek-V4.1-Flash FP4 AgentX 的 Dynamo 前端(KV 路由、会话亲和)+ SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928 配方,DSpark block size 5。" + - "启用 GB300 多节点启动器中 DeepSeek-V4.1-Flash FP4 的 Dynamo + SGLang 路径。" + - "所有 SGLang server role 使用已挂载的共享 Hugging Face cache。" + - "聚合分支:纯 TP4(EP1)c1 最低延迟点(沿用单机配方的 TP4 c1 设置、static ragged verify 与 HBM 内 Engram,置于 Dynamo 之后)以及 TP2/EP2 c64 的 prefill-decode-interval 16/8 变体(GB300 NVL72 实测 153,542 tok/s/GPU @ 83.3 tok/s/user P90 与 159,510 @ 62.7)。" + - "分离式曲线(Mooncake KV 传输),一个 TP2/EP2 prefill worker 服务一个、两个或四个 TP2/EP2 decode worker:1P1D 的 c8/c16/c64/c96(15,724 @ 326.1、28,810 @ 288.3、114,682 @ 147.5、144,448 @ 132.2 tok/s/GPU @ tok/s/user P90;P90 TTFT 0.6-4.1 s)、1P2D 的 c64(80,214 @ 202.9,2.0 s)与 1P4D 的 c64(49,846 @ 253.9,1.9 s)。完整九点 sweep 已通过,包括所有已配置 eval 点的 GSM8K 评测。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3598 From fc1913547de6914019944e5b37b931ae48d62fd5 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 18:55:55 -0700 Subject: [PATCH 2/6] feat(agentx): add the missing GB300 DSV4.1 Flash frontier recipes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cover every vertex of the measured Dynamo+SGLang Pareto curve in the two AgentX recipe files: add the two-GPU TP2/EP2 c1 arm, the 1P2D c48 and spread c16 cells, and the HiCache prefill tier at 1P1D c160 and 2P1D c256; move the multi-decode cells (1P2D c48/c64, 1P4D c64, 2P1D c256) to the 1 s session-affinity TTL they were measured and GSM8K-gated with; refresh the measured numbers in the file headers. 在两个 AgentX 配方文件中覆盖已测得的 Dynamo+SGLang Pareto 曲线的每个顶点:新增双 GPU 的 TP2/EP2 c1 配置、1P2D c48 与跨节点 c16 单元,以及 1P1D c160 与 2P1D c256 的 HiCache prefill 层;将多 decode 单元(1P2D c48/c64、1P4D c64、2P1D c256)改为其实际测量与 GSM8K 验证所用的 1 秒会话亲和 TTL;同时更新文件头中的测量数据。 Co-Authored-By: Claude Fable 5.1 --- .../sglang/gb300-fp4/agentx/agg-variants.yaml | 38 ++++- .../gb300-fp4/agentx/disagg-variants.yaml | 140 ++++++++++++++++-- 2 files changed, 155 insertions(+), 23 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml index 9b5232f7fd..a9bb2d6d50 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -5,14 +5,18 @@ # One SGLang worker serves prefill and decode with the checkpoint's bundled DSpark # draft (block size 5); the Dynamo frontend routes with the KV-aware router and # session affinity. The aggregated arms cover the two ends of the curve; the -# middle band (c8-c96) comes from the disaggregated recipes in disagg-variants.yaml. -# Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, P90 TTFT): -# override_tp2_pdi16_c64 : 153,542 @ 83.3 tok/s/user, 15.2 s (2 GPUs; GSM8K 0.9719) -# override_tp2_pdi8_c64 : 159,510 @ 62.7 tok/s/user, 6.0 s (2 GPUs; GSM8K 0.9712) -# override_tp4_c1 : lowest-latency arm, the standalone recipe's pure TP4 (EP1) -# c1 settings (../../gb300-fp4-mtp/agentic.yaml) behind Dynamo; -# not measured in the campaign (4 GPUs) -# The harness applies the golden acceptance length (3.51 at K=5) itself. +# middle band (c16-c256) comes from the disaggregated recipes in disagg-variants.yaml. +# Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, P90 TTFT; all GSM8K +# gates passed): +# override_tp4_c1 : 6,215 @ 385.0 tok/s/user, 1.2 s (4 GPUs; GSM8K 0.9735) +# override_tp2_c1 : 12,501 @ 377.1 tok/s/user, 0.9 s (2 GPUs; GSM8K 0.9735) +# override_tp2_pdi16_c64 : 153,542 @ 83.3 tok/s/user, 15.2 s (2 GPUs; GSM8K 0.9719) +# override_tp2_pdi8_c64 : 159,510 @ 62.7 tok/s/user, 6.0 s (2 GPUs; GSM8K 0.9712) +# The two c1 arms are the highest-interactivity points of the curve (above the +# published MI355X ATOM curve's last vertex at 336.7 tok/s/user); at c1 the AgentX +# client replays the trace's idle gaps, so a single lane is client-bound and a second +# user on the same worker only lowers the P90 (TP2 c2: 12,538 @ 306.8). The harness +# applies the golden acceptance length (3.51 at K=5) itself. schema: 2 @@ -142,6 +146,24 @@ override_tp4_c1: TP: "4" AGENTIC_WARMUP_GRACE_PERIOD: "1800" +# Two-GPU c1 arm: the base TP2/EP2 worker with the Engram tables in host DRAM, static +# ragged verify, 4096-token prefill chunks and admission 2x CONC: 12,501 tokens/s/GPU +# @ 377.1 tok/s/user, P90 TTFT 0.90 s - twice the tokens/s/GPU of the TP4 c1 arm at a +# 2% lower P90 interactivity. +override_tp2_c1: + name: agg-gb300-tp2ep2-c1 + roles: + agg: + args: + chunked-prefill-size: 4096 + max-running-requests: 2 + env: + SGLANG_RAGGED_VERIFY_MODE: static + benchmark: + env: + CONC: "1" + AGENTIC_WARMUP_GRACE_PERIOD: "1800" + # c64 with the default prefill-decode-interval 16 (512 prefill tokens per decode step): # 153,542 tokens/s/GPU @ 83.3 tok/s/user, P90 TTFT 15.2 s. override_tp2_pdi16_c64: diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml index 4f042083d2..bf19ae20be 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -7,18 +7,26 @@ # Dynamo KV router; with the DSpark draft in PD-disaggregated mode both sides must # use the same TP, so capacity is added with more TP2 workers: one prefill worker # feeds one, two or four decode workers (1PxD), which sets the sessions per decode -# worker and with it the interactivity band. Measured on GB300 NVL72 (tokens/s/GPU -# @ P90 interactivity, P90 TTFT): -# override_1p1d_c8 : 15,724 @ 326.1 tok/s/user, 0.6 s ( 4 GPUs; GSM8K gate passed) -# override_1p1d_c16 : 28,810 @ 288.3 tok/s/user, 0.7 s ( 4 GPUs; GSM8K gate passed) -# override_1p1d_c64 : 114,682 @ 147.5 tok/s/user, 1.8 s ( 4 GPUs; GSM8K 0.9742) -# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s ( 4 GPUs; GSM8K gate passed) -# override_1p2d_c64 : 80,214 @ 202.9 tok/s/user, 2.0 s ( 6 GPUs, 2 nodes; GSM8K gate pending) -# override_1p4d_c64 : 49,846 @ 253.9 tok/s/user, 1.9 s (10 GPUs, 3 nodes; GSM8K gate pending) +# worker and with it the interactivity band, and two prefill workers feed one decode +# worker (2P1D) at the throughput end. Measured on GB300 NVL72 (tokens/s/GPU @ P90 +# interactivity, P90 TTFT; every override's GSM8K gate passed): +# override_1p1d_c8 : 15,724 @ 326.1 tok/s/user, 0.6 s ( 4 GPUs, 2 nodes) +# override_1p1d_c16 : 28,810 @ 288.3 tok/s/user, 0.7 s ( 4 GPUs, 2 nodes) +# override_1p1d_c64 : 114,682 @ 147.5 tok/s/user, 1.8 s ( 4 GPUs, 2 nodes) +# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s ( 4 GPUs, 2 nodes) +# override_1p1d_hicache_c160 : 172,192 @ 98.8 tok/s/user, 21.7 s ( 4 GPUs, 2 nodes) +# override_1p2d_spread_c16 : 19,280 @ 327.4 tok/s/user, 0.7 s ( 6 GPUs, 3 nodes) +# override_1p2d_c48 : 59,740 @ 244.3 tok/s/user, 1.4 s ( 6 GPUs, 2 nodes) +# override_1p2d_c64 : 80,583 @ 207.6 tok/s/user, 2.0 s ( 6 GPUs, 2 nodes) +# override_1p4d_c64 : 49,887 @ 254.6 tok/s/user, 2.1 s (10 GPUs, 3 nodes) +# override_2p1d_hicache_c256 : 174,562 @ 68.1 tok/s/user, 18.9 s ( 6 GPUs, 3 nodes) # Against the published MI355X ATOM curve at matched P90 interactivity: 1P1D c16 1.53x, -# c64 2.30x, c96 2.72x, 1P2D c64 2.15x, 1P4D c64 1.90x tokens/s/GPU; 1P1D c8 (1.41x) -# extends the curve to 326 tok/s/user. The harness applies the golden acceptance -# length (3.51 at K=5) itself. +# c64 2.30x, c96 2.72x, HiCache c160 2.40x; 1P2D spread c16 1.73x, c48 2.11x, c64 2.21x; +# 1P4D c64 1.91x; 2P1D HiCache c256 1.94x tokens/s/GPU; 1P1D c8 (1.41x) extends the +# curve to 326 tok/s/user. The multi-decode cells run the Dynamo frontend with a 1 s +# session-affinity TTL (per-turn load-aware decode pick); the single-decode cells keep +# the 3600 s default they were measured with. The harness applies the golden +# acceptance length (3.51 at K=5) itself. schema: 2 @@ -59,6 +67,8 @@ base: DYN_NATS_REQUEST_TIMEOUT_SECS: "1800" args: router-mode: kv + # Session-affinity TTL as measured for the single-decode 1P1D cells; the + # multi-decode overrides below set it to 1 s (per-turn load-aware decode pick). router-session-affinity-ttl-secs: "3600" active-decode-blocks-threshold: "None" active-prefill-tokens-threshold: "None" @@ -200,11 +210,36 @@ override_1p1d_c96: env: CONC: "96" -# 1 prefill TP2/EP2 + 2 decode TP2/EP2 on one node (6 GPUs, 2 nodes), 32 sessions per -# decode worker: 80,214 tokens/s/GPU @ 202.9 tok/s/user, P90 TTFT 2.0 s. The KV -# router kept the two decode workers evenly loaded without per-worker caps. +# 1 prefill TP2/EP2 + 2 decode TP2/EP2 (6 GPUs). Two decode workers halve the sessions +# per decode worker at a given concurrency, which sets the interactivity band. The +# frontend's session-affinity TTL is 1 s, so every turn takes a load-aware decode pick +# instead of pinning the session to one decode worker for an hour: +4% P90 +# interactivity at 8 sessions per decode worker, neutral at 12 and more (c24 31,465 @ +# 299.0 vs 31,339 @ 298.1; c32 41,910 @ 276.1 vs 41,768 @ 277.3 at TTL 3600). +# +# Same-node layout (both decode workers on one node, 2 nodes total): +# c48: 59,740 tokens/s/GPU @ 244.3 tok/s/user, P90 TTFT 1.36 s +# c64: 80,583 tokens/s/GPU @ 207.6 tok/s/user, P90 TTFT 2.01 s (GSM8K 0.9742) +override_1p2d_c48: + name: disagg-gb300-1p2d-tp2ep2-c48 + frontend: + args: + router-session-affinity-ttl-secs: "1" + roles: + decode: + workers: 2 + args: + max-running-requests: 64 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: "48" + override_1p2d_c64: name: disagg-gb300-1p2d-tp2ep2-c64 + frontend: + args: + router-session-affinity-ttl-secs: "1" roles: decode: workers: 2 @@ -215,10 +250,34 @@ override_1p2d_c64: env: CONC: "64" +# Spread layout (one decode worker per node, 3 nodes total), 8 sessions per decode +# worker: 19,280 tokens/s/GPU @ 327.4 tok/s/user, P90 TTFT 0.67 s (GSM8K 0.9735) - +# the highest-interactivity disaggregated point (the same cell on one node at TTL +# 3600 gave 19,227 @ 308.9). +override_1p2d_spread_c16: + name: disagg-gb300-1p2d-spread-tp2ep2-c16 + frontend: + args: + router-session-affinity-ttl-secs: "1" + roles: + decode: + nodes: 2 + workers: 2 + args: + max-running-requests: 64 + cuda-graph-max-bs-decode: 64 + benchmark: + env: + CONC: "16" + # 1 prefill TP2/EP2 + 4 decode TP2/EP2 on two nodes (10 GPUs, 3 nodes), 16 sessions -# per decode worker: 49,846 tokens/s/GPU @ 253.9 tok/s/user, P90 TTFT 1.9 s. +# per decode worker, session-affinity TTL 1 s: 49,887 tokens/s/GPU @ 254.6 tok/s/user, +# P90 TTFT 2.06 s (GSM8K 0.9742; identical to the TTL 3600 measurement 49,846 @ 253.9). override_1p4d_c64: name: disagg-gb300-1p4d-tp2ep2-c64 + frontend: + args: + router-session-affinity-ttl-secs: "1" roles: decode: nodes: 2 @@ -229,3 +288,54 @@ override_1p4d_c64: benchmark: env: CONC: "64" + +# Throughput end. The prefill worker keeps its KV prefix cache in a host-memory tier +# (SGLang HiCache: ratio 1 = one host copy of the device pool, write-back, direct I/O), +# which removes the LRU evictions that pushed P90 TTFT past the 25 s ceiling beyond c96 +# (1P1D c128 without it: 25.4 s). 1P1D c160: 172,192 tokens/s/GPU @ 98.8 tok/s/user, +# P90 TTFT 21.7 s (GSM8K 0.9727). c192 breaches the ceiling (29.2 s): the single +# prefill worker is saturated, so the next step is a second prefill worker (below). +override_1p1d_hicache_c160: + name: disagg-gb300-1p1d-hicache-tp2ep2-c160 + roles: + prefill: + args: + enable-hierarchical-cache: true + hicache-ratio: 1 + hicache-write-policy: write_back + hicache-io-backend: direct + decode: + args: + max-running-requests: 192 + cuda-graph-max-bs-decode: 192 + benchmark: + env: + CONC: "160" + +# Two HiCache prefill workers (one per node) feeding one decode worker (6 GPUs, 3 nodes), +# 128 sessions per prefill worker, decode admission and graph batch 256, session-affinity +# TTL 1 s: 174,562 tokens/s/GPU @ 68.1 tok/s/user, P90 TTFT 18.9 s (GSM8K 0.9712) - the +# highest tokens/s/GPU of the campaign (the same cell without HiCache: 151,671 @ 70.8 +# with a 36 s P90 TTFT). +override_2p1d_hicache_c256: + name: disagg-gb300-2p1d-hicache-tp2ep2-c256 + frontend: + args: + router-session-affinity-ttl-secs: "1" + roles: + prefill: + nodes: 2 + workers: 2 + args: + max-running-requests: 128 + enable-hierarchical-cache: true + hicache-ratio: 1 + hicache-write-policy: write_back + hicache-io-backend: direct + decode: + args: + max-running-requests: 256 + cuda-graph-max-bs-decode: 256 + benchmark: + env: + CONC: "256" From 27684bfb2671bc810a579d7c42aba782ed12b373 Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Tue, 29 Sep 2026 19:05:42 -0700 Subject: [PATCH 3/6] feat(agentx): ship only the eight GB300 DSV4.1 Flash frontier points MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Trim the aggregated and disaggregated AgentX recipes and the nvidia-master sweep to the eight measured Pareto vertices: agg TP4/EP1 c1 and TP2/EP2 c1; disagg 1P2D spread c16, 1P2D c48, 1P2D c64, 1P1D c96, 1P1D HiCache c160 and 2P1D HiCache c256. Drop the dominated aggregated c64 (prefill-decode-interval 8/16), 1P1D c8/c16/c64 and 1P4D c64 arms, add master entries for the new overrides, and update the perf-changelog description to the shipped set. 将聚合与分离式 AgentX 配方以及 nvidia-master sweep 精简为实测 Pareto 曲线的八个顶点:聚合 TP4/EP1 c1 与 TP2/EP2 c1;分离式 1P2D 跨节点 c16、1P2D c48、1P2D c64、1P1D c96、1P1D HiCache c160 与 2P1D HiCache c256。移除被支配的聚合 c64(prefill-decode-interval 8/16)、1P1D c8/c16/c64 与 1P4D c64 配置,为新增 override 添加 master 条目,并将 perf-changelog 描述更新为最终交付集合。 Co-Authored-By: Claude Fable 5.1 --- .../sglang/gb300-fp4/agentx/agg-variants.yaml | 53 +++---------- .../gb300-fp4/agentx/disagg-variants.yaml | 78 +++++-------------- inferencex-e2e/configs/nvidia-master.yaml | 44 ++++------- inferencex-e2e/perf-changelog.yaml | 8 +- 4 files changed, 49 insertions(+), 134 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml index a9bb2d6d50..910d9d1617 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -4,15 +4,16 @@ # # One SGLang worker serves prefill and decode with the checkpoint's bundled DSpark # draft (block size 5); the Dynamo frontend routes with the KV-aware router and -# session affinity. The aggregated arms cover the two ends of the curve; the -# middle band (c16-c256) comes from the disaggregated recipes in disagg-variants.yaml. -# Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, P90 TTFT; all GSM8K -# gates passed): -# override_tp4_c1 : 6,215 @ 385.0 tok/s/user, 1.2 s (4 GPUs; GSM8K 0.9735) -# override_tp2_c1 : 12,501 @ 377.1 tok/s/user, 0.9 s (2 GPUs; GSM8K 0.9735) -# override_tp2_pdi16_c64 : 153,542 @ 83.3 tok/s/user, 15.2 s (2 GPUs; GSM8K 0.9719) -# override_tp2_pdi8_c64 : 159,510 @ 62.7 tok/s/user, 6.0 s (2 GPUs; GSM8K 0.9712) -# The two c1 arms are the highest-interactivity points of the curve (above the +# session affinity. The aggregated arms are the two lowest-latency points of the +# curve; everything from c16 to c256 comes from the disaggregated recipes in +# disagg-variants.yaml. Measured on GB300 NVL72 (tokens/s/GPU @ P90 interactivity, +# P90 TTFT; both GSM8K gates passed): +# override_tp4_c1 : 6,215 @ 385.0 tok/s/user, 1.2 s (4 GPUs; GSM8K 0.9735) +# override_tp2_c1 : 12,501 @ 377.1 tok/s/user, 0.9 s (2 GPUs; GSM8K 0.9735) +# The aggregated c64 cells (prefill-decode-interval 8: 159,510 @ 62.7; interval 16: +# 153,542 @ 83.3) were measured as well but are dominated by the disaggregated +# HiCache cells and are not shipped. The two c1 arms are the highest-interactivity +# points of the curve (above the # published MI355X ATOM curve's last vertex at 336.7 tok/s/user); at c1 the AgentX # client replays the trace's idle gaps, so a single lane is client-bound and a second # user on the same worker only lowers the P90 (TP2 c2: 12,538 @ 306.8). The harness @@ -163,37 +164,3 @@ override_tp2_c1: env: CONC: "1" AGENTIC_WARMUP_GRACE_PERIOD: "1800" - -# c64 with the default prefill-decode-interval 16 (512 prefill tokens per decode step): -# 153,542 tokens/s/GPU @ 83.3 tok/s/user, P90 TTFT 15.2 s. -override_tp2_pdi16_c64: - name: agg-gb300-tp2ep2-pdi16-c64 - roles: - agg: - args: - chunked-prefill-size: 8192 - prefill-decode-interval: 16 - swa-prefix-tails: 4096 - max-running-requests: 64 - benchmark: - env: - CONC: "64" - AGENTIC_WARMUP_GRACE_PERIOD: "3600" - -# c64 with prefill-decode-interval 8 (1024 prefill tokens per decode step): the -# throughput end of the aggregated curve, 159,510 tokens/s/GPU @ 62.7 tok/s/user, -# P90 TTFT 6.0 s (+3.9% tokens/s/GPU and 2.5x lower TTFT than interval 16, at -# -25% interactivity). -override_tp2_pdi8_c64: - name: agg-gb300-tp2ep2-pdi8-c64 - roles: - agg: - args: - chunked-prefill-size: 8192 - prefill-decode-interval: 8 - swa-prefix-tails: 4096 - max-running-requests: 64 - benchmark: - env: - CONC: "64" - AGENTIC_WARMUP_GRACE_PERIOD: "3600" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml index bf19ae20be..dbd3e5d12a 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml @@ -6,27 +6,23 @@ # Prefill and decode are separate TP2/EP2 SGLang workers (two GPUs each) behind the # Dynamo KV router; with the DSpark draft in PD-disaggregated mode both sides must # use the same TP, so capacity is added with more TP2 workers: one prefill worker -# feeds one, two or four decode workers (1PxD), which sets the sessions per decode +# feeds one or two decode workers (1P1D, 1P2D), which sets the sessions per decode # worker and with it the interactivity band, and two prefill workers feed one decode # worker (2P1D) at the throughput end. Measured on GB300 NVL72 (tokens/s/GPU @ P90 # interactivity, P90 TTFT; every override's GSM8K gate passed): -# override_1p1d_c8 : 15,724 @ 326.1 tok/s/user, 0.6 s ( 4 GPUs, 2 nodes) -# override_1p1d_c16 : 28,810 @ 288.3 tok/s/user, 0.7 s ( 4 GPUs, 2 nodes) -# override_1p1d_c64 : 114,682 @ 147.5 tok/s/user, 1.8 s ( 4 GPUs, 2 nodes) -# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s ( 4 GPUs, 2 nodes) -# override_1p1d_hicache_c160 : 172,192 @ 98.8 tok/s/user, 21.7 s ( 4 GPUs, 2 nodes) -# override_1p2d_spread_c16 : 19,280 @ 327.4 tok/s/user, 0.7 s ( 6 GPUs, 3 nodes) -# override_1p2d_c48 : 59,740 @ 244.3 tok/s/user, 1.4 s ( 6 GPUs, 2 nodes) -# override_1p2d_c64 : 80,583 @ 207.6 tok/s/user, 2.0 s ( 6 GPUs, 2 nodes) -# override_1p4d_c64 : 49,887 @ 254.6 tok/s/user, 2.1 s (10 GPUs, 3 nodes) -# override_2p1d_hicache_c256 : 174,562 @ 68.1 tok/s/user, 18.9 s ( 6 GPUs, 3 nodes) -# Against the published MI355X ATOM curve at matched P90 interactivity: 1P1D c16 1.53x, -# c64 2.30x, c96 2.72x, HiCache c160 2.40x; 1P2D spread c16 1.73x, c48 2.11x, c64 2.21x; -# 1P4D c64 1.91x; 2P1D HiCache c256 1.94x tokens/s/GPU; 1P1D c8 (1.41x) extends the -# curve to 326 tok/s/user. The multi-decode cells run the Dynamo frontend with a 1 s -# session-affinity TTL (per-turn load-aware decode pick); the single-decode cells keep -# the 3600 s default they were measured with. The harness applies the golden -# acceptance length (3.51 at K=5) itself. +# override_1p2d_spread_c16 : 19,280 @ 327.4 tok/s/user, 0.7 s (6 GPUs, 3 nodes) +# override_1p2d_c48 : 59,740 @ 244.3 tok/s/user, 1.4 s (6 GPUs, 2 nodes) +# override_1p2d_c64 : 80,583 @ 207.6 tok/s/user, 2.0 s (6 GPUs, 2 nodes) +# override_1p1d_c96 : 144,448 @ 132.2 tok/s/user, 4.1 s (4 GPUs, 2 nodes) +# override_1p1d_hicache_c160 : 172,192 @ 98.8 tok/s/user, 21.7 s (4 GPUs, 2 nodes) +# override_2p1d_hicache_c256 : 174,562 @ 68.1 tok/s/user, 18.9 s (6 GPUs, 3 nodes) +# Against the published MI355X ATOM curve at matched P90 interactivity: 1P2D spread +# c16 1.73x, c48 2.11x, c64 2.21x; 1P1D c96 2.72x, HiCache c160 2.40x; 2P1D HiCache +# c256 1.94x tokens/s/GPU. Other measured cells (1P1D c8/c16/c64, 1P2D c24/c32, 1P4D +# c64) lie on or under the line through these vertices and are not shipped. The +# multi-decode cells run the Dynamo frontend with a 1 s session-affinity TTL (per-turn +# load-aware decode pick); the single-decode cells keep the 3600 s default they were +# measured with. The harness applies the golden acceptance length (3.51 at K=5) itself. schema: 2 @@ -181,29 +177,10 @@ base: AGENTIC_WARMUP_GRACE_PERIOD: "3600" # 1 prefill TP2/EP2 + 1 decode TP2/EP2 (4 GPUs, 2 nodes). The decode worker alone -# sets the interactivity: 8 sessions give 326 tok/s/user, 16 give 288, 64 give -# 147.5 and 96 give 132; the single prefill worker holds P90 TTFT under 4.1 s up -# to c96 (it saturates between c96 and c128). -override_1p1d_c8: - name: disagg-gb300-1p1d-tp2ep2-c8 - benchmark: - env: - CONC: "8" - -override_1p1d_c16: - name: disagg-gb300-1p1d-tp2ep2-c16 - benchmark: - env: - CONC: "16" - -# 114,682 tokens/s/GPU @ 147.5 tok/s/user with a 1.8 s P90 TTFT. -override_1p1d_c64: - name: disagg-gb300-1p1d-tp2ep2-c64 - benchmark: - env: - CONC: "64" - -# 144,448 tokens/s/GPU @ 132.2 tok/s/user with a 4.1 s P90 TTFT. +# sets the interactivity (96 sessions give 132 tok/s/user); the single prefill worker +# holds P90 TTFT under 4.1 s up to c96 and saturates between c96 and c128, where the +# HiCache override below takes over. 144,448 tokens/s/GPU @ 132.2 tok/s/user with a +# 4.1 s P90 TTFT. override_1p1d_c96: name: disagg-gb300-1p1d-tp2ep2-c96 benchmark: @@ -270,25 +247,6 @@ override_1p2d_spread_c16: env: CONC: "16" -# 1 prefill TP2/EP2 + 4 decode TP2/EP2 on two nodes (10 GPUs, 3 nodes), 16 sessions -# per decode worker, session-affinity TTL 1 s: 49,887 tokens/s/GPU @ 254.6 tok/s/user, -# P90 TTFT 2.06 s (GSM8K 0.9742; identical to the TTL 3600 measurement 49,846 @ 253.9). -override_1p4d_c64: - name: disagg-gb300-1p4d-tp2ep2-c64 - frontend: - args: - router-session-affinity-ttl-secs: "1" - roles: - decode: - nodes: 2 - workers: 4 - args: - max-running-requests: 32 - cuda-graph-max-bs-decode: 32 - benchmark: - env: - CONC: "64" - # Throughput end. The prefill worker keeps its KV prefix cache in a host-memory tier # (SGLang HiCache: ratio 1 = one host copy of the device pool, write-back, direct I/O), # which removes the LRU evictions that pushed P90 TTFT past the 25 s ceiling beyond c96 diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 8fba7f5782..4a35d9349d 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8964,17 +8964,7 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-agg: additional-settings: - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp4_c1" - spec-decoding: mtp - conc-list: [64] - num-nodes: 1 - worker: - num-worker: 1 - tp: 2 - ep: 2 - dp-attn: false - additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_pdi16_c64" - - spec-decoding: mtp - conc-list: [64] + conc-list: [1] num-nodes: 1 worker: num-worker: 1 @@ -8982,7 +8972,7 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-agg: ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_pdi8_c64" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml:override_tp2_c1" dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: image: lmsysorg/sglang:nightly-dev-20260928-81f27fb3@sha256:d9e4917808cfaa4b3be033a0c85a3a73d71eb93c17accbfa2ac3e96f551f3a33 model: deepseek-ai/DeepSeek-V4.1-Flash @@ -8999,30 +8989,30 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: - dram-utilization: 0.80 search-space: - spec-decoding: mtp - conc-list: [8] + conc-list: [16] prefill: num-worker: 1 tp: 2 ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c8" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_spread_c16" decode: - num-worker: 1 + num-worker: 2 tp: 2 ep: 2 dp-attn: false - spec-decoding: mtp - conc-list: [16] + conc-list: [48] prefill: num-worker: 1 tp: 2 ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c16" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_c48" decode: - num-worker: 1 + num-worker: 2 tp: 2 ep: 2 dp-attn: false @@ -9034,9 +9024,9 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_c64" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_c64" decode: - num-worker: 1 + num-worker: 2 tp: 2 ep: 2 dp-attn: false @@ -9055,30 +9045,30 @@ dsv41flash-fp4-gb300-dynamo-sglang-agentic-disagg: ep: 2 dp-attn: false - spec-decoding: mtp - conc-list: [64] + conc-list: [160] prefill: num-worker: 1 tp: 2 ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p2d_c64" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p1d_hicache_c160" decode: - num-worker: 2 + num-worker: 1 tp: 2 ep: 2 dp-attn: false - spec-decoding: mtp - conc-list: [64] + conc-list: [256] prefill: - num-worker: 1 + num-worker: 2 tp: 2 ep: 2 dp-attn: false additional-settings: - - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_1p4d_c64" + - "CONFIG_FILE=recipes/dsv41flash/sglang/gb300-fp4/agentx/disagg-variants.yaml:override_2p1d_hicache_c256" decode: - num-worker: 4 + num-worker: 1 tp: 2 ep: 2 dp-attn: false diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 3c6b07793d..2132df7a88 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9114,11 +9114,11 @@ - "Add DeepSeek-V4.1-Flash FP4 AgentX on GB300 with the Dynamo frontend (KV router, session affinity) and SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928, DSpark block size 5." - "Enable the GB300 multi-node launcher route for DeepSeek-V4.1-Flash FP4 with Dynamo + SGLang." - "Use the mounted shared Hugging Face cache for every SGLang server role." - - "Aggregated arms: a pure TP4 (EP1) c1 lowest-latency point (the standalone recipe's TP4 c1 settings, static ragged verify and Engram in HBM, behind Dynamo) and the TP2/EP2 c64 prefill-decode-interval 16 and 8 variants (153,542 tok/s/GPU @ 83.3 tok/s/user P90 and 159,510 @ 62.7 measured on GB300 NVL72)." - - "Disaggregated curve with Mooncake KV transfer, one TP2/EP2 prefill worker feeding one, two or four TP2/EP2 decode workers: 1P1D at c8/c16/c64/c96 (15,724 @ 326.1, 28,810 @ 288.3, 114,682 @ 147.5 and 144,448 @ 132.2 tok/s/GPU @ tok/s/user P90; P90 TTFT 0.6-4.1 s), 1P2D at c64 (80,214 @ 202.9, 2.0 s) and 1P4D at c64 (49,846 @ 253.9, 1.9 s). The complete nine-point sweep passed, including GSM8K evaluation for all configured eval points." + - "Aggregated arms: the two lowest-latency points of the curve, a pure TP4 (EP1) c1 point (the standalone recipe's TP4 c1 settings, static ragged verify and Engram in HBM, behind Dynamo; 6,215 tok/s/GPU @ 385.0 tok/s/user P90) and a TP2/EP2 c1 point with static ragged verify and Engram in host DRAM (12,501 @ 377.1, P90 TTFT 0.9 s); both passed their GSM8K gates." + - "Disaggregated curve with Mooncake KV transfer and TP2/EP2 workers: 1P2D with a 1 s Dynamo session-affinity TTL at c16 (one decode worker per node; 19,280 tok/s/GPU @ 327.4 tok/s/user P90, P90 TTFT 0.7 s), c48 (59,740 @ 244.3, 1.4 s) and c64 (80,583 @ 207.6, 2.0 s); 1P1D at c96 (144,448 @ 132.2, 4.1 s) and, with the SGLang HiCache host-memory prefix-cache tier on the prefill worker, at c160 (172,192 @ 98.8, 21.7 s); 2P1D HiCache at c256 (174,562 @ 68.1, 18.9 s). Every point passed its GSM8K gate; at matched P90 interactivity the curve is 1.7-2.7x the published MI355X ATOM tokens/s/GPU." - "新增 GB300 上 DeepSeek-V4.1-Flash FP4 AgentX 的 Dynamo 前端(KV 路由、会话亲和)+ SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928 配方,DSpark block size 5。" - "启用 GB300 多节点启动器中 DeepSeek-V4.1-Flash FP4 的 Dynamo + SGLang 路径。" - "所有 SGLang server role 使用已挂载的共享 Hugging Face cache。" - - "聚合分支:纯 TP4(EP1)c1 最低延迟点(沿用单机配方的 TP4 c1 设置、static ragged verify 与 HBM 内 Engram,置于 Dynamo 之后)以及 TP2/EP2 c64 的 prefill-decode-interval 16/8 变体(GB300 NVL72 实测 153,542 tok/s/GPU @ 83.3 tok/s/user P90 与 159,510 @ 62.7)。" - - "分离式曲线(Mooncake KV 传输),一个 TP2/EP2 prefill worker 服务一个、两个或四个 TP2/EP2 decode worker:1P1D 的 c8/c16/c64/c96(15,724 @ 326.1、28,810 @ 288.3、114,682 @ 147.5、144,448 @ 132.2 tok/s/GPU @ tok/s/user P90;P90 TTFT 0.6-4.1 s)、1P2D 的 c64(80,214 @ 202.9,2.0 s)与 1P4D 的 c64(49,846 @ 253.9,1.9 s)。完整九点 sweep 已通过,包括所有已配置 eval 点的 GSM8K 评测。" + - "聚合分支:曲线上延迟最低的两个点——纯 TP4(EP1)c1(沿用单机配方的 TP4 c1 设置、static ragged verify 与 HBM 内 Engram,置于 Dynamo 之后;6,215 tok/s/GPU @ 385.0 tok/s/user P90)与采用 static ragged verify、Engram 置于主机内存的 TP2/EP2 c1(12,501 @ 377.1,P90 TTFT 0.9 s);两者均通过 GSM8K 验证。" + - "分离式曲线(Mooncake KV 传输,TP2/EP2 worker):1P2D 配合 1 秒 Dynamo 会话亲和 TTL 的 c16(每节点一个 decode worker;19,280 tok/s/GPU @ 327.4 tok/s/user P90,P90 TTFT 0.7 s)、c48(59,740 @ 244.3,1.4 s)与 c64(80,583 @ 207.6,2.0 s);1P1D 的 c96(144,448 @ 132.2,4.1 s)以及在 prefill worker 上启用 SGLang HiCache 主机内存前缀缓存层后的 c160(172,192 @ 98.8,21.7 s);2P1D HiCache 的 c256(174,562 @ 68.1,18.9 s)。所有点均通过 GSM8K 验证;在相同 P90 交互性下,该曲线为已发布 MI355X ATOM tokens/s/GPU 的 1.7-2.7 倍。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3598 From fde7b9da34d7859e26ad5c6f8402393109194fea Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Thu, 1 Oct 2026 02:27:56 -0700 Subject: [PATCH 4/6] fix(agentx): run GB300 DSV4.1 Flash Dynamo SGLang on the default Slurm route Drop the named `restricted` GB300 Slurm route (batch_2 partition, restricted account, alternate shared caches). The CI runners are not associated with that account, so the canary failed at image import with "Invalid account or account/partition combination specified". The recipes now use the default gb300-nv partition, account and shared caches, like the other GB300 DeepSeek-V4.1-Flash configs. Revert the route plumbing in the Slurm settings, SRT launcher, tests and docs, and drop the matching changelog bullet. Co-Authored-By: Claude Opus 5.5 --- inferencex-e2e/configs/CONFIGS.md | 4 --- inferencex-e2e/configs/runners.yaml | 9 ------ .../docs/configuration-procedures.md | 4 --- .../docs/configuration-procedures_zh.md | 3 -- inferencex-e2e/infx/clusters/slurm.py | 30 ------------------- .../infx/launch/drivers/srt/__init__.py | 9 ++---- .../infx/launch/drivers/srt/lanes.py | 20 ------------- inferencex-e2e/infx/launch/drivers/srt/run.py | 12 -------- .../infx/tests/launch/test_srt_driver.py | 13 -------- inferencex-e2e/perf-changelog.yaml | 2 -- 10 files changed, 2 insertions(+), 104 deletions(-) diff --git a/inferencex-e2e/configs/CONFIGS.md b/inferencex-e2e/configs/CONFIGS.md index 211fc2ee8c..03a6cce4c1 100644 --- a/inferencex-e2e/configs/CONFIGS.md +++ b/inferencex-e2e/configs/CONFIGS.md @@ -200,10 +200,6 @@ schema; unknown keys fail. - `slurm:` ([`infx/clusters/slurm.py`](../infx/clusters/slurm.py)) holds the partition, account, exclusivity, GRES, excluded nodes and extra `srun`/`salloc` options; its volumes are host `path`s that jobs see at the same place. -- `slurm.routes` declares named alternate partition/account/storage routes for the same - physical runner pool. A route may replace selected volume paths and the squash-cache - directory. Workload matching stays in the applicable named launcher policy table; the - route itself contains only cluster facts. - `slurm.squash` is the Pyxis squash cache: `dir`, `visibility`, `lock-timeout-s`, `key-style` (`underscore`, `plus` or `plus-strip-nvcr`) and `import`: `submit-host` (on the launching host), `compute` (once on one compute node), `all-nodes` (on every diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index caaad93b91..cf67501a63 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -621,15 +621,6 @@ clusters: Qwen3.5-397B-A17B-NVFP4: {root: scratch, dir: Qwen3.5-397B-A17B-NVFP4-V2} scheduler: slurm slurm: - routes: - restricted: - partition: batch_2 - account: restricted - volumes: - hf-hub-cache: {path: /data/home/slurm-shared/gharunners/hf-hub-cache} - aiperf-cache: {path: /data/home/slurm-shared/gharunners/ai-perf-cache} - dynamo-wheels: {path: /data/home/slurm-shared/gharunners/dynamo-wheels} - squash-dir: /data/home/slurm-shared/gharunners/squash partition: batch_1 account: benchmark exclusive: false diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index a4e3f36d80..d0ed2d739b 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -169,10 +169,6 @@ Setup source: [`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNE ### Repository registration 1. Add the fleet's `clusters.` record to [`configs/runners.yaml`](../configs/runners.yaml) (node shape, workload env, models, and the scheduler sub-record: for Slurm the partition, volumes, squash cache and srt-slurm facts; schema in [`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners)). Put launch rules that depend on model, framework, precision or recipe in [`infx/launch/policy.py`](../infx/launch/policy.py) or beside the one driver that reads them, never in a driver branch on the cluster id. A cluster on a new scheduler needs that scheduler's settings model under [`infx/clusters/`](../infx/clusters) and its backend under [`infx/launch/backends/`](../infx/launch/backends), one registry entry each, and no driver change; it runs only script-driver (`BENCH_SCRIPT_OVERRIDE`) points. - When one physical Slurm pool exposes an alternate partition/account and storage root, - declare those facts as a named `slurm.routes` entry and select it from the applicable - named workload-policy table. Do not duplicate runner ownership or embed the alternate - cluster facts in driver control flow. 2. Add each exact registered runner name under the intended `labels:` key in [`configs/runners.yaml`](../configs/runners.yaml). New names use `_` with zero-padded indices. 3. Add every runner name to exactly one `cluster:` label matching that record. `python -m infx.launch run` resolves the cluster from the runner name, so a runner outside every cluster label fails validation and fails at launch. 4. Master entries whose facts depend on one physical fleet use that exact `cluster:` label. Agentic configs require it. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 1f74c8a5c5..a659d68c46 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -150,9 +150,6 @@ STP(Single Token Prediction,单 Token 预测)是每次前向传播生成 ### 仓库注册 1. 在 [`configs/runners.yaml`](../configs/runners.yaml) 中为 fleet 添加 `clusters.` 记录(节点形状、工作负载环境、模型以及调度器子记录:Slurm 为分区、卷、squash 缓存和 srt-slurm 事实;schema 见 [`configs/CONFIGS.md#runners`](../configs/CONFIGS.md#runners))。依赖模型、框架、精度或配方的启动规则写进 [`infx/launch/policy.py`](../infx/launch/policy.py) 或唯一读取它的驱动旁边,绝不在驱动中按集群 id 分支。新调度器上的集群需要在 [`infx/clusters/`](../infx/clusters) 下新增该调度器的设置模型、在 [`infx/launch/backends/`](../infx/launch/backends) 下新增其后端,各登记一行,无需修改驱动;这类集群只运行 script 驱动(`BENCH_SCRIPT_OVERRIDE`)的点。 - 当同一个物理 Slurm 池提供备用分区/账号和存储根目录时,请将这些事实声明为命名的 - `slurm.routes` 条目,并在相应的命名工作负载策略表中选择它。不要重复 runner 所有权, - 也不要把备用集群事实嵌入驱动控制流。 2. 在 [`configs/runners.yaml`](../configs/runners.yaml) 预期的 `labels:` key 下添加每个精确的已注册 runner 名称。新名称使用 `_`,索引必须两位补零。 3. 把每个 runner 名称加入且仅加入一个与该记录对应的 `cluster:` 标签。`python -m infx.launch run` 通过 runner 名称解析集群,因此不属于任何集群标签的 runner 会导致校验失败,并在启动时失败。 4. 事实依赖某个物理 fleet 的主条目使用对应的精确 `cluster:` 标签;agentic 配置强制要求该标签。 diff --git a/inferencex-e2e/infx/clusters/slurm.py b/inferencex-e2e/infx/clusters/slurm.py index 492f19e29c..2fbc41f75e 100644 --- a/inferencex-e2e/infx/clusters/slurm.py +++ b/inferencex-e2e/infx/clusters/slurm.py @@ -177,15 +177,6 @@ class SrtSlurmSettings(Record): extra: dict[str, Any] = Field(default_factory=dict) -class SlurmRoute(Record): - """A workload-selected route through another partition and shared-storage root.""" - - partition: str = Field(min_length=1) - account: str = Field(min_length=1) - volumes: dict[str, HostVolume] = Field(default_factory=dict) - squash_dir: HostPath | None = Field(default=None, alias="squash-dir") - - class SlurmSettings(SchedulerSettings): """Scheduler facts shared by every Slurm submission on the cluster.""" @@ -201,7 +192,6 @@ class SlurmSettings(SchedulerSettings): salloc_args: tuple[LongOption, ...] = Field(default=(), alias="salloc-args") squash: SquashCache | None = None srt_slurm: SrtSlurmSettings | None = Field(default=None, alias="srt-slurm") - routes: dict[str, SlurmRoute] = Field(default_factory=dict) @field_validator("gres") @classmethod @@ -249,26 +239,6 @@ def path(self, volume: str) -> Path | None: declared = self.volumes.get(volume) return None if declared is None else declared.path - def routed(self, name: str) -> Self: - """Return these settings with the named route's scheduler and storage facts.""" - try: - route = self.routes[name] - except KeyError: - raise ValueError(f"unknown Slurm route {name!r}") from None - squash = self.squash - if route.squash_dir is not None: - if squash is None: - raise ValueError(f"Slurm route {name!r} sets squash-dir without a squash cache") - squash = squash.model_copy(update={"dir": route.squash_dir}) - return self.model_copy( - update={ - "partition": route.partition, - "account": route.account, - "volumes": {**self.volumes, **route.volumes}, - "squash": squash, - } - ) - def slurm_settings(cluster: Cluster) -> SlurmSettings: """The Slurm sub-record of ``cluster``; a cluster on another scheduler is a caller bug.""" diff --git a/inferencex-e2e/infx/launch/drivers/srt/__init__.py b/inferencex-e2e/infx/launch/drivers/srt/__init__.py index 765665a579..4a3aa0cd3d 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/__init__.py +++ b/inferencex-e2e/infx/launch/drivers/srt/__init__.py @@ -15,7 +15,7 @@ from pathlib import Path from typing import TYPE_CHECKING -from infx.clusters.slurm import SlurmSettings, slurm_settings +from infx.clusters.slurm import SlurmSettings from infx.launch import policy from infx.launch.backends.base import BackendError from infx.launch.backends.slurm import srtctl_job_name @@ -30,7 +30,7 @@ run_setup, ) from infx.launch.drivers.srt.recipe import eval_overrides, prepare_recipe -from infx.launch.drivers.srt.run import SrtRun, require, routed_launch, slurm_backend +from infx.launch.drivers.srt.run import SrtRun, require, slurm_backend from infx.launch.request import BATCH_REENTRY_ENV, RequestError, SingleNodeRequest, SrtRequest if TYPE_CHECKING: @@ -124,7 +124,6 @@ def run_multinode(launch: Launch) -> int: lane = lanes.srt_lane(launch.cluster.id, launch.path) request = SrtRequest.from_env(launch.request.env) lanes.check_request(lane, request) - launch = routed_launch(launch, lanes.scheduler_route(lane, request)) config_file = lanes.config_file(request) decision = power.resolve_power(launch.cluster.id, launch.path, request) model = models.checkpoint(launch.cluster, request) @@ -191,10 +190,6 @@ def volume(where: str, cluster_id: str, name: str) -> None: where = f"SRT_LANES[{cluster_id!r}, {path}]" for mount in lane.mounts: volume(where, cluster_id, mount.volume) - settings = slurm_settings(clusters[cluster_id]) - for _, route in lane.scheduler_routes: - if route not in settings.routes: - problems.append(f"{where}: no Slurm route {route!r}") if lane.shared_run_root and srt.shared_run_root is None: problems.append(f"{where}: no srt-slurm.shared-run-root") if srt.default_time_limit is not None and (lane.time_limit or lane.long_time_limit): diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index 46ed24f3d3..e8217eb607 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -41,7 +41,6 @@ class SrtLane: time_limit: str | None = None long_time_limit: str | None = None long_time: Match | None = None - scheduler_routes: tuple[tuple[Match, str], ...] = () _DYNAMO = any_of("dynamo-sglang", "dynamo-trt", "dynamo-vllm") @@ -92,17 +91,6 @@ class SrtLane: long_time=Match( any_of("dsv4"), frameworks=any_of("dynamo-sglang", "dynamo-trt"), agentic=True ), - scheduler_routes=( - ( - Match( - any_of("dsv41flash"), - any_of("fp4"), - frameworks=any_of("dynamo-sglang"), - agentic=True, - ), - "restricted", - ), - ), ), ("h100-dgxc", LaunchPath.SRT_MULTI): SrtLane(frameworks=any_of("dynamo-sglang", "dynamo-trt")), ("h200-dgxc", LaunchPath.SRT_MULTI): SrtLane( @@ -142,14 +130,6 @@ def check_request(lane: SrtLane, request: SrtRequest) -> None: raise LaunchError(f"{message} (FRAMEWORK={framework})") -def scheduler_route(lane: SrtLane, request: SrtRequest) -> str | None: - """The one named scheduler route selected for this request, if any.""" - selected = [name for match, name in lane.scheduler_routes if match(request)] - if len(selected) > 1: - raise LaunchError(f"request selects several scheduler routes: {', '.join(selected)}") - return selected[0] if selected else None - - def config_file(request: SrtRequest) -> str: """CONFIG_FILE, or on an eval-only run its real-verification EVAL_CONFIG_FILE.""" if request.eval_only and request.eval_config_file: diff --git a/inferencex-e2e/infx/launch/drivers/srt/run.py b/inferencex-e2e/infx/launch/drivers/srt/run.py index f060a6f37e..e1f1e92c4b 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/run.py +++ b/inferencex-e2e/infx/launch/drivers/srt/run.py @@ -8,7 +8,6 @@ from pathlib import Path from typing import TYPE_CHECKING -from infx.clusters.slurm import slurm_settings from infx.config import repository_root from infx.launch.backends.slurm import SlurmBackend, cli from infx.launch.context import Launch, LaunchError @@ -70,14 +69,3 @@ def slurm_backend(launch: Launch) -> SlurmBackend: f"{launch.path} needs the Slurm backend, got {type(launch.backend).__name__}" ) return launch.backend - - -def routed_launch(launch: Launch, route: str | None) -> Launch: - """Apply a named cluster-declared Slurm route without changing the physical cluster.""" - if route is None: - return launch - settings = slurm_settings(launch.cluster).routed(route) - cluster = launch.cluster.model_copy(update={"scheduler_settings": settings}) - cluster.bind_id(launch.cluster.id) - backend = SlurmBackend(cluster, launch.request, launch.life) - return Launch(cluster, backend, launch.request, launch.life, launch.path) diff --git a/inferencex-e2e/infx/tests/launch/test_srt_driver.py b/inferencex-e2e/infx/tests/launch/test_srt_driver.py index 1ae1b0b499..da513d1dd4 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_driver.py @@ -176,19 +176,16 @@ def test_single_node_failed_allocation_fails_the_launch(harness): lane=SrtLane( setup_scripts={"dynamo-sglang": "setup.sh"}, mounts=(LaneMount(Match(), "cache", "/cache"),), time_limit="2:00:00", - scheduler_routes=((Match(), "restricted"),), ), env=dict(FRAMEWORK="dynamo-sglang"), model="nvme/model", preflight=False, tag="lab,dsr1,fp8,1024x1024,", setup_script="setup.sh", served="served-model", dist_timeout=True, time="2:00:00", mounts=("/cache",), staging="import", - partition="p2", account="restricted", cache="restricted-cache", squash="restricted-squash", ), "lab-b": dict( lane=SrtLane(shared_run_root=(Match(),)), env=dict(FRAMEWORK="dynamo-vllm", IS_AGENTIC="1", ISL="0", OSL="0", FAKE_RESULTS="agentic"), model="models/model", preflight=True, tag=None, setup_script=None, served=None, dist_timeout=False, time="10", mounts=(), staging="registry", shared_checkout=True, - partition="p", account=None, cache=None, squash=None, ), } # fmt: skip @@ -202,11 +199,6 @@ def lab_config(tmp: Path) -> Path: "volumes": {"nvme": {"path": str(tmp / "nvme"), "visibility": "node-local"}, "cache": {"path": str(tmp / "cache")}}, "squash": {"dir": str(tmp / "squash"), "import": "submit-host"}, - "routes": {"restricted": { - "partition": "p2", "account": "restricted", - "volumes": {"cache": {"path": str(tmp / "restricted-cache")}}, - "squash-dir": str(tmp / "restricted-squash"), - }}, "srt-slurm": {"network-interface": "", "job-tag": "lab", "dist-timeout-s": 1800}, }}, "lab-b": {**common, "models": {"entries": {"Model": {"root": "models", "dir": "model"}}}, "slurm": { @@ -283,15 +275,10 @@ def test_multinode_lane_stages_workflow_artifacts(harness, monkeypatch, cluster_ config = srtslurm(checkout) assert config["model_paths"] == {"alias": str(tmp / lab["model"])} assert config["default_time_limit"] == lab["time"] - assert config["default_partition"] == lab["partition"] - assert config.get("default_account") == lab["account"] assert set(lab["mounts"]) <= set(config.get("default_mounts", {}).values()) - if lab["cache"] is not None: - assert config["default_mounts"][str(tmp / lab["cache"])] == "/cache" imported = [line.split()[-1] for line in lines(harness.logs, "enroot")] if lab["staging"] == "import": assert config["containers"][env["IMAGE"]].endswith(".sqsh") and imported == ["docker://test:tag"] - assert Path(config["containers"][env["IMAGE"]]).parent == tmp / lab["squash"] else: assert config["containers"][env["IMAGE"]] == env["IMAGE"] and imported == [] outputs = Path(json.loads((workspace / "srt-submission.json").read_text())["output_dir"]) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 1f56e93a3e..0e38a32609 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9128,12 +9128,10 @@ - agentic-coding description: - "Add DeepSeek-V4.1-Flash FP4 AgentX on GB300 with the Dynamo frontend (KV router, session affinity) and SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928, DSpark block size 5." - - "Enable the GB300 multi-node launcher route for DeepSeek-V4.1-Flash FP4 with Dynamo + SGLang." - "Use the mounted shared Hugging Face cache for every SGLang server role." - "Aggregated arms: the two lowest-latency points of the curve, a pure TP4 (EP1) c1 point (the standalone recipe's TP4 c1 settings, static ragged verify and Engram in HBM, behind Dynamo; 6,215 tok/s/GPU @ 385.0 tok/s/user P90) and a TP2/EP2 c1 point with static ragged verify and Engram in host DRAM (12,501 @ 377.1, P90 TTFT 0.9 s); both passed their GSM8K gates." - "Disaggregated curve with Mooncake KV transfer and TP2/EP2 workers: 1P2D with a 1 s Dynamo session-affinity TTL at c16 (one decode worker per node; 19,280 tok/s/GPU @ 327.4 tok/s/user P90, P90 TTFT 0.7 s), c48 (59,740 @ 244.3, 1.4 s) and c64 (80,583 @ 207.6, 2.0 s); 1P1D at c96 (144,448 @ 132.2, 4.1 s) and, with the SGLang HiCache host-memory prefix-cache tier on the prefill worker, at c160 (172,192 @ 98.8, 21.7 s); 2P1D HiCache at c256 (174,562 @ 68.1, 18.9 s). Every point passed its GSM8K gate; at matched P90 interactivity the curve is 1.7-2.7x the published MI355X ATOM tokens/s/GPU." - "新增 GB300 上 DeepSeek-V4.1-Flash FP4 AgentX 的 Dynamo 前端(KV 路由、会话亲和)+ SGLang nightly-dev-20260928-81f27fb3 / ai-dynamo 1.6.0.dev20260928 配方,DSpark block size 5。" - - "启用 GB300 多节点启动器中 DeepSeek-V4.1-Flash FP4 的 Dynamo + SGLang 路径。" - "所有 SGLang server role 使用已挂载的共享 Hugging Face cache。" - "聚合分支:曲线上延迟最低的两个点——纯 TP4(EP1)c1(沿用单机配方的 TP4 c1 设置、static ragged verify 与 HBM 内 Engram,置于 Dynamo 之后;6,215 tok/s/GPU @ 385.0 tok/s/user P90)与采用 static ragged verify、Engram 置于主机内存的 TP2/EP2 c1(12,501 @ 377.1,P90 TTFT 0.9 s);两者均通过 GSM8K 验证。" - "分离式曲线(Mooncake KV 传输,TP2/EP2 worker):1P2D 配合 1 秒 Dynamo 会话亲和 TTL 的 c16(每节点一个 decode worker;19,280 tok/s/GPU @ 327.4 tok/s/user P90,P90 TTFT 0.7 s)、c48(59,740 @ 244.3,1.4 s)与 c64(80,583 @ 207.6,2.0 s);1P1D 的 c96(144,448 @ 132.2,4.1 s)以及在 prefill worker 上启用 SGLang HiCache 主机内存前缀缓存层后的 c160(172,192 @ 98.8,21.7 s);2P1D HiCache 的 c256(174,562 @ 68.1,18.9 s)。所有点均通过 GSM8K 验证;在相同 P90 交互性下,该曲线为已发布 MI355X ATOM tokens/s/GPU 的 1.7-2.7 倍。" From 59530ec7a89080029a315c8d1c1b5269c8360caf Mon Sep 17 00:00:00 2001 From: Po-Han Huang Date: Thu, 1 Oct 2026 03:14:51 -0700 Subject: [PATCH 5/6] fix(agentx): set PP_SIZE and PCP_SIZE for the GB300 DSV4.1 Flash aggregated recipe The aggregated AgentX power path now requires TP, PP_SIZE and PCP_SIZE before replay. Set PP_SIZE=1 and PCP_SIZE=1 in the base benchmark env, as the Qwen3.5 GB300 aggregated recipe does, so the replay no longer exits on missing inputs. Co-Authored-By: Claude Opus 5.5 --- .../dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml index 910d9d1617..e855c7db4c 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4/agentx/agg-variants.yaml @@ -124,6 +124,8 @@ base: # Warmup can leave bytes unacknowledged past AIPerf's 30 s default. AIPERF_HTTP_TCP_USER_TIMEOUT: "900000" TP: "2" + PP_SIZE: "1" + PCP_SIZE: "1" # Lowest-latency arm: one pure TP4 (EP1) worker on all four GPUs at c1, the standalone # recipe's override_tp4_c1 behind the Dynamo frontend: static ragged verify, Engram From b82ca4d35a79a5f56381735abb621d21eb1b95f7 Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Thu, 1 Oct 2026 21:01:30 -0500 Subject: [PATCH 6/6] Add GB300 partition routing under one cluster profile (#3658) * feat: route GB300 runners by partition labels within one cluster * fix: use sequential names for all GB300 runner instances --- inferencex-e2e/configs/CONFIGS.md | 8 ++ inferencex-e2e/configs/runners.yaml | 76 +++++++++++++++++++ inferencex-e2e/infx/clusters/__init__.py | 30 ++++++++ inferencex-e2e/infx/clusters/slurm.py | 1 + .../tests/clusters/test_cluster_config.py | 29 +++++++ .../utils/runner_setup/RUNNER_SETUP.md | 35 +++++++++ 6 files changed, 179 insertions(+) diff --git a/inferencex-e2e/configs/CONFIGS.md b/inferencex-e2e/configs/CONFIGS.md index 6a29f4a9fe..96980159bf 100644 --- a/inferencex-e2e/configs/CONFIGS.md +++ b/inferencex-e2e/configs/CONFIGS.md @@ -200,6 +200,14 @@ schema; unknown keys fail. - `slurm:` ([`infx/clusters/slurm.py`](../infx/clusters/slurm.py)) holds the partition, account, exclusivity, GRES, excluded nodes and extra `srun`/`salloc` options; its volumes are host `path`s that jobs see at the same place. +- Optional `slurm.partitions` lists independent allocation partitions inside one + cluster. When nonempty, every runner in that cluster must have exactly one + `partition:` inventory label naming an allowed partition, and the default + `slurm.partition` must be in the allowlist. The launcher resolves its actual + partition from the selected anchor's `RUNNER_NAME`, without changing the cluster + identity, model paths, or workload policies. Inventory labels must match GitHub. + The dashboard controller must support partition-aware leases before enabling + these runners; it must never combine capacity across partition labels. - `slurm.squash` is the Pyxis squash cache: `dir`, `visibility`, `lock-timeout-s`, `key-style` (`underscore`, `plus` or `plus-strip-nvcr`) and `import`: `submit-host` (on the launching host), `compute` (once on one compute node), `all-nodes` (on every diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index 8036d3de69..e4bf55018a 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -153,6 +153,24 @@ labels: - gb300-nv_15 - gb300-nv_16 - gb300-nv_17 + - gb300-nv_18 + - gb300-nv_19 + - gb300-nv_20 + - gb300-nv_21 + - gb300-nv_22 + - gb300-nv_23 + - gb300-nv_24 + - gb300-nv_25 + - gb300-nv_26 + - gb300-nv_27 + - gb300-nv_28 + - gb300-nv_29 + - gb300-nv_30 + - gb300-nv_31 + - gb300-nv_32 + - gb300-nv_33 + - gb300-nv_34 + - gb300-nv_35 cluster:h100-cw: - h100-cw_00 - h100-cw_01 @@ -269,6 +287,62 @@ labels: - gb300-nv_15 - gb300-nv_16 - gb300-nv_17 + - gb300-nv_18 + - gb300-nv_19 + - gb300-nv_20 + - gb300-nv_21 + - gb300-nv_22 + - gb300-nv_23 + - gb300-nv_24 + - gb300-nv_25 + - gb300-nv_26 + - gb300-nv_27 + - gb300-nv_28 + - gb300-nv_29 + - gb300-nv_30 + - gb300-nv_31 + - gb300-nv_32 + - gb300-nv_33 + - gb300-nv_34 + - gb300-nv_35 + partition:batch_1: + - gb300-nv_00 + - gb300-nv_01 + - gb300-nv_02 + - gb300-nv_03 + - gb300-nv_04 + - gb300-nv_05 + - gb300-nv_06 + - gb300-nv_07 + - gb300-nv_08 + - gb300-nv_09 + - gb300-nv_10 + - gb300-nv_11 + - gb300-nv_12 + - gb300-nv_13 + - gb300-nv_14 + - gb300-nv_15 + - gb300-nv_16 + - gb300-nv_17 + partition:batch_3: + - gb300-nv_18 + - gb300-nv_19 + - gb300-nv_20 + - gb300-nv_21 + - gb300-nv_22 + - gb300-nv_23 + - gb300-nv_24 + - gb300-nv_25 + - gb300-nv_26 + - gb300-nv_27 + - gb300-nv_28 + - gb300-nv_29 + - gb300-nv_30 + - gb300-nv_31 + - gb300-nv_32 + - gb300-nv_33 + - gb300-nv_34 + - gb300-nv_35 cluster:mi300x-amd: - mi300x-amd_00 - mi300x-amd_01 @@ -669,6 +743,8 @@ clusters: scheduler: slurm slurm: partition: batch_1 + # Each runner's partition: label selects one of these allocation domains. + partitions: [batch_1, batch_3] account: benchmark exclusive: false volumes: diff --git a/inferencex-e2e/infx/clusters/__init__.py b/inferencex-e2e/infx/clusters/__init__.py index 4fc7a60cfa..d17cdc3f53 100644 --- a/inferencex-e2e/infx/clusters/__init__.py +++ b/inferencex-e2e/infx/clusters/__init__.py @@ -160,6 +160,12 @@ def _one_cluster_per_runner(self) -> Self: if orphans: raise ValueError(f"runners without a cluster: label: {orphans}") for cluster_id, cluster in self.clusters.items(): + settings = cluster.scheduler_settings + if isinstance(settings, SlurmSettings) and settings.partitions: + if settings.partition not in settings.partitions: + raise ValueError(f"cluster {cluster_id!r} default partition is not allowed") + for runner in cluster_labels[cluster_id]: + self.partition_for(runner, settings.partitions) cluster.bind_id(cluster_id) return self @@ -175,6 +181,30 @@ def cluster_for(self, runner_name: str) -> Cluster: raise ValueError( f"runner {runner_name!r} must be in exactly one cluster label, found {found}" ) + cluster = matches[0] + settings = cluster.scheduler_settings + if isinstance(settings, SlurmSettings) and settings.partitions: + # Return a job-local profile; never mutate the shared inventory. + cluster = cluster.model_copy( + update={ + "scheduler_settings": settings.model_copy( + update={"partition": self.partition_for(runner_name, settings.partitions)} + ) + } + ) + return cluster + + def partition_for(self, runner_name: str, allowed: tuple[str, ...]) -> str: + """Resolve exactly one allowed partition from the runner's inventory labels.""" + matches = [ + label.removeprefix("partition:") + for label, runners in self.labels.items() + if label.startswith("partition:") and runner_name in runners + ] + if len(matches) != 1 or matches[0] not in allowed: + raise ValueError( + f"runner {runner_name!r} needs exactly one allowed partition label; found {matches}" + ) return matches[0] diff --git a/inferencex-e2e/infx/clusters/slurm.py b/inferencex-e2e/infx/clusters/slurm.py index 2fbc41f75e..72bafde63e 100644 --- a/inferencex-e2e/infx/clusters/slurm.py +++ b/inferencex-e2e/infx/clusters/slurm.py @@ -182,6 +182,7 @@ class SlurmSettings(SchedulerSettings): volumes: dict[str, HostVolume] = Field(default_factory=dict) partition: str = Field(min_length=1) + partitions: tuple[Annotated[str, Field(pattern=r"^[A-Za-z0-9][A-Za-z0-9_-]*$")], ...] = () account: str | None = Field(default=None, min_length=1) exclusive: bool exclude: tuple[str, ...] = () diff --git a/inferencex-e2e/infx/tests/clusters/test_cluster_config.py b/inferencex-e2e/infx/tests/clusters/test_cluster_config.py index fb2d680c36..ce69e5efbd 100644 --- a/inferencex-e2e/infx/tests/clusters/test_cluster_config.py +++ b/inferencex-e2e/infx/tests/clusters/test_cluster_config.py @@ -65,6 +65,35 @@ def test_scheduler_record_errors_carry_the_record_name(): assert error["loc"] == ("clusters", "alpha", "slurm", "gres") +def test_runner_partition_resolution_preserves_cluster_and_shared_profile(): + data = inventory(alpha=with_change( + "slurm.partitions", ["batch", "batch_1", "batch_3"] + )) + data["labels"].update({"partition:batch_1": ["alpha_0"], "partition:batch_3": ["alpha_1"]}) + loaded = load_inventory(data) + first = loaded.cluster_for("alpha_0") + third = loaded.cluster_for("alpha_1") + assert first.id == third.id == "alpha" + assert first.scheduler_settings.partition == "batch_1" + assert third.scheduler_settings.partition == "batch_3" + assert loaded.clusters["alpha"].scheduler_settings.partition == "batch" + assert loaded.cluster_for("alpha_0").scheduler_settings.partition == "batch_1" + assert str(third.scheduler_settings.squash.dir) == "/shared/squash" + + +@pytest.mark.parametrize("labels", [ + {"partition:batch_1": ["alpha_0"]}, + {"partition:batch_1": ["alpha_0", "alpha_1"], "partition:batch_3": ["alpha_1"]}, + {"partition:unknown": ["alpha_0", "alpha_1"]}, + {"partition:batch_1,batch_3": ["alpha_0", "alpha_1"]}, +]) +def test_partition_routes_reject_incomplete_unknown_or_multi_partition_targets(labels): + data = inventory(alpha=with_change("slurm.partitions", ["batch", "batch_1", "batch_3"])) + data["labels"].update(labels) + with pytest.raises(ValidationError): + load_inventory(data) + + def test_a_registered_scheduler_parses_its_own_record_and_volumes(monkeypatch): monkeypatch.setitem(SCHEDULERS, "fake", FakeSettings) fake = { diff --git a/inferencex-e2e/utils/runner_setup/RUNNER_SETUP.md b/inferencex-e2e/utils/runner_setup/RUNNER_SETUP.md index ab31fadf08..8355aaee02 100644 --- a/inferencex-e2e/utils/runner_setup/RUNNER_SETUP.md +++ b/inferencex-e2e/utils/runner_setup/RUNNER_SETUP.md @@ -233,6 +233,41 @@ and allocations submitted outside this admission path. ## Storage layout +### GB300: one cluster, two allocation partitions + +All 36 GB300 runners belong to `cluster:gb300-nv` and use the same runtime +profile and launcher policies. `gb300-nv_00` through `gb300-nv_17` carry +`partition:batch_1`; the additional `gb300-nv_18` through `gb300-nv_35` +carry `partition:batch_3`. Both ranges also carry `gb300` and `slurm`. + +The partitions are separate 18-node NVLink domains. The partition-aware +dashboard controller keeps each complete lease inside one partition and +checks its own Slurm availability, while grouping fleet/queue data under +`gb300-nv`. A four-node job cannot use two idle nodes from each partition. +Normal jobs still request `gb300` or `cluster:gb300-nv`; no master-config +duplication is needed. The launcher reads the anchor's `partition:` label +from this inventory and submits exclusively to that partition. + +New listeners run as `sa-shared@im-gb300-login-02`, using persistent NFS at +`/data/home/sa-shared/gharunners-batch3`. Shared caches and staged models stay +at the existing GB300 paths. Provision with `setup.sh` and tags +`slurm,gb300,cluster:gb300-nv,partition:batch_3`. Keep new listeners stopped +until both the dashboard's partition-aware admission and this inventory +routing are deployed, all runner labels agree, and the existing `gb300-nv` +collector reports `batch_1` and `batch_3` (never their overlapping `batch_All`). +Relabel existing runners only outside active leases. + +After those checks, start the new range with isolated nine-pane tmux sessions: + +```sh +bash start_runners.sh 18 26 /data/home/sa-shared/gharunners-batch3 gb300-runners-batch3-a +bash start_runners.sh 27 35 /data/home/sa-shared/gharunners-batch3 gb300-runners-batch3-b +``` + +Do not replace the existing `github-actions` tmux session. No custom runner +service or second collector identity is needed. Existing historical telemetry +for `gb300-nv` is preserved. + The login node (where the runners live) and the Slurm compute nodes (where benchmarks run) exchange everything through the filesystem, so every path the CI touches must be visible from the compute node that the job lands on. Each path must either live on