From da35b40f831baaff835882c1c743a64465364789 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 15:26:51 -0500 Subject: [PATCH 1/7] refactor(tilert): port MI355X AgentX to native srt-slurm --- .../agentic/glm5.3_fp8_mi355x_tilert.sh | 172 ------ .../benchmarks/multi_node/amd_utils/env.sh | 27 +- .../benchmarks/multi_node/amd_utils/job.slurm | 55 +- .../multi_node/amd_utils/models_tilert.yaml | 6 - .../benchmarks/multi_node/amd_utils/server.sh | 3 - .../multi_node/amd_utils/server_atom.sh | 1 - .../multi_node/amd_utils/server_sglang.sh | 1 - .../multi_node/amd_utils/server_tilert.sh | 573 ------------------ .../multi_node/amd_utils/setup_deps.sh | 136 ----- .../configs/glm5.3-tilert-rocm.sh | 68 +++ .../agentx/disagg-1p1d-tp8-mtp.yaml | 103 ++++ inferencex-e2e/configs/amd-master.yaml | 23 +- inferencex-e2e/perf-changelog.yaml | 6 + inferencex-e2e/runners/launch_mi355x-amds.sh | 6 +- 14 files changed, 192 insertions(+), 988 deletions(-) delete mode 100644 inferencex-e2e/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh delete mode 100644 inferencex-e2e/benchmarks/multi_node/amd_utils/models_tilert.yaml delete mode 100644 inferencex-e2e/benchmarks/multi_node/amd_utils/server_tilert.sh delete mode 100644 inferencex-e2e/benchmarks/multi_node/amd_utils/setup_deps.sh create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml diff --git a/inferencex-e2e/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh b/inferencex-e2e/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh deleted file mode 100644 index 82e81aee32..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh +++ /dev/null @@ -1,172 +0,0 @@ -#!/usr/bin/env bash - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -source "$SCRIPT_DIR/../../benchmark_lib.sh" - -check_env_vars \ - CONC_LIST \ - ISL \ - OSL \ - IMAGE \ - SPEC_DECODING \ - MODEL_PATH \ - MODEL_NAME \ - PREFILL_NUM_WORKERS \ - PREFILL_TP \ - PREFILL_EP \ - PREFILL_DP_ATTN \ - DECODE_NUM_WORKERS \ - DECODE_TP \ - DECODE_EP \ - DECODE_DP_ATTN \ - PREFILL_NODES \ - DECODE_NODES \ - RANDOM_RANGE_RATIO \ - DURATION \ - MODEL_PREFIX \ - PRECISION \ - RESULT_FILENAME \ - KV_OFFLOADING \ - IS_AGENTIC \ - FRAMEWORK \ - PREFILL_IMAGE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -set -x - -cd "$GITHUB_WORKSPACE/benchmarks/multi_node/amd_utils" || exit 1 - -export TIME_LIMIT=08:00:00 -export MODEL_PATH=$MODEL_PATH -export MODEL_NAME=$MODEL_NAME -export CONTAINER_IMAGE=$IMAGE -export PREFILL_IMAGE - -export RESULT_FILENAME - -if [[ "$PREFILL_NODES" -ne 1 || "$DECODE_NODES" -ne 1 || \ - "$PREFILL_NUM_WORKERS" -ne 1 || "$DECODE_NUM_WORKERS" -ne 1 ]]; then - echo "Error: tilert supports exactly 1 prefill node/worker + 1 decode node/worker" \ - "(got PREFILL_NODES=$PREFILL_NODES x$PREFILL_NUM_WORKERS, DECODE_NODES=$DECODE_NODES x$DECODE_NUM_WORKERS)" >&2 - exit 1 -fi - -if [[ "$KV_OFFLOADING" != "none" ]]; then - echo "Error: tilert has no KV offload backend; kv-offloading must be 'none' (got '$KV_OFFLOADING')" >&2 - exit 1 -fi - -# TileRT configuration. Every value is explicit here: server_tilert.sh -# validates each one with check_env_vars and supplies no defaults of its own. -export TILERT_VERSION=0.1.6.post2 -export TILERT_PROFILE=glm5_2 # decode_server --model (TileRT model profile) -export TILERT_MODEL_TYPE=glm-5 # weight_converter --model_type (fallback converter) -export TILERT_MODEL_PKG=glm_5_2_rocm # per-model converter package, preferred when importable -export SERVED_MODEL_NAME=glm5_2 -# GLM-5.3's full context window (config.json max_position_embeddings), as every -# in-tree GLM-5.2 recipe serves. (202752 was GLM-5.1's, inherited from the B200 -# TileRT recipe this mirrors.) -# -# Memory at this context, per rank, bf16 wire layout (verified against the -# tilert 0.1.6 and vLLM 0.24.0 sources and the MI355X logs, 287.98 GiB cards; -# the undivided PD buffer sizes are what 0.1.6 allocated on one card): -# decode : weights 90.72 GiB + engine cache window 93.25 GiB -# + PD receive buffer 99.06 GiB (receive_server.py, dense in max_seq_len) -# prefill: weights 90.45 GiB + profiling/non-torch 40.3 GiB + vLLM KV 91.71 GiB -# + PD staging buffer 99.06 GiB (prefill_connector.py, TP rank 0, -# allocated OUTSIDE vLLM's gpu-memory-utilization budget) -# Undivided, neither side starts: the decode rank is node-marginal (~283 of -# 288 GiB) and the prefill rank cannot fit at any utilization (~321 GiB). -# tilert 0.1.6.post1 keeps both buffers on the GPU but shards them by layer -# across the eight devices (layer lid on device lid % 8, TILERT_PD_SHARDS, -# default on), so each card holds 12.54 GiB instead of 99.06 GiB on one. -# convert() dequantises each layer on the device that received it, which spreads -# its transients too (108.6 KiB/token, 82.9 GiB at 800k tokens) instead of -# leaving them on cuda:0. Measured on 2x8 MI350X at this context with bf16 KV: -# decode peaks at 202.1 GiB per card, prefill at 269.2 GiB of 287.69 GiB, and -# the KV path stays device-to-device at 108 GB/s (81 GB in 751 ms, 54% of the -# 4x400 GbE line rate). No host hop and no patch: the wheel runs as shipped. -export TILERT_MAX_MODEL_LEN=1048576 -export TILERT_TRANSPORT=mooncake -export TILERT_PARSER=none -export TILERT_RDMA_STRICT=0 -export TILERT_CONVERT_LOCK_WAIT=21600 -export TILERT_SIMULATE_ACC_METHOD=match-expected -export TILERT_WEIGHTS_DIR="/models/${MODEL_NAME}-tilert-tp${DECODE_TP}" -# bf16 MLA KV on both roles. This is the only layout TileRT 0.1.6 can consume -# from vLLM on ROCm: MlaNsaProfile.classify_layers infers the layout from the -# cache tensor stride and accepts exactly 1152 B/token (bf16) or 656 B/token -# (fp8_ds_mla). vLLM's ROCM_AITER_MLA_SPARSE backend has no fp8_ds_mla; its -# plain "fp8" writes a flat 576 B/token row, which the connector rejects at -# register_kv_caches. Explicit bfloat16 rather than auto so the stride does not -# depend on the model dtype. Never float16: it passes the 1152 B check and is -# then read as bf16. -export PREFILL_KV_DTYPE=bfloat16 -# The ROCm backend supports block sizes [1, 64] and vLLM picks 1, which makes -# the connector's KI plane copy fail and MLA address the wrong rows. -export PREFILL_BLOCK_SIZE=64 -export DECODE_KV_DTYPE=bf16 -# The PD staging shard sits outside vLLM's budget, so vLLM needs 90.45 (weights) -# + 40.3 (profiling) + 91.71 GiB (KV for one 1048576-token request) = 222.5 GiB -# inside it: 0.85 x 287.98 = 244.8 GiB leaves 22 GiB of KV margin and 43 GiB -# outside the budget for the 12.54 GiB staging shard plus the ~6.3 GiB non-torch -# baseline measured on the decode OOM node (287.98 - 95.94 free - 184.17 - 1.58 -# reserved). 0.75 (216 GiB) refuses with "91.71 GiB KV cache is needed ... -# available 85.25 GiB". -export GPU_MEM_UTIL=0.85 -export SKIP_CONTAINER_BARRIER=0 -# Two images, one per rank, ~32 GB each. On a node that has neither cached the -# pull alone outlasts the SGLang path's 300s default and the 1800s this script -# used to hardcode, and the rank that comes up first waits out the whole -# timeout while its peer is still pulling. -export CONTAINER_BARRIER_TIMEOUT=5400 -export ROUTER_PORT=30000 -export PREFILL_PORT=8000 -export DECODE_CTRL_PORT=5556 -export DECODE_HTTP_PORT=5557 -export DECODE_WAIT=7200 # prefill waits for the decode ctrl port -export PREFILL_WAIT=3600 # prefill waits for its own vLLM port -export ROUTER_WAIT=10800 # decode waits for the router port to open - -if [[ "$SPEC_DECODING" == "mtp" ]]; then - # TileRT decode drafts at depth 3 (the only depth the ROCm GLM profile - # builds) and the golden acceptance curve is keyed on it. The vLLM prefill - # rank only has to materialise the MTP layer's KV, so it runs at 1. - export DECODE_MTP_SIZE=3 - export PREFILL_SPEC_TOKENS=1 -else - export DECODE_MTP_SIZE=0 - export PREFILL_SPEC_TOKENS=0 -fi -export TILERT_QUEUE_TIMEOUT=1800 # requests wait on the bs=1 decode engine -export THINKING_MODE=thinking_on - -if [[ "$PREFILL_EP" -ne 1 || "$DECODE_EP" -ne 1 || \ - "$PREFILL_DP_ATTN" == "true" || "$DECODE_DP_ATTN" == "true" ]]; then - echo "Error: tilert runs pure TP8 on both roles; ep must be 1 and dp-attn false" >&2 - exit 1 -fi -export PREFILL_ENABLE_EP=false -export PREFILL_ENABLE_DP=false -export DECODE_ENABLE_EP=false -export DECODE_ENABLE_DP=false - -JOB_ID=$(bash ./submit.sh $PREFILL_NODES \ - $PREFILL_NUM_WORKERS \ - $DECODE_NODES \ - $DECODE_NUM_WORKERS \ - $ISL $OSL "${CONC_LIST// /x}" inf \ - ${PREFILL_ENABLE_EP} ${PREFILL_ENABLE_DP} \ - ${DECODE_ENABLE_EP} ${DECODE_ENABLE_DP} \ - ${PREFILL_TP} ${DECODE_TP} \ - ${RANDOM_RANGE_RATIO}) - -if [[ $? -ne 0 ]]; then - echo "Failed to submit job" >&2 - exit 1 -fi - -echo "$JOB_ID" diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/env.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/env.sh index ba48879cb4..04adb875ac 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/env.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/env.sh @@ -2,21 +2,12 @@ source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only check_env_vars ENGINE -# MoRI-IO queue-pair tuning, the UCX RoCE GID index, SGLang router logging and the -# SGLang decode cuda-graph NCCL workaround. Only the SGLang MoRI KV path -# below reads these. ENGINE=tilert moves KV over mooncake and starts no SGLang -# router, so it is neither given nor reads them: validating them there would force -# the recipe to invent MoRI tuning for a transport it never uses. -if [[ "$ENGINE" != "tilert" ]]; then - check_env_vars \ - MORI_IO_SQ_BACKOFF_TIMEOUT_US MORI_IO_QP_MAX_SEND_WR MORI_IO_QP_MAX_CQE MORI_IO_QP_MAX_SGE MORI_IO_TC_DISABLE \ - UCX_IB_GID_INDEX MORI_APP_LOG_LEVEL SGLANG_ROUTER_STDOUT_LOGS TORCH_NCCL_BLOCKING_WAIT NCCL_BLOCKING_WAIT \ - SGLANG_OPT_USE_AITER_INDEXER -fi -# Dual-engine environment setup for multi-node disaggregated serving. -# -# ENGINE=sglang-disagg or tilert selects the engine-specific block. -# +# SGLang MoRI environment. +check_env_vars \ + MORI_IO_SQ_BACKOFF_TIMEOUT_US MORI_IO_QP_MAX_SEND_WR MORI_IO_QP_MAX_CQE MORI_IO_QP_MAX_SGE MORI_IO_TC_DISABLE \ + UCX_IB_GID_INDEX MORI_APP_LOG_LEVEL SGLANG_ROUTER_STDOUT_LOGS TORCH_NCCL_BLOCKING_WAIT NCCL_BLOCKING_WAIT \ + SGLANG_OPT_USE_AITER_INDEXER + # REQUIRED ENVIRONMENT VARIABLES: # IBDEVICES - RDMA/InfiniBand device names (e.g., ionic_0,ionic_1,... or mlx5_0,mlx5_1,...) # Set by runner or auto-detected from hostname. @@ -119,10 +110,6 @@ else fi fi -if [[ "$ENGINE" == "tilert" ]]; then - echo "[INFO] tilert: IBDEVICES=$IBDEVICES NCCL_SOCKET_IFNAME=$NCCL_SOCKET_IFNAME NCCL_IB_HCA=$NCCL_IB_HCA" - -else export SGLANG_USE_AITER=1 export AITER_LOG_LEVEL=ERROR @@ -237,5 +224,3 @@ else export GPU_MAX_HW_QUEUES=2 fi fi - -fi diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm b/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm index cecc989a0f..ef46cfc859 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm @@ -36,8 +36,6 @@ echo "" # at runtime, but the CWD remains the submit-time directory (amd_utils/). if [[ "$ENGINE" == "atom-disagg" ]]; then MODELS_YAML="$(pwd)/models_atom.yaml" -elif [[ "$ENGINE" == "tilert" ]]; then - MODELS_YAML="$(pwd)/models_tilert.yaml" else MODELS_YAML="$(pwd)/models.yaml" fi @@ -52,16 +50,6 @@ if [[ -z "${DOCKER_IMAGE_NAME:-}" ]]; then exit 1 fi -if [[ "$ENGINE" == "tilert" && -z "${PREFILL_IMAGE:-}" ]]; then - echo "Error: ENGINE=tilert requires PREFILL_IMAGE (e.g. PREFILL_IMAGE=vllm/vllm-openai-rocm:nightly- in prefill.additional-settings)." - exit 1 -fi -if [[ "$ENGINE" == "tilert" ]]; then - # server_tilert.sh takes the container-creation barrier timeout from the - # recipe (no 300s default as on the SGLang path); fail here, before sbatch - # work is done, rather than inside the container. - check_env_vars CONTAINER_BARRIER_TIMEOUT -fi # Resolve the models.yaml entry the same way server_sglang.sh does: agentic runs # (IS_AGENTIC) use the '-AgentX' recipe, non-agentic disaggregated runs use @@ -564,41 +552,6 @@ if [[ "$ENGINE" == "atom-disagg" ]]; then -e EXTRA_SERVER_ARGS=\${EXTRA_SERVER_ARGS:-} -e IBDEVICES=${IBDEVICES:-} ) -elif [[ "$ENGINE" == "tilert" ]]; then - DOCKER_ENV_ENGINE=( - -e MODEL_PATH=$DOCKER_MODEL_PATH - -e PREFILL_IMAGE=${PREFILL_IMAGE} - -e TILERT_VERSION=${TILERT_VERSION} - -e TILERT_PROFILE=${TILERT_PROFILE} - -e TILERT_MODEL_TYPE=${TILERT_MODEL_TYPE} - -e TILERT_MODEL_PKG=${TILERT_MODEL_PKG} - -e TILERT_MAX_MODEL_LEN=${TILERT_MAX_MODEL_LEN} - -e TILERT_TRANSPORT=${TILERT_TRANSPORT} - -e TILERT_PARSER=${TILERT_PARSER} - -e TILERT_QUEUE_TIMEOUT=${TILERT_QUEUE_TIMEOUT} - -e TILERT_WEIGHTS_DIR=${TILERT_WEIGHTS_DIR} - -e TILERT_RDMA_STRICT=${TILERT_RDMA_STRICT} - -e TILERT_CONVERT_LOCK_WAIT=${TILERT_CONVERT_LOCK_WAIT} - -e TILERT_SIMULATE_ACC_METHOD=${TILERT_SIMULATE_ACC_METHOD} - -e \"TILERT_EXTRA_ENV=${TILERT_EXTRA_ENV:-}\" - -e SERVED_MODEL_NAME=${SERVED_MODEL_NAME} - -e PREFILL_KV_DTYPE=${PREFILL_KV_DTYPE} - -e PREFILL_BLOCK_SIZE=${PREFILL_BLOCK_SIZE} - -e PREFILL_SPEC_TOKENS=${PREFILL_SPEC_TOKENS} - -e DECODE_KV_DTYPE=${DECODE_KV_DTYPE} - -e GPU_MEM_UTIL=${GPU_MEM_UTIL} - -e PREFILL_PORT=${PREFILL_PORT} - -e DECODE_CTRL_PORT=${DECODE_CTRL_PORT} - -e DECODE_HTTP_PORT=${DECODE_HTTP_PORT} - -e DECODE_WAIT=${DECODE_WAIT} - -e PREFILL_WAIT=${PREFILL_WAIT} - -e ROUTER_WAIT=${ROUTER_WAIT} - -e SKIP_CONTAINER_BARRIER=${SKIP_CONTAINER_BARRIER} - # Golden-acceptance selection on agentic MTP runs; unset elsewhere. - -e THINKING_MODE=${THINKING_MODE:-} - -e IBDEVICES=${IBDEVICES:-} - -e PYTHONPYCACHEPREFIX=/tmp/pycache - ) else DOCKER_ENV_ENGINE=( -e SGLANG_WS_PATH=${WS_PATH} @@ -751,12 +704,6 @@ else fi fi # end: if ENGINE == atom-disagg -RANK_IMAGE= -if [[ \"$ENGINE\" == \"tilert\" && \"\$SLURM_PROCID\" -lt \"$xP\" ]]; then - RANK_IMAGE=\"$PREFILL_IMAGE\" - echo \"[tilert] rank \$SLURM_PROCID is a prefill rank; using PREFILL_IMAGE=\$RANK_IMAGE\" -fi - exec \$DOCKER_CMD run \ --init \ --stop-timeout 10 \ @@ -798,7 +745,7 @@ exec \$DOCKER_CMD run \ ${CLIENT_DOCKER_ENV} \ --name \"$DOCKER_CONT_NAME\" \ --entrypoint \"\" \ - \"\${RANK_IMAGE:-$DOCKER_IMAGE_NAME}\" bash -lc ' + \"$DOCKER_IMAGE_NAME\" bash -lc ' set -o pipefail mkdir -p /run_logs/slurm_job-'\"\$SLURM_JOB_ID\"' '"$RUN_FILE_FULL"' 2>&1 | tee /run_logs/slurm_job-'\"\$SLURM_JOB_ID\"'/server_\$(hostname).log diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/models_tilert.yaml b/inferencex-e2e/benchmarks/multi_node/amd_utils/models_tilert.yaml deleted file mode 100644 index 1697059d54..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/models_tilert.yaml +++ /dev/null @@ -1,6 +0,0 @@ -# Model-owned engine settings for ENGINE=tilert. Everything the caller owns -# (profile, topology, ports, dtypes, draft depth) is set by the recipe, not here. -GLM-5.3: - prefill_env: "VLLM_ROCM_USE_AITER=1 VLLM_ROCM_USE_AITER_MOE=1 VLLM_ROCM_USE_AITER_FUSION_SHARED_EXPERTS=1 VLLM_ENGINE_READY_TIMEOUT_S=10800" - prefill_extra_flags: "" - decode_extra_flags: "" diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh index f304f53b40..a693fb92cc 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh @@ -5,7 +5,6 @@ source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only # Dispatches to the engine-specific server launcher based on ENGINE env var. # ENGINE=sglang-disagg (default) -> server_sglang.sh (SGLang + MoRI) # ENGINE=atom-disagg -> server_atom.sh (ATOM + mooncake) -# ENGINE=tilert -> server_tilert.sh (vLLM prefill + TileRT decode) check_env_vars ENGINE WS_PATH if [[ -f /config/hicache_mc.env ]]; then @@ -20,8 +19,6 @@ echo "[DISPATCHER] ENGINE=$ENGINE WS_PATH=$WS_PATH" if [[ "$ENGINE" == "atom-disagg" ]]; then export ATOM_WS_PATH="$WS_PATH" source "$WS_PATH/server_atom.sh" -elif [[ "$ENGINE" == "tilert" ]]; then - source "$WS_PATH/server_tilert.sh" else source "$WS_PATH/server_sglang.sh" fi diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh index caa187b6bc..4cc51efea8 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh @@ -16,7 +16,6 @@ check_env_vars \ EXTRA_SERVER_ARGS="${EXTRA_SERVER_ARGS:-}" -source $ATOM_WS_PATH/setup_deps.sh source $ATOM_WS_PATH/env_atom.sh # lm-eval with high num_concurrent exhausts the default 1024 FD limit. diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_sglang.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_sglang.sh index bd6403f8c9..571d1799e0 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_sglang.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_sglang.sh @@ -20,7 +20,6 @@ BENCH_MAX_CONC_VALUE=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | t # can resolve formulas like "BENCH_MAX_CONC_VALUE*2" for max_running_requests. export BENCH_MAX_CONC_VALUE -source $SGLANG_WS_PATH/setup_deps.sh source $SGLANG_WS_PATH/env.sh # Install before starting UMBP or serving processes. Early readiness failures must diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_tilert.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_tilert.sh deleted file mode 100644 index 3a66c926d7..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_tilert.sh +++ /dev/null @@ -1,573 +0,0 @@ -#!/bin/bash -# TileRT disaggregated launcher: upstream vLLM ROCm prefill (TileRTConnector, -# kv_producer) + TileRT decode_server + the OpenAI-compatible pd_router. -# Every value below is supplied by the recipe through job.slurm; this script -# validates them and never invents a default for caller-owned configuration. - -source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only - -check_env_vars \ - NODE0_ADDR NODE_RANK MODEL_DIR MODEL_NAME MODEL_PATH xP yD IPADDRS \ - DRY_RUN GPUS_PER_NODE PREFILL_TP_SIZE DECODE_TP_SIZE \ - BENCH_INPUT_LEN BENCH_OUTPUT_LEN BENCH_MAX_CONCURRENCY \ - RUN_EVAL EVAL_ONLY EVAL_FRAMEWORK BENCHMARK_LOGS_DIR WS_PATH \ - SLURM_JOB_ID SPEC_DECODING \ - TILERT_PROFILE TILERT_MODEL_TYPE TILERT_MODEL_PKG TILERT_MAX_MODEL_LEN \ - TILERT_TRANSPORT TILERT_PARSER TILERT_QUEUE_TIMEOUT TILERT_WEIGHTS_DIR \ - TILERT_RDMA_STRICT TILERT_CONVERT_LOCK_WAIT TILERT_SIMULATE_ACC_METHOD \ - PREFILL_KV_DTYPE PREFILL_BLOCK_SIZE PREFILL_SPEC_TOKENS DECODE_KV_DTYPE \ - DECODE_MTP_SIZE GPU_MEM_UTIL SERVED_MODEL_NAME \ - DECODE_CTRL_PORT DECODE_HTTP_PORT PREFILL_PORT ROUTER_PORT \ - DECODE_WAIT PREFILL_WAIT ROUTER_WAIT SKIP_CONTAINER_BARRIER \ - CONTAINER_BARRIER_TIMEOUT - -LOG_DIR="/run_logs/slurm_job-${SLURM_JOB_ID}" -SHARED_LOG_DIR="${BENCHMARK_LOGS_DIR}/logs/slurm_job-${SLURM_JOB_ID}" -mkdir -p "$LOG_DIR" - -if [[ "$xP" -ne 1 || "$yD" -ne 1 ]]; then - echo "ERROR: tilert supports exactly 1 prefill + 1 decode worker (got xP=$xP yD=$yD)" >&2 - exit 1 -fi -if [[ "$NODE_RANK" -lt "$xP" ]]; then - TILERT_ROLE=prefill -else - TILERT_ROLE=decode -fi -export TILERT_ROLE - -source "$WS_PATH/setup_deps.sh" -source "$WS_PATH/env.sh" -# benchmark_lib.sh derives AIPERF_DIR from this at source time, so -# it must be set before the library is loaded, not in run_agentic_replay. The -# AgentX replay runs in this container, where the repo is mounted at /workspace. -export INFMAX_CONTAINER_WORKSPACE=/workspace -source /workspace/benchmarks/benchmark_lib.sh - -# Model-specific engine environment (not caller configuration): the prefill -# vLLM env block lives with the model. Everything else is passed in by the recipe. -MODELS_YAML="${WS_PATH}/models_tilert.yaml" -eval "$("$PY" - "$MODELS_YAML" "$MODEL_NAME" <<'PYEOF' -import shlex, sys, yaml -path, name = sys.argv[1], sys.argv[2] -with open(path) as f: - models = yaml.safe_load(f) or {} -if name not in models: - sys.exit(f"model '{name}' is not present in {path}") -m = models[name] or {} -for key, var in (("prefill_env", "TILERT_PREFILL_ENV"), - ("prefill_extra_flags", "TILERT_PREFILL_EXTRA_FLAGS"), - ("decode_extra_flags", "TILERT_DECODE_EXTRA_FLAGS")): - print(f"{var}={shlex.quote(str(m.get(key) or ''))}") -PYEOF -)" || { echo "ERROR: cannot read the tilert model entry for '$MODEL_NAME' from $MODELS_YAML" >&2; exit 1; } -echo "[tilert] model entry '$MODEL_NAME' loaded from $MODELS_YAML" - -export ROUTER_PORT -export SERVED_MODEL_NAME - -PREFILL_SPEC=() -DECODE_MTP=() -if [[ "$SPEC_DECODING" == "mtp" ]]; then - # The prefill rank only has to build the MTP layer's KV; TileRT decode owns - # the draft depth (DECODE_MTP_SIZE), so the two counts differ by design. - PREFILL_SPEC=(--speculative-config "{\"method\":\"mtp\",\"num_speculative_tokens\":${PREFILL_SPEC_TOKENS}}") - # decode_server only accepts depth 3 today, but pass it explicitly so the - # converter's --num_mtp, the golden-curve key and the engine depth agree by - # data flow rather than by coincidence of defaults. - DECODE_MTP=(--with-mtp --num-mtp "$DECODE_MTP_SIZE") -fi - -# The only TileRT recipe on this cluster is AgentX (agentic-coding). -if [[ "${IS_AGENTIC:-0}" != "1" && "${IS_AGENTIC:-}" != "true" && "${SCENARIO_TYPE:-}" != "agentic-coding" ]]; then - echo "ERROR: server_tilert.sh only runs agentic-coding (IS_AGENTIC=${IS_AGENTIC:-} SCENARIO_TYPE=${SCENARIO_TYPE:-})" >&2 - exit 1 -fi - -IFS=',' read -ra IP_ARRAY <<< "$IPADDRS" -PREFILL_HOST="${IP_ARRAY[0]:-$NODE0_ADDR}" -DECODE_HOST="${IP_ARRAY[$xP]:-}" -if [[ -z "$DECODE_HOST" ]]; then - echo "ERROR: cannot resolve the decode node IP from IPADDRS='$IPADDRS' (xP=$xP)" >&2 - exit 1 -fi -host_ip=$(ip route get 1.1.1.1 2>/dev/null | awk '/src/ {print $7}') -host_name=$(hostname) - -echo "[tilert] ROLE=$TILERT_ROLE rank=$NODE_RANK host=$host_name ($host_ip)" -echo "[tilert] PREFILL_HOST=$PREFILL_HOST:$PREFILL_PORT DECODE_HOST=$DECODE_HOST:$DECODE_CTRL_PORT/$DECODE_HTTP_PORT ROUTER=:$ROUTER_PORT" -echo "[tilert] MODEL_PATH=$MODEL_PATH profile=$TILERT_PROFILE served=$SERVED_MODEL_NAME max_len=$TILERT_MAX_MODEL_LEN transport=$TILERT_TRANSPORT kv=${PREFILL_KV_DTYPE}->${DECODE_KV_DTYPE} mtp=${SPEC_DECODING}" - -# Enable libibverbs fork safety on both ranks before any verbs context exists. -# Without it, ibv_fork_init() can fail in these containers while Mooncake -# initialization still reports success, silently falling back from RDMA to TCP. -# Slower prefill-to-decode KV transfer degrades TTFT; TPOT is unaffected. -export RDMAV_FORK_SAFE=1 - -for env_pair in ${TILERT_EXTRA_ENV}; do - export "${env_pair?}" - echo "[tilert][EXTRA_ENV] $env_pair" -done - -log_and_run_bg() { - local label="$1" logfile="$2"; shift 2 - { printf '===== [%s] %s =====\n' "$label" "$(date '+%F %T')" - printf '[cmd]'; printf ' %q' "$@"; printf '\n' - printf '[cwd] %s\n[host] %s\n\n' "$PWD" "$host_name" - } | tee -a "$logfile" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: [$label] not started" - LAST_BG_PID="" - return 0 - fi - "$@" >>"$logfile" 2>&1 & - LAST_BG_PID=$! - echo "[$label] pid=$LAST_BG_PID log=$logfile" -} - -rdma_preflight() { - local warn=0 - echo "[rdma] role=$TILERT_ROLE IBDEVICES=${IBDEVICES:-} NCCL_SOCKET_IFNAME=${NCCL_SOCKET_IFNAME:-}" - local uverbs=(/dev/infiniband/uverbs*) - if [[ -e "${uverbs[0]}" ]]; then - echo "[rdma] verbs devices: ${uverbs[*]}" - else - echo "[rdma] WARNING: /dev/infiniband/uverbs* missing -- the container has no RDMA device nodes (job.slurm passes --device /dev/infiniband)" >&2 - warn=1 - fi - local ml; ml="$(ulimit -l 2>/dev/null)" - if [[ "$ml" == "unlimited" ]]; then - echo "[rdma] memlock: unlimited" - else - echo "[rdma] WARNING: memlock=$ml (not unlimited) -- pinning memory for RDMA may fail (job.slurm passes --ulimit memlock=-1)" >&2 - warn=1 - fi - if command -v ibv_devices >/dev/null 2>&1; then - echo "[rdma] ibv_devices:"; ibv_devices 2>&1 | sed 's/^/[rdma] /' - fi - if (( warn )) && [[ "$TILERT_RDMA_STRICT" == "1" ]]; then - echo "[rdma] TILERT_RDMA_STRICT=1 and preflight did not fully pass -- aborting" >&2 - return 1 - fi - return 0 -} - -stage_tokenizer_files() { - local staged=0 f b - for f in "$MODEL_PATH"/*; do - [[ -f "$f" ]] || continue - b="$(basename "$f")" - [[ "$b" == *.safetensors ]] && continue - [[ "$b" == "model.safetensors.index.json" ]] && continue - [[ -e "$TILERT_WEIGHTS_DIR/$b" ]] && continue - cp -p "$f" "$TILERT_WEIGHTS_DIR/$b" && staged=$((staged+1)) - done - echo "[stage_tokenizer] staged $staged auxiliary file(s) from $MODEL_PATH -> $TILERT_WEIGHTS_DIR" - local missing=() - [[ -f "$TILERT_WEIGHTS_DIR/chat_template.jinja" ]] || missing+=(chat_template.jinja) - [[ -f "$TILERT_WEIGHTS_DIR/tokenizer_config.json" || -f "$TILERT_WEIGHTS_DIR/tokenizer.json" ]] \ - || missing+=("tokenizer.json/tokenizer_config.json") - if (( ${#missing[@]} )); then - echo "[stage_tokenizer] ERROR: $TILERT_WEIGHTS_DIR is missing ${missing[*]}; check that MODEL_PATH=$MODEL_PATH is an HF directory with the tokenizer" >&2 - return 1 - fi - return 0 -} - -_tilert_weights_cached() { - local r - for r in $(seq 0 $((DECODE_TP_SIZE - 1))); do - [[ -f "$TILERT_WEIGHTS_DIR/rank${r}/model.safetensors.index.json" ]] || return 1 - done - # The engine refuses a cache converted without the MTP module (end2end.py - # checks tilert_meta.json num_mtp), but only after loading ~90 GiB of - # weights. Check the same field here so a stale non-MTP cache is - # re-converted instead of failing late. - [[ -f "$TILERT_WEIGHTS_DIR/tilert_meta.json" ]] || return 1 - if [[ "$SPEC_DECODING" == "mtp" ]]; then - "$PY" - "$TILERT_WEIGHTS_DIR/tilert_meta.json" <<'PYEOF' || return 1 -import json, sys -sys.exit(0 if int(json.load(open(sys.argv[1])).get("num_mtp", 0)) >= 1 else 1) -PYEOF - fi - return 0 -} - -convert_weights() { - if _tilert_weights_cached; then - echo "[weight_converter] cache hit (${DECODE_TP_SIZE}/${DECODE_TP_SIZE} rank index.json), skipping conversion: $TILERT_WEIGHTS_DIR" - return 0 - fi - mkdir -p "$TILERT_WEIGHTS_DIR" || { echo "[weight_converter] ERROR: cannot create $TILERT_WEIGHTS_DIR (set TILERT_WEIGHTS_DIR to a writable shared path)" >&2; return 1; } - exec 9>"$TILERT_WEIGHTS_DIR/.convert.lock" - flock -w "$TILERT_CONVERT_LOCK_WAIT" 9 || { - echo "[weight_converter] timed out waiting for the conversion lock (another job still converting?)" >&2; return 1; } - if _tilert_weights_cached; then - echo "[weight_converter] cache produced by a concurrent job, skipping conversion"; exec 9>&-; return 0 - fi - if [[ -n "$(ls -A "$TILERT_WEIGHTS_DIR" 2>/dev/null | grep -v '^\.convert\.lock$')" ]]; then - echo "[weight_converter] leftovers without index.json (previous conversion incomplete); cleaning and re-converting" - find "$TILERT_WEIGHTS_DIR" -mindepth 1 ! -name '.convert.lock' -delete - fi - echo "[weight_converter] $MODEL_PATH -> $TILERT_WEIGHTS_DIR (model_type=$TILERT_MODEL_TYPE)" - local conv_mod conv_args - if "$PY" -c "import tilert.models.${TILERT_MODEL_PKG}.weight_converter" 2>/dev/null; then - conv_mod="tilert.models.${TILERT_MODEL_PKG}.weight_converter" - conv_args=(--model_dir "$MODEL_PATH" --save_dir "$TILERT_WEIGHTS_DIR" - --device "cuda:$((GPUS_PER_NODE - 1))") - [[ "$SPEC_DECODING" == "mtp" ]] && conv_args+=(--num_mtp "$DECODE_MTP_SIZE") - else - conv_mod="tilert.models.preprocess.weight_converter" - conv_args=(--model_type "$TILERT_MODEL_TYPE" --model_dir "$MODEL_PATH" --save_dir "$TILERT_WEIGHTS_DIR") - fi - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $PY -m $conv_mod ${conv_args[*]}" - exec 9>&-; return 0 - fi - echo "[weight_converter] using $conv_mod" - "$PY" -m "$conv_mod" "${conv_args[@]}" \ - 2>&1 | tee "$LOG_DIR/tilert_weight_converter_${host_name}.log" - local rc=${PIPESTATUS[0]} - exec 9>&- - if [[ $rc -ne 0 ]] || ! _tilert_weights_cached; then - echo "[weight_converter] ERROR: conversion failed (rc=$rc, per-rank index.json complete: $(_tilert_weights_cached && echo yes || echo no))" >&2 - return 1 - fi - echo "[weight_converter] conversion done and cached: $TILERT_WEIGHTS_DIR" -} - -start_decode() { - # shellcheck disable=SC2206 - local extra=( ${TILERT_DECODE_EXTRA_FLAGS} ) - if [[ "$SPEC_DECODING" == "mtp" && "$EVAL_ONLY" != "true" && "$RUN_EVAL" != "true" ]]; then - check_env_vars MODEL_PREFIX THINKING_MODE - local curve="${WS_PATH%/benchmarks/*}/infx/golden_al_distribution/${MODEL_PREFIX}_mtp.yaml" - TILERT_SIMULATE_ACC_LEN="$("$PY" - "$curve" "$THINKING_MODE" "$DECODE_MTP_SIZE" <<'PYEOF' -import sys, yaml -path, thinking, tokens = sys.argv[1], sys.argv[2], int(sys.argv[3]) -data = yaml.safe_load(open(path)) -if not isinstance(data, dict) or len(data) != 1: - sys.exit(f"golden curve {path} must hold exactly one model key") -model, modes = next(iter(data.items())) -try: - value = float(modes[thinking][tokens]) -except (KeyError, TypeError, ValueError): - sys.exit(f"no golden acceptance for {model}/{thinking}/{tokens} draft tokens in {path}") -if not 1 <= value <= tokens + 1: - sys.exit(f"golden acceptance {value} out of range for {tokens} draft tokens") -print(f"{value:g}") -PYEOF -)" || { echo "[tilert] ERROR: golden AL lookup failed (curve=$curve)" >&2; exit 1; } - echo "[tilert] golden AL ${TILERT_SIMULATE_ACC_LEN} from $(basename "$curve") ($THINKING_MODE, K=${DECODE_MTP_SIZE})" - fi - - if [[ -n "${TILERT_SIMULATE_ACC_LEN:-}" && "$EVAL_ONLY" != "true" ]]; then - export TILERT_SIMULATE_ACC_LEN - export TILERT_SIMULATE_ACC_METHOD - echo "[decode] simulated acceptance: TILERT_SIMULATE_ACC_LEN=${TILERT_SIMULATE_ACC_LEN}" \ - "method=${TILERT_SIMULATE_ACC_METHOD} (output text is meaningless by design)" - else - unset TILERT_SIMULATE_ACC_LEN TILERT_SIMULATE_ACC_METHOD - echo "[decode] real MTP verification (no simulated acceptance)" - fi - local cmd=("$PY" -m tilert.pd_vllm.decode_server - --engine tilert --model "$TILERT_PROFILE" - --model-weights-dir "$TILERT_WEIGHTS_DIR" - --max-seq-len "$TILERT_MAX_MODEL_LEN" - --kv-cache-dtype "$DECODE_KV_DTYPE" --transport "$TILERT_TRANSPORT" - --ctrl-port "$DECODE_CTRL_PORT" --http-port "$DECODE_HTTP_PORT" - "${DECODE_MTP[@]}" "${extra[@]}") - log_and_run_bg decode "$LOG_DIR/decode_${host_name}.log" "${cmd[@]}" - DECODE_PID=$LAST_BG_PID -} - -start_prefill() { - for env_pair in ${TILERT_PREFILL_ENV}; do - export "${env_pair?}" - echo "[PREFILL_ENV] $env_pair" - done - local served=("$SERVED_MODEL_NAME") - [[ -n "$MODEL_NAME" && "$MODEL_NAME" != "$SERVED_MODEL_NAME" ]] && served+=("$MODEL_NAME") - # shellcheck disable=SC2206 - local extra=( ${TILERT_PREFILL_EXTRA_FLAGS} ) - local kv_cfg - kv_cfg=$(printf '{"kv_connector":"TileRTConnector","kv_connector_module_path":"tilert.pd_vllm.prefill_connector","kv_role":"kv_producer","kv_connector_extra_config":{"tilert_host":"%s","tilert_ctrl_port":%s,"tilert_model":"%s","tilert_max_seq_len":%s,"tilert_transport":"%s"}}' \ - "$DECODE_HOST" "$DECODE_CTRL_PORT" "$TILERT_PROFILE" "$TILERT_MAX_MODEL_LEN" "$TILERT_TRANSPORT") - local cmd=(vllm serve "$MODEL_PATH" - --served-model-name "${served[@]}" --port "$PREFILL_PORT" - --tensor-parallel-size "$PREFILL_TP_SIZE" --max-model-len "$TILERT_MAX_MODEL_LEN" - --enforce-eager --trust-remote-code --return-tokens-as-token-ids - --gpu-memory-utilization "$GPU_MEM_UTIL" --kv-cache-dtype "$PREFILL_KV_DTYPE" - --block-size "$PREFILL_BLOCK_SIZE" - "${PREFILL_SPEC[@]}" - --kv-transfer-config "$kv_cfg" - "${extra[@]}") - log_and_run_bg prefill "$LOG_DIR/prefill_${host_name}.log" "${cmd[@]}" - PREFILL_PID=$LAST_BG_PID -} - -start_router() { - local cmd=(env HIP_VISIBLE_DEVICES= ROCR_VISIBLE_DEVICES= CUDA_VISIBLE_DEVICES= - "$PY" -m tilert.pd_vllm.pd_router - --vllm-url "http://$PREFILL_HOST:$PREFILL_PORT" - --decode "$DECODE_HOST:$DECODE_CTRL_PORT:$DECODE_HTTP_PORT" - --host 0.0.0.0 --port "$ROUTER_PORT" --model-path "$MODEL_PATH" --parser "$TILERT_PARSER" - --queue-timeout "$TILERT_QUEUE_TIMEOUT") - log_and_run_bg router "$LOG_DIR/router_${host_name}.log" "${cmd[@]}" - ROUTER_PID=$LAST_BG_PID -} - -tcp_open() { (exec 3<>"/dev/tcp/$1/$2") 2>/dev/null; } - -wait_for_tcp() { - local host="$1" port="$2" timeout="${3:-600}" pid="${4:-}" - local deadline=$(( SECONDS + timeout )) - until tcp_open "$host" "$port"; do - if [[ -n "$pid" ]] && ! kill -0 "$pid" 2>/dev/null; then - echo "[wait_for_tcp] process $pid exited before $host:$port opened" >&2; return 2 - fi - if [[ $SECONDS -ge $deadline ]]; then - echo "[wait_for_tcp] timeout: $host:$port not open after ${timeout}s" >&2; return 1 - fi - sleep 5 - done - echo "[wait_for_tcp] $host:$port ready" -} - -wait_for_tcp_close() { - local host="$1" port="$2" pid="${3:-}" - while tcp_open "$host" "$port"; do - if [[ -n "$pid" ]] && ! kill -0 "$pid" 2>/dev/null; then - echo "[wait_for_tcp_close] process $pid exited while $host:$port is still open" >&2; return 2 - fi - sleep 10 - done - echo "[wait_for_tcp_close] $host:$port closed" -} - -copy_logs_to_shared() { - [[ "$DRY_RUN" -eq 0 ]] || return 0 - mkdir -p "$SHARED_LOG_DIR" && cp -r "$LOG_DIR"/. "$SHARED_LOG_DIR"/ \ - && echo "Copied $LOG_DIR -> $SHARED_LOG_DIR" \ - || echo "WARNING: failed to copy $LOG_DIR to $SHARED_LOG_DIR" >&2 -} -trap copy_logs_to_shared EXIT - -run_lm_eval_on_router() { - echo "Running lm-eval evaluation on the router..." - local ok=false _attempt - for _attempt in 1 2 3; do - if curl -sf --max-time 10 "http://0.0.0.0:${ROUTER_PORT}/health" >/dev/null 2>&1; then ok=true; break; fi - echo "Eval health check attempt $_attempt failed, retrying in 10s..."; sleep 10 - done - if [[ "$ok" != "true" ]]; then - echo "ERROR: router health check failed after 3 attempts; skipping eval" >&2 - return 1 - fi - local eval_failed=0 - pushd /workspace >/dev/null || return 1 - if [[ -n "${EVAL_CONC:-}" ]]; then - export EVAL_CONCURRENT_REQUESTS="${EVAL_CONC}" - else - export EVAL_CONCURRENT_REQUESTS=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | tail -1) - fi - # run_lm_eval reads the endpoint from PORT (check_env_vars) and names the - # model from MODEL_NAME, which the vLLM prefill also serves next to - # SERVED_MODEL_NAME; MODEL stays the local HF dir for the context lookup. - export PORT="$ROUTER_PORT" - export MODEL="$MODEL_PATH" - export MAX_MODEL_LEN="$TILERT_MAX_MODEL_LEN" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: run_eval --port $ROUTER_PORT (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS})" - else - run_eval --port "$ROUTER_PORT" - local eval_rc=$? - if [[ $eval_rc -ne 0 ]]; then - echo "ERROR: run_eval exited rc=$eval_rc; preserving failure artifacts" >&2 - eval_failed=1 - else - export TP="${PREFILL_TP_SIZE}" CONC="${EVAL_CONCURRENT_REQUESTS}" EP_SIZE=1 - export PREFILL_TP="${PREFILL_TP_SIZE}" PREFILL_EP=1 PREFILL_NUM_WORKERS="${xP}" - export DECODE_TP="${DECODE_TP_SIZE}" DECODE_EP=1 DECODE_NUM_WORKERS="${yD}" - export DP_ATTENTION=false PREFILL_DP_ATTENTION=false DECODE_DP_ATTENTION=false - export ISL="${BENCH_INPUT_LEN}" OSL="${BENCH_OUTPUT_LEN}" - # As on the SGLang path: rewrite meta_env.json from the exports above, - # then stage unless run_eval already did (eval-only). - rewrite_lm_eval_meta_env - if [[ "$EVAL_ONLY" != "true" ]]; then - append_lm_eval_summary - fi - fi - local eval_copy_dir="$LOG_DIR/eval_results" - if stage_eval_artifacts "$eval_copy_dir" /workspace "${EVAL_RESULT_DIR:-}"; then - echo "Eval artifacts staged in $eval_copy_dir" - else - echo "ERROR: failed to stage eval artifacts in $eval_copy_dir" >&2 - eval_failed=1 - fi - fi - popd >/dev/null || true - return $eval_failed -} - -run_agentic_replay() { - local rc=0 - wait_for_server_ready --port "$ROUTER_PORT" --server-log "$LOG_DIR/router_${host_name}.log" --server-pid "$ROUTER_PID" - cd /workspace || return 1 - - export PORT="$ROUTER_PORT" - export MODEL="$MODEL_PATH" # aiperf --tokenizer (local HF dir) - export SERVED_MODEL_NAME # aiperf --model (name the router/vLLM serve) - check_env_vars DURATION RESULT_FILENAME INFMAX_CONTAINER_WORKSPACE - export MAX_MODEL_LEN="$TILERT_MAX_MODEL_LEN" - # TileRT decode exposes no /metrics route; only the vLLM prefill is scraped. - export AIPERF_SERVER_METRICS_URLS="http://${PREFILL_HOST}:${PREFILL_PORT}/metrics" - export TRANSFORMERS_VERBOSITY=error TOKENIZERS_PARALLELISM=false - # Keep the trace corpus and aiperf's HF downloads on the node's /tmp mount - # instead of the container's ephemeral ~/.cache, as the SGLang client does. - export HF_HOME=/run_logs/hf_cache - - local result_dir="$LOG_DIR/agentic" - local result_filename_base="$RESULT_FILENAME" - mkdir -p "$result_dir" - - # Neither server.sh nor this script runs with errexit; a failed bootstrap - # must not fall through into the replay loop and its misleading cascade. - resolve_trace_source || return 1 - install_agentic_deps || return 1 - - local conc conc_result_dir - for conc in ${BENCH_MAX_CONCURRENCY//x/ }; do - echo "==========================================" - echo "Agentic trace replay: conc=$conc" - echo "==========================================" - conc_result_dir="$result_dir/conc_${conc}" - mkdir -p "$conc_result_dir" - export CONC="$conc" USERS="$conc" - build_replay_cmd "$conc_result_dir" - export RESULT_FILENAME="${result_filename_base}_conc${conc}" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $REPLAY_CMD" - elif ! run_agentic_replay_and_write_outputs "$conc_result_dir"; then - echo "WARNING: agentic trace replay for conc=$conc failed (replay or validation) after writing available results" >&2 - rc=1 - fi - echo "-----------------------------------------" - done - export RESULT_FILENAME="$result_filename_base" - return $rc -} - -echo "Waiting at the container creation barrier on $host_name" -if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: skipping container creation barrier" -elif [[ "$SKIP_CONTAINER_BARRIER" == "1" ]]; then - echo "SKIP_CONTAINER_BARRIER=1: caller asserts all containers are up" -else - # --grace 60: after the barrier passes, sync.py keeps the port open for - # max(60, timeout/2) seconds in the foreground so a peer one poll behind - # still sees it. At CONTAINER_BARRIER_TIMEOUT=5400 that is a 45-minute idle - # sleep on every rank (jobs 45373/45374 slept 06:28-07:13). Both ranks pass - # within one 5 s poll of each other and the stages below have their own - # readiness waits, so 60 s is plenty. - "$PY" "$WS_PATH/sync.py" barrier \ - --local-ip "${host_ip}" --local-port 5000 --enable-port \ - --node-ips "${IPADDRS}" --node-ports 5000 \ - --wait-for-all-ports --timeout "$CONTAINER_BARRIER_TIMEOUT" --grace 60 \ - || { echo "ERROR: container creation barrier failed after ${CONTAINER_BARRIER_TIMEOUT}s -- the peer rank never opened port 5000." \ - "A cold image pull is the usual cause: this recipe pulls two ~32 GB images, one per rank, and the rank that" \ - "comes up first waits out the whole timeout while the other is still pulling." >&2; exit 1; } -fi - -case "$TILERT_ROLE" in - decode) - echo "${host_name}:${host_ip} is the TileRT Decode Node (Model: ${MODEL_NAME}, profile: ${TILERT_PROFILE})" - rdma_preflight || exit 1 - convert_weights || exit 1 - stage_tokenizer_files || exit 1 - start_decode - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: decode role complete"; exit 0 - fi - echo "Waiting for the router port ${PREFILL_HOST}:${ROUTER_PORT} to open (timeout ${ROUTER_WAIT}s)..." - wait_for_tcp "$PREFILL_HOST" "$ROUTER_PORT" "$ROUTER_WAIT" "$DECODE_PID"; wrc=$? - if [[ $wrc -eq 2 ]]; then - echo "ERROR: decode_server exited before the router came up (see $LOG_DIR/decode_${host_name}.log)" >&2 - tail -50 "$LOG_DIR/decode_${host_name}.log" >&2 || true - copy_logs_to_shared; exit 1 - elif [[ $wrc -ne 0 ]]; then - echo "WARNING: router never opened within ${ROUTER_WAIT}s; shutting down decode" >&2 - kill "$DECODE_PID" 2>/dev/null || true - copy_logs_to_shared; exit 1 - fi - echo "Waiting until the router port closes..." - wait_for_tcp_close "$PREFILL_HOST" "$ROUTER_PORT" "$DECODE_PID"; wrc=$? - if [[ $wrc -eq 2 ]]; then - echo "ERROR: decode_server died while the benchmark was running (see $LOG_DIR/decode_${host_name}.log)" >&2 - tail -50 "$LOG_DIR/decode_${host_name}.log" >&2 || true - copy_logs_to_shared; exit 1 - fi - echo "Killing the decode server" - kill "$DECODE_PID" 2>/dev/null || true - sleep 2 - copy_logs_to_shared - ;; - prefill) - echo "NODE INFO =======================================" - echo "Node List : ${SLURM_JOB_NODELIST:-}" - echo "Node IPs : ${IPADDRS}" - echo "Model : ${MODEL_NAME}" - echo "${host_name}:${host_ip} is the Prefill Node (vLLM + TileRTConnector) and Router Node" - echo "================================================" - rdma_preflight || exit 1 - echo "Waiting for the decode ctrl port ${DECODE_HOST}:${DECODE_CTRL_PORT} (timeout ${DECODE_WAIT}s)..." - if [[ "$DRY_RUN" -eq 0 ]]; then - wait_for_tcp "$DECODE_HOST" "$DECODE_CTRL_PORT" "$DECODE_WAIT" \ - || echo "WARNING: timed out waiting for the decode ctrl port; starting prefill anyway" >&2 - fi - start_prefill - if [[ "$DRY_RUN" -eq 0 ]]; then - wait_for_tcp "$PREFILL_HOST" "$PREFILL_PORT" "$PREFILL_WAIT" "$PREFILL_PID"; wrc=$? - if [[ $wrc -ne 0 ]]; then - echo "ERROR: vLLM prefill did not open ${PREFILL_HOST}:${PREFILL_PORT} (rc=$wrc, see $LOG_DIR/prefill_${host_name}.log)" >&2 - tail -50 "$LOG_DIR/prefill_${host_name}.log" >&2 || true - kill "$PREFILL_PID" 2>/dev/null || true - copy_logs_to_shared; exit 1 - fi - fi - start_router - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: prefill/router role complete"; exit 0 - fi - echo "Ready for benchmarking on ${host_name}:${host_ip}" - cd "$WS_PATH" || exit 1 - # EVAL_ONLY skips the AgentX replay and runs GSM8K on the same router; - # RUN_EVAL after a replay runs it once the replay has finished. - if [[ "$EVAL_ONLY" == "true" ]]; then - echo "EVAL_ONLY mode: skipping the AgentX replay" - wait_for_server_ready --port "$ROUTER_PORT" --server-log "$LOG_DIR/router_${host_name}.log" --server-pid "$ROUTER_PID" - export TRANSFORMERS_VERBOSITY=error TOKENIZERS_PARALLELISM=false - run_lm_eval_on_router; BENCH_RC=$? - else - run_agentic_replay; BENCH_RC=$? - if [[ "$RUN_EVAL" == "true" ]]; then - run_lm_eval_on_router || BENCH_RC=1 - fi - fi - copy_logs_to_shared - echo "Killing the router and the prefill server" - kill "$ROUTER_PID" "$PREFILL_PID" 2>/dev/null || true - sleep 2 - pkill -f "tilert.pd_vllm.pd_router" 2>/dev/null || true - pkill -f "vllm serve" 2>/dev/null || true - if [[ "$BENCH_RC" -ne 0 ]]; then - echo "ERROR: benchmark/eval reported rc=$BENCH_RC" >&2 - exit "$BENCH_RC" - fi - ;; - *) - echo "ERROR: unknown TILERT_ROLE='$TILERT_ROLE'" >&2; exit 2 ;; -esac - -echo "Script completed successfully" -exit 0 diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/setup_deps.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/setup_deps.sh deleted file mode 100644 index 1a5c5cfacf..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/setup_deps.sh +++ /dev/null @@ -1,136 +0,0 @@ -#!/bin/bash - -source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only -# Install missing disagg dependencies at container start. Each installer is -# idempotent and gated on $ENGINE. - -_SETUP_START=$(date +%s) -_SETUP_INSTALLED=() - -# Pinned by the recipe (TILERT_VERSION); the rest are fixed properties of the -# TileRT 0.1.x runtime rather than caller configuration. -TILERT_PACKAGE=tilert -TILERT_HTTP_DEPS="fastapi uvicorn httpx" -TILERT_TRANSPORT_DEPS="mooncake-transfer-engine-rocm>=0.3.13" -TILERT_TRANSFORMERS_SPEC="transformers>=4.56" - -_tilert_resolve_python() { - if [[ -n "${PY:-}" ]] && command -v "$PY" >/dev/null 2>&1; then :; else - PY="" - local c - for c in python3 python; do command -v "$c" >/dev/null 2>&1 && { PY="$c"; break; }; done - fi - [[ -n "$PY" ]] || { echo "[SETUP] ERROR: neither python3 nor python found"; exit 1; } - export PY - echo "[SETUP] interpreter PY=$PY ($(command -v "$PY"))" -} - -_tilert_installed_version() { - "$PY" - "$1" <<'PYEOF' 2>/dev/null -import sys -from importlib.metadata import version, PackageNotFoundError -try: - print(version(sys.argv[1])) -except PackageNotFoundError: - pass -PYEOF -} - -_tilert_pip() { - "$PY" -m pip install --quiet --no-cache-dir "$@" -} - -_tilert_install_missing() { - local probe="$1"; shift - [[ $# -gt 0 ]] || return 0 - if "$PY" -c "import $probe" 2>/dev/null; then - echo "[SETUP] $probe already present, skipping ($*)" - return 0 - fi - echo "[SETUP] installing $* (probe module '$probe' missing)" - _tilert_pip "$@" || { echo "[SETUP] ERROR: failed to install: $*"; exit 1; } - _SETUP_INSTALLED+=("$*") -} - -install_tilert_container_tools() { - if command -v ip >/dev/null 2>&1 && command -v curl >/dev/null 2>&1 \ - && command -v ibv_devices >/dev/null 2>&1; then - echo "[SETUP] Container RDMA/net tools already present" - return 0 - fi - echo "[SETUP] Installing iproute2 + curl + ibverbs userspace in container..." - apt-get update -q -y && apt-get install -q -y --no-install-recommends \ - iproute2 curl ibverbs-utils libibverbs1 librdmacm1 ibverbs-providers \ - && rm -rf /var/lib/apt/lists/* - if ! command -v ip >/dev/null 2>&1 || ! command -v curl >/dev/null 2>&1; then - echo "[SETUP] ERROR: failed to install iproute2/curl"; exit 1 - fi - _SETUP_INSTALLED+=("iproute2+curl+ibverbs") -} - -_tilert_install_wheel() { - local mode="$1" # full | no-deps - local have; have="$(_tilert_installed_version tilert)" - if [[ "$have" == "$TILERT_VERSION" ]]; then - echo "[SETUP] tilert $have already installed, skipping" - return 0 - fi - [[ -n "$have" ]] && echo "[SETUP] tilert $have installed, switching to pinned $TILERT_VERSION" - if [[ "$mode" == "no-deps" ]]; then - echo "[SETUP] installing $TILERT_PIP_SPEC --no-deps (connector plugin + router on top of the image's vLLM)" - _tilert_pip --no-deps "$TILERT_PIP_SPEC" || { echo "[SETUP] ERROR: failed to install $TILERT_PIP_SPEC (--no-deps)"; exit 1; } - else - echo "[SETUP] installing $TILERT_PIP_SPEC (TileRT ROCm build, official PyPI wheel)" - _tilert_pip "$TILERT_PIP_SPEC" || { echo "[SETUP] ERROR: failed to install $TILERT_PIP_SPEC"; exit 1; } - fi - have="$(_tilert_installed_version tilert)" - [[ "$have" == "$TILERT_VERSION" ]] || { - echo "[SETUP] ERROR: tilert is ${have:-not installed} after install, expected $TILERT_VERSION"; exit 1; } - _SETUP_INSTALLED+=("$TILERT_PACKAGE==$TILERT_VERSION($mode)") -} - -install_tilert_decode() { - install_tilert_container_tools - _tilert_install_wheel full - _tilert_install_missing uvicorn $TILERT_HTTP_DEPS - _tilert_install_missing mooncake.engine "$TILERT_TRANSPORT_DEPS" - _tilert_install_missing transformers "$TILERT_TRANSFORMERS_SPEC" - "$PY" -c "import tilert.pd_vllm.decode_server" 2>/dev/null || { - echo "[SETUP] ERROR: import tilert.pd_vllm.decode_server failed:" - "$PY" -c "import tilert.pd_vllm.decode_server" 2>&1 | tail -3 - exit 1; } - echo "[SETUP] tilert.pd_vllm.decode_server imports OK" -} - -install_tilert_prefill() { - local vllm_v; vllm_v="$(_tilert_installed_version vllm)" - if [[ -z "$vllm_v" ]]; then - echo "[SETUP] ERROR: no vLLM in the prefill image (PREFILL_IMAGE must be a vllm/vllm-openai-rocm image)." - exit 1 - fi - echo "[SETUP] prefill-side vLLM $vllm_v" - install_tilert_container_tools - _tilert_install_wheel no-deps - _tilert_install_missing mooncake.engine "$TILERT_TRANSPORT_DEPS" - "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>/dev/null || { - echo "[SETUP] WARN: import tilert.pd_vllm.prefill_connector failed (vLLM will report again when loading the connector plugin):" - "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>&1 | tail -3; } -} - -if [[ "$ENGINE" == "tilert" ]]; then - check_env_vars TILERT_VERSION - TILERT_PIP_SPEC="$TILERT_PACKAGE==$TILERT_VERSION" - _tilert_resolve_python - case "${TILERT_ROLE:-}" in - decode) install_tilert_decode ;; - prefill) install_tilert_prefill ;; - *) echo "[SETUP] ERROR: ENGINE=tilert needs TILERT_ROLE=decode|prefill (got '${TILERT_ROLE:-}')"; exit 1 ;; - esac -fi - -_SETUP_END=$(date +%s) -if [[ ${#_SETUP_INSTALLED[@]} -eq 0 ]]; then - echo "[SETUP] All dependencies already present ($(( _SETUP_END - _SETUP_START ))s wallclock)" -else - echo "[SETUP] Installed: ${_SETUP_INSTALLED[*]} in $(( _SETUP_END - _SETUP_START ))s" -fi diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh new file mode 100644 index 0000000000..1c636f8229 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh @@ -0,0 +1,68 @@ +#!/usr/bin/env bash +set -eo pipefail + +source /infmax-workspace/benchmarks/benchmark_lib.sh --validation-only +check_env_vars TILERT_VERSION TILERT_ROLE +case "$TILERT_ROLE" in + prefill|decode|router) ;; + *) echo "Unknown TileRT role: $TILERT_ROLE" >&2; exit 1 ;; +esac + +install_args=(--quiet --no-cache-dir) +if [[ "$TILERT_ROLE" == prefill ]]; then + # Preserve the prefill image's vLLM/Torch dependency set. + install_args+=(--no-deps) +fi +python3 -m pip install "${install_args[@]}" "tilert==$TILERT_VERSION" + +if ! python3 -c 'import mooncake.engine' >/dev/null 2>&1; then + python3 -m pip install --quiet --no-cache-dir 'mooncake-transfer-engine-rocm>=0.3.13' +fi +if [[ "$TILERT_ROLE" != prefill ]]; then + if ! python3 -c 'import uvicorn' >/dev/null 2>&1; then + python3 -m pip install --quiet --no-cache-dir fastapi uvicorn httpx + fi + if ! python3 -c 'import transformers' >/dev/null 2>&1; then + python3 -m pip install --quiet --no-cache-dir 'transformers>=4.56' + fi +fi + +if [[ "$TILERT_ROLE" != decode ]]; then + exit 0 +fi +check_env_vars TILERT_WEIGHTS_DIR TILERT_CONVERT_LOCK_WAIT + +weights_ready() { + local rank + for rank in {0..7}; do + [[ -f "$TILERT_WEIGHTS_DIR/rank$rank/model.safetensors.index.json" ]] || return 1 + done + [[ -f "$TILERT_WEIGHTS_DIR/tilert_meta.json" ]] || return 1 + python3 - "$TILERT_WEIGHTS_DIR/tilert_meta.json" <<'PY' +import json +import sys + +with open(sys.argv[1]) as source: + metadata = json.load(source) +sys.exit(0 if int(metadata.get("num_mtp", 0)) >= 1 else 1) +PY +} + +mkdir -p "$TILERT_WEIGHTS_DIR" +exec 9>"$TILERT_WEIGHTS_DIR/.convert.lock" +flock -w "$TILERT_CONVERT_LOCK_WAIT" 9 +if ! weights_ready; then + python3 -m tilert.models.glm_5_2_rocm.weight_converter \ + --model_dir /model --save_dir "$TILERT_WEIGHTS_DIR" --device cuda:7 --num_mtp 3 + weights_ready +fi + +for file in /model/*; do + [[ -f "$file" ]] || continue + name="${file##*/}" + [[ "$name" == *.safetensors || "$name" == model.safetensors.index.json ]] && continue + [[ -e "$TILERT_WEIGHTS_DIR/$name" ]] || cp -p "$file" "$TILERT_WEIGHTS_DIR/$name" +done +[[ -f "$TILERT_WEIGHTS_DIR/chat_template.jinja" ]] +[[ -f "$TILERT_WEIGHTS_DIR/tokenizer_config.json" || -f "$TILERT_WEIGHTS_DIR/tokenizer.json" ]] +echo "TileRT TP8/MTP weights ready: $TILERT_WEIGHTS_DIR" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml new file mode 100644 index 0000000000..b9215a3759 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml @@ -0,0 +1,103 @@ +schema: 2 +name: glm5.3-mi355x-tilert-agentx +model: + path: GLM-5.3 + container: ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 + precision: fp8 +slurm: + time_limit: "08:00:00" +resources: + gpu_type: mi355x + gpus_per_node: 8 +setup_script: glm5.3-tilert-rocm.sh +environment: + TILERT_VERSION: "0.1.6.post2" + RDMAV_FORK_SAFE: "1" + PYTHONDONTWRITEBYTECODE: "1" + NCCL_SOCKET_IFNAME: eno0 + GLOO_SOCKET_IFNAME: eno0 + NCCL_IB_HCA: rdma0,rdma1,rdma2,rdma3,rdma4,rdma5,rdma6,rdma7 +roles: + prefill: + engine: + type: vllm + set_visible_devices: true + container: ghcr.io/tile-ai/tilert-rocm-prefill:0.1.6 + nodes: 1 + workers: 1 + gpus: 8 + env: + TILERT_ROLE: prefill + VLLM_ROCM_USE_AITER: "1" + VLLM_ROCM_USE_AITER_MOE: "1" + VLLM_ROCM_USE_AITER_FUSION_SHARED_EXPERTS: "1" + VLLM_ENGINE_READY_TIMEOUT_S: "10800" + args: + served-model-name: [glm5_2, GLM-5.3] + tensor-parallel-size: 8 + max-model-len: 1048576 + enforce-eager: true + trust-remote-code: true + return-tokens-as-token-ids: true + gpu-memory-utilization: 0.85 + kv-cache-dtype: bfloat16 + block-size: 64 + speculative-config: '{"method":"mtp","num_speculative_tokens":1}' + kv-transfer-config: >- + {"kv_connector":"TileRTConnector","kv_connector_module_path":"tilert.pd_vllm.prefill_connector", + "kv_role":"kv_producer","kv_connector_extra_config":{ + "tilert_model":"glm5_2","tilert_max_seq_len":1048576,"tilert_transport":"mooncake"}} + decode: + engine: + type: tilert + served_model_name: glm5_2 + container: ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 + nodes: 1 + workers: 1 + gpus: 8 + env: + TILERT_ROLE: decode + TILERT_WEIGHTS_DIR: /models/GLM-5.3-tilert-tp8 + TILERT_CONVERT_LOCK_WAIT: "21600" + GLM5_AR_N: "2" + args: + model: glm5_2 + model-weights-dir: /models/GLM-5.3-tilert-tp8 + max-seq-len: 1048576 + kv-cache-dtype: bf16 + transport: mooncake + with-mtp: true + num-mtp: 3 +frontend: + type: tilert-router + container_image: ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 + enable_multiple_frontends: false + env: + TILERT_ROLE: router + args: + parser: none + model-path: /model + queue-timeout: 1800 +sbatch_directives: + cpus-per-task: "128" + mem: "0" +srun_options: + mem: "0" + container-writable: "" + container-remap-root: "" +health_check: + max_attempts: 2160 + interval_seconds: 5 +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + CONC: "1" + SERVED_MODEL_NAME: glm5_2 + RESULT_DIR: /infmax-workspace/LOGS/agentic + AGENTIC_OUTPUT_DIR: /infmax-workspace + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" + HF_HOME: /logs/hf_cache + TRANSFORMERS_VERBOSITY: error + TOKENIZERS_PARALLELISM: "false" diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index 74c709949e..71cadbc3af 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -1406,21 +1406,8 @@ dsv41flash-fp4-mi355x-vllm-agentic-dspark: - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml } - { tp: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml } -# Speculative decoding on an agentic scenario must run with simulated -# synthetic acceptance at the committed golden AL for this model, thinking mode -# and draft length (docs/PR_REVIEW_CHECKLIST.md), and a submission may not -# substitute its own target. Nothing about that target is written here: -# AGENTS.md forbids hard-coding an acceptance length in a master config, so -# server_tilert.sh reads infx/golden_al_distribution/glm5.3_mtp.yaml at launch and -# fails the run if the curve or the draft length is missing. -# The curve is consumed in the same units as every other framework here, and -# AgentX replays run with thinking on. -# -# NOTE FOR REVIEWERS: that curve is GLM-5.2's, copied because no SPEED-Bench run -# on GLM-5.3 exists and the two share a base. It is committed as provisional and -# labelled as such in the file. Flagging it rather than letting it read as a -# measured 5.3 curve -- please say if you would rather see a measured curve, a -# non-MTP agentic entry, or a waiver. +# The golden acceptance curve remains the provisional GLM-5.2-derived curve +# documented in infx/golden_al_distribution/glm5.3_mtp.yaml. glm5.3-fp8-mi355x-tilert-agentic: image: ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 model: zai-org/GLM-5.3 @@ -1443,16 +1430,12 @@ glm5.3-fp8-mi355x-tilert-agentic: ep: 1 dp-attn: false additional-settings: - - "PREFILL_IMAGE=ghcr.io/tile-ai/tilert-rocm-prefill:0.1.6" - - "PREFILL_NODES=1" + - "CONFIG_FILE=recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml" decode: num-worker: 1 tp: 8 ep: 1 dp-attn: false - additional-settings: - - "DECODE_NODES=1" - - "TILERT_EXTRA_ENV=GLM5_AR_N=2" # User-selected official DeepSeek V4.1 MI35X model preview, pinned by digest. # Official MI350X TP4/EP4 candidate; compare against the separate TP4/EP1 vLLM baseline. diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index a0ecd2fc71..4ea1b57c50 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9003,3 +9003,9 @@ description: - "Update B300 vLLM AgentX to DSpark6 on a new image with a sampled concurrency grid and per-mode --kv-cache-memory-bytes pins." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3477 + +- config-keys: + - glm5.3-fp8-mi355x-tilert-agentic + description: + - "Port MI355X GLM-5.3 TileRT AgentX to native srt-slurm per-role engines." + pr-link: TBD diff --git a/inferencex-e2e/runners/launch_mi355x-amds.sh b/inferencex-e2e/runners/launch_mi355x-amds.sh index 7f313f8fb9..4e3bf2222e 100644 --- a/inferencex-e2e/runners/launch_mi355x-amds.sh +++ b/inferencex-e2e/runners/launch_mi355x-amds.sh @@ -56,11 +56,15 @@ if [[ "$EXECUTION_PATH" == multinode && -n "${CONFIG_FILE:-}" ]]; then # Reuse a provisioned image when one exists; otherwise Pyxis imports it. SQUASH_FILE="$SRT_SHARED_BASE/containers/$(printf '%s' "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" [[ -f "$SQUASH_FILE" ]] || SQUASH_FILE="$IMAGE" + SRT_CLUSTER_ARGS=() + if [[ "$FRAMEWORK" == tilert ]]; then + SRT_CLUSTER_ARGS+=(--model GLM-5.3 /it-share/data/GLM-5.3 --mount /it-share/data /models) + fi SLURM_ACCOUNT="$USER" SLURM_PARTITION=compute NGINX_SQUASH_FILE=nginx:1.27.4 \ write_srt_cluster_config mi355x-amds srtslurm.yaml 0 \ --var SRT_DEFAULT_TIME_LIMIT 01:00:00 --var GITHUB_WORKSPACE "$GITHUB_WORKSPACE" \ --container "$IMAGE" "$SQUASH_FILE" \ - --mount /it-share/aiperf-cache /aiperf_mmap_cache || exit 1 + --mount /it-share/aiperf-cache /aiperf_mmap_cache "${SRT_CLUSTER_ARGS[@]}" || exit 1 make setup ARCH=x86_64 export INFMAX_WORKSPACE="$GITHUB_WORKSPACE" From e67168459a2073e1c3ba19f8d51050e7f9c66eb0 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 15:27:34 -0500 Subject: [PATCH 2/7] chore: link port validation PR --- inferencex-e2e/perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 4ea1b57c50..694509bf90 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9008,4 +9008,4 @@ - glm5.3-fp8-mi355x-tilert-agentic description: - "Port MI355X GLM-5.3 TileRT AgentX to native srt-slurm per-role engines." - pr-link: TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3552 From d8c13305757103d141409801abff05a2591e4773 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 15:44:08 -0500 Subject: [PATCH 3/7] fix(tilert): reuse prepared AMD weights without a write lock --- .../configs/glm5.3-tilert-rocm.sh | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh index 1c636f8229..fe58737801 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh @@ -48,13 +48,16 @@ sys.exit(0 if int(metadata.get("num_mtp", 0)) >= 1 else 1) PY } -mkdir -p "$TILERT_WEIGHTS_DIR" -exec 9>"$TILERT_WEIGHTS_DIR/.convert.lock" -flock -w "$TILERT_CONVERT_LOCK_WAIT" 9 if ! weights_ready; then - python3 -m tilert.models.glm_5_2_rocm.weight_converter \ - --model_dir /model --save_dir "$TILERT_WEIGHTS_DIR" --device cuda:7 --num_mtp 3 - weights_ready + mkdir -p "$TILERT_WEIGHTS_DIR" + exec 9>"$TILERT_WEIGHTS_DIR/.convert.lock" + flock -w "$TILERT_CONVERT_LOCK_WAIT" 9 + if ! weights_ready; then + python3 -m tilert.models.glm_5_2_rocm.weight_converter \ + --model_dir /model --save_dir "$TILERT_WEIGHTS_DIR" --device cuda:7 --num_mtp 3 + weights_ready + fi + exec 9>&- fi for file in /model/*; do From c2f90607a33d4a5f2483c3a5b6feffe5a2f8ab06 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 15:52:10 -0500 Subject: [PATCH 4/7] refactor(tilert): keep AMD setup limited to installation --- .../configs/glm5.3-tilert-rocm.sh | 43 ------------------- .../agentx/disagg-1p1d-tp8-mtp.yaml | 2 - inferencex-e2e/runners/launch_mi355x-amds.sh | 4 ++ 3 files changed, 4 insertions(+), 45 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh index fe58737801..e9906ebb2c 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh @@ -26,46 +26,3 @@ if [[ "$TILERT_ROLE" != prefill ]]; then python3 -m pip install --quiet --no-cache-dir 'transformers>=4.56' fi fi - -if [[ "$TILERT_ROLE" != decode ]]; then - exit 0 -fi -check_env_vars TILERT_WEIGHTS_DIR TILERT_CONVERT_LOCK_WAIT - -weights_ready() { - local rank - for rank in {0..7}; do - [[ -f "$TILERT_WEIGHTS_DIR/rank$rank/model.safetensors.index.json" ]] || return 1 - done - [[ -f "$TILERT_WEIGHTS_DIR/tilert_meta.json" ]] || return 1 - python3 - "$TILERT_WEIGHTS_DIR/tilert_meta.json" <<'PY' -import json -import sys - -with open(sys.argv[1]) as source: - metadata = json.load(source) -sys.exit(0 if int(metadata.get("num_mtp", 0)) >= 1 else 1) -PY -} - -if ! weights_ready; then - mkdir -p "$TILERT_WEIGHTS_DIR" - exec 9>"$TILERT_WEIGHTS_DIR/.convert.lock" - flock -w "$TILERT_CONVERT_LOCK_WAIT" 9 - if ! weights_ready; then - python3 -m tilert.models.glm_5_2_rocm.weight_converter \ - --model_dir /model --save_dir "$TILERT_WEIGHTS_DIR" --device cuda:7 --num_mtp 3 - weights_ready - fi - exec 9>&- -fi - -for file in /model/*; do - [[ -f "$file" ]] || continue - name="${file##*/}" - [[ "$name" == *.safetensors || "$name" == model.safetensors.index.json ]] && continue - [[ -e "$TILERT_WEIGHTS_DIR/$name" ]] || cp -p "$file" "$TILERT_WEIGHTS_DIR/$name" -done -[[ -f "$TILERT_WEIGHTS_DIR/chat_template.jinja" ]] -[[ -f "$TILERT_WEIGHTS_DIR/tokenizer_config.json" || -f "$TILERT_WEIGHTS_DIR/tokenizer.json" ]] -echo "TileRT TP8/MTP weights ready: $TILERT_WEIGHTS_DIR" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml index b9215a3759..68b8a81220 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml @@ -57,8 +57,6 @@ roles: gpus: 8 env: TILERT_ROLE: decode - TILERT_WEIGHTS_DIR: /models/GLM-5.3-tilert-tp8 - TILERT_CONVERT_LOCK_WAIT: "21600" GLM5_AR_N: "2" args: model: glm5_2 diff --git a/inferencex-e2e/runners/launch_mi355x-amds.sh b/inferencex-e2e/runners/launch_mi355x-amds.sh index 4e3bf2222e..efd0383db0 100644 --- a/inferencex-e2e/runners/launch_mi355x-amds.sh +++ b/inferencex-e2e/runners/launch_mi355x-amds.sh @@ -58,6 +58,10 @@ if [[ "$EXECUTION_PATH" == multinode && -n "${CONFIG_FILE:-}" ]]; then [[ -f "$SQUASH_FILE" ]] || SQUASH_FILE="$IMAGE" SRT_CLUSTER_ARGS=() if [[ "$FRAMEWORK" == tilert ]]; then + if [[ ! -r /it-share/data/GLM-5.3-tilert-tp8/tilert_meta.json ]]; then + echo "Missing prepared TileRT checkpoint: /it-share/data/GLM-5.3-tilert-tp8" >&2 + exit 1 + fi SRT_CLUSTER_ARGS+=(--model GLM-5.3 /it-share/data/GLM-5.3 --mount /it-share/data /models) fi SLURM_ACCOUNT="$USER" SLURM_PARTITION=compute NGINX_SQUASH_FILE=nginx:1.27.4 \ From 15b341df57e4629ee0773b56f96660532fa9ca3f Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 18:31:39 -0500 Subject: [PATCH 5/7] fix(tilert): allow AgentX client to fetch its dataset --- .../glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml index 68b8a81220..5c30a63054 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml @@ -97,5 +97,6 @@ benchmark: AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" HF_HOME: /logs/hf_cache + HF_HUB_OFFLINE: "0" TRANSFORMERS_VERBOSITY: error TOKENIZERS_PARALLELISM: "false" From 34217c72ebc1869129dbb23bcd99598438065454 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 20:35:16 -0500 Subject: [PATCH 6/7] fix(tilert): clear inherited offline mode for AgentX --- .../glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml index 5c30a63054..4882a1c910 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml @@ -97,6 +97,6 @@ benchmark: AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" HF_HOME: /logs/hf_cache - HF_HUB_OFFLINE: "0" + HF_HUB_OFFLINE: "" TRANSFORMERS_VERBOSITY: error TOKENIZERS_PARALLELISM: "false" From cdb425edba274924da2cc5c4b147c51628625f1d Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 21:28:38 -0500 Subject: [PATCH 7/7] fix(tilert): preserve AgentX tokenizer and context selection --- .../glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml index 4882a1c910..066183afa7 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml @@ -91,7 +91,9 @@ benchmark: command: bash /infmax-workspace/benchmarks/srt_agentic.sh env: CONC: "1" + MODEL: /model SERVED_MODEL_NAME: glm5_2 + AIPERF_MAX_CONTEXT_LENGTH: "1048576" RESULT_DIR: /infmax-workspace/LOGS/agentic AGENTIC_OUTPUT_DIR: /infmax-workspace AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache