diff --git a/MODELS.md b/MODELS.md index d461f1657c..619e3fefc9 100644 --- a/MODELS.md +++ b/MODELS.md @@ -155,6 +155,7 @@ Other offloading tiers, including NVMe KV cache offloading, are outside the init | Model architecture class | Prefix | Date added | Active scenarios | Deprecated scenarios | |---|---|---|---|---| | DeepSeek-V4.1-Flash | `dsv41flash` | 2026-09-10 | Agentic coding (vLLM: DSpark, Engram UVA offload; SGLang: DSpark arms added per SKU from 2026-09-17; GPU validation pending) | | +| GLM-5.3 | `glm5.3` | 2026-09-22 ([#3366](https://github.com/SemiAnalysisAI/InferenceX/pull/3366)) | Agentic coding (MTP only, per the Deprecation Notice) | Single-turn 1k1k (deprecated for all models before this model was added; never run) | | Qwen3.8-Flash-Next | `qwen3.8next` | 2026-08-26 ([#2742](https://github.com/SemiAnalysisAI/InferenceX/pull/2742)) | Agentic coding | | | Kimi-K3 | `kimik3` | 2026-07-27 ([#2391](https://github.com/SemiAnalysisAI/InferenceX/pull/2391)) | Agentic coding (DSpark may be disabled for better Pareto points) | Standalone non-DSpark A/B baseline (not required from day 0) | | GLM-5.2 | `glm5.2` | 2026-07-18 ([#2268](https://github.com/SemiAnalysisAI/InferenceX/pull/2268)) | Agentic coding (non-MTP points remain eligible under the Pareto policy; see Deprecation Notice) | | diff --git a/MODELS_zh.md b/MODELS_zh.md index dd52db9a2d..da8c72b08f 100644 --- a/MODELS_zh.md +++ b/MODELS_zh.md @@ -155,6 +155,7 @@ InferenceX 支持 SGLang 和 vLLM 双方的维护者,并响应 AI 实验室和 | 模型架构类别 | 前缀 | 加入日期 | 启用场景 | 已弃用场景 | |---|---|---|---|---| | DeepSeek-V4.1-Flash | `dsv41flash` | 2026-09-10 | 智能体编码(DSpark、Engram UVA 卸载;GPU 待验证) | | +| GLM-5.3 | `glm5.3` | 2026-09-22([#3366](https://github.com/SemiAnalysisAI/InferenceX/pull/3366)) | 智能体编码(仅 MTP,见弃用公告) | 单轮 1k1k(在本模型加入前已对所有模型弃用,从未运行) | | Qwen3.8-Flash-Next | `qwen3.8next` | 2026-08-26([#2742](https://github.com/SemiAnalysisAI/InferenceX/pull/2742)) | 智能体编码 | | | Kimi-K3 | `kimik3` | 2026-07-27 ([#2391](https://github.com/SemiAnalysisAI/InferenceX/pull/2391)) | 智能体编码(可关闭 DSpark 以获得更优帕累托点) | 独立非 DSpark A/B 基线(自第 0 天起即不要求) | | GLM-5.2 | `glm5.2` | 2026-07-18([#2268](https://github.com/SemiAnalysisAI/InferenceX/pull/2268)) | 智能体编码(非 MTP 数据点仍可按帕累托策略参与发布;见弃用公告) | | diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index 60bdfecead..e5765862ed 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -3162,7 +3162,7 @@ resolve_trace_source() { # corpus; 1M-context families take the unfiltered variant, others 256k. local default_loader case "${MODEL_PREFIX:-}" in - dsv4*|glm5.2*|minimaxm3*|kimik3*) + dsv4*|glm5.2*|glm5.3*|minimaxm3*|kimik3*) default_loader="semianalysis_cc_traces_weka_062126" ;; *) diff --git a/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh b/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh new file mode 100644 index 0000000000..a8f6c3ac9f --- /dev/null +++ b/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh @@ -0,0 +1,172 @@ +#!/usr/bin/env bash + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +source "$SCRIPT_DIR/../../benchmark_lib.sh" + +check_env_vars \ + CONC_LIST \ + ISL \ + OSL \ + IMAGE \ + SPEC_DECODING \ + MODEL_PATH \ + MODEL_NAME \ + PREFILL_NUM_WORKERS \ + PREFILL_TP \ + PREFILL_EP \ + PREFILL_DP_ATTN \ + DECODE_NUM_WORKERS \ + DECODE_TP \ + DECODE_EP \ + DECODE_DP_ATTN \ + PREFILL_NODES \ + DECODE_NODES \ + RANDOM_RANGE_RATIO \ + DURATION \ + MODEL_PREFIX \ + PRECISION \ + RESULT_FILENAME \ + KV_OFFLOADING \ + IS_AGENTIC \ + FRAMEWORK \ + PREFILL_IMAGE + +if [[ -n "$SLURM_JOB_ID" ]]; then + echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" +fi + +set -x + +cd "$GITHUB_WORKSPACE/benchmarks/multi_node/amd_utils" || exit 1 + +export TIME_LIMIT=08:00:00 +export MODEL_PATH=$MODEL_PATH +export MODEL_NAME=$MODEL_NAME +export CONTAINER_IMAGE=$IMAGE +export PREFILL_IMAGE + +export RESULT_FILENAME + +if [[ "$PREFILL_NODES" -ne 1 || "$DECODE_NODES" -ne 1 || \ + "$PREFILL_NUM_WORKERS" -ne 1 || "$DECODE_NUM_WORKERS" -ne 1 ]]; then + echo "Error: tilert supports exactly 1 prefill node/worker + 1 decode node/worker" \ + "(got PREFILL_NODES=$PREFILL_NODES x$PREFILL_NUM_WORKERS, DECODE_NODES=$DECODE_NODES x$DECODE_NUM_WORKERS)" >&2 + exit 1 +fi + +if [[ "$KV_OFFLOADING" != "none" ]]; then + echo "Error: tilert has no KV offload backend; kv-offloading must be 'none' (got '$KV_OFFLOADING')" >&2 + exit 1 +fi + +# TileRT configuration. Every value is explicit here: server_tilert.sh +# validates each one with check_env_vars and supplies no defaults of its own. +export TILERT_VERSION=0.1.6.post1 +export TILERT_PROFILE=glm5_2 # decode_server --model (TileRT model profile) +export TILERT_MODEL_TYPE=glm-5 # weight_converter --model_type (fallback converter) +export TILERT_MODEL_PKG=glm_5_2_rocm # per-model converter package, preferred when importable +export SERVED_MODEL_NAME=glm5_2 +# GLM-5.3's full context window (config.json max_position_embeddings), as every +# in-tree GLM-5.2 recipe serves. (202752 was GLM-5.1's, inherited from the B200 +# TileRT recipe this mirrors.) +# +# Memory at this context, per rank, bf16 wire layout (verified against the +# tilert 0.1.6 and vLLM 0.24.0 sources and the MI355X logs, 287.98 GiB cards; +# the undivided PD buffer sizes are what 0.1.6 allocated on one card): +# decode : weights 90.72 GiB + engine cache window 93.25 GiB +# + PD receive buffer 99.06 GiB (receive_server.py, dense in max_seq_len) +# prefill: weights 90.45 GiB + profiling/non-torch 40.3 GiB + vLLM KV 91.71 GiB +# + PD staging buffer 99.06 GiB (prefill_connector.py, TP rank 0, +# allocated OUTSIDE vLLM's gpu-memory-utilization budget) +# Undivided, neither side starts: the decode rank is node-marginal (~283 of +# 288 GiB) and the prefill rank cannot fit at any utilization (~321 GiB). +# tilert 0.1.6.post1 keeps both buffers on the GPU but shards them by layer +# across the eight devices (layer lid on device lid % 8, TILERT_PD_SHARDS, +# default on), so each card holds 12.54 GiB instead of 99.06 GiB on one. +# convert() dequantises each layer on the device that received it, which spreads +# its transients too (108.6 KiB/token, 82.9 GiB at 800k tokens) instead of +# leaving them on cuda:0. Measured on 2x8 MI350X at this context with bf16 KV: +# decode peaks at 202.1 GiB per card, prefill at 269.2 GiB of 287.69 GiB, and +# the KV path stays device-to-device at 108 GB/s (81 GB in 751 ms, 54% of the +# 4x400 GbE line rate). No host hop and no patch: the wheel runs as shipped. +export TILERT_MAX_MODEL_LEN=1048576 +export TILERT_TRANSPORT=mooncake +export TILERT_PARSER=none +export TILERT_RDMA_STRICT=0 +export TILERT_CONVERT_LOCK_WAIT=21600 +export TILERT_SIMULATE_ACC_METHOD=match-expected +export TILERT_WEIGHTS_DIR="/models/${MODEL_NAME}-tilert-tp${DECODE_TP}" +# bf16 MLA KV on both roles. This is the only layout TileRT 0.1.6 can consume +# from vLLM on ROCm: MlaNsaProfile.classify_layers infers the layout from the +# cache tensor stride and accepts exactly 1152 B/token (bf16) or 656 B/token +# (fp8_ds_mla). vLLM's ROCM_AITER_MLA_SPARSE backend has no fp8_ds_mla; its +# plain "fp8" writes a flat 576 B/token row, which the connector rejects at +# register_kv_caches. Explicit bfloat16 rather than auto so the stride does not +# depend on the model dtype. Never float16: it passes the 1152 B check and is +# then read as bf16. +export PREFILL_KV_DTYPE=bfloat16 +# The ROCm backend supports block sizes [1, 64] and vLLM picks 1, which makes +# the connector's KI plane copy fail and MLA address the wrong rows. +export PREFILL_BLOCK_SIZE=64 +export DECODE_KV_DTYPE=bf16 +# The PD staging shard sits outside vLLM's budget, so vLLM needs 90.45 (weights) +# + 40.3 (profiling) + 91.71 GiB (KV for one 1048576-token request) = 222.5 GiB +# inside it: 0.85 x 287.98 = 244.8 GiB leaves 22 GiB of KV margin and 43 GiB +# outside the budget for the 12.54 GiB staging shard plus the ~6.3 GiB non-torch +# baseline measured on the decode OOM node (287.98 - 95.94 free - 184.17 - 1.58 +# reserved). 0.75 (216 GiB) refuses with "91.71 GiB KV cache is needed ... +# available 85.25 GiB". +export GPU_MEM_UTIL=0.85 +export SKIP_CONTAINER_BARRIER=0 +# Two images, one per rank, ~32 GB each. On a node that has neither cached the +# pull alone outlasts the SGLang path's 300s default and the 1800s this script +# used to hardcode, and the rank that comes up first waits out the whole +# timeout while its peer is still pulling. +export CONTAINER_BARRIER_TIMEOUT=5400 +export ROUTER_PORT=30000 +export PREFILL_PORT=8000 +export DECODE_CTRL_PORT=5556 +export DECODE_HTTP_PORT=5557 +export DECODE_WAIT=7200 # prefill waits for the decode ctrl port +export PREFILL_WAIT=3600 # prefill waits for its own vLLM port +export ROUTER_WAIT=10800 # decode waits for the router port to open + +if [[ "$SPEC_DECODING" == "mtp" ]]; then + # TileRT decode drafts at depth 3 (the only depth the ROCm GLM profile + # builds) and the golden acceptance curve is keyed on it. The vLLM prefill + # rank only has to materialise the MTP layer's KV, so it runs at 1. + export DECODE_MTP_SIZE=3 + export PREFILL_SPEC_TOKENS=1 +else + export DECODE_MTP_SIZE=0 + export PREFILL_SPEC_TOKENS=0 +fi +export TILERT_QUEUE_TIMEOUT=1800 # requests wait on the bs=1 decode engine +export THINKING_MODE=thinking_on + +if [[ "$PREFILL_EP" -ne 1 || "$DECODE_EP" -ne 1 || \ + "$PREFILL_DP_ATTN" == "true" || "$DECODE_DP_ATTN" == "true" ]]; then + echo "Error: tilert runs pure TP8 on both roles; ep must be 1 and dp-attn false" >&2 + exit 1 +fi +export PREFILL_ENABLE_EP=false +export PREFILL_ENABLE_DP=false +export DECODE_ENABLE_EP=false +export DECODE_ENABLE_DP=false + +JOB_ID=$(bash ./submit.sh $PREFILL_NODES \ + $PREFILL_NUM_WORKERS \ + $DECODE_NODES \ + $DECODE_NUM_WORKERS \ + $ISL $OSL "${CONC_LIST// /x}" inf \ + ${PREFILL_ENABLE_EP} ${PREFILL_ENABLE_DP} \ + ${DECODE_ENABLE_EP} ${DECODE_ENABLE_DP} \ + ${PREFILL_TP} ${DECODE_TP} \ + ${RANDOM_RANGE_RATIO}) + +if [[ $? -ne 0 ]]; then + echo "Failed to submit job" >&2 + exit 1 +fi + +echo "$JOB_ID" diff --git a/benchmarks/multi_node/amd_utils/env.sh b/benchmarks/multi_node/amd_utils/env.sh index dbe93fa88d..551a55ac2b 100755 --- a/benchmarks/multi_node/amd_utils/env.sh +++ b/benchmarks/multi_node/amd_utils/env.sh @@ -1,10 +1,18 @@ #!/bin/bash source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only -check_env_vars \ - MORI_IO_SQ_BACKOFF_TIMEOUT_US MORI_IO_QP_MAX_SEND_WR MORI_IO_QP_MAX_CQE MORI_IO_QP_MAX_SGE MORI_IO_TC_DISABLE \ - UCX_IB_GID_INDEX MORI_APP_LOG_LEVEL SGLANG_ROUTER_STDOUT_LOGS TORCH_NCCL_BLOCKING_WAIT NCCL_BLOCKING_WAIT \ - SGLANG_OPT_USE_AITER_INDEXER +check_env_vars ENGINE +# MoRI-IO queue-pair tuning, the UCX RoCE GID index, SGLang router logging and the +# SGLang decode cuda-graph NCCL workaround. Only the SGLang and vLLM MoRI KV paths +# below read these. ENGINE=tilert moves KV over mooncake and starts no SGLang +# router, so it is neither given nor reads them: validating them there would force +# the recipe to invent MoRI tuning for a transport it never uses. +if [[ "$ENGINE" != "tilert" ]]; then + check_env_vars \ + MORI_IO_SQ_BACKOFF_TIMEOUT_US MORI_IO_QP_MAX_SEND_WR MORI_IO_QP_MAX_CQE MORI_IO_QP_MAX_SGE MORI_IO_TC_DISABLE \ + UCX_IB_GID_INDEX MORI_APP_LOG_LEVEL SGLANG_ROUTER_STDOUT_LOGS TORCH_NCCL_BLOCKING_WAIT NCCL_BLOCKING_WAIT \ + SGLANG_OPT_USE_AITER_INDEXER +fi # Dual-engine environment setup for multi-node disaggregated serving. # # ENGINE=sglang-disagg or vllm-disagg selects the engine-specific block. @@ -180,6 +188,9 @@ $1 == "DSCP" && $2 == ":" && $NF == p { set +x echo "[INFO] IBDEVICES=$IBDEVICES UCX_NET_DEVICES=$UCX_NET_DEVICES NCCL_SOCKET_IFNAME=$NCCL_SOCKET_IFNAME UCX_IB_GID_INDEX=$UCX_IB_GID_INDEX UCX_IB_TRAFFIC_CLASS=${UCX_IB_TRAFFIC_CLASS:-unset}" +elif [[ "$ENGINE" == "tilert" ]]; then + echo "[INFO] tilert: IBDEVICES=$IBDEVICES NCCL_SOCKET_IFNAME=$NCCL_SOCKET_IFNAME NCCL_IB_HCA=$NCCL_IB_HCA" + else export SGLANG_USE_AITER=1 diff --git a/benchmarks/multi_node/amd_utils/job.slurm b/benchmarks/multi_node/amd_utils/job.slurm index d1f5ea11a7..0c779476ea 100755 --- a/benchmarks/multi_node/amd_utils/job.slurm +++ b/benchmarks/multi_node/amd_utils/job.slurm @@ -38,6 +38,8 @@ if [[ "$ENGINE" == "vllm-disagg" ]]; then MODELS_YAML="$(pwd)/models_vllm.yaml" elif [[ "$ENGINE" == "atom-disagg" ]]; then MODELS_YAML="$(pwd)/models_atom.yaml" +elif [[ "$ENGINE" == "tilert" ]]; then + MODELS_YAML="$(pwd)/models_tilert.yaml" else MODELS_YAML="$(pwd)/models.yaml" fi @@ -52,6 +54,17 @@ if [[ -z "${DOCKER_IMAGE_NAME:-}" ]]; then exit 1 fi +if [[ "$ENGINE" == "tilert" && -z "${PREFILL_IMAGE:-}" ]]; then + echo "Error: ENGINE=tilert requires PREFILL_IMAGE (e.g. PREFILL_IMAGE=vllm/vllm-openai-rocm:nightly- in prefill.additional-settings)." + exit 1 +fi +if [[ "$ENGINE" == "tilert" ]]; then + # server_tilert.sh takes the container-creation barrier timeout from the + # recipe (no 300s default as on the SGLang path); fail here, before sbatch + # work is done, rather than inside the container. + check_env_vars CONTAINER_BARRIER_TIMEOUT +fi + # Resolve the models.yaml entry the same way server_sglang.sh does: agentic runs # (IS_AGENTIC) use the '-AgentX' recipe, non-agentic disaggregated runs use # '-DI'. Fall back to the bare model name if the variant key is absent. @@ -522,6 +535,41 @@ elif [[ "$ENGINE" == "atom-disagg" ]]; then -e EXTRA_SERVER_ARGS=\${EXTRA_SERVER_ARGS:-} -e IBDEVICES=${IBDEVICES:-} ) +elif [[ "$ENGINE" == "tilert" ]]; then + DOCKER_ENV_ENGINE=( + -e MODEL_PATH=$DOCKER_MODEL_PATH + -e PREFILL_IMAGE=${PREFILL_IMAGE} + -e TILERT_VERSION=${TILERT_VERSION} + -e TILERT_PROFILE=${TILERT_PROFILE} + -e TILERT_MODEL_TYPE=${TILERT_MODEL_TYPE} + -e TILERT_MODEL_PKG=${TILERT_MODEL_PKG} + -e TILERT_MAX_MODEL_LEN=${TILERT_MAX_MODEL_LEN} + -e TILERT_TRANSPORT=${TILERT_TRANSPORT} + -e TILERT_PARSER=${TILERT_PARSER} + -e TILERT_QUEUE_TIMEOUT=${TILERT_QUEUE_TIMEOUT} + -e TILERT_WEIGHTS_DIR=${TILERT_WEIGHTS_DIR} + -e TILERT_RDMA_STRICT=${TILERT_RDMA_STRICT} + -e TILERT_CONVERT_LOCK_WAIT=${TILERT_CONVERT_LOCK_WAIT} + -e TILERT_SIMULATE_ACC_METHOD=${TILERT_SIMULATE_ACC_METHOD} + -e \"TILERT_EXTRA_ENV=${TILERT_EXTRA_ENV:-}\" + -e SERVED_MODEL_NAME=${SERVED_MODEL_NAME} + -e PREFILL_KV_DTYPE=${PREFILL_KV_DTYPE} + -e PREFILL_BLOCK_SIZE=${PREFILL_BLOCK_SIZE} + -e PREFILL_SPEC_TOKENS=${PREFILL_SPEC_TOKENS} + -e DECODE_KV_DTYPE=${DECODE_KV_DTYPE} + -e GPU_MEM_UTIL=${GPU_MEM_UTIL} + -e PREFILL_PORT=${PREFILL_PORT} + -e DECODE_CTRL_PORT=${DECODE_CTRL_PORT} + -e DECODE_HTTP_PORT=${DECODE_HTTP_PORT} + -e DECODE_WAIT=${DECODE_WAIT} + -e PREFILL_WAIT=${PREFILL_WAIT} + -e ROUTER_WAIT=${ROUTER_WAIT} + -e SKIP_CONTAINER_BARRIER=${SKIP_CONTAINER_BARRIER} + # Golden-acceptance selection on agentic MTP runs; unset elsewhere. + -e THINKING_MODE=${THINKING_MODE:-} + -e IBDEVICES=${IBDEVICES:-} + -e PYTHONPYCACHEPREFIX=/tmp/pycache + ) else DOCKER_ENV_ENGINE=( -e SGLANG_WS_PATH=${WS_PATH} @@ -704,6 +752,12 @@ if [[ \"$ENGINE\" == \"vllm-disagg\" && \"$ROUTER_TYPE\" == \"vllm-router\" && \ set +e fi +RANK_IMAGE= +if [[ \"$ENGINE\" == \"tilert\" && \"\$SLURM_PROCID\" -lt \"$xP\" ]]; then + RANK_IMAGE=\"$PREFILL_IMAGE\" + echo \"[tilert] rank \$SLURM_PROCID is a prefill rank; using PREFILL_IMAGE=\$RANK_IMAGE\" +fi + \$MAYBE_EXEC \$DOCKER_CMD run \ --init \ --stop-timeout 10 \ @@ -745,7 +799,7 @@ fi ${CLIENT_DOCKER_ENV} \ --name \"$DOCKER_CONT_NAME\" \ --entrypoint \"\" \ - \"$DOCKER_IMAGE_NAME\" bash -lc ' + \"\${RANK_IMAGE:-$DOCKER_IMAGE_NAME}\" bash -lc ' set -o pipefail mkdir -p /run_logs/slurm_job-'\"\$SLURM_JOB_ID\"' '"$RUN_FILE_FULL"' 2>&1 | tee /run_logs/slurm_job-'\"\$SLURM_JOB_ID\"'/server_\$(hostname).log diff --git a/benchmarks/multi_node/amd_utils/models_tilert.yaml b/benchmarks/multi_node/amd_utils/models_tilert.yaml new file mode 100644 index 0000000000..1697059d54 --- /dev/null +++ b/benchmarks/multi_node/amd_utils/models_tilert.yaml @@ -0,0 +1,6 @@ +# Model-owned engine settings for ENGINE=tilert. Everything the caller owns +# (profile, topology, ports, dtypes, draft depth) is set by the recipe, not here. +GLM-5.3: + prefill_env: "VLLM_ROCM_USE_AITER=1 VLLM_ROCM_USE_AITER_MOE=1 VLLM_ROCM_USE_AITER_FUSION_SHARED_EXPERTS=1 VLLM_ENGINE_READY_TIMEOUT_S=10800" + prefill_extra_flags: "" + decode_extra_flags: "" diff --git a/benchmarks/multi_node/amd_utils/server.sh b/benchmarks/multi_node/amd_utils/server.sh index 6b65281c35..08cb65d2a9 100755 --- a/benchmarks/multi_node/amd_utils/server.sh +++ b/benchmarks/multi_node/amd_utils/server.sh @@ -6,6 +6,7 @@ source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only # ENGINE=sglang-disagg (default) -> server_sglang.sh (SGLang + MoRI) # ENGINE=vllm-disagg -> server_vllm.sh (vLLM + Nixl/MoRI-IO) # ENGINE=atom-disagg -> server_atom.sh (ATOM + mooncake) +# ENGINE=tilert -> server_tilert.sh (vLLM prefill + TileRT decode) check_env_vars ENGINE WS_PATH if [[ -f /config/hicache_mc.env ]]; then @@ -22,6 +23,8 @@ if [[ "$ENGINE" == "vllm-disagg" ]]; then elif [[ "$ENGINE" == "atom-disagg" ]]; then export ATOM_WS_PATH="$WS_PATH" source "$WS_PATH/server_atom.sh" +elif [[ "$ENGINE" == "tilert" ]]; then + source "$WS_PATH/server_tilert.sh" else source "$WS_PATH/server_sglang.sh" fi diff --git a/benchmarks/multi_node/amd_utils/server_tilert.sh b/benchmarks/multi_node/amd_utils/server_tilert.sh new file mode 100644 index 0000000000..a22cd5c433 --- /dev/null +++ b/benchmarks/multi_node/amd_utils/server_tilert.sh @@ -0,0 +1,574 @@ +#!/bin/bash +# TileRT disaggregated launcher: upstream vLLM ROCm prefill (TileRTConnector, +# kv_producer) + TileRT decode_server + the OpenAI-compatible pd_router. +# Every value below is supplied by the recipe through job.slurm; this script +# validates them and never invents a default for caller-owned configuration. + +source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only + +check_env_vars \ + NODE0_ADDR NODE_RANK MODEL_DIR MODEL_NAME MODEL_PATH xP yD IPADDRS \ + DRY_RUN GPUS_PER_NODE PREFILL_TP_SIZE DECODE_TP_SIZE \ + BENCH_INPUT_LEN BENCH_OUTPUT_LEN BENCH_MAX_CONCURRENCY \ + RUN_EVAL EVAL_ONLY EVAL_FRAMEWORK BENCHMARK_LOGS_DIR WS_PATH \ + SLURM_JOB_ID SPEC_DECODING \ + TILERT_PROFILE TILERT_MODEL_TYPE TILERT_MODEL_PKG TILERT_MAX_MODEL_LEN \ + TILERT_TRANSPORT TILERT_PARSER TILERT_QUEUE_TIMEOUT TILERT_WEIGHTS_DIR \ + TILERT_RDMA_STRICT TILERT_CONVERT_LOCK_WAIT TILERT_SIMULATE_ACC_METHOD \ + PREFILL_KV_DTYPE PREFILL_BLOCK_SIZE PREFILL_SPEC_TOKENS DECODE_KV_DTYPE \ + DECODE_MTP_SIZE GPU_MEM_UTIL SERVED_MODEL_NAME \ + DECODE_CTRL_PORT DECODE_HTTP_PORT PREFILL_PORT ROUTER_PORT \ + DECODE_WAIT PREFILL_WAIT ROUTER_WAIT SKIP_CONTAINER_BARRIER \ + CONTAINER_BARRIER_TIMEOUT + +LOG_DIR="/run_logs/slurm_job-${SLURM_JOB_ID}" +SHARED_LOG_DIR="${BENCHMARK_LOGS_DIR}/logs/slurm_job-${SLURM_JOB_ID}" +mkdir -p "$LOG_DIR" + +if [[ "$xP" -ne 1 || "$yD" -ne 1 ]]; then + echo "ERROR: tilert supports exactly 1 prefill + 1 decode worker (got xP=$xP yD=$yD)" >&2 + exit 1 +fi +if [[ "$NODE_RANK" -lt "$xP" ]]; then + TILERT_ROLE=prefill +else + TILERT_ROLE=decode +fi +export TILERT_ROLE + +source "$WS_PATH/setup_deps.sh" +source "$WS_PATH/env.sh" +# benchmark_lib.sh derives AGENTIC_DIR/AIPERF_DIR from this at source time, so +# it must be set before the library is loaded, not in run_agentic_replay. The +# AgentX replay runs in this container, where the repo is mounted at /workspace. +export INFMAX_CONTAINER_WORKSPACE=/workspace +source /workspace/benchmarks/benchmark_lib.sh + +# Model-specific engine environment (not caller configuration): the prefill +# vLLM env block lives with the model, exactly as models_atom.yaml carries the +# ATOM `env` string. Everything else is passed in by the recipe. +MODELS_YAML="${WS_PATH}/models_tilert.yaml" +eval "$("$PY" - "$MODELS_YAML" "$MODEL_NAME" <<'PYEOF' +import shlex, sys, yaml +path, name = sys.argv[1], sys.argv[2] +with open(path) as f: + models = yaml.safe_load(f) or {} +if name not in models: + sys.exit(f"model '{name}' is not present in {path}") +m = models[name] or {} +for key, var in (("prefill_env", "TILERT_PREFILL_ENV"), + ("prefill_extra_flags", "TILERT_PREFILL_EXTRA_FLAGS"), + ("decode_extra_flags", "TILERT_DECODE_EXTRA_FLAGS")): + print(f"{var}={shlex.quote(str(m.get(key) or ''))}") +PYEOF +)" || { echo "ERROR: cannot read the tilert model entry for '$MODEL_NAME' from $MODELS_YAML" >&2; exit 1; } +echo "[tilert] model entry '$MODEL_NAME' loaded from $MODELS_YAML" + +export ROUTER_PORT +export SERVED_MODEL_NAME + +PREFILL_SPEC=() +DECODE_MTP=() +if [[ "$SPEC_DECODING" == "mtp" ]]; then + # The prefill rank only has to build the MTP layer's KV; TileRT decode owns + # the draft depth (DECODE_MTP_SIZE), so the two counts differ by design. + PREFILL_SPEC=(--speculative-config "{\"method\":\"mtp\",\"num_speculative_tokens\":${PREFILL_SPEC_TOKENS}}") + # decode_server only accepts depth 3 today, but pass it explicitly so the + # converter's --num_mtp, the golden-curve key and the engine depth agree by + # data flow rather than by coincidence of defaults. + DECODE_MTP=(--with-mtp --num-mtp "$DECODE_MTP_SIZE") +fi + +# The only TileRT recipe on this cluster is AgentX (agentic-coding). +if [[ "${IS_AGENTIC:-0}" != "1" && "${IS_AGENTIC:-}" != "true" && "${SCENARIO_TYPE:-}" != "agentic-coding" ]]; then + echo "ERROR: server_tilert.sh only runs agentic-coding (IS_AGENTIC=${IS_AGENTIC:-} SCENARIO_TYPE=${SCENARIO_TYPE:-})" >&2 + exit 1 +fi + +IFS=',' read -ra IP_ARRAY <<< "$IPADDRS" +PREFILL_HOST="${IP_ARRAY[0]:-$NODE0_ADDR}" +DECODE_HOST="${IP_ARRAY[$xP]:-}" +if [[ -z "$DECODE_HOST" ]]; then + echo "ERROR: cannot resolve the decode node IP from IPADDRS='$IPADDRS' (xP=$xP)" >&2 + exit 1 +fi +host_ip=$(ip route get 1.1.1.1 2>/dev/null | awk '/src/ {print $7}') +host_name=$(hostname) + +echo "[tilert] ROLE=$TILERT_ROLE rank=$NODE_RANK host=$host_name ($host_ip)" +echo "[tilert] PREFILL_HOST=$PREFILL_HOST:$PREFILL_PORT DECODE_HOST=$DECODE_HOST:$DECODE_CTRL_PORT/$DECODE_HTTP_PORT ROUTER=:$ROUTER_PORT" +echo "[tilert] MODEL_PATH=$MODEL_PATH profile=$TILERT_PROFILE served=$SERVED_MODEL_NAME max_len=$TILERT_MAX_MODEL_LEN transport=$TILERT_TRANSPORT kv=${PREFILL_KV_DTYPE}->${DECODE_KV_DTYPE} mtp=${SPEC_DECODING}" + +# Enable libibverbs fork safety on both ranks before any verbs context exists. +# Without it, ibv_fork_init() can fail in these containers while Mooncake +# initialization still reports success, silently falling back from RDMA to TCP. +# Slower prefill-to-decode KV transfer degrades TTFT; TPOT is unaffected. +export RDMAV_FORK_SAFE=1 + +for env_pair in ${TILERT_EXTRA_ENV}; do + export "${env_pair?}" + echo "[tilert][EXTRA_ENV] $env_pair" +done + +log_and_run_bg() { + local label="$1" logfile="$2"; shift 2 + { printf '===== [%s] %s =====\n' "$label" "$(date '+%F %T')" + printf '[cmd]'; printf ' %q' "$@"; printf '\n' + printf '[cwd] %s\n[host] %s\n\n' "$PWD" "$host_name" + } | tee -a "$logfile" + if [[ "$DRY_RUN" -eq 1 ]]; then + echo "DRY RUN: [$label] not started" + LAST_BG_PID="" + return 0 + fi + "$@" >>"$logfile" 2>&1 & + LAST_BG_PID=$! + echo "[$label] pid=$LAST_BG_PID log=$logfile" +} + +rdma_preflight() { + local warn=0 + echo "[rdma] role=$TILERT_ROLE IBDEVICES=${IBDEVICES:-} NCCL_SOCKET_IFNAME=${NCCL_SOCKET_IFNAME:-}" + local uverbs=(/dev/infiniband/uverbs*) + if [[ -e "${uverbs[0]}" ]]; then + echo "[rdma] verbs devices: ${uverbs[*]}" + else + echo "[rdma] WARNING: /dev/infiniband/uverbs* missing -- the container has no RDMA device nodes (job.slurm passes --device /dev/infiniband)" >&2 + warn=1 + fi + local ml; ml="$(ulimit -l 2>/dev/null)" + if [[ "$ml" == "unlimited" ]]; then + echo "[rdma] memlock: unlimited" + else + echo "[rdma] WARNING: memlock=$ml (not unlimited) -- pinning memory for RDMA may fail (job.slurm passes --ulimit memlock=-1)" >&2 + warn=1 + fi + if command -v ibv_devices >/dev/null 2>&1; then + echo "[rdma] ibv_devices:"; ibv_devices 2>&1 | sed 's/^/[rdma] /' + fi + if (( warn )) && [[ "$TILERT_RDMA_STRICT" == "1" ]]; then + echo "[rdma] TILERT_RDMA_STRICT=1 and preflight did not fully pass -- aborting" >&2 + return 1 + fi + return 0 +} + +stage_tokenizer_files() { + local staged=0 f b + for f in "$MODEL_PATH"/*; do + [[ -f "$f" ]] || continue + b="$(basename "$f")" + [[ "$b" == *.safetensors ]] && continue + [[ "$b" == "model.safetensors.index.json" ]] && continue + [[ -e "$TILERT_WEIGHTS_DIR/$b" ]] && continue + cp -p "$f" "$TILERT_WEIGHTS_DIR/$b" && staged=$((staged+1)) + done + echo "[stage_tokenizer] staged $staged auxiliary file(s) from $MODEL_PATH -> $TILERT_WEIGHTS_DIR" + local missing=() + [[ -f "$TILERT_WEIGHTS_DIR/chat_template.jinja" ]] || missing+=(chat_template.jinja) + [[ -f "$TILERT_WEIGHTS_DIR/tokenizer_config.json" || -f "$TILERT_WEIGHTS_DIR/tokenizer.json" ]] \ + || missing+=("tokenizer.json/tokenizer_config.json") + if (( ${#missing[@]} )); then + echo "[stage_tokenizer] ERROR: $TILERT_WEIGHTS_DIR is missing ${missing[*]}; check that MODEL_PATH=$MODEL_PATH is an HF directory with the tokenizer" >&2 + return 1 + fi + return 0 +} + +_tilert_weights_cached() { + local r + for r in $(seq 0 $((DECODE_TP_SIZE - 1))); do + [[ -f "$TILERT_WEIGHTS_DIR/rank${r}/model.safetensors.index.json" ]] || return 1 + done + # The engine refuses a cache converted without the MTP module (end2end.py + # checks tilert_meta.json num_mtp), but only after loading ~90 GiB of + # weights. Check the same field here so a stale non-MTP cache is + # re-converted instead of failing late. + [[ -f "$TILERT_WEIGHTS_DIR/tilert_meta.json" ]] || return 1 + if [[ "$SPEC_DECODING" == "mtp" ]]; then + "$PY" - "$TILERT_WEIGHTS_DIR/tilert_meta.json" <<'PYEOF' || return 1 +import json, sys +sys.exit(0 if int(json.load(open(sys.argv[1])).get("num_mtp", 0)) >= 1 else 1) +PYEOF + fi + return 0 +} + +convert_weights() { + if _tilert_weights_cached; then + echo "[weight_converter] cache hit (${DECODE_TP_SIZE}/${DECODE_TP_SIZE} rank index.json), skipping conversion: $TILERT_WEIGHTS_DIR" + return 0 + fi + mkdir -p "$TILERT_WEIGHTS_DIR" || { echo "[weight_converter] ERROR: cannot create $TILERT_WEIGHTS_DIR (set TILERT_WEIGHTS_DIR to a writable shared path)" >&2; return 1; } + exec 9>"$TILERT_WEIGHTS_DIR/.convert.lock" + flock -w "$TILERT_CONVERT_LOCK_WAIT" 9 || { + echo "[weight_converter] timed out waiting for the conversion lock (another job still converting?)" >&2; return 1; } + if _tilert_weights_cached; then + echo "[weight_converter] cache produced by a concurrent job, skipping conversion"; exec 9>&-; return 0 + fi + if [[ -n "$(ls -A "$TILERT_WEIGHTS_DIR" 2>/dev/null | grep -v '^\.convert\.lock$')" ]]; then + echo "[weight_converter] leftovers without index.json (previous conversion incomplete); cleaning and re-converting" + find "$TILERT_WEIGHTS_DIR" -mindepth 1 ! -name '.convert.lock' -delete + fi + echo "[weight_converter] $MODEL_PATH -> $TILERT_WEIGHTS_DIR (model_type=$TILERT_MODEL_TYPE)" + local conv_mod conv_args + if "$PY" -c "import tilert.models.${TILERT_MODEL_PKG}.weight_converter" 2>/dev/null; then + conv_mod="tilert.models.${TILERT_MODEL_PKG}.weight_converter" + conv_args=(--model_dir "$MODEL_PATH" --save_dir "$TILERT_WEIGHTS_DIR" + --device "cuda:$((GPUS_PER_NODE - 1))") + [[ "$SPEC_DECODING" == "mtp" ]] && conv_args+=(--num_mtp "$DECODE_MTP_SIZE") + else + conv_mod="tilert.models.preprocess.weight_converter" + conv_args=(--model_type "$TILERT_MODEL_TYPE" --model_dir "$MODEL_PATH" --save_dir "$TILERT_WEIGHTS_DIR") + fi + if [[ "$DRY_RUN" -eq 1 ]]; then + echo "DRY RUN: $PY -m $conv_mod ${conv_args[*]}" + exec 9>&-; return 0 + fi + echo "[weight_converter] using $conv_mod" + "$PY" -m "$conv_mod" "${conv_args[@]}" \ + 2>&1 | tee "$LOG_DIR/tilert_weight_converter_${host_name}.log" + local rc=${PIPESTATUS[0]} + exec 9>&- + if [[ $rc -ne 0 ]] || ! _tilert_weights_cached; then + echo "[weight_converter] ERROR: conversion failed (rc=$rc, per-rank index.json complete: $(_tilert_weights_cached && echo yes || echo no))" >&2 + return 1 + fi + echo "[weight_converter] conversion done and cached: $TILERT_WEIGHTS_DIR" +} + +start_decode() { + # shellcheck disable=SC2206 + local extra=( ${TILERT_DECODE_EXTRA_FLAGS} ) + if [[ "$SPEC_DECODING" == "mtp" && "$EVAL_ONLY" != "true" && "$RUN_EVAL" != "true" ]]; then + check_env_vars MODEL_PREFIX THINKING_MODE + local curve="${WS_PATH%/benchmarks/*}/golden_al_distribution/${MODEL_PREFIX}_mtp.yaml" + TILERT_SIMULATE_ACC_LEN="$("$PY" - "$curve" "$THINKING_MODE" "$DECODE_MTP_SIZE" <<'PYEOF' +import sys, yaml +path, thinking, tokens = sys.argv[1], sys.argv[2], int(sys.argv[3]) +data = yaml.safe_load(open(path)) +if not isinstance(data, dict) or len(data) != 1: + sys.exit(f"golden curve {path} must hold exactly one model key") +model, modes = next(iter(data.items())) +try: + value = float(modes[thinking][tokens]) +except (KeyError, TypeError, ValueError): + sys.exit(f"no golden acceptance for {model}/{thinking}/{tokens} draft tokens in {path}") +if not 1 <= value <= tokens + 1: + sys.exit(f"golden acceptance {value} out of range for {tokens} draft tokens") +print(f"{value:g}") +PYEOF +)" || { echo "[tilert] ERROR: golden AL lookup failed (curve=$curve)" >&2; exit 1; } + echo "[tilert] golden AL ${TILERT_SIMULATE_ACC_LEN} from $(basename "$curve") ($THINKING_MODE, K=${DECODE_MTP_SIZE})" + fi + + if [[ -n "${TILERT_SIMULATE_ACC_LEN:-}" && "$EVAL_ONLY" != "true" ]]; then + export TILERT_SIMULATE_ACC_LEN + export TILERT_SIMULATE_ACC_METHOD + echo "[decode] simulated acceptance: TILERT_SIMULATE_ACC_LEN=${TILERT_SIMULATE_ACC_LEN}" \ + "method=${TILERT_SIMULATE_ACC_METHOD} (output text is meaningless by design)" + else + unset TILERT_SIMULATE_ACC_LEN TILERT_SIMULATE_ACC_METHOD + echo "[decode] real MTP verification (no simulated acceptance)" + fi + local cmd=("$PY" -m tilert.pd_vllm.decode_server + --engine tilert --model "$TILERT_PROFILE" + --model-weights-dir "$TILERT_WEIGHTS_DIR" + --max-seq-len "$TILERT_MAX_MODEL_LEN" + --kv-cache-dtype "$DECODE_KV_DTYPE" --transport "$TILERT_TRANSPORT" + --ctrl-port "$DECODE_CTRL_PORT" --http-port "$DECODE_HTTP_PORT" + "${DECODE_MTP[@]}" "${extra[@]}") + log_and_run_bg decode "$LOG_DIR/decode_${host_name}.log" "${cmd[@]}" + DECODE_PID=$LAST_BG_PID +} + +start_prefill() { + for env_pair in ${TILERT_PREFILL_ENV}; do + export "${env_pair?}" + echo "[PREFILL_ENV] $env_pair" + done + local served=("$SERVED_MODEL_NAME") + [[ -n "$MODEL_NAME" && "$MODEL_NAME" != "$SERVED_MODEL_NAME" ]] && served+=("$MODEL_NAME") + # shellcheck disable=SC2206 + local extra=( ${TILERT_PREFILL_EXTRA_FLAGS} ) + local kv_cfg + kv_cfg=$(printf '{"kv_connector":"TileRTConnector","kv_connector_module_path":"tilert.pd_vllm.prefill_connector","kv_role":"kv_producer","kv_connector_extra_config":{"tilert_host":"%s","tilert_ctrl_port":%s,"tilert_model":"%s","tilert_max_seq_len":%s,"tilert_transport":"%s"}}' \ + "$DECODE_HOST" "$DECODE_CTRL_PORT" "$TILERT_PROFILE" "$TILERT_MAX_MODEL_LEN" "$TILERT_TRANSPORT") + local cmd=(vllm serve "$MODEL_PATH" + --served-model-name "${served[@]}" --port "$PREFILL_PORT" + --tensor-parallel-size "$PREFILL_TP_SIZE" --max-model-len "$TILERT_MAX_MODEL_LEN" + --enforce-eager --trust-remote-code --return-tokens-as-token-ids + --gpu-memory-utilization "$GPU_MEM_UTIL" --kv-cache-dtype "$PREFILL_KV_DTYPE" + --block-size "$PREFILL_BLOCK_SIZE" + "${PREFILL_SPEC[@]}" + --kv-transfer-config "$kv_cfg" + "${extra[@]}") + log_and_run_bg prefill "$LOG_DIR/prefill_${host_name}.log" "${cmd[@]}" + PREFILL_PID=$LAST_BG_PID +} + +start_router() { + local cmd=(env HIP_VISIBLE_DEVICES= ROCR_VISIBLE_DEVICES= CUDA_VISIBLE_DEVICES= + "$PY" -m tilert.pd_vllm.pd_router + --vllm-url "http://$PREFILL_HOST:$PREFILL_PORT" + --decode "$DECODE_HOST:$DECODE_CTRL_PORT:$DECODE_HTTP_PORT" + --host 0.0.0.0 --port "$ROUTER_PORT" --model-path "$MODEL_PATH" --parser "$TILERT_PARSER" + --queue-timeout "$TILERT_QUEUE_TIMEOUT") + log_and_run_bg router "$LOG_DIR/router_${host_name}.log" "${cmd[@]}" + ROUTER_PID=$LAST_BG_PID +} + +tcp_open() { (exec 3<>"/dev/tcp/$1/$2") 2>/dev/null; } + +wait_for_tcp() { + local host="$1" port="$2" timeout="${3:-600}" pid="${4:-}" + local deadline=$(( SECONDS + timeout )) + until tcp_open "$host" "$port"; do + if [[ -n "$pid" ]] && ! kill -0 "$pid" 2>/dev/null; then + echo "[wait_for_tcp] process $pid exited before $host:$port opened" >&2; return 2 + fi + if [[ $SECONDS -ge $deadline ]]; then + echo "[wait_for_tcp] timeout: $host:$port not open after ${timeout}s" >&2; return 1 + fi + sleep 5 + done + echo "[wait_for_tcp] $host:$port ready" +} + +wait_for_tcp_close() { + local host="$1" port="$2" pid="${3:-}" + while tcp_open "$host" "$port"; do + if [[ -n "$pid" ]] && ! kill -0 "$pid" 2>/dev/null; then + echo "[wait_for_tcp_close] process $pid exited while $host:$port is still open" >&2; return 2 + fi + sleep 10 + done + echo "[wait_for_tcp_close] $host:$port closed" +} + +copy_logs_to_shared() { + [[ "$DRY_RUN" -eq 0 ]] || return 0 + mkdir -p "$SHARED_LOG_DIR" && cp -r "$LOG_DIR"/. "$SHARED_LOG_DIR"/ \ + && echo "Copied $LOG_DIR -> $SHARED_LOG_DIR" \ + || echo "WARNING: failed to copy $LOG_DIR to $SHARED_LOG_DIR" >&2 +} +trap copy_logs_to_shared EXIT + +run_lm_eval_on_router() { + echo "Running lm-eval evaluation on the router..." + local ok=false _attempt + for _attempt in 1 2 3; do + if curl -sf --max-time 10 "http://0.0.0.0:${ROUTER_PORT}/health" >/dev/null 2>&1; then ok=true; break; fi + echo "Eval health check attempt $_attempt failed, retrying in 10s..."; sleep 10 + done + if [[ "$ok" != "true" ]]; then + echo "ERROR: router health check failed after 3 attempts; skipping eval" >&2 + return 1 + fi + local eval_failed=0 + pushd /workspace >/dev/null || return 1 + if [[ -n "${EVAL_CONC:-}" ]]; then + export EVAL_CONCURRENT_REQUESTS="${EVAL_CONC}" + else + export EVAL_CONCURRENT_REQUESTS=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | tail -1) + fi + # run_lm_eval reads the endpoint from PORT (check_env_vars) and names the + # model from MODEL_NAME, which the vLLM prefill also serves next to + # SERVED_MODEL_NAME; MODEL stays the local HF dir for the context lookup. + export PORT="$ROUTER_PORT" + export MODEL="$MODEL_PATH" + export MAX_MODEL_LEN="$TILERT_MAX_MODEL_LEN" + if [[ "$DRY_RUN" -eq 1 ]]; then + echo "DRY RUN: run_eval --port $ROUTER_PORT (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS})" + else + run_eval --port "$ROUTER_PORT" + local eval_rc=$? + if [[ $eval_rc -ne 0 ]]; then + echo "ERROR: run_eval exited rc=$eval_rc; preserving failure artifacts" >&2 + eval_failed=1 + else + export TP="${PREFILL_TP_SIZE}" CONC="${EVAL_CONCURRENT_REQUESTS}" EP_SIZE=1 + export PREFILL_TP="${PREFILL_TP_SIZE}" PREFILL_EP=1 PREFILL_NUM_WORKERS="${xP}" + export DECODE_TP="${DECODE_TP_SIZE}" DECODE_EP=1 DECODE_NUM_WORKERS="${yD}" + export DP_ATTENTION=false PREFILL_DP_ATTENTION=false DECODE_DP_ATTENTION=false + export ISL="${BENCH_INPUT_LEN}" OSL="${BENCH_OUTPUT_LEN}" + # As on the SGLang path: rewrite meta_env.json from the exports above, + # then stage unless run_eval already did (eval-only). + rewrite_lm_eval_meta_env + if [[ "$EVAL_ONLY" != "true" ]]; then + append_lm_eval_summary + fi + fi + local eval_copy_dir="$LOG_DIR/eval_results" + if stage_eval_artifacts "$eval_copy_dir" /workspace "${EVAL_RESULT_DIR:-}"; then + echo "Eval artifacts staged in $eval_copy_dir" + else + echo "ERROR: failed to stage eval artifacts in $eval_copy_dir" >&2 + eval_failed=1 + fi + fi + popd >/dev/null || true + return $eval_failed +} + +run_agentic_replay() { + local rc=0 + wait_for_server_ready --port "$ROUTER_PORT" --server-log "$LOG_DIR/router_${host_name}.log" --server-pid "$ROUTER_PID" + cd /workspace || return 1 + + export PORT="$ROUTER_PORT" + export MODEL="$MODEL_PATH" # aiperf --tokenizer (local HF dir) + export SERVED_MODEL_NAME # aiperf --model (name the router/vLLM serve) + check_env_vars DURATION RESULT_FILENAME INFMAX_CONTAINER_WORKSPACE + export MAX_MODEL_LEN="$TILERT_MAX_MODEL_LEN" + # TileRT decode exposes no /metrics route; only the vLLM prefill is scraped. + export AIPERF_SERVER_METRICS_URLS="http://${PREFILL_HOST}:${PREFILL_PORT}/metrics" + export TRANSFORMERS_VERBOSITY=error TOKENIZERS_PARALLELISM=false + # Keep the trace corpus and aiperf's HF downloads on the node's /tmp mount + # instead of the container's ephemeral ~/.cache, as the SGLang client does. + export HF_HOME=/run_logs/hf_cache + + local result_dir="$LOG_DIR/agentic" + local result_filename_base="$RESULT_FILENAME" + mkdir -p "$result_dir" + + # Neither server.sh nor this script runs with errexit; a failed bootstrap + # must not fall through into the replay loop and its misleading cascade. + resolve_trace_source || return 1 + install_agentic_deps || return 1 + + local conc conc_result_dir + for conc in ${BENCH_MAX_CONCURRENCY//x/ }; do + echo "==========================================" + echo "Agentic trace replay: conc=$conc" + echo "==========================================" + conc_result_dir="$result_dir/conc_${conc}" + mkdir -p "$conc_result_dir" + export CONC="$conc" USERS="$conc" + build_replay_cmd "$conc_result_dir" + export RESULT_FILENAME="${result_filename_base}_conc${conc}" + if [[ "$DRY_RUN" -eq 1 ]]; then + echo "DRY RUN: $REPLAY_CMD" + elif ! run_agentic_replay_and_write_outputs "$conc_result_dir"; then + echo "WARNING: agentic trace replay for conc=$conc failed (replay or validation) after writing available results" >&2 + rc=1 + fi + echo "-----------------------------------------" + done + export RESULT_FILENAME="$result_filename_base" + return $rc +} + +echo "Waiting at the container creation barrier on $host_name" +if [[ "$DRY_RUN" -eq 1 ]]; then + echo "DRY RUN: skipping container creation barrier" +elif [[ "$SKIP_CONTAINER_BARRIER" == "1" ]]; then + echo "SKIP_CONTAINER_BARRIER=1: caller asserts all containers are up" +else + # --grace 60: after the barrier passes, sync.py keeps the port open for + # max(60, timeout/2) seconds in the foreground so a peer one poll behind + # still sees it. At CONTAINER_BARRIER_TIMEOUT=5400 that is a 45-minute idle + # sleep on every rank (jobs 45373/45374 slept 06:28-07:13). Both ranks pass + # within one 5 s poll of each other and the stages below have their own + # readiness waits, so 60 s is plenty. + "$PY" "$WS_PATH/sync.py" barrier \ + --local-ip "${host_ip}" --local-port 5000 --enable-port \ + --node-ips "${IPADDRS}" --node-ports 5000 \ + --wait-for-all-ports --timeout "$CONTAINER_BARRIER_TIMEOUT" --grace 60 \ + || { echo "ERROR: container creation barrier failed after ${CONTAINER_BARRIER_TIMEOUT}s -- the peer rank never opened port 5000." \ + "A cold image pull is the usual cause: this recipe pulls two ~32 GB images, one per rank, and the rank that" \ + "comes up first waits out the whole timeout while the other is still pulling." >&2; exit 1; } +fi + +case "$TILERT_ROLE" in + decode) + echo "${host_name}:${host_ip} is the TileRT Decode Node (Model: ${MODEL_NAME}, profile: ${TILERT_PROFILE})" + rdma_preflight || exit 1 + convert_weights || exit 1 + stage_tokenizer_files || exit 1 + start_decode + if [[ "$DRY_RUN" -eq 1 ]]; then + echo "DRY RUN: decode role complete"; exit 0 + fi + echo "Waiting for the router port ${PREFILL_HOST}:${ROUTER_PORT} to open (timeout ${ROUTER_WAIT}s)..." + wait_for_tcp "$PREFILL_HOST" "$ROUTER_PORT" "$ROUTER_WAIT" "$DECODE_PID"; wrc=$? + if [[ $wrc -eq 2 ]]; then + echo "ERROR: decode_server exited before the router came up (see $LOG_DIR/decode_${host_name}.log)" >&2 + tail -50 "$LOG_DIR/decode_${host_name}.log" >&2 || true + copy_logs_to_shared; exit 1 + elif [[ $wrc -ne 0 ]]; then + echo "WARNING: router never opened within ${ROUTER_WAIT}s; shutting down decode" >&2 + kill "$DECODE_PID" 2>/dev/null || true + copy_logs_to_shared; exit 1 + fi + echo "Waiting until the router port closes..." + wait_for_tcp_close "$PREFILL_HOST" "$ROUTER_PORT" "$DECODE_PID"; wrc=$? + if [[ $wrc -eq 2 ]]; then + echo "ERROR: decode_server died while the benchmark was running (see $LOG_DIR/decode_${host_name}.log)" >&2 + tail -50 "$LOG_DIR/decode_${host_name}.log" >&2 || true + copy_logs_to_shared; exit 1 + fi + echo "Killing the decode server" + kill "$DECODE_PID" 2>/dev/null || true + sleep 2 + copy_logs_to_shared + ;; + prefill) + echo "NODE INFO =======================================" + echo "Node List : ${SLURM_JOB_NODELIST:-}" + echo "Node IPs : ${IPADDRS}" + echo "Model : ${MODEL_NAME}" + echo "${host_name}:${host_ip} is the Prefill Node (vLLM + TileRTConnector) and Router Node" + echo "================================================" + rdma_preflight || exit 1 + echo "Waiting for the decode ctrl port ${DECODE_HOST}:${DECODE_CTRL_PORT} (timeout ${DECODE_WAIT}s)..." + if [[ "$DRY_RUN" -eq 0 ]]; then + wait_for_tcp "$DECODE_HOST" "$DECODE_CTRL_PORT" "$DECODE_WAIT" \ + || echo "WARNING: timed out waiting for the decode ctrl port; starting prefill anyway" >&2 + fi + start_prefill + if [[ "$DRY_RUN" -eq 0 ]]; then + wait_for_tcp "$PREFILL_HOST" "$PREFILL_PORT" "$PREFILL_WAIT" "$PREFILL_PID"; wrc=$? + if [[ $wrc -ne 0 ]]; then + echo "ERROR: vLLM prefill did not open ${PREFILL_HOST}:${PREFILL_PORT} (rc=$wrc, see $LOG_DIR/prefill_${host_name}.log)" >&2 + tail -50 "$LOG_DIR/prefill_${host_name}.log" >&2 || true + kill "$PREFILL_PID" 2>/dev/null || true + copy_logs_to_shared; exit 1 + fi + fi + start_router + if [[ "$DRY_RUN" -eq 1 ]]; then + echo "DRY RUN: prefill/router role complete"; exit 0 + fi + echo "Ready for benchmarking on ${host_name}:${host_ip}" + cd "$WS_PATH" || exit 1 + # EVAL_ONLY skips the AgentX replay and runs GSM8K on the same router; + # RUN_EVAL after a replay runs it once the replay has finished. + if [[ "$EVAL_ONLY" == "true" ]]; then + echo "EVAL_ONLY mode: skipping the AgentX replay" + wait_for_server_ready --port "$ROUTER_PORT" --server-log "$LOG_DIR/router_${host_name}.log" --server-pid "$ROUTER_PID" + export TRANSFORMERS_VERBOSITY=error TOKENIZERS_PARALLELISM=false + run_lm_eval_on_router; BENCH_RC=$? + else + run_agentic_replay; BENCH_RC=$? + if [[ "$RUN_EVAL" == "true" ]]; then + run_lm_eval_on_router || BENCH_RC=1 + fi + fi + copy_logs_to_shared + echo "Killing the router and the prefill server" + kill "$ROUTER_PID" "$PREFILL_PID" 2>/dev/null || true + sleep 2 + pkill -f "tilert.pd_vllm.pd_router" 2>/dev/null || true + pkill -f "vllm serve" 2>/dev/null || true + if [[ "$BENCH_RC" -ne 0 ]]; then + echo "ERROR: benchmark/eval reported rc=$BENCH_RC" >&2 + exit "$BENCH_RC" + fi + ;; + *) + echo "ERROR: unknown TILERT_ROLE='$TILERT_ROLE'" >&2; exit 2 ;; +esac + +echo "Script completed successfully" +exit 0 diff --git a/benchmarks/multi_node/amd_utils/setup_deps.sh b/benchmarks/multi_node/amd_utils/setup_deps.sh index 1cc50a15ea..8adb438681 100644 --- a/benchmarks/multi_node/amd_utils/setup_deps.sh +++ b/benchmarks/multi_node/amd_utils/setup_deps.sh @@ -45,6 +45,116 @@ install_amd_quark() { _SETUP_INSTALLED+=("amd-quark") } +# Pinned by the recipe (TILERT_VERSION); the rest are fixed properties of the +# TileRT 0.1.x runtime rather than caller configuration. +TILERT_PACKAGE=tilert +TILERT_HTTP_DEPS="fastapi uvicorn httpx" +TILERT_TRANSPORT_DEPS="mooncake-transfer-engine-rocm>=0.3.13" +TILERT_TRANSFORMERS_SPEC="transformers>=4.56" + +_tilert_resolve_python() { + if [[ -n "${PY:-}" ]] && command -v "$PY" >/dev/null 2>&1; then :; else + PY="" + local c + for c in python3 python; do command -v "$c" >/dev/null 2>&1 && { PY="$c"; break; }; done + fi + [[ -n "$PY" ]] || { echo "[SETUP] ERROR: neither python3 nor python found"; exit 1; } + export PY + echo "[SETUP] interpreter PY=$PY ($(command -v "$PY"))" +} + +_tilert_installed_version() { + "$PY" - "$1" <<'PYEOF' 2>/dev/null +import sys +from importlib.metadata import version, PackageNotFoundError +try: + print(version(sys.argv[1])) +except PackageNotFoundError: + pass +PYEOF +} + +_tilert_pip() { + "$PY" -m pip install --quiet --no-cache-dir "$@" +} + +_tilert_install_missing() { + local probe="$1"; shift + [[ $# -gt 0 ]] || return 0 + if "$PY" -c "import $probe" 2>/dev/null; then + echo "[SETUP] $probe already present, skipping ($*)" + return 0 + fi + echo "[SETUP] installing $* (probe module '$probe' missing)" + _tilert_pip "$@" || { echo "[SETUP] ERROR: failed to install: $*"; exit 1; } + _SETUP_INSTALLED+=("$*") +} + +install_tilert_container_tools() { + if command -v ip >/dev/null 2>&1 && command -v curl >/dev/null 2>&1 \ + && command -v ibv_devices >/dev/null 2>&1; then + echo "[SETUP] Container RDMA/net tools already present" + return 0 + fi + echo "[SETUP] Installing iproute2 + curl + ibverbs userspace in container..." + apt-get update -q -y && apt-get install -q -y --no-install-recommends \ + iproute2 curl ibverbs-utils libibverbs1 librdmacm1 ibverbs-providers \ + && rm -rf /var/lib/apt/lists/* + if ! command -v ip >/dev/null 2>&1 || ! command -v curl >/dev/null 2>&1; then + echo "[SETUP] ERROR: failed to install iproute2/curl"; exit 1 + fi + _SETUP_INSTALLED+=("iproute2+curl+ibverbs") +} + +_tilert_install_wheel() { + local mode="$1" # full | no-deps + local have; have="$(_tilert_installed_version tilert)" + if [[ "$have" == "$TILERT_VERSION" ]]; then + echo "[SETUP] tilert $have already installed, skipping" + return 0 + fi + [[ -n "$have" ]] && echo "[SETUP] tilert $have installed, switching to pinned $TILERT_VERSION" + if [[ "$mode" == "no-deps" ]]; then + echo "[SETUP] installing $TILERT_PIP_SPEC --no-deps (connector plugin + router on top of the image's vLLM)" + _tilert_pip --no-deps "$TILERT_PIP_SPEC" || { echo "[SETUP] ERROR: failed to install $TILERT_PIP_SPEC (--no-deps)"; exit 1; } + else + echo "[SETUP] installing $TILERT_PIP_SPEC (TileRT ROCm build, official PyPI wheel)" + _tilert_pip "$TILERT_PIP_SPEC" || { echo "[SETUP] ERROR: failed to install $TILERT_PIP_SPEC"; exit 1; } + fi + have="$(_tilert_installed_version tilert)" + [[ "$have" == "$TILERT_VERSION" ]] || { + echo "[SETUP] ERROR: tilert is ${have:-not installed} after install, expected $TILERT_VERSION"; exit 1; } + _SETUP_INSTALLED+=("$TILERT_PACKAGE==$TILERT_VERSION($mode)") +} + +install_tilert_decode() { + install_tilert_container_tools + _tilert_install_wheel full + _tilert_install_missing uvicorn $TILERT_HTTP_DEPS + _tilert_install_missing mooncake.engine "$TILERT_TRANSPORT_DEPS" + _tilert_install_missing transformers "$TILERT_TRANSFORMERS_SPEC" + "$PY" -c "import tilert.pd_vllm.decode_server" 2>/dev/null || { + echo "[SETUP] ERROR: import tilert.pd_vllm.decode_server failed:" + "$PY" -c "import tilert.pd_vllm.decode_server" 2>&1 | tail -3 + exit 1; } + echo "[SETUP] tilert.pd_vllm.decode_server imports OK" +} + +install_tilert_prefill() { + local vllm_v; vllm_v="$(_tilert_installed_version vllm)" + if [[ -z "$vllm_v" ]]; then + echo "[SETUP] ERROR: no vLLM in the prefill image (PREFILL_IMAGE must be a vllm/vllm-openai-rocm image)." + exit 1 + fi + echo "[SETUP] prefill-side vLLM $vllm_v" + install_tilert_container_tools + _tilert_install_wheel no-deps + _tilert_install_missing mooncake.engine "$TILERT_TRANSPORT_DEPS" + "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>/dev/null || { + echo "[SETUP] WARN: import tilert.pd_vllm.prefill_connector failed (vLLM will report again when loading the connector plugin):" + "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>&1 | tail -3; } +} + if [[ "$ENGINE" == "vllm-disagg" ]]; then install_recipe_deps install_amd_quark @@ -54,6 +164,15 @@ if [[ "$ENGINE" == "vllm-disagg" ]]; then export RIXL_HOME export PATH="${UCX_HOME}/bin:/usr/local/bin/etcd:/root/.cargo/bin:${PATH}" export LD_LIBRARY_PATH="${UCX_HOME}/lib:${RIXL_HOME}/lib:${RIXL_HOME}/lib/x86_64-linux-gnu:${LD_LIBRARY_PATH:-}" +elif [[ "$ENGINE" == "tilert" ]]; then + check_env_vars TILERT_VERSION + TILERT_PIP_SPEC="$TILERT_PACKAGE==$TILERT_VERSION" + _tilert_resolve_python + case "${TILERT_ROLE:-}" in + decode) install_tilert_decode ;; + prefill) install_tilert_prefill ;; + *) echo "[SETUP] ERROR: ENGINE=tilert needs TILERT_ROLE=decode|prefill (got '${TILERT_ROLE:-}')"; exit 1 ;; + esac fi _SETUP_END=$(date +%s) diff --git a/benchmarks/multi_node/amd_utils/sync.py b/benchmarks/multi_node/amd_utils/sync.py index 96e94c1b08..fb618d2f51 100755 --- a/benchmarks/multi_node/amd_utils/sync.py +++ b/benchmarks/multi_node/amd_utils/sync.py @@ -161,7 +161,10 @@ def close_port(): if args.enable_port: # Keep the port open long enough for slow nodes to pass their barrier. # The previous 30s was too short when setup times vary by minutes. - grace = max(60, args.timeout // 2) if args.timeout > 0 else 300 + if args.grace is not None: + grace = args.grace + else: + grace = max(60, args.timeout // 2) if args.timeout > 0 else 300 time.sleep(grace) close_port() @@ -198,6 +201,10 @@ def main(): bp.add_argument("--node-ports", required=True, help="Comma-separated list of ports to check.") bp.add_argument("--timeout", type=int, default=600, help="Timeout in seconds (default: 600). Set to 0 for no timeout.") + bp.add_argument("--grace", type=int, default=None, + help="Seconds to keep the local port open after the barrier passes so peers one poll " + "behind still see it. Default max(60, timeout // 2). Callers whose later stages " + "have their own readiness waits can pass a small value.") bp.add_argument("--wait-for-all-ports", action="store_true", help="Wait until all node ports are open (TCP).") bp.add_argument("--wait-for-all-health", action="store_true", diff --git a/benchmarks/multi_node/runtime_settings.sh b/benchmarks/multi_node/runtime_settings.sh index 5bf43deb6e..725efcdba6 100644 --- a/benchmarks/multi_node/runtime_settings.sh +++ b/benchmarks/multi_node/runtime_settings.sh @@ -48,7 +48,9 @@ case "$FRAMEWORK" in fi ;; tilert) - check_env_vars GITHUB_WORKSPACE + # RUNNER_TYPE selects the AMD block below, so a missing value must fail + # here rather than silently skip it. + check_env_vars GITHUB_WORKSPACE RUNNER_TYPE export BENCHMARK_LOGS_DIR="$GITHUB_WORKSPACE" RESULT_DIR=/workspace export GPU_MEM_UTIL=0.75 DECODE_CTRL_PORT=5556 DECODE_HTTP_PORT=5557 PREFILL_PORT=8000 export DECODE_WAIT=3600 PREFILL_WAIT=3600 TILERT_QUEUE_TIMEOUT=0 @@ -58,6 +60,30 @@ case "$FRAMEWORK" in if [[ "$IS_AGENTIC" == 1 || "$IS_AGENTIC" == true ]]; then export TILERT_QUEUE_TIMEOUT=1800 fi + # The MI355X TileRT recipe runs through the shared amd_utils chain + # (submit.sh -> job.slurm -> server.sh -> setup_deps.sh), which validates + # the same orchestration inputs the AMD SGLang/vLLM/ATOM arms receive. + # Without them submit.sh exits before sbatch and the launcher never gets + # a job id. The B200 TileRT lane goes through srt-slurm and reads none of + # these, so they are scoped to the AMD pool. + if [[ "$RUNNER_TYPE" == *mi355x-amds* ]]; then + # Validated by job.slurm; only the vllm-disagg router branch reads + # VLLM_ROUTER_IMAGE, which ENGINE=tilert never enters. + export VLLM_ROUTER_IMAGE=vllm/vllm-router:nightly-20260716-1fbcde7 + export SKIP_RDMA_CHECK=0 SKIP_GPU_SANITY=0 + # The B200 profile above points BENCHMARK_LOGS_DIR at the workspace + # itself; launch_mi355x-amds.sh's EXIT trap does `rm -rf + # "$BENCHMARK_LOGS_DIR"`, which then deleted the whole checkout, + # results included (sweep 35704948491). Use the AMD launcher's own + # convention from runners/runtime_settings.sh. + export BENCHMARK_LOGS_DIR="$GITHUB_WORKSPACE/benchmark_logs" + export ROUTER_TYPE=tilert-pd-router ROUTER_PORT=30000 PROXY_PING_PORT=36367 + export HEADNODE_PORT=20000 SERVER_PORT=2584 PROXY_STREAM_IDLE_TIMEOUT=300 + export ENABLE_METRICS=0 PREFILL_ROUTER_POLICY=random DECODE_ROUTER_POLICY=random + export FLUSH_DRAIN_TIMEOUT=120 CLEAR_CACHE_BETWEEN_CONC=1 + export DECODE_MTP_SIZE=0 + export ROCM_PATH=/opt/rocm UCX_HOME=/usr/local/ucx RIXL_HOME=/usr/local/rixl + fi ;; llmd-vllm) export LLMD_CONTAINER_ENGINE=docker VLLM_RANDOMIZE_DP_DUMMY_INPUTS=1 diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index 141eb6e82c..b26f8afda6 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -1507,3 +1507,51 @@ dsv41flash-fp4-mi355x-vllm-agentic-dspark: # 1-32 measured only on the superseded nightly-eed1f3d0 pin. - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } - { tp: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128] } + +# Speculative decoding on an agentic scenario must run with simulated +# synthetic acceptance at the committed golden AL for this model, thinking mode +# and draft length (docs/PR_REVIEW_CHECKLIST.md), and a submission may not +# substitute its own target. Nothing about that target is written here: +# AGENTS.md forbids hard-coding an acceptance length in a master config, so +# server_tilert.sh reads golden_al_distribution/glm5.3_mtp.yaml at launch and +# fails the run if the curve or the draft length is missing. +# The curve is consumed in the same units as every other framework here, and +# AgentX replays run with thinking on. +# +# NOTE FOR REVIEWERS: that curve is GLM-5.2's, copied because no SPEED-Bench run +# on GLM-5.3 exists and the two share a base. It is committed as provisional and +# labelled as such in the file. Flagging it rather than letting it read as a +# measured 5.3 curve -- please say if you would rather see a measured curve, a +# non-MTP agentic entry, or a waiver. +glm5.3-fp8-mi355x-tilert-agentic: + image: ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 + model: zai-org/GLM-5.3 + model-prefix: glm5.3 + runner: cluster:mi355x-amds + precision: fp8 + framework: tilert + router: { name: tilert-pd-router, version: "0.1.6.post1" } + multinode: true + disagg: true + kv-p2p-transfer: mooncake + scenarios: + agentic-coding: + - search-space: + - spec-decoding: "mtp" + conc-list: [1] + prefill: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "PREFILL_IMAGE=ghcr.io/tile-ai/tilert-rocm-prefill:0.1.6" + - "PREFILL_NODES=1" + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + additional-settings: + - "DECODE_NODES=1" + - "TILERT_EXTRA_ENV=GLM5_AR_N=2" diff --git a/golden_al_distribution/glm5.3_mtp.yaml b/golden_al_distribution/glm5.3_mtp.yaml new file mode 100644 index 0000000000..4933511703 --- /dev/null +++ b/golden_al_distribution/glm5.3_mtp.yaml @@ -0,0 +1,19 @@ +# PROVISIONAL — copied from glm5.2_mtp.yaml; NOT measured on GLM-5.3. +# +# GLM-5.3 shares GLM-5.2's base and layer topology (its config.json differs only +# by the fp8 quantization block), so the GLM-5.2 curve is the closest committed +# reference. Only the K=3 cell is carried over, because the ROCm TileRT GLM +# profile builds MTP at draft depth 3 and no other cell can be exercised. +# Replace this file with a SPEED-Bench run on glm-5.3-fp8 when one exists. +# +# Inherited value provenance (GLM-5.2, unchanged): +# Source GitHub Actions run: https://github.com/SemiAnalysisAI/InferenceX/actions/runs/28058352479 +# Acceptance Length (AL) reference values measured with SPEED-Bench. +# dataset: coding | temperature: 1.0 | top_p: 0.95 | output_len: 4096 +# thinking_on chat_template_kwargs: {"enable_thinking": true} +# Measured on glm-5.2-fp8 (B300, vLLM MTP), per num_speculative_tokens. +# +# key = num_speculative_tokens (MTP level); value = golden AL +glm-5.3-fp8: + thinking_on: + 3: 2.99 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 80f83a2bc3..bce3819e20 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8710,3 +8710,51 @@ - "删除显式的 direct/page_first_direct HiCache 覆盖,使 TP4/EP4 HiCache arm 运行 SGLang 默认的 kernel/page_first 路径;处于休眠状态的 DP-attention 与 Mooncake 分支同样删除这两个覆盖。该路径在 ROCm 上的正确性来自 sgl-project/sglang#35233——它把 registered host pool 的 device-accessible alias 交给 transfer kernel;上游在同一个 TP4/EP4 MXFP4 MI355X arm 上实测并发 14 时 kernel/page_first 为 338 tok/s,direct/page_first_direct 为 291 tok/s。" - "将 dram-utilization 从 0.80 提升到 0.85,并把 TP4/EP4 arm 的 ratio-1.0 host KV pool(实测约 115.19 GB/rank)改为固定 180 GB/rank;DSA indexer 同比扩展到约 41.25 GB/rank,在 SA 的 1,274 GB 预算内保留约 125 GB 启动余量。该 arm 的并发 sweep 为 4、8、10、12、14、16,TP8/EP1 GPU-resident arm 仅保留 1、2、4。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3329 + +- config-keys: + - glm5.3-fp8-mi355x-tilert-agentic + scenario-type: + - agentic-coding + description: + - "Add GLM-5.3 FP8 MI355X AgentX (agentic-coding) via TileRT prefill/decode disaggregation with MTP, 1 prefill node (TP8) + 1 decode node (TP8), concurrency [1], mirroring the B200 glm5.1-fp8-b200-tilert-agentic entry on the amd_utils orchestration." + - "Two images per job, both first-party and neither carrying the TileRT wheel: the decode rank runs ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 (ROCm PyTorch runtime plus the mooncake ROCm transfer engine built from source) and the prefill rank runs ghcr.io/tile-ai/tilert-rocm-prefill:0.1.6 (the upstream vllm/vllm-openai-rocm release image plus that same transfer engine, with vLLM itself unpatched and no file inside it modified), selected per rank via PREFILL_IMAGE in prefill.additional-settings. setup_deps.sh pip-installs the official tilert==0.1.6 wheel into both ranks at container start (full install plus the fastapi/uvicorn/httpx router deps on decode, --no-deps connector plugin on prefill, whose vLLM image already ships those), so neither image is pinned to a wheel version; this mirrors the B200 glm5.1-fp8-b200-tilert pattern of a first-party TileRT decode image. The transfer engine is built from source because the published mooncake ROCm wheels are compiled without ENABLE_MULTI_PROTOCOL, which leaves the cross-host locality gate (kvcache-ai/Mooncake#2753) out of the binary so every cross-node KV transfer is treated as node-local and fails in hipIpcOpenMemHandle." + - "KV moves P->D over mooncake (kv-p2p-transfer: mooncake); both roles use a bf16 MLA KV cache (prefill --kv-cache-dtype bfloat16, decode bf16). This is the only layout TileRT 0.1.6 can consume from vLLM on ROCm: TileRT infers the layout from the cache tensor stride and accepts 1152 B/token (bf16) or 656 B/token (fp8_ds_mla); vLLM's ROCM_AITER_MLA_SPARSE backend has no fp8_ds_mla and its plain fp8 writes a flat 576 B/token row that the connector rejects at register_kv_caches. max-model-len 1048576 on both roles (GLM-5.3's full context window); MTP wired via spec-decoding=mtp: TileRT decode drafts at depth 3 (DECODE_MTP_SIZE, the only depth the ROCm GLM profile builds, and the key the golden acceptance curve is read at), while the vLLM prefill rank only materialises the MTP layer's KV and runs speculative-config mtp with 1 draft token. TileRT decode is bs=1 only, so conc-list is [1]; topology 1 prefill node (TP8) + 1 decode node (TP8)." + - "Orchestration: runners/launch_mi355x-amds.sh dispatches framework=tilert through the existing amd_utils chain (submit.sh -> job.slurm -> server.sh -> new server_tilert.sh); job.slurm selects the image per rank and a tilert env block, setup_deps.sh gains a per-role tilert install branch, models_tilert.yaml holds the GLM-5.3 prefill environment (the VLLM_ROCM_USE_AITER* block) and extra-flag hooks. TileRT decode weights are converted once from the HF checkpoint under a flock and cached next to it on the shared model volume. Cross-node teardown uses the amd_utils router-port-close barrier." + - "The pd_router runs with --parser none and a 1800-second queue timeout so requests wait on the bs=1 decode engine instead of failing fast; AgentX traces are filtered at the 1048576-token full GLM context both roles are launched with; AIPerf scrapes only the vLLM prefill Prometheus endpoint (TileRT decode exposes no /metrics route)." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3330 + +- config-keys: + - glm5.3-fp8-mi355x-tilert-agentic + scenario-type: + - agentic-coding + description: + - "Set RDMAV_FORK_SAFE=1 on both TileRT ranks for AgentX, preventing silent RDMA-to-TCP fallback that degrades KV-transfer latency and TTFT; TPOT is unaffected." + - "为 AgentX 的 TileRT 两侧设置 RDMAV_FORK_SAFE=1,避免静默从 RDMA 回退到 TCP 导致 KV 传输延迟和 TTFT 恶化;TPOT 不受影响。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3330 + +- config-keys: + - glm5.3-fp8-mi355x-tilert-agentic + scenario-type: + - agentic-coding + description: + - "Serve GLM-5.3's full 1048576-token context on both TileRT ranks with the bf16 MLA wire layout, gpu-memory-utilization 0.85 on the vLLM prefill, and the container-creation barrier timeout taken from CONTAINER_BARRIER_TIMEOUT (5400s) instead of a hardcoded 1800s. bf16 is the only layout TileRT 0.1.6 can consume from vLLM on ROCm: TileRT infers the layout from the cache tensor stride and accepts 1152 B/token (bf16) or 656 B/token (fp8_ds_mla), while ROCM_AITER_MLA_SPARSE has no fp8_ds_mla and its plain fp8 writes a flat 576 B/token row that the connector rejects at register_kv_caches. Per rank at this context (287.98 GiB MI355X): decode holds 90.72 GiB of weights, a 93.25 GiB engine cache window and a 99.06 GiB PD receive buffer; prefill holds 90.45 GiB of weights, 40.3 GiB of profiling/non-torch memory, 91.71 GiB of vLLM KV and a 99.06 GiB PD staging buffer on TP rank 0 outside vLLM's budget. Both PD buffers are dense in max_seq_len and must live in pinned host memory for this context to start: on the GPU the decode side is node-marginal (torch.OutOfMemoryError on one node, 99.06 GiB requested with 95.94 GiB free; served on another) and the prefill side cannot fit at any utilization. setup_deps.sh therefore applies patches/tilert-0.1.6-pd-buffers-in-dram.patch to the pip-installed tilert 0.1.6 at container start on both ranks (engine-patch waiver docs/waiver/3330.md; vLLM stays unpatched), placing both PD buffers in 2 MiB-backed pinned host memory behind TILERT_PD_BUFFER_DEVICE=cpu. The backing is verified before registration because the ionic RDMA VFs cap 4 KiB-page registrations at ~3.9 GiB per HCA (a ~2^20 page-table-entry budget), while 2 MiB-backed regions of 100 GiB register and take cross-node mooncake writes at ~21 GiB/s. The patch is removed when a TileRT release carries DRAM PD buffers. The barrier change covers cold image pulls: each rank pulls its own ~32 GB image, and the decode image alone took about 25 minutes of the old 30-minute budget on an uncached node. The launcher also passes sync.py --grace 60: the barrier's post-pass port-hold defaults to max(60, timeout/2) seconds in the foreground, which at 5400s was a 45-minute idle sleep on every rank (jobs 45373/45374)." + - "以 bf16 MLA 线上布局在两侧 TileRT rank 提供 GLM-5.3 完整的 1048576 上下文,vLLM prefill 的 gpu-memory-utilization 设为 0.85,容器创建屏障超时改为由 CONTAINER_BARRIER_TIMEOUT(5400 秒)提供而非硬编码 1800 秒。bf16 是 TileRT 0.1.6 在 ROCm 上唯一能从 vLLM 读取的布局:TileRT 依据缓存张量步长推断布局,只接受每 token 1152 字节(bf16)或 656 字节(fp8_ds_mla),而 ROCM_AITER_MLA_SPARSE 没有 fp8_ds_mla,其普通 fp8 写出每 token 576 字节的扁平行,会在 register_kv_caches 被 connector 拒绝。该上下文下每 rank(287.98 GiB 的 MI355X):decode 侧含 90.72 GiB 权重、93.25 GiB 引擎缓存窗口和 99.06 GiB 的 PD 接收缓冲;prefill 侧含 90.45 GiB 权重、40.3 GiB 分析/非 torch 内存、91.71 GiB vLLM KV,以及 TP rank 0 上位于 vLLM 预算之外的 99.06 GiB PD 暂存缓冲。两个 PD 缓冲均按 max_seq_len 密集分配,必须放在锁页主机内存中该上下文才能启动:放在 GPU 上时 decode 侧处于节点边缘(一节点 OOM,请求 99.06 GiB 而仅剩 95.94 GiB;另一节点可服务),prefill 侧在任何利用率下都放不下。因此 setup_deps.sh 在两侧容器启动时对 pip 安装的 tilert 0.1.6 应用 patches/tilert-0.1.6-pd-buffers-in-dram.patch(引擎补丁豁免 docs/waiver/3330.md;vLLM 未改动),在 TILERT_PD_BUFFER_DEVICE=cpu 下将两个 PD 缓冲放入 2 MiB 大页支撑的锁页主机内存。注册前会校验大页覆盖:ionic RDMA VF 对 4 KiB 页的注册上限约为每 HCA 3.9 GiB(约 2^20 条页表项),而 2 MiB 大页区域可注册 100 GiB 并以约 21 GiB/s 接收跨节点 mooncake 写入。TileRT 发布自带 DRAM PD 缓冲的版本后移除该补丁。屏障改动用于覆盖镜像冷拉取:每个 rank 各拉取约 32 GB 的镜像,仅 decode 镜像在未缓存节点上就实测约占旧 30 分钟预算的 25 分钟。启动脚本同时向 sync.py 传入 --grace 60:屏障通过后默认在前台保持端口 max(60, timeout/2) 秒,在 5400 秒下即每个 rank 空转 45 分钟(作业 45373/45374)。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3330 + +- config-keys: + - glm5.3-fp8-mi355x-tilert-agentic + scenario-type: + - agentic-coding + description: + - "Add glm5.3* to resolve_trace_source's 1M-context model pattern so GLM-5.3 AgentX uses the unfiltered semianalysis_cc_traces_weka_062126 corpus instead of the 256k-filtered variant. Explicit WEKA_LOADER_OVERRIDE behavior is unchanged." + - "将 glm5.3* 加入 resolve_trace_source 的 1M 上下文模型匹配分支,使 GLM-5.3 AgentX 默认使用未截断的 semianalysis_cc_traces_weka_062126 语料,而非 256k 过滤版本。显式 WEKA_LOADER_OVERRIDE 的行为保持不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3330 + +- config-keys: + - glm5.3-fp8-mi355x-tilert-agentic + scenario-type: + - agentic-coding + description: + - "Replace the PD-buffers-in-DRAM route with tilert==0.1.6.post1 on both TileRT ranks (PyPI, 2026-09-22): the shipped wheel shards the mooncake receive buffer (decode, receive_server.py) and the connector staging buffer (prefill TP rank 0, prefill_connector.py) by layer across the eight devices (TILERT_PD_SHARDS, default 8, capped at device_count; layer lid on device lid % 8; the glm5_2 profile has 79 layers, so the heaviest shard holds 10 of them: 99.06 GiB x 10/79 = 12.54 GiB per card) instead of one dense 99.06 GiB buffer per role, and both fit on the GPU at the full 1048576-token context without host memory. Each shard is registered with mooncake on its own device, hello_layout carries per-shard bases and rdma_plan writes layer by layer into the matching remote shard, so the KV path is device-to-device again (81 GB in 751 ms = 108 GB/s on 2x8 MI350X per CrimsonDump/InferenceX@56a1e5a, against the 20.9 GiB/s measured for the host-to-host hop) and mla_nsa.convert() dequantises each layer on the shard that received it rather than on cuda:0. patches/tilert-0.1.6-pd-buffers-in-dram.patch, docs/waiver/3330.md and TILERT_PD_BUFFER_DEVICE (the AgentX recipe, server_tilert.sh, job.slurm, setup_deps.sh, including the 0.1.6 pin guard and the patch apt package) are removed; neither image is modified. Router metadata 0.1.6 -> 0.1.6.post1: pd_router.py is one of the six files post1 changes, its fixed 600 s request timeouts to vLLM and decode become TILERT_PD_HTTP_TIMEOUT_S (default 3600). GPU_MEM_UTIL stays 0.85; the 12.54 GiB staging shard sits outside vLLM's budget next to the ~6.3 GiB non-torch baseline. server_tilert.sh now honours EVAL_ONLY/RUN_EVAL on the AgentX path (GSM8K on the same router instead of an unconditional trace replay) and exports PORT for benchmark_lib's run_lm_eval; evals run with real MTP verification since simulated acceptance is already disabled under those flags. Measured by CrimsonDump on 2x8 MI350X at this context with bf16 KV, 3600 s AgentX at concurrency 1: submission_valid true, 239 successful requests, 0 errors, TTFT p50 5711.8 ms, ITL p50 2.18 ms, peak device memory 202.1 GiB per decode card and 269.2 GiB per prefill card of 287.69 GiB." + - "在两侧 TileRT rank 上以 tilert==0.1.6.post1(PyPI,2026-09-22)取代 PD 缓冲进 DRAM 的方案:正式 wheel 把 mooncake 接收缓冲(decode,receive_server.py)与 connector 暂存缓冲(prefill TP rank 0,prefill_connector.py)按层分片到 8 张卡(TILERT_PD_SHARDS,默认 8,以 device_count 为上限;第 lid 层落在第 lid % 8 张卡;glm5_2 profile 共 79 层,最重分片含 10 层:99.06 GiB x 10/79 = 每卡 12.54 GiB),不再是每个角色一块致密的 99.06 GiB 缓冲,于是在完整 1048576 上下文下两者都能放进 GPU,无需主机内存。各分片在所在卡上向 mooncake 注册,hello_layout 携带各分片基址,rdma_plan 逐层写入对应的远端分片,KV 通路重回设备到设备(2x8 MI350X 上 81 GB / 751 ms = 108 GB/s,见 CrimsonDump/InferenceX@56a1e5a;对比主机到主机实测 20.9 GiB/s),mla_nsa.convert() 在收到该层的分片上反量化而非集中于 cuda:0。移除 patches/tilert-0.1.6-pd-buffers-in-dram.patch、docs/waiver/3330.md 与 TILERT_PD_BUFFER_DEVICE(AgentX 配方、server_tilert.sh、job.slurm、setup_deps.sh,含 0.1.6 版本锁定检查与 patch apt 包);两侧镜像均无改动。router 元数据 0.1.6 -> 0.1.6.post1:pd_router.py 是 post1 改动的六个文件之一,其对 vLLM 与 decode 固定的 600 秒请求超时改为 TILERT_PD_HTTP_TIMEOUT_S(默认 3600)。GPU_MEM_UTIL 保持 0.85;12.54 GiB 暂存分片与约 6.3 GiB 非 torch 基线同在 vLLM 预算之外。server_tilert.sh 现在在 AgentX 路径也遵循 EVAL_ONLY/RUN_EVAL(在同一 router 上跑 GSM8K,而非无条件回放轨迹),并为 benchmark_lib 的 run_lm_eval 导出 PORT;这些标志下模拟接受已关闭,评测使用真实 MTP 验证。CrimsonDump 在 2x8 MI350X、该上下文、bf16 KV 下实测 3600 秒 AgentX 并发 1:submission_valid true、239 条成功、0 错误、TTFT p50 5711.8 ms、ITL p50 2.18 ms,decode 每卡峰值 202.1 GiB、prefill 每卡 269.2 GiB(卡容量 287.69 GiB)。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3366 diff --git a/runners/launch_mi355x-amds.sh b/runners/launch_mi355x-amds.sh index e27c0178d0..9920bb9d90 100644 --- a/runners/launch_mi355x-amds.sh +++ b/runners/launch_mi355x-amds.sh @@ -47,6 +47,13 @@ if [[ "$IS_MULTINODE" == "true" ]]; then export OSL="$OSL" check_env_vars BENCHMARK_LOGS_DIR + # cleanup_and_save_logs below removes BENCHMARK_LOGS_DIR wholesale. A profile + # that points it at the checkout (or a parent of it) deletes the workspace + # and every result just copied into it; sweep 35704948491 did exactly that. + if [[ "$BENCHMARK_LOGS_DIR" == "$GITHUB_WORKSPACE" || "$GITHUB_WORKSPACE" == "$BENCHMARK_LOGS_DIR"/* ]]; then + echo "ERROR: BENCHMARK_LOGS_DIR ($BENCHMARK_LOGS_DIR) must not be the checkout ($GITHUB_WORKSPACE) or contain it" >&2 + exit 1 + fi mkdir -p "$BENCHMARK_LOGS_DIR" sudo rm -rf "$BENCHMARK_LOGS_DIR/logs" 2>/dev/null || true @@ -74,7 +81,7 @@ if [[ "$IS_MULTINODE" == "true" ]]; then fi SCRIPT_NAME="${EXP_NAME%%_*}_${PRECISION}_mi355x_${FRAMEWORK}.sh" - if [[ "$FRAMEWORK" == "sglang-disagg" ]] || [[ "$FRAMEWORK" == "vllm-disagg" ]] || [[ "$FRAMEWORK" == "atom-disagg" ]]; then + if [[ "$FRAMEWORK" == "sglang-disagg" ]] || [[ "$FRAMEWORK" == "vllm-disagg" ]] || [[ "$FRAMEWORK" == "atom-disagg" ]] || [[ "$FRAMEWORK" == "tilert" ]]; then # Agentic recipes under multi_node/agentic/ export the HiCache tunables; # fixed-seq-len recipes live at the multi_node/ root. if [[ "${SCENARIO_SUBDIR}" == "agentic/" ]]; then @@ -87,6 +94,16 @@ if [[ "$IS_MULTINODE" == "true" ]]; then fi JOB_ID=$(bash "benchmarks/${BENCHMARK_SUBDIR}/${SCRIPT_NAME}") + # An empty JOB_ID means the recipe or submit.sh failed before sbatch. The + # wait loop below would then poll for slurm_job-.out forever, because its + # liveness guard degenerates to `grep -q ""` and matches any job this user + # has queued. Fail here instead of burning the job's whole time limit. + if [[ -z "${JOB_ID//[[:space:]]/}" ]]; then + echo "ERROR: benchmarks/${BENCHMARK_SUBDIR}/${SCRIPT_NAME} returned no Slurm job id;" \ + "the recipe or submit.sh failed before sbatch (see its stderr above)" >&2 + exit 1 + fi + LOG_FILE="$BENCHMARK_LOGS_DIR/slurm_job-${JOB_ID}.out" sleep 10