diff --git a/inferencex-e2e/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh b/inferencex-e2e/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh deleted file mode 100644 index 82e81aee32..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh +++ /dev/null @@ -1,172 +0,0 @@ -#!/usr/bin/env bash - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -source "$SCRIPT_DIR/../../benchmark_lib.sh" - -check_env_vars \ - CONC_LIST \ - ISL \ - OSL \ - IMAGE \ - SPEC_DECODING \ - MODEL_PATH \ - MODEL_NAME \ - PREFILL_NUM_WORKERS \ - PREFILL_TP \ - PREFILL_EP \ - PREFILL_DP_ATTN \ - DECODE_NUM_WORKERS \ - DECODE_TP \ - DECODE_EP \ - DECODE_DP_ATTN \ - PREFILL_NODES \ - DECODE_NODES \ - RANDOM_RANGE_RATIO \ - DURATION \ - MODEL_PREFIX \ - PRECISION \ - RESULT_FILENAME \ - KV_OFFLOADING \ - IS_AGENTIC \ - FRAMEWORK \ - PREFILL_IMAGE - -if [[ -n "$SLURM_JOB_ID" ]]; then - echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" -fi - -set -x - -cd "$GITHUB_WORKSPACE/benchmarks/multi_node/amd_utils" || exit 1 - -export TIME_LIMIT=08:00:00 -export MODEL_PATH=$MODEL_PATH -export MODEL_NAME=$MODEL_NAME -export CONTAINER_IMAGE=$IMAGE -export PREFILL_IMAGE - -export RESULT_FILENAME - -if [[ "$PREFILL_NODES" -ne 1 || "$DECODE_NODES" -ne 1 || \ - "$PREFILL_NUM_WORKERS" -ne 1 || "$DECODE_NUM_WORKERS" -ne 1 ]]; then - echo "Error: tilert supports exactly 1 prefill node/worker + 1 decode node/worker" \ - "(got PREFILL_NODES=$PREFILL_NODES x$PREFILL_NUM_WORKERS, DECODE_NODES=$DECODE_NODES x$DECODE_NUM_WORKERS)" >&2 - exit 1 -fi - -if [[ "$KV_OFFLOADING" != "none" ]]; then - echo "Error: tilert has no KV offload backend; kv-offloading must be 'none' (got '$KV_OFFLOADING')" >&2 - exit 1 -fi - -# TileRT configuration. Every value is explicit here: server_tilert.sh -# validates each one with check_env_vars and supplies no defaults of its own. -export TILERT_VERSION=0.1.6.post2 -export TILERT_PROFILE=glm5_2 # decode_server --model (TileRT model profile) -export TILERT_MODEL_TYPE=glm-5 # weight_converter --model_type (fallback converter) -export TILERT_MODEL_PKG=glm_5_2_rocm # per-model converter package, preferred when importable -export SERVED_MODEL_NAME=glm5_2 -# GLM-5.3's full context window (config.json max_position_embeddings), as every -# in-tree GLM-5.2 recipe serves. (202752 was GLM-5.1's, inherited from the B200 -# TileRT recipe this mirrors.) -# -# Memory at this context, per rank, bf16 wire layout (verified against the -# tilert 0.1.6 and vLLM 0.24.0 sources and the MI355X logs, 287.98 GiB cards; -# the undivided PD buffer sizes are what 0.1.6 allocated on one card): -# decode : weights 90.72 GiB + engine cache window 93.25 GiB -# + PD receive buffer 99.06 GiB (receive_server.py, dense in max_seq_len) -# prefill: weights 90.45 GiB + profiling/non-torch 40.3 GiB + vLLM KV 91.71 GiB -# + PD staging buffer 99.06 GiB (prefill_connector.py, TP rank 0, -# allocated OUTSIDE vLLM's gpu-memory-utilization budget) -# Undivided, neither side starts: the decode rank is node-marginal (~283 of -# 288 GiB) and the prefill rank cannot fit at any utilization (~321 GiB). -# tilert 0.1.6.post1 keeps both buffers on the GPU but shards them by layer -# across the eight devices (layer lid on device lid % 8, TILERT_PD_SHARDS, -# default on), so each card holds 12.54 GiB instead of 99.06 GiB on one. -# convert() dequantises each layer on the device that received it, which spreads -# its transients too (108.6 KiB/token, 82.9 GiB at 800k tokens) instead of -# leaving them on cuda:0. Measured on 2x8 MI350X at this context with bf16 KV: -# decode peaks at 202.1 GiB per card, prefill at 269.2 GiB of 287.69 GiB, and -# the KV path stays device-to-device at 108 GB/s (81 GB in 751 ms, 54% of the -# 4x400 GbE line rate). No host hop and no patch: the wheel runs as shipped. -export TILERT_MAX_MODEL_LEN=1048576 -export TILERT_TRANSPORT=mooncake -export TILERT_PARSER=none -export TILERT_RDMA_STRICT=0 -export TILERT_CONVERT_LOCK_WAIT=21600 -export TILERT_SIMULATE_ACC_METHOD=match-expected -export TILERT_WEIGHTS_DIR="/models/${MODEL_NAME}-tilert-tp${DECODE_TP}" -# bf16 MLA KV on both roles. This is the only layout TileRT 0.1.6 can consume -# from vLLM on ROCm: MlaNsaProfile.classify_layers infers the layout from the -# cache tensor stride and accepts exactly 1152 B/token (bf16) or 656 B/token -# (fp8_ds_mla). vLLM's ROCM_AITER_MLA_SPARSE backend has no fp8_ds_mla; its -# plain "fp8" writes a flat 576 B/token row, which the connector rejects at -# register_kv_caches. Explicit bfloat16 rather than auto so the stride does not -# depend on the model dtype. Never float16: it passes the 1152 B check and is -# then read as bf16. -export PREFILL_KV_DTYPE=bfloat16 -# The ROCm backend supports block sizes [1, 64] and vLLM picks 1, which makes -# the connector's KI plane copy fail and MLA address the wrong rows. -export PREFILL_BLOCK_SIZE=64 -export DECODE_KV_DTYPE=bf16 -# The PD staging shard sits outside vLLM's budget, so vLLM needs 90.45 (weights) -# + 40.3 (profiling) + 91.71 GiB (KV for one 1048576-token request) = 222.5 GiB -# inside it: 0.85 x 287.98 = 244.8 GiB leaves 22 GiB of KV margin and 43 GiB -# outside the budget for the 12.54 GiB staging shard plus the ~6.3 GiB non-torch -# baseline measured on the decode OOM node (287.98 - 95.94 free - 184.17 - 1.58 -# reserved). 0.75 (216 GiB) refuses with "91.71 GiB KV cache is needed ... -# available 85.25 GiB". -export GPU_MEM_UTIL=0.85 -export SKIP_CONTAINER_BARRIER=0 -# Two images, one per rank, ~32 GB each. On a node that has neither cached the -# pull alone outlasts the SGLang path's 300s default and the 1800s this script -# used to hardcode, and the rank that comes up first waits out the whole -# timeout while its peer is still pulling. -export CONTAINER_BARRIER_TIMEOUT=5400 -export ROUTER_PORT=30000 -export PREFILL_PORT=8000 -export DECODE_CTRL_PORT=5556 -export DECODE_HTTP_PORT=5557 -export DECODE_WAIT=7200 # prefill waits for the decode ctrl port -export PREFILL_WAIT=3600 # prefill waits for its own vLLM port -export ROUTER_WAIT=10800 # decode waits for the router port to open - -if [[ "$SPEC_DECODING" == "mtp" ]]; then - # TileRT decode drafts at depth 3 (the only depth the ROCm GLM profile - # builds) and the golden acceptance curve is keyed on it. The vLLM prefill - # rank only has to materialise the MTP layer's KV, so it runs at 1. - export DECODE_MTP_SIZE=3 - export PREFILL_SPEC_TOKENS=1 -else - export DECODE_MTP_SIZE=0 - export PREFILL_SPEC_TOKENS=0 -fi -export TILERT_QUEUE_TIMEOUT=1800 # requests wait on the bs=1 decode engine -export THINKING_MODE=thinking_on - -if [[ "$PREFILL_EP" -ne 1 || "$DECODE_EP" -ne 1 || \ - "$PREFILL_DP_ATTN" == "true" || "$DECODE_DP_ATTN" == "true" ]]; then - echo "Error: tilert runs pure TP8 on both roles; ep must be 1 and dp-attn false" >&2 - exit 1 -fi -export PREFILL_ENABLE_EP=false -export PREFILL_ENABLE_DP=false -export DECODE_ENABLE_EP=false -export DECODE_ENABLE_DP=false - -JOB_ID=$(bash ./submit.sh $PREFILL_NODES \ - $PREFILL_NUM_WORKERS \ - $DECODE_NODES \ - $DECODE_NUM_WORKERS \ - $ISL $OSL "${CONC_LIST// /x}" inf \ - ${PREFILL_ENABLE_EP} ${PREFILL_ENABLE_DP} \ - ${DECODE_ENABLE_EP} ${DECODE_ENABLE_DP} \ - ${PREFILL_TP} ${DECODE_TP} \ - ${RANDOM_RANGE_RATIO}) - -if [[ $? -ne 0 ]]; then - echo "Failed to submit job" >&2 - exit 1 -fi - -echo "$JOB_ID" diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/env.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/env.sh index ba48879cb4..04adb875ac 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/env.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/env.sh @@ -2,21 +2,12 @@ source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only check_env_vars ENGINE -# MoRI-IO queue-pair tuning, the UCX RoCE GID index, SGLang router logging and the -# SGLang decode cuda-graph NCCL workaround. Only the SGLang MoRI KV path -# below reads these. ENGINE=tilert moves KV over mooncake and starts no SGLang -# router, so it is neither given nor reads them: validating them there would force -# the recipe to invent MoRI tuning for a transport it never uses. -if [[ "$ENGINE" != "tilert" ]]; then - check_env_vars \ - MORI_IO_SQ_BACKOFF_TIMEOUT_US MORI_IO_QP_MAX_SEND_WR MORI_IO_QP_MAX_CQE MORI_IO_QP_MAX_SGE MORI_IO_TC_DISABLE \ - UCX_IB_GID_INDEX MORI_APP_LOG_LEVEL SGLANG_ROUTER_STDOUT_LOGS TORCH_NCCL_BLOCKING_WAIT NCCL_BLOCKING_WAIT \ - SGLANG_OPT_USE_AITER_INDEXER -fi -# Dual-engine environment setup for multi-node disaggregated serving. -# -# ENGINE=sglang-disagg or tilert selects the engine-specific block. -# +# SGLang MoRI environment. +check_env_vars \ + MORI_IO_SQ_BACKOFF_TIMEOUT_US MORI_IO_QP_MAX_SEND_WR MORI_IO_QP_MAX_CQE MORI_IO_QP_MAX_SGE MORI_IO_TC_DISABLE \ + UCX_IB_GID_INDEX MORI_APP_LOG_LEVEL SGLANG_ROUTER_STDOUT_LOGS TORCH_NCCL_BLOCKING_WAIT NCCL_BLOCKING_WAIT \ + SGLANG_OPT_USE_AITER_INDEXER + # REQUIRED ENVIRONMENT VARIABLES: # IBDEVICES - RDMA/InfiniBand device names (e.g., ionic_0,ionic_1,... or mlx5_0,mlx5_1,...) # Set by runner or auto-detected from hostname. @@ -119,10 +110,6 @@ else fi fi -if [[ "$ENGINE" == "tilert" ]]; then - echo "[INFO] tilert: IBDEVICES=$IBDEVICES NCCL_SOCKET_IFNAME=$NCCL_SOCKET_IFNAME NCCL_IB_HCA=$NCCL_IB_HCA" - -else export SGLANG_USE_AITER=1 export AITER_LOG_LEVEL=ERROR @@ -237,5 +224,3 @@ else export GPU_MAX_HW_QUEUES=2 fi fi - -fi diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm b/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm index 455e9d3cb0..5051dd793e 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm @@ -36,8 +36,6 @@ echo "" # at runtime, but the CWD remains the submit-time directory (amd_utils/). if [[ "$ENGINE" == "atom-disagg" ]]; then MODELS_YAML="$(pwd)/models_atom.yaml" -elif [[ "$ENGINE" == "tilert" ]]; then - MODELS_YAML="$(pwd)/models_tilert.yaml" else MODELS_YAML="$(pwd)/models.yaml" fi @@ -52,16 +50,6 @@ if [[ -z "${DOCKER_IMAGE_NAME:-}" ]]; then exit 1 fi -if [[ "$ENGINE" == "tilert" && -z "${PREFILL_IMAGE:-}" ]]; then - echo "Error: ENGINE=tilert requires PREFILL_IMAGE (e.g. PREFILL_IMAGE=vllm/vllm-openai-rocm:nightly- in prefill.additional-settings)." - exit 1 -fi -if [[ "$ENGINE" == "tilert" ]]; then - # server_tilert.sh takes the container-creation barrier timeout from the - # recipe (no 300s default as on the SGLang path); fail here, before sbatch - # work is done, rather than inside the container. - check_env_vars CONTAINER_BARRIER_TIMEOUT -fi # Resolve the models.yaml entry the same way server_sglang.sh does: agentic runs # (IS_AGENTIC) use the '-AgentX' recipe, non-agentic disaggregated runs use @@ -562,41 +550,6 @@ if [[ "$ENGINE" == "atom-disagg" ]]; then -e EXTRA_SERVER_ARGS=\${EXTRA_SERVER_ARGS:-} -e IBDEVICES=${IBDEVICES:-} ) -elif [[ "$ENGINE" == "tilert" ]]; then - DOCKER_ENV_ENGINE=( - -e MODEL_PATH=$DOCKER_MODEL_PATH - -e PREFILL_IMAGE=${PREFILL_IMAGE} - -e TILERT_VERSION=${TILERT_VERSION} - -e TILERT_PROFILE=${TILERT_PROFILE} - -e TILERT_MODEL_TYPE=${TILERT_MODEL_TYPE} - -e TILERT_MODEL_PKG=${TILERT_MODEL_PKG} - -e TILERT_MAX_MODEL_LEN=${TILERT_MAX_MODEL_LEN} - -e TILERT_TRANSPORT=${TILERT_TRANSPORT} - -e TILERT_PARSER=${TILERT_PARSER} - -e TILERT_QUEUE_TIMEOUT=${TILERT_QUEUE_TIMEOUT} - -e TILERT_WEIGHTS_DIR=${TILERT_WEIGHTS_DIR} - -e TILERT_RDMA_STRICT=${TILERT_RDMA_STRICT} - -e TILERT_CONVERT_LOCK_WAIT=${TILERT_CONVERT_LOCK_WAIT} - -e TILERT_SIMULATE_ACC_METHOD=${TILERT_SIMULATE_ACC_METHOD} - -e \"TILERT_EXTRA_ENV=${TILERT_EXTRA_ENV:-}\" - -e SERVED_MODEL_NAME=${SERVED_MODEL_NAME} - -e PREFILL_KV_DTYPE=${PREFILL_KV_DTYPE} - -e PREFILL_BLOCK_SIZE=${PREFILL_BLOCK_SIZE} - -e PREFILL_SPEC_TOKENS=${PREFILL_SPEC_TOKENS} - -e DECODE_KV_DTYPE=${DECODE_KV_DTYPE} - -e GPU_MEM_UTIL=${GPU_MEM_UTIL} - -e PREFILL_PORT=${PREFILL_PORT} - -e DECODE_CTRL_PORT=${DECODE_CTRL_PORT} - -e DECODE_HTTP_PORT=${DECODE_HTTP_PORT} - -e DECODE_WAIT=${DECODE_WAIT} - -e PREFILL_WAIT=${PREFILL_WAIT} - -e ROUTER_WAIT=${ROUTER_WAIT} - -e SKIP_CONTAINER_BARRIER=${SKIP_CONTAINER_BARRIER} - # Golden-acceptance selection on agentic MTP runs; unset elsewhere. - -e THINKING_MODE=${THINKING_MODE:-} - -e IBDEVICES=${IBDEVICES:-} - -e PYTHONPYCACHEPREFIX=/tmp/pycache - ) else DOCKER_ENV_ENGINE=( -e SGLANG_WS_PATH=${WS_PATH} @@ -749,12 +702,6 @@ else fi fi # end: if ENGINE == atom-disagg -RANK_IMAGE= -if [[ \"$ENGINE\" == \"tilert\" && \"\$SLURM_PROCID\" -lt \"$xP\" ]]; then - RANK_IMAGE=\"$PREFILL_IMAGE\" - echo \"[tilert] rank \$SLURM_PROCID is a prefill rank; using PREFILL_IMAGE=\$RANK_IMAGE\" -fi - exec \$DOCKER_CMD run \ --init \ --stop-timeout 10 \ @@ -796,7 +743,7 @@ exec \$DOCKER_CMD run \ ${CLIENT_DOCKER_ENV} \ --name \"$DOCKER_CONT_NAME\" \ --entrypoint \"\" \ - \"\${RANK_IMAGE:-$DOCKER_IMAGE_NAME}\" bash -lc ' + \"$DOCKER_IMAGE_NAME\" bash -lc ' set -o pipefail mkdir -p /run_logs/slurm_job-'\"\$SLURM_JOB_ID\"' '"$RUN_FILE_FULL"' 2>&1 | tee /run_logs/slurm_job-'\"\$SLURM_JOB_ID\"'/server_\$(hostname).log diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/models_tilert.yaml b/inferencex-e2e/benchmarks/multi_node/amd_utils/models_tilert.yaml deleted file mode 100644 index 1697059d54..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/models_tilert.yaml +++ /dev/null @@ -1,6 +0,0 @@ -# Model-owned engine settings for ENGINE=tilert. Everything the caller owns -# (profile, topology, ports, dtypes, draft depth) is set by the recipe, not here. -GLM-5.3: - prefill_env: "VLLM_ROCM_USE_AITER=1 VLLM_ROCM_USE_AITER_MOE=1 VLLM_ROCM_USE_AITER_FUSION_SHARED_EXPERTS=1 VLLM_ENGINE_READY_TIMEOUT_S=10800" - prefill_extra_flags: "" - decode_extra_flags: "" diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh index f304f53b40..a693fb92cc 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh @@ -5,7 +5,6 @@ source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only # Dispatches to the engine-specific server launcher based on ENGINE env var. # ENGINE=sglang-disagg (default) -> server_sglang.sh (SGLang + MoRI) # ENGINE=atom-disagg -> server_atom.sh (ATOM + mooncake) -# ENGINE=tilert -> server_tilert.sh (vLLM prefill + TileRT decode) check_env_vars ENGINE WS_PATH if [[ -f /config/hicache_mc.env ]]; then @@ -20,8 +19,6 @@ echo "[DISPATCHER] ENGINE=$ENGINE WS_PATH=$WS_PATH" if [[ "$ENGINE" == "atom-disagg" ]]; then export ATOM_WS_PATH="$WS_PATH" source "$WS_PATH/server_atom.sh" -elif [[ "$ENGINE" == "tilert" ]]; then - source "$WS_PATH/server_tilert.sh" else source "$WS_PATH/server_sglang.sh" fi diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh index b7351bebef..6a0377a316 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_atom.sh @@ -16,7 +16,6 @@ check_env_vars \ EXTRA_SERVER_ARGS="${EXTRA_SERVER_ARGS:-}" -source $ATOM_WS_PATH/setup_deps.sh source $ATOM_WS_PATH/env_atom.sh # lm-eval with high num_concurrent exhausts the default 1024 FD limit. diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_sglang.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_sglang.sh index 3447d0030e..73af40c0d3 100755 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_sglang.sh +++ b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_sglang.sh @@ -20,7 +20,6 @@ BENCH_MAX_CONC_VALUE=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | t # can resolve formulas like "BENCH_MAX_CONC_VALUE*2" for max_running_requests. export BENCH_MAX_CONC_VALUE -source $SGLANG_WS_PATH/setup_deps.sh source $SGLANG_WS_PATH/env.sh # Install before starting UMBP or serving processes. Early readiness failures must diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_tilert.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/server_tilert.sh deleted file mode 100644 index 37bb8fa6ee..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/server_tilert.sh +++ /dev/null @@ -1,572 +0,0 @@ -#!/bin/bash -# TileRT disaggregated launcher: upstream vLLM ROCm prefill (TileRTConnector, -# kv_producer) + TileRT decode_server + the OpenAI-compatible pd_router. -# Every value below is supplied by the recipe through job.slurm; this script -# validates them and never invents a default for caller-owned configuration. - -source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only -if [[ "${EVAL_ONLY:-}" != true ]]; then - validate_agentic_concurrency "${BENCH_MAX_CONCURRENCY:-}" || exit 1 -fi - -check_env_vars \ - NODE0_ADDR NODE_RANK MODEL_DIR MODEL_NAME MODEL_PATH xP yD IPADDRS \ - DRY_RUN GPUS_PER_NODE PREFILL_TP_SIZE DECODE_TP_SIZE \ - BENCH_INPUT_LEN BENCH_OUTPUT_LEN BENCH_MAX_CONCURRENCY \ - RUN_EVAL EVAL_ONLY EVAL_FRAMEWORK BENCHMARK_LOGS_DIR WS_PATH \ - SLURM_JOB_ID SPEC_DECODING \ - TILERT_PROFILE TILERT_MODEL_TYPE TILERT_MODEL_PKG TILERT_MAX_MODEL_LEN \ - TILERT_TRANSPORT TILERT_PARSER TILERT_QUEUE_TIMEOUT TILERT_WEIGHTS_DIR \ - TILERT_RDMA_STRICT TILERT_CONVERT_LOCK_WAIT TILERT_SIMULATE_ACC_METHOD \ - PREFILL_KV_DTYPE PREFILL_BLOCK_SIZE PREFILL_SPEC_TOKENS DECODE_KV_DTYPE \ - DECODE_MTP_SIZE GPU_MEM_UTIL SERVED_MODEL_NAME \ - DECODE_CTRL_PORT DECODE_HTTP_PORT PREFILL_PORT ROUTER_PORT \ - DECODE_WAIT PREFILL_WAIT ROUTER_WAIT SKIP_CONTAINER_BARRIER \ - CONTAINER_BARRIER_TIMEOUT - -LOG_DIR="/run_logs/slurm_job-${SLURM_JOB_ID}" -SHARED_LOG_DIR="${BENCHMARK_LOGS_DIR}/logs/slurm_job-${SLURM_JOB_ID}" -mkdir -p "$LOG_DIR" - -if [[ "$xP" -ne 1 || "$yD" -ne 1 ]]; then - echo "ERROR: tilert supports exactly 1 prefill + 1 decode worker (got xP=$xP yD=$yD)" >&2 - exit 1 -fi -if [[ "$NODE_RANK" -lt "$xP" ]]; then - TILERT_ROLE=prefill -else - TILERT_ROLE=decode -fi -export TILERT_ROLE - -source "$WS_PATH/setup_deps.sh" -source "$WS_PATH/env.sh" -# benchmark_lib.sh derives AIPERF_DIR from this at source time, so -# it must be set before the library is loaded, not in run_agentic_replay. The -# AgentX replay runs in this container, where the repo is mounted at /workspace. -export INFMAX_CONTAINER_WORKSPACE=/workspace -source /workspace/benchmarks/benchmark_lib.sh - -# Model-specific engine environment (not caller configuration): the prefill -# vLLM env block lives with the model. Everything else is passed in by the recipe. -MODELS_YAML="${WS_PATH}/models_tilert.yaml" -eval "$("$PY" - "$MODELS_YAML" "$MODEL_NAME" <<'PYEOF' -import shlex, sys, yaml -path, name = sys.argv[1], sys.argv[2] -with open(path) as f: - models = yaml.safe_load(f) or {} -if name not in models: - sys.exit(f"model '{name}' is not present in {path}") -m = models[name] or {} -for key, var in (("prefill_env", "TILERT_PREFILL_ENV"), - ("prefill_extra_flags", "TILERT_PREFILL_EXTRA_FLAGS"), - ("decode_extra_flags", "TILERT_DECODE_EXTRA_FLAGS")): - print(f"{var}={shlex.quote(str(m.get(key) or ''))}") -PYEOF -)" || { echo "ERROR: cannot read the tilert model entry for '$MODEL_NAME' from $MODELS_YAML" >&2; exit 1; } -echo "[tilert] model entry '$MODEL_NAME' loaded from $MODELS_YAML" - -export ROUTER_PORT -export SERVED_MODEL_NAME - -PREFILL_SPEC=() -DECODE_MTP=() -if [[ "$SPEC_DECODING" == "mtp" ]]; then - # The prefill rank only has to build the MTP layer's KV; TileRT decode owns - # the draft depth (DECODE_MTP_SIZE), so the two counts differ by design. - PREFILL_SPEC=(--speculative-config "{\"method\":\"mtp\",\"num_speculative_tokens\":${PREFILL_SPEC_TOKENS}}") - # decode_server only accepts depth 3 today, but pass it explicitly so the - # converter's --num_mtp, the golden-curve key and the engine depth agree by - # data flow rather than by coincidence of defaults. - DECODE_MTP=(--with-mtp --num-mtp "$DECODE_MTP_SIZE") -fi - -# The only TileRT recipe on this cluster is AgentX (agentic-coding). -if [[ "${IS_AGENTIC:-0}" != "1" && "${IS_AGENTIC:-}" != "true" && "${SCENARIO_TYPE:-}" != "agentic-coding" ]]; then - echo "ERROR: server_tilert.sh only runs agentic-coding (IS_AGENTIC=${IS_AGENTIC:-} SCENARIO_TYPE=${SCENARIO_TYPE:-})" >&2 - exit 1 -fi - -IFS=',' read -ra IP_ARRAY <<< "$IPADDRS" -PREFILL_HOST="${IP_ARRAY[0]:-$NODE0_ADDR}" -DECODE_HOST="${IP_ARRAY[$xP]:-}" -if [[ -z "$DECODE_HOST" ]]; then - echo "ERROR: cannot resolve the decode node IP from IPADDRS='$IPADDRS' (xP=$xP)" >&2 - exit 1 -fi -host_ip=$(ip route get 1.1.1.1 2>/dev/null | awk '/src/ {print $7}') -host_name=$(hostname) - -echo "[tilert] ROLE=$TILERT_ROLE rank=$NODE_RANK host=$host_name ($host_ip)" -echo "[tilert] PREFILL_HOST=$PREFILL_HOST:$PREFILL_PORT DECODE_HOST=$DECODE_HOST:$DECODE_CTRL_PORT/$DECODE_HTTP_PORT ROUTER=:$ROUTER_PORT" -echo "[tilert] MODEL_PATH=$MODEL_PATH profile=$TILERT_PROFILE served=$SERVED_MODEL_NAME max_len=$TILERT_MAX_MODEL_LEN transport=$TILERT_TRANSPORT kv=${PREFILL_KV_DTYPE}->${DECODE_KV_DTYPE} mtp=${SPEC_DECODING}" - -# Enable libibverbs fork safety on both ranks before any verbs context exists. -# Without it, ibv_fork_init() can fail in these containers while Mooncake -# initialization still reports success, silently falling back from RDMA to TCP. -# Slower prefill-to-decode KV transfer degrades TTFT; TPOT is unaffected. -export RDMAV_FORK_SAFE=1 - -for env_pair in ${TILERT_EXTRA_ENV}; do - export "${env_pair?}" - echo "[tilert][EXTRA_ENV] $env_pair" -done - -log_and_run_bg() { - local label="$1" logfile="$2"; shift 2 - { printf '===== [%s] %s =====\n' "$label" "$(date '+%F %T')" - printf '[cmd]'; printf ' %q' "$@"; printf '\n' - printf '[cwd] %s\n[host] %s\n\n' "$PWD" "$host_name" - } | tee -a "$logfile" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: [$label] not started" - LAST_BG_PID="" - return 0 - fi - "$@" >>"$logfile" 2>&1 & - LAST_BG_PID=$! - echo "[$label] pid=$LAST_BG_PID log=$logfile" -} - -rdma_preflight() { - local warn=0 - echo "[rdma] role=$TILERT_ROLE IBDEVICES=${IBDEVICES:-} NCCL_SOCKET_IFNAME=${NCCL_SOCKET_IFNAME:-}" - local uverbs=(/dev/infiniband/uverbs*) - if [[ -e "${uverbs[0]}" ]]; then - echo "[rdma] verbs devices: ${uverbs[*]}" - else - echo "[rdma] WARNING: /dev/infiniband/uverbs* missing -- the container has no RDMA device nodes (job.slurm passes --device /dev/infiniband)" >&2 - warn=1 - fi - local ml; ml="$(ulimit -l 2>/dev/null)" - if [[ "$ml" == "unlimited" ]]; then - echo "[rdma] memlock: unlimited" - else - echo "[rdma] WARNING: memlock=$ml (not unlimited) -- pinning memory for RDMA may fail (job.slurm passes --ulimit memlock=-1)" >&2 - warn=1 - fi - if command -v ibv_devices >/dev/null 2>&1; then - echo "[rdma] ibv_devices:"; ibv_devices 2>&1 | sed 's/^/[rdma] /' - fi - if (( warn )) && [[ "$TILERT_RDMA_STRICT" == "1" ]]; then - echo "[rdma] TILERT_RDMA_STRICT=1 and preflight did not fully pass -- aborting" >&2 - return 1 - fi - return 0 -} - -stage_tokenizer_files() { - local staged=0 f b - for f in "$MODEL_PATH"/*; do - [[ -f "$f" ]] || continue - b="$(basename "$f")" - [[ "$b" == *.safetensors ]] && continue - [[ "$b" == "model.safetensors.index.json" ]] && continue - [[ -e "$TILERT_WEIGHTS_DIR/$b" ]] && continue - cp -p "$f" "$TILERT_WEIGHTS_DIR/$b" && staged=$((staged+1)) - done - echo "[stage_tokenizer] staged $staged auxiliary file(s) from $MODEL_PATH -> $TILERT_WEIGHTS_DIR" - local missing=() - [[ -f "$TILERT_WEIGHTS_DIR/chat_template.jinja" ]] || missing+=(chat_template.jinja) - [[ -f "$TILERT_WEIGHTS_DIR/tokenizer_config.json" || -f "$TILERT_WEIGHTS_DIR/tokenizer.json" ]] \ - || missing+=("tokenizer.json/tokenizer_config.json") - if (( ${#missing[@]} )); then - echo "[stage_tokenizer] ERROR: $TILERT_WEIGHTS_DIR is missing ${missing[*]}; check that MODEL_PATH=$MODEL_PATH is an HF directory with the tokenizer" >&2 - return 1 - fi - return 0 -} - -_tilert_weights_cached() { - local r - for r in $(seq 0 $((DECODE_TP_SIZE - 1))); do - [[ -f "$TILERT_WEIGHTS_DIR/rank${r}/model.safetensors.index.json" ]] || return 1 - done - # The engine refuses a cache converted without the MTP module (end2end.py - # checks tilert_meta.json num_mtp), but only after loading ~90 GiB of - # weights. Check the same field here so a stale non-MTP cache is - # re-converted instead of failing late. - [[ -f "$TILERT_WEIGHTS_DIR/tilert_meta.json" ]] || return 1 - if [[ "$SPEC_DECODING" == "mtp" ]]; then - "$PY" - "$TILERT_WEIGHTS_DIR/tilert_meta.json" <<'PYEOF' || return 1 -import json, sys -sys.exit(0 if int(json.load(open(sys.argv[1])).get("num_mtp", 0)) >= 1 else 1) -PYEOF - fi - return 0 -} - -convert_weights() { - if _tilert_weights_cached; then - echo "[weight_converter] cache hit (${DECODE_TP_SIZE}/${DECODE_TP_SIZE} rank index.json), skipping conversion: $TILERT_WEIGHTS_DIR" - return 0 - fi - mkdir -p "$TILERT_WEIGHTS_DIR" || { echo "[weight_converter] ERROR: cannot create $TILERT_WEIGHTS_DIR (set TILERT_WEIGHTS_DIR to a writable shared path)" >&2; return 1; } - exec 9>"$TILERT_WEIGHTS_DIR/.convert.lock" - flock -w "$TILERT_CONVERT_LOCK_WAIT" 9 || { - echo "[weight_converter] timed out waiting for the conversion lock (another job still converting?)" >&2; return 1; } - if _tilert_weights_cached; then - echo "[weight_converter] cache produced by a concurrent job, skipping conversion"; exec 9>&-; return 0 - fi - if [[ -n "$(ls -A "$TILERT_WEIGHTS_DIR" 2>/dev/null | grep -v '^\.convert\.lock$')" ]]; then - echo "[weight_converter] leftovers without index.json (previous conversion incomplete); cleaning and re-converting" - find "$TILERT_WEIGHTS_DIR" -mindepth 1 ! -name '.convert.lock' -delete - fi - echo "[weight_converter] $MODEL_PATH -> $TILERT_WEIGHTS_DIR (model_type=$TILERT_MODEL_TYPE)" - local conv_mod conv_args - if "$PY" -c "import tilert.models.${TILERT_MODEL_PKG}.weight_converter" 2>/dev/null; then - conv_mod="tilert.models.${TILERT_MODEL_PKG}.weight_converter" - conv_args=(--model_dir "$MODEL_PATH" --save_dir "$TILERT_WEIGHTS_DIR" - --device "cuda:$((GPUS_PER_NODE - 1))") - [[ "$SPEC_DECODING" == "mtp" ]] && conv_args+=(--num_mtp "$DECODE_MTP_SIZE") - else - conv_mod="tilert.models.preprocess.weight_converter" - conv_args=(--model_type "$TILERT_MODEL_TYPE" --model_dir "$MODEL_PATH" --save_dir "$TILERT_WEIGHTS_DIR") - fi - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $PY -m $conv_mod ${conv_args[*]}" - exec 9>&-; return 0 - fi - echo "[weight_converter] using $conv_mod" - "$PY" -m "$conv_mod" "${conv_args[@]}" \ - 2>&1 | tee "$LOG_DIR/tilert_weight_converter_${host_name}.log" - local rc=${PIPESTATUS[0]} - exec 9>&- - if [[ $rc -ne 0 ]] || ! _tilert_weights_cached; then - echo "[weight_converter] ERROR: conversion failed (rc=$rc, per-rank index.json complete: $(_tilert_weights_cached && echo yes || echo no))" >&2 - return 1 - fi - echo "[weight_converter] conversion done and cached: $TILERT_WEIGHTS_DIR" -} - -start_decode() { - # shellcheck disable=SC2206 - local extra=( ${TILERT_DECODE_EXTRA_FLAGS} ) - if [[ "$SPEC_DECODING" == "mtp" && "$EVAL_ONLY" != "true" && "$RUN_EVAL" != "true" ]]; then - check_env_vars MODEL_PREFIX THINKING_MODE - local curve="${WS_PATH%/benchmarks/*}/infx/golden_al_distribution/${MODEL_PREFIX}_mtp.yaml" - TILERT_SIMULATE_ACC_LEN="$("$PY" - "$curve" "$THINKING_MODE" "$DECODE_MTP_SIZE" <<'PYEOF' -import sys, yaml -path, thinking, tokens = sys.argv[1], sys.argv[2], int(sys.argv[3]) -data = yaml.safe_load(open(path)) -if not isinstance(data, dict) or len(data) != 1: - sys.exit(f"golden curve {path} must hold exactly one model key") -model, modes = next(iter(data.items())) -try: - value = float(modes[thinking][tokens]) -except (KeyError, TypeError, ValueError): - sys.exit(f"no golden acceptance for {model}/{thinking}/{tokens} draft tokens in {path}") -if not 1 <= value <= tokens + 1: - sys.exit(f"golden acceptance {value} out of range for {tokens} draft tokens") -print(f"{value:g}") -PYEOF -)" || { echo "[tilert] ERROR: golden AL lookup failed (curve=$curve)" >&2; exit 1; } - echo "[tilert] golden AL ${TILERT_SIMULATE_ACC_LEN} from $(basename "$curve") ($THINKING_MODE, K=${DECODE_MTP_SIZE})" - fi - - if [[ -n "${TILERT_SIMULATE_ACC_LEN:-}" && "$EVAL_ONLY" != "true" ]]; then - export TILERT_SIMULATE_ACC_LEN - export TILERT_SIMULATE_ACC_METHOD - echo "[decode] simulated acceptance: TILERT_SIMULATE_ACC_LEN=${TILERT_SIMULATE_ACC_LEN}" \ - "method=${TILERT_SIMULATE_ACC_METHOD} (output text is meaningless by design)" - else - unset TILERT_SIMULATE_ACC_LEN TILERT_SIMULATE_ACC_METHOD - echo "[decode] real MTP verification (no simulated acceptance)" - fi - local cmd=("$PY" -m tilert.pd_vllm.decode_server - --engine tilert --model "$TILERT_PROFILE" - --model-weights-dir "$TILERT_WEIGHTS_DIR" - --max-seq-len "$TILERT_MAX_MODEL_LEN" - --kv-cache-dtype "$DECODE_KV_DTYPE" --transport "$TILERT_TRANSPORT" - --ctrl-port "$DECODE_CTRL_PORT" --http-port "$DECODE_HTTP_PORT" - "${DECODE_MTP[@]}" "${extra[@]}") - log_and_run_bg decode "$LOG_DIR/decode_${host_name}.log" "${cmd[@]}" - DECODE_PID=$LAST_BG_PID -} - -start_prefill() { - for env_pair in ${TILERT_PREFILL_ENV}; do - export "${env_pair?}" - echo "[PREFILL_ENV] $env_pair" - done - local served=("$SERVED_MODEL_NAME") - [[ -n "$MODEL_NAME" && "$MODEL_NAME" != "$SERVED_MODEL_NAME" ]] && served+=("$MODEL_NAME") - # shellcheck disable=SC2206 - local extra=( ${TILERT_PREFILL_EXTRA_FLAGS} ) - local kv_cfg - kv_cfg=$(printf '{"kv_connector":"TileRTConnector","kv_connector_module_path":"tilert.pd_vllm.prefill_connector","kv_role":"kv_producer","kv_connector_extra_config":{"tilert_host":"%s","tilert_ctrl_port":%s,"tilert_model":"%s","tilert_max_seq_len":%s,"tilert_transport":"%s"}}' \ - "$DECODE_HOST" "$DECODE_CTRL_PORT" "$TILERT_PROFILE" "$TILERT_MAX_MODEL_LEN" "$TILERT_TRANSPORT") - local cmd=(vllm serve "$MODEL_PATH" - --served-model-name "${served[@]}" --port "$PREFILL_PORT" - --tensor-parallel-size "$PREFILL_TP_SIZE" --max-model-len "$TILERT_MAX_MODEL_LEN" - --enforce-eager --trust-remote-code --return-tokens-as-token-ids - --gpu-memory-utilization "$GPU_MEM_UTIL" --kv-cache-dtype "$PREFILL_KV_DTYPE" - --block-size "$PREFILL_BLOCK_SIZE" - "${PREFILL_SPEC[@]}" - --kv-transfer-config "$kv_cfg" - "${extra[@]}") - log_and_run_bg prefill "$LOG_DIR/prefill_${host_name}.log" "${cmd[@]}" - PREFILL_PID=$LAST_BG_PID -} - -start_router() { - local cmd=(env HIP_VISIBLE_DEVICES= ROCR_VISIBLE_DEVICES= CUDA_VISIBLE_DEVICES= - "$PY" -m tilert.pd_vllm.pd_router - --vllm-url "http://$PREFILL_HOST:$PREFILL_PORT" - --decode "$DECODE_HOST:$DECODE_CTRL_PORT:$DECODE_HTTP_PORT" - --host 0.0.0.0 --port "$ROUTER_PORT" --model-path "$MODEL_PATH" --parser "$TILERT_PARSER" - --queue-timeout "$TILERT_QUEUE_TIMEOUT") - log_and_run_bg router "$LOG_DIR/router_${host_name}.log" "${cmd[@]}" - ROUTER_PID=$LAST_BG_PID -} - -tcp_open() { (exec 3<>"/dev/tcp/$1/$2") 2>/dev/null; } - -wait_for_tcp() { - local host="$1" port="$2" timeout="${3:-600}" pid="${4:-}" - local deadline=$(( SECONDS + timeout )) - until tcp_open "$host" "$port"; do - if [[ -n "$pid" ]] && ! kill -0 "$pid" 2>/dev/null; then - echo "[wait_for_tcp] process $pid exited before $host:$port opened" >&2; return 2 - fi - if [[ $SECONDS -ge $deadline ]]; then - echo "[wait_for_tcp] timeout: $host:$port not open after ${timeout}s" >&2; return 1 - fi - sleep 5 - done - echo "[wait_for_tcp] $host:$port ready" -} - -wait_for_tcp_close() { - local host="$1" port="$2" pid="${3:-}" - while tcp_open "$host" "$port"; do - if [[ -n "$pid" ]] && ! kill -0 "$pid" 2>/dev/null; then - echo "[wait_for_tcp_close] process $pid exited while $host:$port is still open" >&2; return 2 - fi - sleep 10 - done - echo "[wait_for_tcp_close] $host:$port closed" -} - -copy_logs_to_shared() { - [[ "$DRY_RUN" -eq 0 ]] || return 0 - mkdir -p "$SHARED_LOG_DIR" && cp -r "$LOG_DIR"/. "$SHARED_LOG_DIR"/ \ - && echo "Copied $LOG_DIR -> $SHARED_LOG_DIR" \ - || echo "WARNING: failed to copy $LOG_DIR to $SHARED_LOG_DIR" >&2 -} -trap copy_logs_to_shared EXIT - -run_lm_eval_on_router() { - echo "Running lm-eval evaluation on the router..." - local ok=false _attempt - for _attempt in 1 2 3; do - if curl -sf --max-time 10 "http://0.0.0.0:${ROUTER_PORT}/health" >/dev/null 2>&1; then ok=true; break; fi - echo "Eval health check attempt $_attempt failed, retrying in 10s..."; sleep 10 - done - if [[ "$ok" != "true" ]]; then - echo "ERROR: router health check failed after 3 attempts; skipping eval" >&2 - return 1 - fi - local eval_failed=0 - pushd /workspace >/dev/null || return 1 - if [[ -n "${EVAL_CONC:-}" ]]; then - export EVAL_CONCURRENT_REQUESTS="${EVAL_CONC}" - else - export EVAL_CONCURRENT_REQUESTS=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | tail -1) - fi - # run_lm_eval reads the endpoint from PORT (check_env_vars) and names the - # model from MODEL_NAME, which the vLLM prefill also serves next to - # SERVED_MODEL_NAME; MODEL stays the local HF dir for the context lookup. - export PORT="$ROUTER_PORT" - export MODEL="$MODEL_PATH" - export MAX_MODEL_LEN="$TILERT_MAX_MODEL_LEN" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: run_eval --port $ROUTER_PORT (framework=${EVAL_FRAMEWORK}, conc=${EVAL_CONCURRENT_REQUESTS})" - else - run_eval --port "$ROUTER_PORT" - local eval_rc=$? - if [[ $eval_rc -ne 0 ]]; then - echo "ERROR: run_eval exited rc=$eval_rc; preserving failure artifacts" >&2 - eval_failed=1 - else - export TP="${PREFILL_TP_SIZE}" CONC="${EVAL_CONCURRENT_REQUESTS}" EP_SIZE=1 - export PREFILL_TP="${PREFILL_TP_SIZE}" PREFILL_EP=1 PREFILL_NUM_WORKERS="${xP}" - export DECODE_TP="${DECODE_TP_SIZE}" DECODE_EP=1 DECODE_NUM_WORKERS="${yD}" - export DP_ATTENTION=false PREFILL_DP_ATTENTION=false DECODE_DP_ATTENTION=false - export ISL="${BENCH_INPUT_LEN}" OSL="${BENCH_OUTPUT_LEN}" - # As on the SGLang path: rewrite meta_env.json from the exports above, - # then stage unless run_eval already did (eval-only). - rewrite_lm_eval_meta_env - if [[ "$EVAL_ONLY" != "true" ]]; then - append_lm_eval_summary - fi - fi - local eval_copy_dir="$LOG_DIR/eval_results" - if stage_eval_artifacts "$eval_copy_dir" /workspace "${EVAL_RESULT_DIR:-}"; then - echo "Eval artifacts staged in $eval_copy_dir" - else - echo "ERROR: failed to stage eval artifacts in $eval_copy_dir" >&2 - eval_failed=1 - fi - fi - popd >/dev/null || true - return $eval_failed -} - -run_agentic_replay() { - local rc=0 - wait_for_server_ready --port "$ROUTER_PORT" --server-log "$LOG_DIR/router_${host_name}.log" --server-pid "$ROUTER_PID" - cd /workspace || return 1 - - export PORT="$ROUTER_PORT" - export MODEL="$MODEL_PATH" # aiperf --tokenizer (local HF dir) - export SERVED_MODEL_NAME # aiperf --model (name the router/vLLM serve) - check_env_vars DURATION RESULT_FILENAME INFMAX_CONTAINER_WORKSPACE - export MAX_MODEL_LEN="$TILERT_MAX_MODEL_LEN" - # TileRT decode exposes no /metrics route; only the vLLM prefill is scraped. - export AIPERF_SERVER_METRICS_URLS="http://${PREFILL_HOST}:${PREFILL_PORT}/metrics" - export TRANSFORMERS_VERBOSITY=error TOKENIZERS_PARALLELISM=false - # Keep the trace corpus and aiperf's HF downloads on the node's /tmp mount - # instead of the container's ephemeral ~/.cache, as the SGLang client does. - export HF_HOME=/run_logs/hf_cache - - local result_dir="$LOG_DIR/agentic" - local result_filename_base="$RESULT_FILENAME" - mkdir -p "$result_dir" - - # Neither server.sh nor this script runs with errexit; a failed bootstrap - # must not fall through into replay and its misleading cascade. - resolve_trace_source || return 1 - install_agentic_deps || return 1 - - local conc="$BENCH_MAX_CONCURRENCY" conc_result_dir - validate_agentic_concurrency "$conc" || return 1 - echo "Agentic trace replay: conc=$conc" - conc_result_dir="$result_dir/conc_${conc}" - mkdir -p "$conc_result_dir" - export CONC="$conc" USERS="$conc" - build_replay_cmd "$conc_result_dir" - export RESULT_FILENAME="${result_filename_base}_conc${conc}" - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: $REPLAY_CMD" - elif ! run_agentic_replay_and_write_outputs "$conc_result_dir"; then - echo "WARNING: agentic trace replay for conc=$conc failed (replay or validation) after writing available results" >&2 - rc=1 - fi - export RESULT_FILENAME="$result_filename_base" - return $rc -} - -echo "Waiting at the container creation barrier on $host_name" -if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: skipping container creation barrier" -elif [[ "$SKIP_CONTAINER_BARRIER" == "1" ]]; then - echo "SKIP_CONTAINER_BARRIER=1: caller asserts all containers are up" -else - # --grace 60: after the barrier passes, sync.py keeps the port open for - # max(60, timeout/2) seconds in the foreground so a peer one poll behind - # still sees it. At CONTAINER_BARRIER_TIMEOUT=5400 that is a 45-minute idle - # sleep on every rank (jobs 45373/45374 slept 06:28-07:13). Both ranks pass - # within one 5 s poll of each other and the stages below have their own - # readiness waits, so 60 s is plenty. - "$PY" "$WS_PATH/sync.py" barrier \ - --local-ip "${host_ip}" --local-port 5000 --enable-port \ - --node-ips "${IPADDRS}" --node-ports 5000 \ - --wait-for-all-ports --timeout "$CONTAINER_BARRIER_TIMEOUT" --grace 60 \ - || { echo "ERROR: container creation barrier failed after ${CONTAINER_BARRIER_TIMEOUT}s -- the peer rank never opened port 5000." \ - "A cold image pull is the usual cause: this recipe pulls two ~32 GB images, one per rank, and the rank that" \ - "comes up first waits out the whole timeout while the other is still pulling." >&2; exit 1; } -fi - -case "$TILERT_ROLE" in - decode) - echo "${host_name}:${host_ip} is the TileRT Decode Node (Model: ${MODEL_NAME}, profile: ${TILERT_PROFILE})" - rdma_preflight || exit 1 - convert_weights || exit 1 - stage_tokenizer_files || exit 1 - start_decode - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: decode role complete"; exit 0 - fi - echo "Waiting for the router port ${PREFILL_HOST}:${ROUTER_PORT} to open (timeout ${ROUTER_WAIT}s)..." - wait_for_tcp "$PREFILL_HOST" "$ROUTER_PORT" "$ROUTER_WAIT" "$DECODE_PID"; wrc=$? - if [[ $wrc -eq 2 ]]; then - echo "ERROR: decode_server exited before the router came up (see $LOG_DIR/decode_${host_name}.log)" >&2 - tail -50 "$LOG_DIR/decode_${host_name}.log" >&2 || true - copy_logs_to_shared; exit 1 - elif [[ $wrc -ne 0 ]]; then - echo "WARNING: router never opened within ${ROUTER_WAIT}s; shutting down decode" >&2 - kill "$DECODE_PID" 2>/dev/null || true - copy_logs_to_shared; exit 1 - fi - echo "Waiting until the router port closes..." - wait_for_tcp_close "$PREFILL_HOST" "$ROUTER_PORT" "$DECODE_PID"; wrc=$? - if [[ $wrc -eq 2 ]]; then - echo "ERROR: decode_server died while the benchmark was running (see $LOG_DIR/decode_${host_name}.log)" >&2 - tail -50 "$LOG_DIR/decode_${host_name}.log" >&2 || true - copy_logs_to_shared; exit 1 - fi - echo "Killing the decode server" - kill "$DECODE_PID" 2>/dev/null || true - sleep 2 - copy_logs_to_shared - ;; - prefill) - echo "NODE INFO =======================================" - echo "Node List : ${SLURM_JOB_NODELIST:-}" - echo "Node IPs : ${IPADDRS}" - echo "Model : ${MODEL_NAME}" - echo "${host_name}:${host_ip} is the Prefill Node (vLLM + TileRTConnector) and Router Node" - echo "================================================" - rdma_preflight || exit 1 - echo "Waiting for the decode ctrl port ${DECODE_HOST}:${DECODE_CTRL_PORT} (timeout ${DECODE_WAIT}s)..." - if [[ "$DRY_RUN" -eq 0 ]]; then - wait_for_tcp "$DECODE_HOST" "$DECODE_CTRL_PORT" "$DECODE_WAIT" \ - || echo "WARNING: timed out waiting for the decode ctrl port; starting prefill anyway" >&2 - fi - start_prefill - if [[ "$DRY_RUN" -eq 0 ]]; then - wait_for_tcp "$PREFILL_HOST" "$PREFILL_PORT" "$PREFILL_WAIT" "$PREFILL_PID"; wrc=$? - if [[ $wrc -ne 0 ]]; then - echo "ERROR: vLLM prefill did not open ${PREFILL_HOST}:${PREFILL_PORT} (rc=$wrc, see $LOG_DIR/prefill_${host_name}.log)" >&2 - tail -50 "$LOG_DIR/prefill_${host_name}.log" >&2 || true - kill "$PREFILL_PID" 2>/dev/null || true - copy_logs_to_shared; exit 1 - fi - fi - start_router - if [[ "$DRY_RUN" -eq 1 ]]; then - echo "DRY RUN: prefill/router role complete"; exit 0 - fi - echo "Ready for benchmarking on ${host_name}:${host_ip}" - cd "$WS_PATH" || exit 1 - # EVAL_ONLY skips the AgentX replay and runs GSM8K on the same router; - # RUN_EVAL after a replay runs it once the replay has finished. - if [[ "$EVAL_ONLY" == "true" ]]; then - echo "EVAL_ONLY mode: skipping the AgentX replay" - wait_for_server_ready --port "$ROUTER_PORT" --server-log "$LOG_DIR/router_${host_name}.log" --server-pid "$ROUTER_PID" - export TRANSFORMERS_VERBOSITY=error TOKENIZERS_PARALLELISM=false - run_lm_eval_on_router; BENCH_RC=$? - else - run_agentic_replay; BENCH_RC=$? - if [[ "$RUN_EVAL" == "true" ]]; then - run_lm_eval_on_router || BENCH_RC=1 - fi - fi - copy_logs_to_shared - echo "Killing the router and the prefill server" - kill "$ROUTER_PID" "$PREFILL_PID" 2>/dev/null || true - sleep 2 - pkill -f "tilert.pd_vllm.pd_router" 2>/dev/null || true - pkill -f "vllm serve" 2>/dev/null || true - if [[ "$BENCH_RC" -ne 0 ]]; then - echo "ERROR: benchmark/eval reported rc=$BENCH_RC" >&2 - exit "$BENCH_RC" - fi - ;; - *) - echo "ERROR: unknown TILERT_ROLE='$TILERT_ROLE'" >&2; exit 2 ;; -esac - -echo "Script completed successfully" -exit 0 diff --git a/inferencex-e2e/benchmarks/multi_node/amd_utils/setup_deps.sh b/inferencex-e2e/benchmarks/multi_node/amd_utils/setup_deps.sh deleted file mode 100644 index 1a5c5cfacf..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/amd_utils/setup_deps.sh +++ /dev/null @@ -1,136 +0,0 @@ -#!/bin/bash - -source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only -# Install missing disagg dependencies at container start. Each installer is -# idempotent and gated on $ENGINE. - -_SETUP_START=$(date +%s) -_SETUP_INSTALLED=() - -# Pinned by the recipe (TILERT_VERSION); the rest are fixed properties of the -# TileRT 0.1.x runtime rather than caller configuration. -TILERT_PACKAGE=tilert -TILERT_HTTP_DEPS="fastapi uvicorn httpx" -TILERT_TRANSPORT_DEPS="mooncake-transfer-engine-rocm>=0.3.13" -TILERT_TRANSFORMERS_SPEC="transformers>=4.56" - -_tilert_resolve_python() { - if [[ -n "${PY:-}" ]] && command -v "$PY" >/dev/null 2>&1; then :; else - PY="" - local c - for c in python3 python; do command -v "$c" >/dev/null 2>&1 && { PY="$c"; break; }; done - fi - [[ -n "$PY" ]] || { echo "[SETUP] ERROR: neither python3 nor python found"; exit 1; } - export PY - echo "[SETUP] interpreter PY=$PY ($(command -v "$PY"))" -} - -_tilert_installed_version() { - "$PY" - "$1" <<'PYEOF' 2>/dev/null -import sys -from importlib.metadata import version, PackageNotFoundError -try: - print(version(sys.argv[1])) -except PackageNotFoundError: - pass -PYEOF -} - -_tilert_pip() { - "$PY" -m pip install --quiet --no-cache-dir "$@" -} - -_tilert_install_missing() { - local probe="$1"; shift - [[ $# -gt 0 ]] || return 0 - if "$PY" -c "import $probe" 2>/dev/null; then - echo "[SETUP] $probe already present, skipping ($*)" - return 0 - fi - echo "[SETUP] installing $* (probe module '$probe' missing)" - _tilert_pip "$@" || { echo "[SETUP] ERROR: failed to install: $*"; exit 1; } - _SETUP_INSTALLED+=("$*") -} - -install_tilert_container_tools() { - if command -v ip >/dev/null 2>&1 && command -v curl >/dev/null 2>&1 \ - && command -v ibv_devices >/dev/null 2>&1; then - echo "[SETUP] Container RDMA/net tools already present" - return 0 - fi - echo "[SETUP] Installing iproute2 + curl + ibverbs userspace in container..." - apt-get update -q -y && apt-get install -q -y --no-install-recommends \ - iproute2 curl ibverbs-utils libibverbs1 librdmacm1 ibverbs-providers \ - && rm -rf /var/lib/apt/lists/* - if ! command -v ip >/dev/null 2>&1 || ! command -v curl >/dev/null 2>&1; then - echo "[SETUP] ERROR: failed to install iproute2/curl"; exit 1 - fi - _SETUP_INSTALLED+=("iproute2+curl+ibverbs") -} - -_tilert_install_wheel() { - local mode="$1" # full | no-deps - local have; have="$(_tilert_installed_version tilert)" - if [[ "$have" == "$TILERT_VERSION" ]]; then - echo "[SETUP] tilert $have already installed, skipping" - return 0 - fi - [[ -n "$have" ]] && echo "[SETUP] tilert $have installed, switching to pinned $TILERT_VERSION" - if [[ "$mode" == "no-deps" ]]; then - echo "[SETUP] installing $TILERT_PIP_SPEC --no-deps (connector plugin + router on top of the image's vLLM)" - _tilert_pip --no-deps "$TILERT_PIP_SPEC" || { echo "[SETUP] ERROR: failed to install $TILERT_PIP_SPEC (--no-deps)"; exit 1; } - else - echo "[SETUP] installing $TILERT_PIP_SPEC (TileRT ROCm build, official PyPI wheel)" - _tilert_pip "$TILERT_PIP_SPEC" || { echo "[SETUP] ERROR: failed to install $TILERT_PIP_SPEC"; exit 1; } - fi - have="$(_tilert_installed_version tilert)" - [[ "$have" == "$TILERT_VERSION" ]] || { - echo "[SETUP] ERROR: tilert is ${have:-not installed} after install, expected $TILERT_VERSION"; exit 1; } - _SETUP_INSTALLED+=("$TILERT_PACKAGE==$TILERT_VERSION($mode)") -} - -install_tilert_decode() { - install_tilert_container_tools - _tilert_install_wheel full - _tilert_install_missing uvicorn $TILERT_HTTP_DEPS - _tilert_install_missing mooncake.engine "$TILERT_TRANSPORT_DEPS" - _tilert_install_missing transformers "$TILERT_TRANSFORMERS_SPEC" - "$PY" -c "import tilert.pd_vllm.decode_server" 2>/dev/null || { - echo "[SETUP] ERROR: import tilert.pd_vllm.decode_server failed:" - "$PY" -c "import tilert.pd_vllm.decode_server" 2>&1 | tail -3 - exit 1; } - echo "[SETUP] tilert.pd_vllm.decode_server imports OK" -} - -install_tilert_prefill() { - local vllm_v; vllm_v="$(_tilert_installed_version vllm)" - if [[ -z "$vllm_v" ]]; then - echo "[SETUP] ERROR: no vLLM in the prefill image (PREFILL_IMAGE must be a vllm/vllm-openai-rocm image)." - exit 1 - fi - echo "[SETUP] prefill-side vLLM $vllm_v" - install_tilert_container_tools - _tilert_install_wheel no-deps - _tilert_install_missing mooncake.engine "$TILERT_TRANSPORT_DEPS" - "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>/dev/null || { - echo "[SETUP] WARN: import tilert.pd_vllm.prefill_connector failed (vLLM will report again when loading the connector plugin):" - "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>&1 | tail -3; } -} - -if [[ "$ENGINE" == "tilert" ]]; then - check_env_vars TILERT_VERSION - TILERT_PIP_SPEC="$TILERT_PACKAGE==$TILERT_VERSION" - _tilert_resolve_python - case "${TILERT_ROLE:-}" in - decode) install_tilert_decode ;; - prefill) install_tilert_prefill ;; - *) echo "[SETUP] ERROR: ENGINE=tilert needs TILERT_ROLE=decode|prefill (got '${TILERT_ROLE:-}')"; exit 1 ;; - esac -fi - -_SETUP_END=$(date +%s) -if [[ ${#_SETUP_INSTALLED[@]} -eq 0 ]]; then - echo "[SETUP] All dependencies already present ($(( _SETUP_END - _SETUP_START ))s wallclock)" -else - echo "[SETUP] Installed: ${_SETUP_INSTALLED[*]} in $(( _SETUP_END - _SETUP_START ))s" -fi diff --git a/inferencex-e2e/benchmarks/multi_node/runtime_settings.sh b/inferencex-e2e/benchmarks/multi_node/runtime_settings.sh index 1590aac836..faf40b4a2b 100644 --- a/inferencex-e2e/benchmarks/multi_node/runtime_settings.sh +++ b/inferencex-e2e/benchmarks/multi_node/runtime_settings.sh @@ -41,9 +41,7 @@ case "$FRAMEWORK" in fi ;; tilert) - # RUNNER_TYPE selects the AMD block below, so a missing value must fail - # here rather than silently skip it. - check_env_vars GITHUB_WORKSPACE RUNNER_TYPE + check_env_vars GITHUB_WORKSPACE export BENCHMARK_LOGS_DIR="$GITHUB_WORKSPACE" RESULT_DIR=/workspace export GPU_MEM_UTIL=0.75 DECODE_CTRL_PORT=5556 DECODE_HTTP_PORT=5557 PREFILL_PORT=8000 export DECODE_WAIT=3600 PREFILL_WAIT=3600 TILERT_QUEUE_TIMEOUT=0 @@ -52,18 +50,6 @@ case "$FRAMEWORK" in if [[ "$IS_AGENTIC" == 1 || "$IS_AGENTIC" == true ]]; then export TILERT_QUEUE_TIMEOUT=1800 fi - # The MI355X TileRT recipe runs through the shared amd_utils chain - # (submit.sh -> job.slurm -> server.sh -> setup_deps.sh), which validates - # the same orchestration inputs the AMD SGLang arm receives. - # Without them submit.sh exits before sbatch and the launcher never gets a job id. - if [[ "$RUNNER_TYPE" == *mi355x-amds* ]]; then - export SKIP_RDMA_CHECK=0 SKIP_GPU_SANITY=0 - export ROUTER_TYPE=tilert-pd-router ROUTER_PORT=30000 PROXY_PING_PORT=36367 - export HEADNODE_PORT=20000 SERVER_PORT=2584 PROXY_STREAM_IDLE_TIMEOUT=300 - export ENABLE_METRICS=0 PREFILL_ROUTER_POLICY=random DECODE_ROUTER_POLICY=random - export DECODE_MTP_SIZE=0 - export ROCM_PATH=/opt/rocm UCX_HOME=/usr/local/ucx RIXL_HOME=/usr/local/rixl - fi ;; llmd-vllm) export LLMD_CONTAINER_ENGINE=docker VLLM_RANDOMIZE_DP_DUMMY_INPUTS=1 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh new file mode 100644 index 0000000000..e9906ebb2c --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +set -eo pipefail + +source /infmax-workspace/benchmarks/benchmark_lib.sh --validation-only +check_env_vars TILERT_VERSION TILERT_ROLE +case "$TILERT_ROLE" in + prefill|decode|router) ;; + *) echo "Unknown TileRT role: $TILERT_ROLE" >&2; exit 1 ;; +esac + +install_args=(--quiet --no-cache-dir) +if [[ "$TILERT_ROLE" == prefill ]]; then + # Preserve the prefill image's vLLM/Torch dependency set. + install_args+=(--no-deps) +fi +python3 -m pip install "${install_args[@]}" "tilert==$TILERT_VERSION" + +if ! python3 -c 'import mooncake.engine' >/dev/null 2>&1; then + python3 -m pip install --quiet --no-cache-dir 'mooncake-transfer-engine-rocm>=0.3.13' +fi +if [[ "$TILERT_ROLE" != prefill ]]; then + if ! python3 -c 'import uvicorn' >/dev/null 2>&1; then + python3 -m pip install --quiet --no-cache-dir fastapi uvicorn httpx + fi + if ! python3 -c 'import transformers' >/dev/null 2>&1; then + python3 -m pip install --quiet --no-cache-dir 'transformers>=4.56' + fi +fi diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml new file mode 100644 index 0000000000..6ee36f441f --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml @@ -0,0 +1,102 @@ +schema: 2 +name: glm5.3-mi355x-tilert-agentx +model: + path: GLM-5.3 + container: ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 + precision: fp8 +slurm: + time_limit: "08:00:00" +resources: + gpu_type: mi355x + gpus_per_node: 8 +setup_script: glm5.3-tilert-rocm.sh +environment: + TILERT_VERSION: "0.1.6.post2" + RDMAV_FORK_SAFE: "1" + PYTHONDONTWRITEBYTECODE: "1" + NCCL_SOCKET_IFNAME: eno0 + GLOO_SOCKET_IFNAME: eno0 + NCCL_IB_HCA: rdma0,rdma1,rdma2,rdma3,rdma4,rdma5,rdma6,rdma7 +roles: + prefill: + engine: + type: vllm + set_visible_devices: true + container: ghcr.io/tile-ai/tilert-rocm-prefill:0.1.6 + nodes: 1 + workers: 1 + gpus: 8 + env: + TILERT_ROLE: prefill + VLLM_ROCM_USE_AITER: "1" + VLLM_ROCM_USE_AITER_MOE: "1" + VLLM_ROCM_USE_AITER_FUSION_SHARED_EXPERTS: "1" + VLLM_ENGINE_READY_TIMEOUT_S: "10800" + args: + served-model-name: glm5_2 + tensor-parallel-size: 8 + max-model-len: 1048576 + enforce-eager: true + trust-remote-code: true + return-tokens-as-token-ids: true + gpu-memory-utilization: 0.85 + kv-cache-dtype: bfloat16 + block-size: 64 + speculative-config: '{"method":"mtp","num_speculative_tokens":1}' + kv-transfer-config: >- + {"kv_connector":"TileRTConnector","kv_connector_module_path":"tilert.pd_vllm.prefill_connector", + "kv_role":"kv_producer","kv_connector_extra_config":{ + "tilert_model":"glm5_2","tilert_max_seq_len":1048576,"tilert_transport":"mooncake"}} + decode: + engine: tilert + container: ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 + nodes: 1 + workers: 1 + gpus: 8 + env: + TILERT_ROLE: decode + GLM5_AR_N: "2" + args: + model: glm5_2 + model-weights-dir: /models/GLM-5.3-tilert-tp8 + max-seq-len: 1048576 + kv-cache-dtype: bf16 + transport: mooncake + with-mtp: true + num-mtp: 3 +frontend: + type: tilert-router + container_image: ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 + enable_multiple_frontends: false + env: + TILERT_ROLE: router + args: + parser: none + model-path: /model + queue-timeout: 1800 +sbatch_directives: + cpus-per-task: "128" + mem: "0" +srun_options: + mem: "0" + container-writable: "" + container-remap-root: "" +health_check: + max_attempts: 2160 + interval_seconds: 5 +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + CONC: "1" + MODEL: /model + SERVED_MODEL_NAME: glm5_2 + AIPERF_MAX_CONTEXT_LENGTH: "1048576" + RESULT_DIR: /infmax-workspace/LOGS/agentic + AGENTIC_OUTPUT_DIR: /infmax-workspace + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" + HF_HOME: /logs/hf_cache + HF_HUB_OFFLINE: "" + TRANSFORMERS_VERBOSITY: error + TOKENIZERS_PARALLELISM: "false" diff --git a/inferencex-e2e/configs/amd-master.yaml b/inferencex-e2e/configs/amd-master.yaml index 6a3fb3a5a3..704b960b06 100644 --- a/inferencex-e2e/configs/amd-master.yaml +++ b/inferencex-e2e/configs/amd-master.yaml @@ -1438,21 +1438,8 @@ dsv41flash-fp4-mi355x-vllm-agentic-dspark: - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml } - { tp: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/vllm/mi355x-fp4-mtp/agentic.yaml } -# Speculative decoding on an agentic scenario must run with simulated -# synthetic acceptance at the committed golden AL for this model, thinking mode -# and draft length (docs/PR_REVIEW_CHECKLIST.md), and a submission may not -# substitute its own target. Nothing about that target is written here: -# AGENTS.md forbids hard-coding an acceptance length in a master config, so -# server_tilert.sh reads infx/golden_al_distribution/glm5.3_mtp.yaml at launch and -# fails the run if the curve or the draft length is missing. -# The curve is consumed in the same units as every other framework here, and -# AgentX replays run with thinking on. -# -# NOTE FOR REVIEWERS: that curve is GLM-5.2's, copied because no SPEED-Bench run -# on GLM-5.3 exists and the two share a base. It is committed as provisional and -# labelled as such in the file. Flagging it rather than letting it read as a -# measured 5.3 curve -- please say if you would rather see a measured curve, a -# non-MTP agentic entry, or a waiver. +# The golden acceptance curve remains the provisional GLM-5.2-derived curve +# documented in infx/golden_al_distribution/glm5.3_mtp.yaml. glm5.3-fp8-mi355x-tilert-agentic: image: ghcr.io/tile-ai/tilert-rocm-decode:0.1.6 model: zai-org/GLM-5.3 @@ -1476,15 +1463,12 @@ glm5.3-fp8-mi355x-tilert-agentic: dp-attn: false additional-settings: - "PREFILL_IMAGE=ghcr.io/tile-ai/tilert-rocm-prefill:0.1.6" - - "PREFILL_NODES=1" + - "CONFIG_FILE=recipes/glm5.3/tilert/mi355x-fp8/agentx/disagg-1p1d-tp8-mtp.yaml" decode: num-worker: 1 tp: 8 ep: 1 dp-attn: false - additional-settings: - - "DECODE_NODES=1" - - "TILERT_EXTRA_ENV=GLM5_AR_N=2" # User-selected official DeepSeek V4.1 MI35X model preview, pinned by digest. # Official MI350X TP4/EP4 candidate; compare against the separate TP4/EP1 vLLM baseline. diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index cf67501a63..c96b90734d 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -720,6 +720,7 @@ clusters: models: entries: DeepSeek-V4-Pro-0813: {root: it-share-data, dir: DeepSeek-V4-Pro-0813} + GLM-5.3: {root: it-share-data, dir: GLM-5.3} Qwen3.5-397B-A17B-FP8: {root: it-share-data, dir: Qwen3.5-397B-A17B-FP8} scheduler: slurm slurm: diff --git a/inferencex-e2e/infx/launch/drivers/srt/lanes.py b/inferencex-e2e/infx/launch/drivers/srt/lanes.py index e8217eb607..7be02aeb4b 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/lanes.py +++ b/inferencex-e2e/infx/launch/drivers/srt/lanes.py @@ -100,7 +100,10 @@ class SrtLane: long_time=Match(any_of("dsv4"), frameworks=any_of("dynamo-sglang"), agentic=True), ), ("mi355x-amds", LaunchPath.SRT_MULTI): SrtLane( - mounts=(LaneMount(Match(), "aiperf-cache", "/aiperf_mmap_cache"),), + mounts=( + LaneMount(Match(), "aiperf-cache", "/aiperf_mmap_cache"), + LaneMount(Match(frameworks=any_of("tilert")), "it-share-data", "/models"), + ), eval_unsets=( "roles.prefill.args.ep-dispatch-algorithm", "roles.decode.args.ep-dispatch-algorithm", diff --git a/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py b/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py index 45a9f2e153..12dfd6fdb7 100644 --- a/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py +++ b/inferencex-e2e/infx/srt_slurm/synthetic_acceptance.py @@ -28,6 +28,7 @@ "trt": "trtllm", "dynamo-trt": "trtllm", "atom": "atom", + "tilert": "tilert", } SGLANG_VARIABLES = ( "SGLANG_SIMULATE_ACC_LEN", @@ -35,10 +36,15 @@ "SGLANG_SIMULATE_ACC_TOKEN_MODE", ) TRT_VARIABLE = "TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS" +TILERT_VARIABLES = ("TILERT_SIMULATE_ACC_LEN", "TILERT_SIMULATE_ACC_METHOD") def spec_parameters(role: Mapping[str, Any], engine: str) -> dict[str, Any]: args = role.get("args", {}) + if engine == "tilert": + if args.get("with-mtp") is not True: + return {} + return {"method": "mtp", "num_speculative_tokens": args.get("num-mtp")} if engine == "atom": method = args.get("method") if not method: @@ -116,7 +122,11 @@ def build_overrides( environment["MODEL_PREFIX"], spec, environment["THINKING_MODE"], golden_dir ) overrides = [] - variables = {"sglang": SGLANG_VARIABLES, "trtllm": (TRT_VARIABLE,)}.get(engine, ()) + variables = { + "sglang": SGLANG_VARIABLES, + "trtllm": (TRT_VARIABLE,), + "tilert": TILERT_VARIABLES, + }.get(engine, ()) # SRT applies recipe-wide environment after role environment. Keep simulation # role-local so global values cannot override the golden AL or leak into evals. for key in variables: @@ -126,13 +136,27 @@ def build_overrides( if name not in ("agg", "prefill", "decode"): continue prefix = f"roles.{name}" - worker_spec = spec_parameters(role, engine) - if engine == "vllm": + worker_engine = engine + worker_al = al + if engine == "tilert": + selected = role.get("engine", recipe.get("engine", "tilert")) + worker_engine = selected.get("type") if isinstance(selected, Mapping) else selected + # TileRT's first token and draft cache come from real vLLM prefill. + # Only its decode runtime simulates acceptance, using environment + # variables (decode_server has no --simulate-acc-* CLI options). + if name != "decode" or worker_engine != "tilert": + worker_al = None + worker_spec = spec_parameters(role, worker_engine) + if engine == "tilert" and (worker_al is None or not worker_spec): + for key in TILERT_VARIABLES: + if key in (role.get("env") or {}): + overrides += ["--unset", f"{prefix}.env.{key}"] + if worker_engine == "vllm": if not worker_spec: continue - if al is not None: + if worker_al is not None: worker_spec.update( - rejection_sample_method="synthetic", synthetic_acceptance_length=al + rejection_sample_method="synthetic", synthetic_acceptance_length=worker_al ) elif ( worker_spec.get("rejection_sample_method") == "synthetic" @@ -146,22 +170,24 @@ def build_overrides( "--set", f"{prefix}.args.speculative-config={json.dumps(worker_spec)}", ] - elif engine == "atom": + elif worker_engine == "atom": # ATOM forces acceptance with a server flag rather than environment. key = "spec-decode-acceptance-length" - if al is not None and worker_spec: - overrides += ["--set", f"{prefix}.args.{key}={al:g}"] + if worker_al is not None and worker_spec: + overrides += ["--set", f"{prefix}.args.{key}={worker_al:g}"] elif key in (role.get("args") or {}): overrides += ["--unset", f"{prefix}.args.{key}"] - elif al is not None and worker_spec: + elif worker_al is not None and worker_spec: values = ( - (f"{al:g}", "match-expected", "real-draft-token") + (f"{worker_al:g}", "match-expected", "real-draft-token") if engine == "sglang" - else (f"{al - 1:g}",) + else (f"{worker_al:g}", "match-expected") + if engine == "tilert" + else (f"{worker_al - 1:g}",) ) for key, value in zip(variables, values, strict=True): overrides += ["--set", f"{prefix}.env.{key}={json.dumps(value)}"] - else: + elif engine != "tilert": for key in variables: if key in (role.get("env") or {}): overrides += ["--unset", f"{prefix}.env.{key}"] diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py b/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py index aa6cbf5be4..6b43d9ac49 100644 --- a/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py +++ b/inferencex-e2e/infx/tests/srt_slurm/test_synthetic_acceptance.py @@ -50,6 +50,7 @@ def golden_dir(tmp_path: Path) -> Path: ), ("minimaxm3_eagle3.yaml", "minimax-m3", 2.5), ("minimaxm3_eagle3_gqa.yaml", "minimax-m3", 2.6), + ("glm5.3_mtp.yaml", "glm-5.3", 3.2), ]: (directory / filename).write_text( yaml.safe_dump( @@ -289,6 +290,118 @@ def test_atom_forces_golden_acceptance_by_server_flag( assert "spec-decode-acceptance-length" not in evaluated["roles"]["agg"]["args"] +def tilert_recipe() -> dict[str, Any]: + return { + "roles": { + "prefill": { + "engine": "vllm", + "args": { + "speculative-config": '{"method":"mtp","num_speculative_tokens":1}', + }, + }, + "decode": { + "engine": {"type": "tilert"}, + "args": {"with-mtp": True, "num-mtp": 3}, + "env": {"GLM5_AR_N": "2"}, + }, + }, + } + + +def test_tilert_plan_uses_caller_decode_depth_and_keeps_prefill_real( + tmp_path: Path, golden_dir: Path +) -> None: + recipe = tilert_recipe() + recipe["roles"]["decode"]["args"]["num-mtp"] = 2 + recipe["environment"] = { + "KEEP": "global", + "TILERT_SIMULATE_ACC_LEN": "99", + "TILERT_SIMULATE_ACC_METHOD": "stale-method", + } + path = tmp_path / "tilert.yaml" + path.write_text(yaml.safe_dump(recipe)) + commands = plan_commands( + str(path), + "tilert", + ["--set", "roles.decode.args.num-mtp=3"], + {**ENV, "MODEL_PREFIX": "glm5.3", "RUN_EVAL": "true"}, + golden_dir=golden_dir, + ) + assert len(commands) == 1 + result = apply_native(recipe, commands[0]) + assert result["roles"]["decode"]["env"] == { + "GLM5_AR_N": "2", + "TILERT_SIMULATE_ACC_LEN": "3.2", + "TILERT_SIMULATE_ACC_METHOD": "match-expected", + } + assert result["roles"]["decode"]["args"] == {"with-mtp": True, "num-mtp": 3} + assert result["environment"] == {"KEEP": "global"} + assert json.loads(result["roles"]["prefill"]["args"]["speculative-config"]) == { + "method": "mtp", + "num_speculative_tokens": 1, + } + + +@pytest.mark.parametrize( + "environment", + [{"EVAL_ONLY": "true"}, {"IS_AGENTIC": "0"}, {"SPEC_DECODING": "none"}, {}], +) +def test_tilert_real_verification_removes_stale_role_and_global_simulation( + tmp_path: Path, environment: dict[str, str] +) -> None: + recipe = tilert_recipe() + if not environment: + recipe["roles"]["decode"]["args"]["with-mtp"] = False + stale = {"TILERT_SIMULATE_ACC_LEN": "99", "TILERT_SIMULATE_ACC_METHOD": "match-expected"} + recipe["environment"] = {"KEEP": "global", **stale} + for role in recipe["roles"].values(): + role.setdefault("env", {}).update(stale) + recipe["roles"]["prefill"]["engine"] = {"type": "vllm"} + recipe["roles"]["prefill"]["args"]["speculative-config"] = json.dumps( + { + "method": "mtp", + "num_speculative_tokens": 1, + "rejection_sample_method": "synthetic", + "synthetic_acceptance_length": 99, + } + ) + result = apply_native( + recipe, + build_overrides( + recipe, + "tilert", + {**ENV, "MODEL_PREFIX": "glm5.3", **environment}, + golden_dir=tmp_path / "missing", + ), + ) + assert result["environment"] == {"KEEP": "global"} + assert result["roles"]["decode"]["env"] == {"GLM5_AR_N": "2"} + assert result["roles"]["prefill"]["env"] == {} + assert json.loads(result["roles"]["prefill"]["args"]["speculative-config"]) == { + "method": "mtp", + "num_speculative_tokens": 1, + "rejection_sample_method": "block", + } + + +@pytest.mark.parametrize("depth", [None, 7]) +def test_tilert_requires_explicit_measured_decode_depth( + golden_dir: Path, depth: int | None +) -> None: + recipe = tilert_recipe() + if depth is None: + del recipe["roles"]["decode"]["args"]["num-mtp"] + else: + recipe["roles"]["decode"]["args"]["num-mtp"] = depth + with pytest.raises(ValueError, match="positive integer draft length|No golden acceptance"): + build_overrides( + recipe, + "tilert", + {**ENV, "MODEL_PREFIX": "glm5.3"}, + golden_dir=golden_dir, + ) + + @pytest.mark.parametrize( "curve", [