From a708446f5966c858979733829c0f985c0f569782 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 15:26:52 -0500 Subject: [PATCH 1/6] refactor(tilert): port B200 fixed-sequence to native srt-slurm --- .../glm5.1_fp8_b200_tilert-disagg.sh | 34 -- .../configs/tilert-b200-setup.sh | 49 +++ .../b200-fp8/1k1k/disagg-1p1d-tp8-mtp.yaml | 88 +++++ .../b200-fp8/8k1k/disagg-1p1d-tp8-mtp.yaml | 98 ++++++ .../multi_node/tilert_utils/run_node.sh | 332 ------------------ .../multi_node/tilert_utils/setup_deps.sh | 143 -------- .../multi_node/tilert_utils/submit.sh | 195 ---------- inferencex-e2e/configs/nvidia-master.yaml | 12 +- inferencex-e2e/perf-changelog.yaml | 6 + .../runners/launch_b200-nscale-slurm.sh | 33 +- inferencex-e2e/runners/runtime_settings.sh | 1 - 11 files changed, 250 insertions(+), 741 deletions(-) delete mode 100755 inferencex-e2e/benchmarks/multi_node/glm5.1_fp8_b200_tilert-disagg.sh create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.1/tilert/b200-fp8/1k1k/disagg-1p1d-tp8-mtp.yaml create mode 100644 inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.1/tilert/b200-fp8/8k1k/disagg-1p1d-tp8-mtp.yaml delete mode 100755 inferencex-e2e/benchmarks/multi_node/tilert_utils/run_node.sh delete mode 100644 inferencex-e2e/benchmarks/multi_node/tilert_utils/setup_deps.sh delete mode 100755 inferencex-e2e/benchmarks/multi_node/tilert_utils/submit.sh diff --git a/inferencex-e2e/benchmarks/multi_node/glm5.1_fp8_b200_tilert-disagg.sh b/inferencex-e2e/benchmarks/multi_node/glm5.1_fp8_b200_tilert-disagg.sh deleted file mode 100755 index 8ba34f5c83..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/glm5.1_fp8_b200_tilert-disagg.sh +++ /dev/null @@ -1,34 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../benchmark_lib.sh" - -check_env_vars \ - CONC_LIST \ - ISL \ - OSL \ - IMAGE \ - SPEC_DECODING \ - MODEL_PATH \ - PREFILL_NUM_WORKERS \ - PREFILL_TP \ - PREFILL_EP \ - PREFILL_DP_ATTN \ - DECODE_NUM_WORKERS \ - DECODE_TP \ - DECODE_EP \ - DECODE_DP_ATTN \ - PREFILL_NODES \ - DECODE_NODES \ - RANDOM_RANGE_RATIO \ - FRAMEWORK - -export MODEL_NAME=glm5 -export TILERT_MODEL_TYPE=glm-5 -check_env_vars MAX_MODEL_LEN - -export DECODE_KV_DTYPE=fp8 -export PREFILL_KV_DTYPE=fp8_ds_mla - -export TILERT_PARSER=none - -exec bash "$(dirname "$0")/tilert_utils/submit.sh" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh new file mode 100644 index 0000000000..59197a65d9 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh @@ -0,0 +1,49 @@ +#!/usr/bin/env bash +set -eo pipefail +source /infmax-workspace/benchmarks/benchmark_lib.sh --validation-only +check_env_vars TILERT_ROLE + +case "$TILERT_ROLE" in + prefill) + python -m pip install --quiet --no-cache-dir --no-deps tilert==0.1.5.post3 + if ! python -c 'import nixl' 2>/dev/null; then + python -m pip install --quiet --no-cache-dir nixl==1.3.1 + fi + ;; + decode|router) + python -m pip install --quiet --no-cache-dir tilert==0.1.5.post3 + if ! python -c 'import uvicorn' 2>/dev/null; then + python -m pip install --quiet --no-cache-dir fastapi uvicorn httpx + fi + if ! python -c 'import nixl' 2>/dev/null; then + python -m pip install --quiet --no-cache-dir nixl==1.3.1 + fi + if ! python -c 'from importlib.metadata import version; assert int(version("transformers").split(".")[0]) >= 5'; then + python -m pip install --quiet --no-cache-dir 'transformers>=5.4.0' + fi + ;; + *) + echo "Unknown TileRT setup role: $TILERT_ROLE" >&2 + exit 1 + ;; +esac + +if [[ "$TILERT_ROLE" == decode ]]; then + # Keep the shared converted checkpoint reusable across allocations. + mkdir -p /tilert_weights + exec 9>/tilert_weights/.convert.lock + flock -w 21600 9 + if [[ ! -f /tilert_weights/model.safetensors.index.json ]]; then + python -m tilert.models.preprocess.weight_converter \ + --model_type glm-5 --model_dir /model --save_dir /tilert_weights + test -f /tilert_weights/model.safetensors.index.json + fi + for file in /model/*; do + [[ -f "$file" ]] || continue + name="${file##*/}" + [[ "$name" == *.safetensors || "$name" == model.safetensors.index.json ]] && continue + [[ -e "/tilert_weights/$name" ]] || cp -p "$file" "/tilert_weights/$name" + done + test -f /tilert_weights/chat_template.jinja + test -f /tilert_weights/tokenizer_config.json || test -f /tilert_weights/tokenizer.json +fi diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.1/tilert/b200-fp8/1k1k/disagg-1p1d-tp8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.1/tilert/b200-fp8/1k1k/disagg-1p1d-tp8-mtp.yaml new file mode 100644 index 0000000000..689996877b --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.1/tilert/b200-fp8/1k1k/disagg-1p1d-tp8-mtp.yaml @@ -0,0 +1,88 @@ +schema: 2 +name: glm5.1-tilert-b200-1k1k-1p1d-tp8-mtp +model: + path: glm5.1-fp8 + container: ghcr.io/tile-ai/tilert:0.1.5 + precision: fp8 +slurm: + time_limit: "00:45:00" +resources: + gpu_type: b200 + gpus_per_node: 8 +setup_script: tilert-b200-setup.sh +environment: + NVIDIA_VISIBLE_DEVICES: all + NVIDIA_DRIVER_CAPABILITIES: compute,utility +roles: + prefill: + engine: + type: vllm + set_visible_devices: true + container: vllm/vllm-openai:v0.26.0 + nodes: 1 + workers: 1 + gpus: 8 + env: + TILERT_ROLE: prefill + args: + served-model-name: [glm5, zai-org/GLM-5.1-FP8] + tensor-parallel-size: 8 + max-model-len: 2304 + enforce-eager: true + trust-remote-code: true + return-tokens-as-token-ids: true + gpu-memory-utilization: 0.75 + kv-cache-dtype: fp8_ds_mla + speculative-config: '{"method":"mtp","num_speculative_tokens":1}' + kv-transfer-config: >- + {"kv_connector":"TileRTConnector","kv_connector_module_path":"tilert.pd_vllm.prefill_connector", + "kv_role":"kv_producer","kv_connector_extra_config":{ + "tilert_model":"glm5","tilert_max_seq_len":2304,"tilert_transport":"nixl"}} + decode: + engine: + type: tilert + served_model_name: glm5 + container: ghcr.io/tile-ai/tilert:0.1.5 + nodes: 1 + workers: 1 + gpus: 8 + env: + TILERT_ROLE: decode + PATH: /opt/conda/envs/tilert/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin + args: + model: glm5 + model-weights-dir: /tilert_weights + max-seq-len: 2304 + kv-cache-dtype: fp8 + transport: nixl + with-mtp: true +frontend: + type: tilert-router + container_image: ghcr.io/tile-ai/tilert:0.1.5 + enable_multiple_frontends: false + env: + TILERT_ROLE: router + PATH: /opt/conda/envs/tilert/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin + args: + model-path: /model + parser: none + queue-timeout: 0 +srun_options: + container-writable: "" + container-remap-root: "" +health_check: + max_attempts: 720 + interval_seconds: 5 +benchmark: + type: custom + container_image: vllm/vllm-openai:v0.26.0 + concurrencies: [1] + command: bash /infmax-workspace/benchmarks/multi_node/srt_fixed_sequence.sh + env: + ISL: "1024" + OSL: "1024" + CLIENT_BACKEND: openai-chat + BENCHMARK_SERVED_MODEL_NAME: glm5 + USE_CHAT_TEMPLATE: "true" + TOKENIZER: /model + NUM_PROMPTS: "16" diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.1/tilert/b200-fp8/8k1k/disagg-1p1d-tp8-mtp.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.1/tilert/b200-fp8/8k1k/disagg-1p1d-tp8-mtp.yaml new file mode 100644 index 0000000000..ebb0a4d7c6 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.1/tilert/b200-fp8/8k1k/disagg-1p1d-tp8-mtp.yaml @@ -0,0 +1,98 @@ +schema: 2 +name: glm5.1-tilert-b200-8k1k-1p1d-tp8-mtp +model: + path: glm5.1-fp8 + container: ghcr.io/tile-ai/tilert:0.1.5 + precision: fp8 +slurm: + time_limit: "01:30:00" +resources: + gpu_type: b200 + gpus_per_node: 8 +setup_script: tilert-b200-setup.sh +environment: + NVIDIA_VISIBLE_DEVICES: all + NVIDIA_DRIVER_CAPABILITIES: compute,utility +roles: + prefill: + engine: + type: vllm + set_visible_devices: true + container: vllm/vllm-openai:v0.26.0 + nodes: 1 + workers: 1 + gpus: 8 + env: + TILERT_ROLE: prefill + args: + served-model-name: [glm5, zai-org/GLM-5.1-FP8] + tensor-parallel-size: 8 + max-model-len: 9472 + enforce-eager: true + trust-remote-code: true + return-tokens-as-token-ids: true + gpu-memory-utilization: 0.75 + kv-cache-dtype: fp8_ds_mla + speculative-config: '{"method":"mtp","num_speculative_tokens":1}' + kv-transfer-config: >- + {"kv_connector":"TileRTConnector","kv_connector_module_path":"tilert.pd_vllm.prefill_connector", + "kv_role":"kv_producer","kv_connector_extra_config":{ + "tilert_model":"glm5","tilert_max_seq_len":9472,"tilert_transport":"nixl"}} + decode: + engine: + type: tilert + served_model_name: glm5 + container: ghcr.io/tile-ai/tilert:0.1.5 + nodes: 1 + workers: 1 + gpus: 8 + env: + TILERT_ROLE: decode + PATH: /opt/conda/envs/tilert/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin + args: + model: glm5 + model-weights-dir: /tilert_weights + max-seq-len: 9472 + kv-cache-dtype: fp8 + transport: nixl + with-mtp: true +frontend: + type: tilert-router + container_image: ghcr.io/tile-ai/tilert:0.1.5 + enable_multiple_frontends: false + env: + TILERT_ROLE: router + PATH: /opt/conda/envs/tilert/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin + args: + model-path: /model + parser: none + queue-timeout: 0 +srun_options: + container-writable: "" + container-remap-root: "" +health_check: + max_attempts: 720 + interval_seconds: 5 +telemetry: + enabled: true + collect_interval_ms: 1000 + storage_subdir: power + required: true + startup_timeout_seconds: 120 + request_timeout_seconds: 2 + dcgm_exporter: + container_image: dcgm-exporter + port: 9401 +benchmark: + type: custom + container_image: vllm/vllm-openai:v0.26.0 + concurrencies: [1] + command: bash /infmax-workspace/benchmarks/multi_node/srt_fixed_sequence.sh + env: + ISL: "8192" + OSL: "1024" + CLIENT_BACKEND: openai-chat + BENCHMARK_SERVED_MODEL_NAME: glm5 + USE_CHAT_TEMPLATE: "true" + TOKENIZER: /model + NUM_PROMPTS: "16" diff --git a/inferencex-e2e/benchmarks/multi_node/tilert_utils/run_node.sh b/inferencex-e2e/benchmarks/multi_node/tilert_utils/run_node.sh deleted file mode 100755 index 7d5cb39734..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/tilert_utils/run_node.sh +++ /dev/null @@ -1,332 +0,0 @@ -#!/usr/bin/env bash - -source "$(dirname "$0")/../../benchmark_lib.sh" --validation-only - -check_env_vars \ - MODEL_NAME MAX_MODEL_LEN GPU_MEM_UTIL RESULT_DIR BENCHMARK_LOGS_DIR \ - DECODE_CTRL_PORT DECODE_HTTP_PORT PREFILL_PORT PORT DECODE_WAIT \ - KV_P2P_TRANSFER TILERT_WEIGHTS_DIR TILERT_MODEL_TYPE TILERT_PARSER DECODE_KV_DTYPE \ - PREFILL_KV_DTYPE IS_AGENTIC TILERT_QUEUE_TIMEOUT SLURM_JOB_ID POWERX_NATIVE_ENABLED \ - PREFILL_TP PREFILL_NUM_WORKERS DECODE_TP DECODE_NUM_WORKERS TILERT_RDMA_STRICT \ - TILERT_CONVERT_LOCK_WAIT PREFILL_WAIT EVAL_ONLY DECODE_HOST PREFILL_HOST \ - TILERT_ROLE -SERVER_MAX_MODEL_LEN="$MAX_MODEL_LEN" -source "$(dirname "$0")/../../benchmark_lib.sh" - -ROUTER_PORT=${PORT} - -PREFILL_SPEC=(--speculative-config '{"method":"mtp","num_speculative_tokens":1}') -DECODE_MTP=(--with-mtp) - -TILERT_IS_AGENTIC=0 -if [[ "${IS_AGENTIC}" == "1" || "${SCENARIO_TYPE:-}" == "agentic-coding" ]]; then - TILERT_IS_AGENTIC=1 -fi - -AGENTIC_LOGS_DIR=${AGENTIC_LOGS_DIR:-$RESULT_DIR/LOGS/agentic} - -mkdir -p "$BENCHMARK_LOGS_DIR" - -DONE_SENTINEL="$BENCHMARK_LOGS_DIR/.tilert_done.${SLURM_JOB_ID}" -source "$(dirname "$0")/../../native_power_lifecycle.sh" -finish_tilert_node() { - local rc=$? pid - trap - EXIT - if [[ "${POWERX_NATIVE_ENABLED}" == 1 && -n "${POWERX_COLLECTOR_PID:-}" ]]; then - if [[ "$TILERT_ROLE" == prefill || "$rc" != 0 ]]; then - powerx_stop_collectors || rc=$? - else - powerx_reap_collector || rc=$? - fi - fi - if [[ "$TILERT_ROLE" == prefill ]]; then - printf '%s\n' "$rc" > "$DONE_SENTINEL.tmp" || rc=1 - if [[ -n "${POWERX_HOST_UID:-}" ]]; then - chown "$POWERX_HOST_UID:$POWERX_HOST_GID" "$DONE_SENTINEL.tmp" || rc=1 - fi - mv -f "$DONE_SENTINEL.tmp" "$DONE_SENTINEL" || rc=1 - fi - for pid in "${ROUTER_PID:-}" "${PREFILL_PID:-}" "${DECODE_PID:-}"; do - [[ -z "$pid" ]] || kill -TERM "$pid" 2>/dev/null || true - done - exit "$rc" -} -trap finish_tilert_node EXIT -trap 'exit 143' TERM HUP -trap 'exit 130' INT -if [[ "${POWERX_NATIVE_ENABLED}" == 1 ]]; then - powerx_start_collector "/powerx_native/node-$POWERX_RANK" "/powerx_control" \ - nvidia "$POWERX_RANK" "$TILERT_ROLE" "$POWERX_GPU_COUNT" 2 -fi -echo "[tilert-run_node] ROLE=$TILERT_ROLE host=$(hostname) DECODE_HOST=$DECODE_HOST PREFILL_HOST=$PREFILL_HOST" - -log_and_run_bg() { - local label="$1" logfile="$2"; shift 2 - local _xtrace=0; [[ $- == *x* ]] && _xtrace=1 - { set +x; } 2>/dev/null - { printf '===== [%s] %s =====\n' "$label" "$(date '+%F %T')" - printf '[cmd]'; printf ' %q' "$@"; printf '\n' - printf '[cwd] %s\n[host] %s\n\n' "$PWD" "$(hostname)" - } | tee -a "$logfile" - "$@" >>"$logfile" 2>&1 & - LAST_BG_PID=$! - echo "[$label] pid=$LAST_BG_PID log=$logfile" - (( _xtrace )) && set -x - return 0 -} - -bench_result_stem() { - local conc="$1" - local pg=$(( ${PREFILL_TP} * ${PREFILL_NUM_WORKERS} )) - local dg=$(( ${DECODE_TP} * ${DECODE_NUM_WORKERS} )) - printf '%s_c%s_gpus_%s_ctx_%s_gen_%s' \ - "${RESULT_FILENAME}" "$conc" "$(( pg + dg ))" "$pg" "$dg" -} - -rdma_preflight() { - local warn=0 - echo "[rdma] role=$TILERT_ROLE UCX_NET_DEVICES=${UCX_NET_DEVICES:-} UCX_MEMTYPE_CACHE=${UCX_MEMTYPE_CACHE:-} UCX_MEMTYPE_REG_WHOLE=${UCX_MEMTYPE_REG_WHOLE:-}" - - local uverbs=(/dev/infiniband/uverbs*) - if [[ -e "${uverbs[0]}" ]]; then - echo "[rdma] verbs devices: ${uverbs[*]}" - else - echo "[rdma] WARNING: /dev/infiniband/uverbs* missing -- the container has no RDMA device nodes." >&2 - echo "[rdma] docker: add --device /dev/infiniband; pyxis: needs cluster-side passthrough" >&2 - warn=1 - fi - - local ml; ml="$(ulimit -l 2>/dev/null)" - if [[ "$ml" == "unlimited" ]]; then - echo "[rdma] memlock: unlimited" - else - echo "[rdma] WARNING: memlock=$ml (not unlimited) -- pinning memory for RDMA may fail." >&2 - echo "[rdma] docker: add --cap-add CAP_IPC_LOCK (or --ulimit memlock=-1)" >&2 - warn=1 - fi - - if command -v ibv_devices >/dev/null 2>&1; then - echo "[rdma] ibv_devices:"; ibv_devices 2>&1 | sed 's/^/[rdma] /' - fi - - if (( warn )) && [[ "${TILERT_RDMA_STRICT}" == "1" ]]; then - echo "[rdma] TILERT_RDMA_STRICT=1 and preflight did not fully pass -- aborting" >&2 - return 1 - fi - return 0 -} - -stage_tokenizer_files() { - local staged=0 f b - for f in "$MODEL_PATH"/*; do - [[ -f "$f" ]] || continue - b="$(basename "$f")" - [[ "$b" == *.safetensors ]] && continue - [[ "$b" == "model.safetensors.index.json" ]] && continue - [[ -e "$TILERT_WEIGHTS_DIR/$b" ]] && continue - cp -p "$f" "$TILERT_WEIGHTS_DIR/$b" && staged=$((staged+1)) - done - echo "[stage_tokenizer] staged $staged auxiliary file(s) from $MODEL_PATH" - local missing=() - [[ -f "$TILERT_WEIGHTS_DIR/chat_template.jinja" ]] || missing+=(chat_template.jinja) - [[ -f "$TILERT_WEIGHTS_DIR/tokenizer_config.json" || -f "$TILERT_WEIGHTS_DIR/tokenizer.json" ]] \ - || missing+=("tokenizer.json/tokenizer_config.json") - if (( ${#missing[@]} )); then - echo "[stage_tokenizer] ERROR: $TILERT_WEIGHTS_DIR is missing ${missing[*]}" - echo "[stage_tokenizer] decode_server loads the tokenizer and chat template from that directory." - echo "[stage_tokenizer] Check that MODEL_PATH=$MODEL_PATH is an HF directory containing the tokenizer." - return 1 - fi - return 0 -} - -convert_weights() { - local index_json="$TILERT_WEIGHTS_DIR/model.safetensors.index.json" - if [[ -f "$index_json" ]]; then - echo "[weight_converter] cache hit (index.json present), skipping conversion: $TILERT_WEIGHTS_DIR" - return 0 - fi - - mkdir -p "$TILERT_WEIGHTS_DIR" - exec 9>"$TILERT_WEIGHTS_DIR/.convert.lock" - flock -w "${TILERT_CONVERT_LOCK_WAIT}" 9 || { - echo "[weight_converter] timed out waiting for the conversion lock (another job still converting?)"; return 1; } - if [[ -f "$index_json" ]]; then - echo "[weight_converter] cache produced by a concurrent job, skipping conversion"; exec 9>&-; return 0 - fi - if [[ -n "$(ls -A "$TILERT_WEIGHTS_DIR" 2>/dev/null | grep -v '^\.convert\.lock$')" ]]; then - echo "[weight_converter] found leftovers without index.json (previous conversion incomplete), cleaning and re-converting" - find "$TILERT_WEIGHTS_DIR" -mindepth 1 ! -name '.convert.lock' -delete - fi - - echo "[weight_converter] $MODEL_PATH -> $TILERT_WEIGHTS_DIR (model_type=$TILERT_MODEL_TYPE)" - "$PY" -m tilert.models.preprocess.weight_converter \ - --model_type "$TILERT_MODEL_TYPE" --model_dir "$MODEL_PATH" --save_dir "$TILERT_WEIGHTS_DIR" - local rc=$? - exec 9>&- - if [[ $rc -ne 0 || ! -f "$index_json" ]]; then - echo "[weight_converter] conversion failed (rc=$rc, no index.json produced): $TILERT_WEIGHTS_DIR" - return 1 - fi - echo "[weight_converter] conversion done and cached: $TILERT_WEIGHTS_DIR" -} - -start_decode() { - local cmd=("$PY" -m tilert.pd_vllm.decode_server - --engine tilert --model "$MODEL_NAME" - --model-weights-dir "$TILERT_WEIGHTS_DIR" - --max-seq-len "$SERVER_MAX_MODEL_LEN" - --kv-cache-dtype "$DECODE_KV_DTYPE" --transport "$KV_P2P_TRANSFER" - --ctrl-port "$DECODE_CTRL_PORT" --http-port "$DECODE_HTTP_PORT" - "${DECODE_MTP[@]}") - log_and_run_bg decode "$BENCHMARK_LOGS_DIR/tilert_decode.log" "${cmd[@]}" - DECODE_PID=$LAST_BG_PID -} - -start_prefill() { - local served=("$MODEL_NAME") - [[ -n "${MODEL:-}" && "$MODEL" != "$MODEL_NAME" ]] && served+=("$MODEL") - local cmd=(vllm serve "$MODEL_PATH" - --served-model-name "${served[@]}" --port "$PREFILL_PORT" - --tensor-parallel-size "$PREFILL_TP" --max-model-len "$SERVER_MAX_MODEL_LEN" - --enforce-eager --trust-remote-code --return-tokens-as-token-ids - --gpu-memory-utilization "$GPU_MEM_UTIL" --kv-cache-dtype "$PREFILL_KV_DTYPE" - "${PREFILL_SPEC[@]}" - --kv-transfer-config "{\"kv_connector\":\"TileRTConnector\",\"kv_connector_module_path\":\"tilert.pd_vllm.prefill_connector\",\"kv_role\":\"kv_producer\",\"kv_connector_extra_config\":{\"tilert_host\":\"$DECODE_HOST\",\"tilert_ctrl_port\":$DECODE_CTRL_PORT,\"tilert_model\":\"$MODEL_NAME\",\"tilert_max_seq_len\":$SERVER_MAX_MODEL_LEN,\"tilert_transport\":\"$KV_P2P_TRANSFER\"}}") - log_and_run_bg prefill "$BENCHMARK_LOGS_DIR/tilert_prefill.log" "${cmd[@]}" - PREFILL_PID=$LAST_BG_PID -} - -start_router() { - local cmd=(env CUDA_VISIBLE_DEVICES= "$PY" -m tilert.pd_vllm.pd_router - --vllm-url "http://$PREFILL_HOST:$PREFILL_PORT" - --decode "$DECODE_HOST:$DECODE_CTRL_PORT:$DECODE_HTTP_PORT" - --port "$ROUTER_PORT" --model-path "$MODEL_PATH" --parser "$TILERT_PARSER" - --queue-timeout "$TILERT_QUEUE_TIMEOUT") - log_and_run_bg router "$BENCHMARK_LOGS_DIR/tilert_router.log" "${cmd[@]}" - ROUTER_PID=$LAST_BG_PID -} - -wait_for_tcp() { - local host="$1" port="$2" deadline=$(( SECONDS + ${3:-600} )) - local _xtrace=0; [[ $- == *x* ]] && _xtrace=1 - { set +x; } 2>/dev/null - local rc=0 - until (exec 3<>"/dev/tcp/$host/$port") 2>/dev/null; do - if [[ $SECONDS -ge $deadline ]]; then - echo "[wait_for_tcp] timeout $host:$port after ${3:-600}s"; rc=1; break - fi - sleep 5 - done - [[ $rc -eq 0 ]] && echo "[wait_for_tcp] $host:$port ready" - (( _xtrace )) && set -x - return $rc -} - -run_bench_and_eval() { - wait_for_server_ready --port "$ROUTER_PORT" \ - --server-log "$BENCHMARK_LOGS_DIR/tilert_router.log" --server-pid "$ROUTER_PID" - local rc=0 conc np - if [[ "${POWERX_NATIVE_ENABLED}" == 1 ]]; then - powerx_wait_collectors ready || return $? - fi - if [[ "${EVAL_ONLY}" != "true" ]]; then - for conc in $CONC_LIST; do - np=$(( conc * 10 )) - [[ "$np" -lt 16 ]] && np=16 - run_benchmark_serving \ - --bench-serving-dir /workspace \ - --model "$MODEL_NAME" --port "$ROUTER_PORT" \ - --backend openai-chat --endpoint /v1/chat/completions \ - --input-len "$ISL" --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$np" --max-concurrency "$conc" \ - --use-chat-template --server-pid "$ROUTER_PID" \ - --tokenizer "$MODEL_PATH" --trust-remote-code \ - --result-filename "$(bench_result_stem "$conc")" --result-dir "$RESULT_DIR" \ - || { rc=$?; echo "[bench] WARNING: conc=$conc failed/timed out (rc=$rc)"; } - done - fi - run_tilert_eval || rc=$? - return $rc -} - -run_tilert_eval() { - [[ "${RUN_EVAL}" = "true" ]] || return 0 - if [[ -n "${EVAL_CONC:-}" ]]; then - export EVAL_CONCURRENT_REQUESTS="$EVAL_CONC" - else - export EVAL_CONCURRENT_REQUESTS="$(tr ' ' '\n' <<< "$CONC_LIST" | sort -n | tail -1)" - fi - export CONC="$EVAL_CONCURRENT_REQUESTS" - local eval_rc=0 stage_rc=0 - run_eval --port "$ROUTER_PORT" || eval_rc=$? - append_lm_eval_summary || stage_rc=$? - if [[ "$eval_rc" -ne 0 ]]; then - return "$eval_rc" - fi - return "$stage_rc" -} - -run_agentic_replay() { - wait_for_server_ready --port "$ROUTER_PORT" \ - --server-log "$BENCHMARK_LOGS_DIR/tilert_router.log" --server-pid "$ROUTER_PID" - local rc=0 conc conc_result_dir - local result_filename_base="$RESULT_FILENAME" - for conc in $CONC_LIST; do - conc_result_dir="$AGENTIC_LOGS_DIR/conc_${conc}" - mkdir -p "$conc_result_dir" - export CONC="$conc" - export RESULT_FILENAME="${result_filename_base}_conc${conc}" - build_replay_cmd "$conc_result_dir" - run_agentic_replay_and_write_outputs "$conc_result_dir" \ - || { rc=$?; echo "[agentic] WARNING: conc=$conc failed/timed out (rc=$rc)"; } - done - export RESULT_FILENAME="$result_filename_base" - return $rc -} - -# shellcheck source=./setup_deps.sh -source "$(dirname "$0")/setup_deps.sh" - -set -x -case "$TILERT_ROLE" in - decode) - rdma_preflight || exit 1 - convert_weights || exit 1 - stage_tokenizer_files || exit 1 - start_decode - { set +x; } 2>/dev/null - while kill -0 "$DECODE_PID" 2>/dev/null; do - [[ -f "$DONE_SENTINEL" ]] && break - sleep 5 - done - if [[ -f "$DONE_SENTINEL" ]]; then - echo "[decode] done sentinel received, shutting down" - exit "$(cat "$DONE_SENTINEL")" - fi - echo "[decode] decode_server exited early (see $BENCHMARK_LOGS_DIR/tilert_decode.log)"; exit 1 - ;; - prefill) - rdma_preflight || exit 1 - if [[ "$TILERT_IS_AGENTIC" == "1" ]]; then - resolve_trace_source - install_agentic_deps - fi - wait_for_tcp "$DECODE_HOST" "$DECODE_CTRL_PORT" "$DECODE_WAIT" \ - || echo "[prefill] WARNING: timed out waiting for the decode ctrl port ($DECODE_HOST:$DECODE_CTRL_PORT), starting anyway" - start_prefill - wait_for_tcp "$PREFILL_HOST" "$PREFILL_PORT" "${PREFILL_WAIT}" \ - || echo "[prefill] WARNING: timed out waiting for the vLLM port ($PREFILL_HOST:$PREFILL_PORT), continuing (see $BENCHMARK_LOGS_DIR/tilert_prefill.log)" - start_router - if [[ "$TILERT_IS_AGENTIC" == "1" ]]; then - run_agentic_replay; BENCH_RC=$? - else - run_bench_and_eval; BENCH_RC=$? - fi - # EXIT drains both collectors before the decode role sees completion. - exit $BENCH_RC - ;; - *) - echo "unknown ROLE=$TILERT_ROLE"; exit 2 ;; -esac diff --git a/inferencex-e2e/benchmarks/multi_node/tilert_utils/setup_deps.sh b/inferencex-e2e/benchmarks/multi_node/tilert_utils/setup_deps.sh deleted file mode 100644 index d2d3fea4f0..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/tilert_utils/setup_deps.sh +++ /dev/null @@ -1,143 +0,0 @@ -#!/bin/bash - -check_env_vars TILERT_VERSION TILERT_HTTP_DEPS TILERT_NIXL_VERSION - -TILERT_PIP_INDEX_URL="${TILERT_PIP_INDEX_URL:-}" - -if [[ -z "${TILERT_TRANSPORT_DEPS:-}" ]]; then - TILERT_TRANSPORT_DEPS="nixl==$TILERT_NIXL_VERSION" -fi - -_SETUP_INSTALLED=() - -activate_tilert_env() { - local env_dir=/opt/conda/envs/tilert - [[ -d "$env_dir" ]] || return 0 - if [[ "$(command -v python)" == "$env_dir/bin/python" ]]; then - echo "[SETUP] conda env 'tilert' already active" - return 0 - fi - echo "[SETUP] activating conda env 'tilert' (enroot does not run the image ENTRYPOINT)" - # shellcheck disable=SC1091 - if [[ -f /opt/conda/etc/profile.d/conda.sh ]]; then - . /opt/conda/etc/profile.d/conda.sh && conda activate tilert - fi - [[ "$(command -v python)" == "$env_dir/bin/python" ]] || export PATH="$env_dir/bin:$PATH" - echo "[SETUP] python -> $(command -v python)" -} - -_installed_version() { - "$PY" - "$1" <<'PY' 2>/dev/null -import sys -from importlib.metadata import version, PackageNotFoundError -try: - print(version(sys.argv[1])) -except PackageNotFoundError: - pass -PY -} - -_pip_args() { - local a=(--quiet --no-cache-dir) - [[ -n "$TILERT_PIP_INDEX_URL" ]] && a+=(--index-url "$TILERT_PIP_INDEX_URL") - printf '%s\n' "${a[@]}" -} - -install_tilert_decode() { - mapfile -t _pa < <(_pip_args) - local have; have="$(_installed_version tilert)" - if [[ "$have" == "$TILERT_VERSION" ]]; then - echo "[SETUP] tilert $have already installed, skipping" - else - [[ -n "$have" ]] && echo "[SETUP] tilert $have installed, switching to pinned $TILERT_VERSION" - echo "[SETUP] installing tilert==$TILERT_VERSION (official PyPI release wheel)" - "$PY" -m pip install "${_pa[@]}" "tilert==$TILERT_VERSION" || { - echo "[SETUP] ERROR: failed to install tilert==$TILERT_VERSION"; exit 1; } - have="$(_installed_version tilert)" - [[ "$have" == "$TILERT_VERSION" ]] || { - echo "[SETUP] ERROR: still not $TILERT_VERSION after install (actual: ${have:-not installed})"; exit 1; } - _SETUP_INSTALLED+=("tilert==$TILERT_VERSION") - fi - _install_missing "uvicorn" "$TILERT_HTTP_DEPS" - _install_missing "nixl" "$TILERT_TRANSPORT_DEPS" - local tv; tv="$(_installed_version transformers)" - if [[ -n "$tv" && "${tv%%.*}" -lt 5 ]]; then - echo "[SETUP] transformers $tv < 5 -- cannot load the official checkpoint's TokenizersBackend, upgrading" - "$PY" -m pip install "${_pa[@]}" -U "transformers>=5.4.0" || { - echo "[SETUP] ERROR: failed to upgrade transformers"; exit 1; } - _SETUP_INSTALLED+=("transformers>=5.4.0(upgrade from $tv)") - fi - "$PY" -c "import tilert.pd_vllm.decode_server" 2>/dev/null || { - echo "[SETUP] ERROR: import tilert.pd_vllm.decode_server failed; actual error:" - "$PY" -c "import tilert.pd_vllm.decode_server" 2>&1 | tail -3 - exit 1; } - echo "[SETUP] tilert.pd_vllm.decode_server imports OK" -} - -_install_missing() { - local probe="$1" pkgs="$2" - [[ -n "$pkgs" ]] || return 0 - if "$PY" -c "import $probe" 2>/dev/null; then - echo "[SETUP] $probe already present, skipping ($pkgs)" - return 0 - fi - echo "[SETUP] installing $pkgs (probe module $probe missing)" - mapfile -t _pa < <(_pip_args) - # shellcheck disable=SC2086 - "$PY" -m pip install "${_pa[@]}" $pkgs || { - echo "[SETUP] ERROR: failed to install: $pkgs"; exit 1; } - _SETUP_INSTALLED+=("$pkgs") -} - -install_tilert_prefill() { - local vllm_v; vllm_v="$(_installed_version vllm)" - if [[ -z "$vllm_v" ]]; then - echo "[SETUP] ERROR: no vLLM in the prefill image." - echo "[SETUP] prefill needs an image with V1 disaggregation + GLM-5/5.1 (DSA) +" - echo "[SETUP] --kv-cache-dtype fp8_ds_mla support; the official tilert image has no vLLM," - echo "[SETUP] and its dependency set conflicts with vLLM, so P and D must use different images." - exit 1 - fi - echo "[SETUP] prefill-side vLLM $vllm_v" - local have; have="$(_installed_version tilert)" - if [[ "$have" == "$TILERT_VERSION" ]]; then - echo "[SETUP] tilert $have already installed, skipping" - else - echo "[SETUP] installing tilert==$TILERT_VERSION --no-deps (connector plugin only; leaves transformers untouched)" - mapfile -t _pa < <(_pip_args) - "$PY" -m pip install "${_pa[@]}" --no-deps "tilert==$TILERT_VERSION" || { - echo "[SETUP] ERROR: failed to install tilert==$TILERT_VERSION (--no-deps)"; exit 1; } - have="$(_installed_version tilert)" - [[ "$have" == "$TILERT_VERSION" ]] || { - echo "[SETUP] ERROR: still not $TILERT_VERSION after install (actual: ${have:-not installed})"; exit 1; } - _SETUP_INSTALLED+=("tilert==$TILERT_VERSION(--no-deps)") - fi - _install_missing "nixl" "$TILERT_TRANSPORT_DEPS" - "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>/dev/null || { - echo "[SETUP] WARN: import tilert.pd_vllm.prefill_connector failed (vLLM will report again when loading the connector plugin):" - "$PY" -c "import tilert.pd_vllm.prefill_connector" 2>&1 | tail -3; } -} - -resolve_python() { - if [[ -n "${PY:-}" ]] && command -v "$PY" >/dev/null 2>&1; then : - else - PY="" - for c in python python3; do command -v "$c" >/dev/null 2>&1 && { PY="$c"; break; }; done - fi - [[ -n "$PY" ]] || { echo "[SETUP] ERROR: neither python nor python3 found"; exit 1; } - export PY - echo "[SETUP] interpreter PY=$PY ($(command -v "$PY"))" -} - -activate_tilert_env -resolve_python -case "${TILERT_ROLE:-}" in - decode) install_tilert_decode ;; - prefill) install_tilert_prefill ;; - *) echo "[SETUP] ERROR: unknown ROLE='${TILERT_ROLE:-}'"; exit 1 ;; -esac -if (( ${#_SETUP_INSTALLED[@]} )); then - echo "[SETUP] installed this run: ${_SETUP_INSTALLED[*]}" -else - echo "[SETUP] nothing to install (dependencies already satisfied)" -fi diff --git a/inferencex-e2e/benchmarks/multi_node/tilert_utils/submit.sh b/inferencex-e2e/benchmarks/multi_node/tilert_utils/submit.sh deleted file mode 100755 index 54ba55adfe..0000000000 --- a/inferencex-e2e/benchmarks/multi_node/tilert_utils/submit.sh +++ /dev/null @@ -1,195 +0,0 @@ -#!/usr/bin/env bash -set -eo pipefail -source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only -check_env_vars \ - PREFILL_NODES DECODE_NODES PREFILL_TP DECODE_TP B200_SQUASH_DIR \ - SALLOC_TIME_LIMIT REQUIRE_POWER IS_AGENTIC EVAL_ONLY BENCHMARK_LOGS_DIR \ - TILERT_DECODE_DRAIN MODEL_PREFIX PRECISION PORT GITHUB_WORKSPACE \ - IMAGE PREFILL_IMAGE SLURM_PARTITION SLURM_ACCOUNT RUNNER_NAME \ - MODEL_PATH ISL OSL TILERT_WEIGHTS_DIR -set -x -NODES=$(( ${PREFILL_NODES} + ${DECODE_NODES} )) -if [[ "${PREFILL_NODES}" != 1 || "${DECODE_NODES}" != 1 ]]; then - echo "TileRT native launcher requires one physical node per role" >&2 - exit 1 -fi -if [[ -z "${GPUS_PER_NODE:-}" ]]; then - if [[ -n "${GPU_COUNT:-}" ]]; then - GPUS_PER_NODE="$GPU_COUNT" - else - GPUS_PER_NODE=$(( PREFILL_TP > DECODE_TP ? PREFILL_TP : DECODE_TP )) - fi -fi - -SQUASH_DIR="${B200_SQUASH_DIR}" -{ mkdir -p "$SQUASH_DIR" 2>/dev/null && [[ -w "$SQUASH_DIR" ]]; } || SQUASH_DIR="$GITHUB_WORKSPACE/.container-squash" -mkdir -p "$SQUASH_DIR" -chmod a+rx "$SQUASH_DIR" || true - -if [[ -z "${DECODE_IMAGE:-}" ]]; then - DECODE_IMAGE="$IMAGE" -fi - -squash_path() { echo "$SQUASH_DIR/$(echo "$1" | sed 's/[\/:@#]/_/g').sqsh"; } -DECODE_SQUASH="$(squash_path "$DECODE_IMAGE")" -PREFILL_SQUASH="$(squash_path "$PREFILL_IMAGE")" -MODEL_MOUNTS="$MODEL_PATH:$MODEL_PATH" -CONTAINER_BENCHMARK_LOGS_DIR="${BENCHMARK_LOGS_DIR/#$GITHUB_WORKSPACE//workspace}" -if [[ -n "${HF_HUB_CACHE_HOST_PATH:-}" ]]; then - # HF snapshots link to sibling blobs outside the snapshot directory. - MODEL_MOUNTS="$HF_HUB_CACHE_HOST_PATH:$HF_HUB_CACHE_HOST_PATH,$MODEL_MOUNTS" -fi - -if [[ "${TILERT_IN_ALLOCATION:-0}" != 1 ]]; then - # Run inside the allocation returned by this request. Looking up a runner - # name can attach to an older job; salloc supplies the authoritative ID. - export TILERT_IN_ALLOCATION=1 - exec salloc --partition="$SLURM_PARTITION" --account="$SLURM_ACCOUNT" \ - --nodes="$NODES" --gres=gpu:"$GPUS_PER_NODE" --exclusive --mem=0 \ - --time="${SALLOC_TIME_LIMIT}" --job-name="$RUNNER_NAME" \ - "$BASH" "$0" "$@" -fi -JOB_ID="${SLURM_JOB_ID:?salloc did not provide its allocation ID}" -mapfile -t HOSTS < <(scontrol show hostnames "${SLURM_JOB_NODELIST:?salloc did not provide its nodes}") -[[ "${#HOSTS[@]}" -eq 2 ]] || { echo "expected 2 nodes, got: ${HOSTS[*]}"; exit 1; } -export DECODE_HOST="${HOSTS[0]}" PREFILL_HOST="${HOSTS[1]}" -export POWERX_NATIVE_ENABLED=0 -if [[ "${REQUIRE_POWER}" =~ ^(1|true|TRUE|yes|YES)$ && "$ISL" == 8192 && "$OSL" == 1024 && "${IS_AGENTIC}" != 1 && "${SCENARIO_TYPE:-}" != agentic-coding && "${EVAL_ONLY}" != true ]]; then - export POWERX_NATIVE_ENABLED=1 -fi -export SLURM_JOB_ID="$JOB_ID" -POWERX_MOUNTS="" -POWERX_ENV="POWERX_NATIVE_ENABLED" -if [[ "$POWERX_NATIVE_ENABLED" == 1 ]]; then - export POWERX_HOST_UID="$(id -u)" POWERX_HOST_GID="$(id -g)" - export POWERX_COLLECTOR_REVISION="$(git -C "$GITHUB_WORKSPACE" rev-parse HEAD)" - if [[ -z "${POWERX_RAW_ROOT:-}" ]]; then - POWERX_RAW_ROOT="/tmp/inferencex-native-$JOB_ID" - fi - export POWERX_RAW_ROOT - export POWERX_CONTROL_ROOT="$GITHUB_WORKSPACE/LOGS/power_control-$JOB_ID" - mkdir -p "$POWERX_CONTROL_ROOT" "$GITHUB_WORKSPACE/LOGS/native_power" - chmod 777 "$POWERX_CONTROL_ROOT" - srun --jobid="$JOB_ID" --nodes="$NODES" --ntasks-per-node=1 mkdir -p "$POWERX_RAW_ROOT" - srun --jobid="$JOB_ID" --nodes="$NODES" --ntasks-per-node=1 chmod 777 "$POWERX_RAW_ROOT" - POWERX_MOUNTS=",$POWERX_RAW_ROOT:/powerx_native,$POWERX_CONTROL_ROOT:/powerx_control" - POWERX_ENV+=",CUDA_VISIBLE_DEVICES,POWERX_HOST_UID,POWERX_HOST_GID,POWERX_COLLECTOR_REVISION,POWERX_NODE_NAME,POWERX_CLOCK_SYNCHRONIZED,POWERX_RANK,POWERX_GPU_COUNT" -fi -rm -f "${BENCHMARK_LOGS_DIR}/.tilert_done.$JOB_ID" -# Keep node-local receipts inside the allocation until both serving steps drain. -# A caught cancellation exits through the same staging path as normal completion. -wait_owned_step() { - local pid="$1" deadline=$((SECONDS + ${TILERT_DECODE_DRAIN})) rc=0 - while kill -0 "$pid" 2>/dev/null; do - if (( SECONDS >= deadline )); then - echo "[submit] step $pid did not drain before timeout" >&2 - kill -TERM "$pid" 2>/dev/null || true - sleep 2 - kill -KILL "$pid" 2>/dev/null || true - rc=1 - break - fi - sleep 1 - done - wait "$pid" || rc=$? - return "$rc" -} - -finish_tilert_submit() { - local rc=$? step_rc pid role_rank - trap - EXIT - trap '' TERM HUP INT - if [[ "$POWERX_NATIVE_ENABLED" == 1 ]]; then - printf 'stop\n' > "$POWERX_CONTROL_ROOT/stop" || { [[ "$rc" != 0 ]] || rc=1; } - fi - if [[ "$rc" != 0 ]]; then - for pid in "${PREFILL_SRUN_PID:-}" "${DECODE_SRUN_PID:-}"; do - [[ -z "$pid" ]] || kill -TERM "$pid" 2>/dev/null || true - done - fi - for pid in "${PREFILL_SRUN_PID:-}" "${DECODE_SRUN_PID:-}"; do - [[ -n "$pid" ]] || continue - step_rc=0 - wait_owned_step "$pid" || step_rc=$? - [[ "$rc" != 0 ]] || rc=$step_rc - done - if [[ "$POWERX_NATIVE_ENABLED" == 1 ]]; then - for role_rank in 0 1; do - # A failed or missing rank must not suppress the other rank's audit. - srun --jobid="$JOB_ID" --nodelist="${HOSTS[$role_rank]}" --ntasks=1 \ - bash -c 'cp -R "$POWERX_RAW_ROOT/node-$1" "$GITHUB_WORKSPACE/LOGS/native_power/"' bash "$role_rank" & - step_rc=0 - wait_owned_step "$!" || step_rc=$? - [[ "$rc" != 0 ]] || rc=$step_rc - done - fi - exit "$rc" -} -trap finish_tilert_submit EXIT -trap 'exit 143' TERM HUP -trap 'exit 130' INT - -import_image() { - local image_ref="$1" squash_file="$2" host="$3" - local enroot_ref="${image_ref#docker://}" - local registry="${enroot_ref%%/*}" - # Enroot needs '#' for an explicit registry; '/' alone targets Docker Hub. - if [[ "$enroot_ref" != *#* && "$enroot_ref" == */* && ( - "$registry" == *.* || "$registry" == *:* || "$registry" == localhost - ) ]]; then - enroot_ref="$registry#${enroot_ref#*/}" - fi - local image_key; image_key=$(echo "$image_ref" | sed 's/[\/:@#]/_/g') - local lock_file="$SQUASH_DIR/.locks/${image_key}.lock" - mkdir -p "$SQUASH_DIR/.locks" - srun --jobid="$JOB_ID" --nodelist="$host" --ntasks=1 bash -c " - export ENROOT_CACHE_PATH=\$HOME/.cache/enroot; mkdir -p \$ENROOT_CACHE_PATH - exec 9>\"$lock_file\"; flock -w 600 9 || exit 1 - unsquashfs -l \"$squash_file\" >/dev/null 2>&1 || { - rm -f \"$squash_file\" && enroot import -o \"$squash_file\" \"docker://$enroot_ref\" - } - " -} -import_image "$DECODE_IMAGE" "$DECODE_SQUASH" "$DECODE_HOST" || exit 1 -import_image "$PREFILL_IMAGE" "$PREFILL_SQUASH" "$PREFILL_HOST" || exit 1 - -export TILERT_WEIGHTS_DIR -mkdir -p "$TILERT_WEIGHTS_DIR" - -run_role() { - local role="$1" host="$2" squash_file="$3" - local rank=0 gpu_count="${DECODE_TP}" - if [[ "$role" == prefill ]]; then rank=1; gpu_count="${PREFILL_TP}"; fi - if [[ "$POWERX_NATIVE_ENABLED" == 1 ]]; then - export POWERX_RANK="$rank" POWERX_GPU_COUNT="$gpu_count" POWERX_NODE_NAME="$host" - export CUDA_VISIBLE_DEVICES="$(seq -s, 0 "$((gpu_count - 1))")" - export POWERX_CLOCK_SYNCHRONIZED - POWERX_CLOCK_SYNCHRONIZED=$(srun --jobid="$JOB_ID" --nodelist="$host" --ntasks=1 \ - bash -c 'timedatectl show -p NTPSynchronized --value 2>/dev/null || echo false') - fi - # The tilert image bakes no NVIDIA_VISIBLE_DEVICES (unlike vllm-openai), and - # enroot's nvidia hook only injects the driver when it is set — without it the - # decode container has no libcuda and torch dies with "Found no NVIDIA driver". - # docker --gpus sets this implicitly, which is why the image works elsewhere. - # Exported here (not in --export) because the capabilities value contains a - # comma, which srun's --export parsing would split on. - export NVIDIA_VISIBLE_DEVICES=all NVIDIA_DRIVER_CAPABILITIES=compute,utility - exec srun --jobid="$JOB_ID" --nodelist="$host" --ntasks=1 \ - --container-image="$squash_file" \ - --container-mounts="$GITHUB_WORKSPACE:/workspace,$MODEL_MOUNTS,$TILERT_WEIGHTS_DIR:$TILERT_WEIGHTS_DIR$POWERX_MOUNTS" \ - --container-workdir=/workspace --no-container-entrypoint \ - --container-env="$POWERX_ENV" \ - --export=ALL,TILERT_ROLE="$role",DECODE_HOST="$DECODE_HOST",PREFILL_HOST="$PREFILL_HOST",PORT="${PORT}",BENCHMARK_LOGS_DIR="$CONTAINER_BENCHMARK_LOGS_DIR" \ - bash "/workspace/benchmarks/multi_node/tilert_utils/run_node.sh" -} - -run_role decode "$DECODE_HOST" "$DECODE_SQUASH" & -DECODE_SRUN_PID=$! - -# Both roles run as owned children so a signal interrupts the shell's wait and -# reaches EXIT cleanup immediately; run_role execs srun to preserve that PID. -run_role prefill "$PREFILL_HOST" "$PREFILL_SQUASH" & -PREFILL_SRUN_PID=$! -PREFILL_RC=0 -wait "$PREFILL_SRUN_PID" || PREFILL_RC=$? -exit "$PREFILL_RC" diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 46976f0f0f..9ed9d8e401 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8172,10 +8172,8 @@ glm5.1-fp8-b200-tilert: ep: 1 dp-attn: false additional-settings: + - "CONFIG_FILE=recipes/glm5.1/tilert/b200-fp8/1k1k/disagg-1p1d-tp8-mtp.yaml" - "PREFILL_IMAGE=vllm/vllm-openai:v0.26.0" - - "PREFILL_NODES=1" - - "SALLOC_TIME_LIMIT=45" - - "B200_SQUASH_DIR=/data/home/sa-shared/containers" - "MODEL_PATH=/data/home/sa-shared/gharunners/hf-hub-cache/hub/models--zai-org--GLM-5.1-FP8/snapshots/f396cf805182f4ca10fa675e1a99815b3ca384db" - "HF_HUB_CACHE_HOST_PATH=/data/home/sa-shared/gharunners/hf-hub-cache" - "TILERT_WEIGHTS_DIR=/data/home/sa-shared/gharunners/tilert-cache/glm5.1-fp8-8shard" @@ -8184,8 +8182,6 @@ glm5.1-fp8-b200-tilert: tp: 8 ep: 1 dp-attn: false - additional-settings: - - "DECODE_NODES=1" - isl: 8192 osl: 1024 require-power: true @@ -8198,10 +8194,8 @@ glm5.1-fp8-b200-tilert: ep: 1 dp-attn: false additional-settings: + - "CONFIG_FILE=recipes/glm5.1/tilert/b200-fp8/8k1k/disagg-1p1d-tp8-mtp.yaml" - "PREFILL_IMAGE=vllm/vllm-openai:v0.26.0" - - "PREFILL_NODES=1" - - "SALLOC_TIME_LIMIT=90" - - "B200_SQUASH_DIR=/data/home/sa-shared/containers" - "MODEL_PATH=/data/home/sa-shared/gharunners/hf-hub-cache/hub/models--zai-org--GLM-5.1-FP8/snapshots/f396cf805182f4ca10fa675e1a99815b3ca384db" - "HF_HUB_CACHE_HOST_PATH=/data/home/sa-shared/gharunners/hf-hub-cache" - "TILERT_WEIGHTS_DIR=/data/home/sa-shared/gharunners/tilert-cache/glm5.1-fp8-8shard" @@ -8210,8 +8204,6 @@ glm5.1-fp8-b200-tilert: tp: 8 ep: 1 dp-attn: false - additional-settings: - - "DECODE_NODES=1" # H100 AgentX arm for DeepSeek-V4.1-Flash. H100 is not in the upstream hardware # table (h200, gb200, gb300, mi350x are). It needs its own script, not the diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index a0ecd2fc71..e738365acb 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9003,3 +9003,9 @@ description: - "Update B300 vLLM AgentX to DSpark6 on a new image with a sampled concurrency grid and per-mode --kv-cache-memory-bytes pins." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3477 + +- config-keys: + - glm5.1-fp8-b200-tilert + description: + - "Port B200 GLM-5.1 TileRT 1k1k and 8k1k to native srt-slurm per-role engines." + pr-link: TBD diff --git a/inferencex-e2e/runners/launch_b200-nscale-slurm.sh b/inferencex-e2e/runners/launch_b200-nscale-slurm.sh index 1c6c7afc04..7a70db6cd5 100755 --- a/inferencex-e2e/runners/launch_b200-nscale-slurm.sh +++ b/inferencex-e2e/runners/launch_b200-nscale-slurm.sh @@ -30,9 +30,6 @@ export AIPERF_MMAP_CACHE_HOST_PATH="/data/home/sa-shared/gharunners/aiperf-cache uses_native_srt_lane() { [[ "$IS_MULTINODE" == "true" ]] || return 1 - if [[ "$FRAMEWORK" == "tilert" && "${IS_AGENTIC}" != "1" ]]; then - return 1 - fi case "${MODEL_PREFIX}/${PRECISION}" in dsv4/fp4|kimik3/fp4|glm5.2/fp4) ;; glm5.1/fp8) [[ "$FRAMEWORK" == "tilert" ]] || return 1 ;; @@ -266,7 +263,7 @@ run_native_srt_lane() { "$PRECISION" != "fp4" || ( "$MODEL_PREFIX" == "dsv4" && "$FRAMEWORK" != "dynamo-sglang" && "$FRAMEWORK" != "dynamo-vllm" ) || "$MODEL_PREFIX" != "dsv4" - ) ]]; then + ) && "$FRAMEWORK" != tilert ]]; then echo "Error: B200 nscale dcgm-power requires a supported fixed-sequence lane or Kimi-K3 AgentX vLLM" >&2 exit 1 fi @@ -303,12 +300,11 @@ run_native_srt_lane() { PREFILL_SQUASH_FILE="" SRT_CLUSTER_ARGS=() if [[ $FRAMEWORK == "tilert" ]]; then - : "${PREFILL_IMAGE:?PREFILL_IMAGE is required for TileRT prefill}" + check_env_vars PREFILL_IMAGE PREFILL_SQUASH_FILE="$SQUASH_DIR/$(echo "$PREFILL_IMAGE" | sed 's/[\/:@#]/_/g').sqsh" import_squash "$PREFILL_SQUASH_FILE" "$PREFILL_IMAGE" || exit 1 SRT_CLUSTER_ARGS+=( - --container tilert-decode "$SQUASH_FILE" - --container tilert-prefill "$PREFILL_SQUASH_FILE" + --container "$PREFILL_IMAGE" "$PREFILL_SQUASH_FILE" ) fi @@ -335,11 +331,11 @@ run_native_srt_lane() { ) fi if [[ $FRAMEWORK == "tilert" ]]; then - TILERT_WEIGHTS_HOST_PATH="/data/home/sa-shared/gharunners/tilert-cache" - mkdir -p "$TILERT_WEIGHTS_HOST_PATH" + check_env_vars TILERT_WEIGHTS_DIR + mkdir -p "$TILERT_WEIGHTS_DIR" SRT_CLUSTER_ARGS+=( - --mount "$GITHUB_WORKSPACE" /infmax-workspace - --mount "$TILERT_WEIGHTS_HOST_PATH" "$TILERT_WEIGHTS_HOST_PATH" + --mount "$TILERT_WEIGHTS_DIR" /tilert_weights + --mount "$HF_HUB_CACHE_HOST_PATH" "$HF_HUB_CACHE_HOST_PATH" ) fi @@ -512,21 +508,6 @@ run_native_srt_lane() { # --------------------------------------------------------------------------- run_multinode_srt() { - if [[ "$FRAMEWORK" == "tilert" ]]; then - export SLURM_PARTITION SLURM_ACCOUNT - check_env_vars TILERT_WEIGHTS_DIR - # Nscale exposes eight RoCE HCAs, mlx5_0..mlx5_7. - check_env_vars UCX_NET_DEVICES - check_env_vars UCX_MEMTYPE_CACHE - check_env_vars UCX_MEMTYPE_REG_WHOLE - TILERT_SUBDIR="multi_node" - [[ "${SCENARIO_SUBDIR}" == "agentic/" ]] && TILERT_SUBDIR="multi_node/agentic" - TILERT_DISAGG="$GITHUB_WORKSPACE/benchmarks/${TILERT_SUBDIR}/${EXP_NAME%%_*}_${PRECISION}_b200_${FRAMEWORK}-disagg.sh" - [[ -f "$TILERT_DISAGG" ]] || { echo "tilert disagg script not found: $TILERT_DISAGG"; exit 1; } - exec bash "$TILERT_DISAGG" - exit 1 - fi - if [[ $FRAMEWORK != "dynamo-sglang" && $FRAMEWORK != "dynamo-trt" && $FRAMEWORK != "dynamo-vllm" ]]; then echo "Unsupported framework: $FRAMEWORK. Supported frameworks are: dynamo-trt, dynamo-sglang, dynamo-vllm" exit 1 diff --git a/inferencex-e2e/runners/runtime_settings.sh b/inferencex-e2e/runners/runtime_settings.sh index f39004be9c..39a2fc25d8 100644 --- a/inferencex-e2e/runners/runtime_settings.sh +++ b/inferencex-e2e/runners/runtime_settings.sh @@ -21,7 +21,6 @@ case "${RUNNER_NAME%%_*}" in glm5.2) export MODEL_PATH=/scratch/models/GLM-5.2-NVFP4 ;; esac if [[ "$FRAMEWORK" == tilert ]]; then - export TILERT_WEIGHTS_DIR="/scratch/models/${MODEL_PREFIX}-${PRECISION}-tilert-8shard" export UCX_NET_DEVICES=mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_6:1,mlx5_7:1 export UCX_MEMTYPE_CACHE=n UCX_MEMTYPE_REG_WHOLE=n fi From 9528353cc191e316949c7d83dcd7a91877719895 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 15:27:35 -0500 Subject: [PATCH 2/6] chore: link port validation PR --- inferencex-e2e/perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index e738365acb..a45e8ad04b 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9008,4 +9008,4 @@ - glm5.1-fp8-b200-tilert description: - "Port B200 GLM-5.1 TileRT 1k1k and 8k1k to native srt-slurm per-role engines." - pr-link: TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3553 From c94a50fcd2c5da6cb0c280ec63a420b9bb5a753e Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 15:33:08 -0500 Subject: [PATCH 3/6] fix(tilert): use the prefill image Python executable --- .../configs/tilert-b200-setup.sh | 22 +++++++++---------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh index 59197a65d9..bf77236d92 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh @@ -5,21 +5,21 @@ check_env_vars TILERT_ROLE case "$TILERT_ROLE" in prefill) - python -m pip install --quiet --no-cache-dir --no-deps tilert==0.1.5.post3 - if ! python -c 'import nixl' 2>/dev/null; then - python -m pip install --quiet --no-cache-dir nixl==1.3.1 + python3 -m pip install --quiet --no-cache-dir --no-deps tilert==0.1.5.post3 + if ! python3 -c 'import nixl' 2>/dev/null; then + python3 -m pip install --quiet --no-cache-dir nixl==1.3.1 fi ;; decode|router) - python -m pip install --quiet --no-cache-dir tilert==0.1.5.post3 - if ! python -c 'import uvicorn' 2>/dev/null; then - python -m pip install --quiet --no-cache-dir fastapi uvicorn httpx + python3 -m pip install --quiet --no-cache-dir tilert==0.1.5.post3 + if ! python3 -c 'import uvicorn' 2>/dev/null; then + python3 -m pip install --quiet --no-cache-dir fastapi uvicorn httpx fi - if ! python -c 'import nixl' 2>/dev/null; then - python -m pip install --quiet --no-cache-dir nixl==1.3.1 + if ! python3 -c 'import nixl' 2>/dev/null; then + python3 -m pip install --quiet --no-cache-dir nixl==1.3.1 fi - if ! python -c 'from importlib.metadata import version; assert int(version("transformers").split(".")[0]) >= 5'; then - python -m pip install --quiet --no-cache-dir 'transformers>=5.4.0' + if ! python3 -c 'from importlib.metadata import version; assert int(version("transformers").split(".")[0]) >= 5'; then + python3 -m pip install --quiet --no-cache-dir 'transformers>=5.4.0' fi ;; *) @@ -34,7 +34,7 @@ if [[ "$TILERT_ROLE" == decode ]]; then exec 9>/tilert_weights/.convert.lock flock -w 21600 9 if [[ ! -f /tilert_weights/model.safetensors.index.json ]]; then - python -m tilert.models.preprocess.weight_converter \ + python3 -m tilert.models.preprocess.weight_converter \ --model_type glm-5 --model_dir /model --save_dir /tilert_weights test -f /tilert_weights/model.safetensors.index.json fi From 430185e68668cdcb7c3006be126604472a6adfb5 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 15:39:14 -0500 Subject: [PATCH 4/6] docs: describe the native TileRT launch path --- inferencex-e2e/docs/configuration-procedures.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 50986d4c72..113757c9ad 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -192,17 +192,17 @@ Concurrent cells serialize draft staging with a per-model lock. Each cell lets `hf download` validate or resume the existing cache before serving; a nonempty directory is not a completion signal. -## Native TileRT power +## TileRT on B200 TileRT's shared importer preserves Docker Hub image names and converts explicit registries such as `ghcr.io/team/image:tag` to Enroot's `docker://ghcr.io#team/image:tag` syntax. Existing `#` references are preserved. Valid cached squash images are reused without importing; a cache hit does not validate the registry import path. Invalid cached images are removed under the import lock before retrying the import. The GLM-5.1 B200 Nscale 1k1k and 8k1k recipes select the prepared shared checkpoint, converted TileRT weights and squash cache, with allocation limits of 45 minutes for 1k1k and 90 minutes for 8k1k, including its full GSM8K eval. Since C1 is below automatic eval selection, use the PR `all-evals` label alongside `full-sweep-fail-fast` for full qualification. TileRT was added after the general GLM-5.1 retirement in [#2533](https://github.com/SemiAnalysisAI/InferenceX/pull/2533); [MODELS.md](MODELS.md) records this retained scope. Changes still require the normal PR sweep, applicable quality evidence, sign-off and reuse before publication. -TileRT's eval wrapper calls the shared `run_eval` dispatcher without overriding its `run_lm_eval` client. It stages available artifacts after evaluation and preserves failures from either evaluation or staging. TCP readiness probes keep their socket inside a subshell and preserve the caller's diagnostic streams. +The recipes use vLLM prefill, TileRT decode, and the TileRT router through srt-slurm. The shared `srt_fixed_sequence.sh` runs the benchmark; `srt_eval.sh` handles selected evaluations. srt-slurm owns worker startup, readiness, and teardown. -For GLM-5.1 on B200 Nscale, `MODEL_PATH` can select an existing shared checkpoint instead of the default `/scratch/models/GLM-5.1-FP8`. When it selects an HF snapshot, also set `HF_HUB_CACHE_HOST_PATH` to the existing cache root; TileRT mounts that root at the same absolute path so snapshot links to sibling blobs remain readable. Keep `TILERT_WEIGHTS_DIR` pointed at the separately converted decode weights. +`MODEL_PATH` selects the shared checkpoint. The launcher mounts its HF cache at the same absolute path so snapshot links to sibling blobs remain readable. `TILERT_WEIGHTS_DIR` selects the separately converted decode weights, mounted at `/tilert_weights`. -Only fixed 8192/1024 `glm5.1-fp8-b200-tilert` requires native power. TileRT runs inside its returned `salloc` allocation, retains both role exit codes and drains collectors before staging audits. Exactly one physical node per role is supported. Other sequence lengths, AgentX and eval-only do not enable this collector. Hardware qualification and publication remain pending. +The 8k1k recipe requires srt-slurm DCGM telemetry on both worker nodes. The 1k1k recipe does not require power collection. Both recipes allocate one node per role. Hardware qualification and publication remain pending. ## Register an srt-slurm recipe From cabd9855b17bfebc49d89a2084ffb08937c81f45 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 15:52:11 -0500 Subject: [PATCH 5/6] refactor(tilert): keep B200 setup limited to installation --- .../configs/tilert-b200-setup.sh | 20 ------------------- .../runners/launch_b200-nscale-slurm.sh | 5 ++++- 2 files changed, 4 insertions(+), 21 deletions(-) diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh index bf77236d92..810a8b000d 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/tilert-b200-setup.sh @@ -27,23 +27,3 @@ case "$TILERT_ROLE" in exit 1 ;; esac - -if [[ "$TILERT_ROLE" == decode ]]; then - # Keep the shared converted checkpoint reusable across allocations. - mkdir -p /tilert_weights - exec 9>/tilert_weights/.convert.lock - flock -w 21600 9 - if [[ ! -f /tilert_weights/model.safetensors.index.json ]]; then - python3 -m tilert.models.preprocess.weight_converter \ - --model_type glm-5 --model_dir /model --save_dir /tilert_weights - test -f /tilert_weights/model.safetensors.index.json - fi - for file in /model/*; do - [[ -f "$file" ]] || continue - name="${file##*/}" - [[ "$name" == *.safetensors || "$name" == model.safetensors.index.json ]] && continue - [[ -e "/tilert_weights/$name" ]] || cp -p "$file" "/tilert_weights/$name" - done - test -f /tilert_weights/chat_template.jinja - test -f /tilert_weights/tokenizer_config.json || test -f /tilert_weights/tokenizer.json -fi diff --git a/inferencex-e2e/runners/launch_b200-nscale-slurm.sh b/inferencex-e2e/runners/launch_b200-nscale-slurm.sh index 7a70db6cd5..326b098d01 100755 --- a/inferencex-e2e/runners/launch_b200-nscale-slurm.sh +++ b/inferencex-e2e/runners/launch_b200-nscale-slurm.sh @@ -332,7 +332,10 @@ run_native_srt_lane() { fi if [[ $FRAMEWORK == "tilert" ]]; then check_env_vars TILERT_WEIGHTS_DIR - mkdir -p "$TILERT_WEIGHTS_DIR" + if [[ ! -r "$TILERT_WEIGHTS_DIR/model.safetensors.index.json" ]]; then + echo "Missing prepared TileRT checkpoint: $TILERT_WEIGHTS_DIR" >&2 + exit 1 + fi SRT_CLUSTER_ARGS+=( --mount "$TILERT_WEIGHTS_DIR" /tilert_weights --mount "$HF_HUB_CACHE_HOST_PATH" "$HF_HUB_CACHE_HOST_PATH" From dfca01321a3ce3415b0e11ae278c6435bde3ec91 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 17:27:58 -0500 Subject: [PATCH 6/6] fix(tilert): load B200 checkpoints from shared model storage --- inferencex-e2e/configs/nvidia-master.yaml | 10 ++++------ inferencex-e2e/docs/configuration-procedures.md | 2 +- inferencex-e2e/runners/launch_b200-nscale-slurm.sh | 1 - 3 files changed, 5 insertions(+), 8 deletions(-) diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 9ed9d8e401..90c717c3c8 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8174,9 +8174,8 @@ glm5.1-fp8-b200-tilert: additional-settings: - "CONFIG_FILE=recipes/glm5.1/tilert/b200-fp8/1k1k/disagg-1p1d-tp8-mtp.yaml" - "PREFILL_IMAGE=vllm/vllm-openai:v0.26.0" - - "MODEL_PATH=/data/home/sa-shared/gharunners/hf-hub-cache/hub/models--zai-org--GLM-5.1-FP8/snapshots/f396cf805182f4ca10fa675e1a99815b3ca384db" - - "HF_HUB_CACHE_HOST_PATH=/data/home/sa-shared/gharunners/hf-hub-cache" - - "TILERT_WEIGHTS_DIR=/data/home/sa-shared/gharunners/tilert-cache/glm5.1-fp8-8shard" + - "MODEL_PATH=/data/home/sa-shared/gharunners/models/GLM-5.1-FP8-f396cf805182" + - "TILERT_WEIGHTS_DIR=/data/home/sa-shared/gharunners/models/GLM-5.1-FP8-TileRT-8shard" decode: num-worker: 1 tp: 8 @@ -8196,9 +8195,8 @@ glm5.1-fp8-b200-tilert: additional-settings: - "CONFIG_FILE=recipes/glm5.1/tilert/b200-fp8/8k1k/disagg-1p1d-tp8-mtp.yaml" - "PREFILL_IMAGE=vllm/vllm-openai:v0.26.0" - - "MODEL_PATH=/data/home/sa-shared/gharunners/hf-hub-cache/hub/models--zai-org--GLM-5.1-FP8/snapshots/f396cf805182f4ca10fa675e1a99815b3ca384db" - - "HF_HUB_CACHE_HOST_PATH=/data/home/sa-shared/gharunners/hf-hub-cache" - - "TILERT_WEIGHTS_DIR=/data/home/sa-shared/gharunners/tilert-cache/glm5.1-fp8-8shard" + - "MODEL_PATH=/data/home/sa-shared/gharunners/models/GLM-5.1-FP8-f396cf805182" + - "TILERT_WEIGHTS_DIR=/data/home/sa-shared/gharunners/models/GLM-5.1-FP8-TileRT-8shard" decode: num-worker: 1 tp: 8 diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 113757c9ad..6a5dde35ee 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -200,7 +200,7 @@ The GLM-5.1 B200 Nscale 1k1k and 8k1k recipes select the prepared shared checkpo The recipes use vLLM prefill, TileRT decode, and the TileRT router through srt-slurm. The shared `srt_fixed_sequence.sh` runs the benchmark; `srt_eval.sh` handles selected evaluations. srt-slurm owns worker startup, readiness, and teardown. -`MODEL_PATH` selects the shared checkpoint. The launcher mounts its HF cache at the same absolute path so snapshot links to sibling blobs remain readable. `TILERT_WEIGHTS_DIR` selects the separately converted decode weights, mounted at `/tilert_weights`. +`MODEL_PATH` selects the prepared checkpoint on shared model storage, mounted at `/model`. It must not contain relative symlinks outside that directory. `TILERT_WEIGHTS_DIR` selects the separately converted decode weights, mounted at `/tilert_weights`. Recipe setup scripts only install dependencies; checkpoint preparation happens before benchmark submission. The 8k1k recipe requires srt-slurm DCGM telemetry on both worker nodes. The 1k1k recipe does not require power collection. Both recipes allocate one node per role. Hardware qualification and publication remain pending. diff --git a/inferencex-e2e/runners/launch_b200-nscale-slurm.sh b/inferencex-e2e/runners/launch_b200-nscale-slurm.sh index 326b098d01..85715f1849 100755 --- a/inferencex-e2e/runners/launch_b200-nscale-slurm.sh +++ b/inferencex-e2e/runners/launch_b200-nscale-slurm.sh @@ -338,7 +338,6 @@ run_native_srt_lane() { fi SRT_CLUSTER_ARGS+=( --mount "$TILERT_WEIGHTS_DIR" /tilert_weights - --mount "$HF_HUB_CACHE_HOST_PATH" "$HF_HUB_CACHE_HOST_PATH" ) fi