Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view

This file was deleted.

27 changes: 6 additions & 21 deletions inferencex-e2e/benchmarks/multi_node/amd_utils/env.sh
Original file line number Diff line number Diff line change
Expand Up @@ -2,21 +2,12 @@

source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only
check_env_vars ENGINE
# MoRI-IO queue-pair tuning, the UCX RoCE GID index, SGLang router logging and the
# SGLang decode cuda-graph NCCL workaround. Only the SGLang MoRI KV path
# below reads these. ENGINE=tilert moves KV over mooncake and starts no SGLang
# router, so it is neither given nor reads them: validating them there would force
# the recipe to invent MoRI tuning for a transport it never uses.
if [[ "$ENGINE" != "tilert" ]]; then
check_env_vars \
MORI_IO_SQ_BACKOFF_TIMEOUT_US MORI_IO_QP_MAX_SEND_WR MORI_IO_QP_MAX_CQE MORI_IO_QP_MAX_SGE MORI_IO_TC_DISABLE \
UCX_IB_GID_INDEX MORI_APP_LOG_LEVEL SGLANG_ROUTER_STDOUT_LOGS TORCH_NCCL_BLOCKING_WAIT NCCL_BLOCKING_WAIT \
SGLANG_OPT_USE_AITER_INDEXER
fi
# Dual-engine environment setup for multi-node disaggregated serving.
#
# ENGINE=sglang-disagg or tilert selects the engine-specific block.
#
# SGLang MoRI environment.
check_env_vars \
MORI_IO_SQ_BACKOFF_TIMEOUT_US MORI_IO_QP_MAX_SEND_WR MORI_IO_QP_MAX_CQE MORI_IO_QP_MAX_SGE MORI_IO_TC_DISABLE \
UCX_IB_GID_INDEX MORI_APP_LOG_LEVEL SGLANG_ROUTER_STDOUT_LOGS TORCH_NCCL_BLOCKING_WAIT NCCL_BLOCKING_WAIT \
SGLANG_OPT_USE_AITER_INDEXER

# REQUIRED ENVIRONMENT VARIABLES:
# IBDEVICES - RDMA/InfiniBand device names (e.g., ionic_0,ionic_1,... or mlx5_0,mlx5_1,...)
# Set by runner or auto-detected from hostname.
Expand Down Expand Up @@ -119,10 +110,6 @@ else
fi
fi

if [[ "$ENGINE" == "tilert" ]]; then
echo "[INFO] tilert: IBDEVICES=$IBDEVICES NCCL_SOCKET_IFNAME=$NCCL_SOCKET_IFNAME NCCL_IB_HCA=$NCCL_IB_HCA"

else

export SGLANG_USE_AITER=1
export AITER_LOG_LEVEL=ERROR
Expand Down Expand Up @@ -237,5 +224,3 @@ else
export GPU_MAX_HW_QUEUES=2
fi
fi

fi
55 changes: 1 addition & 54 deletions inferencex-e2e/benchmarks/multi_node/amd_utils/job.slurm
Original file line number Diff line number Diff line change
Expand Up @@ -36,8 +36,6 @@ echo ""
# at runtime, but the CWD remains the submit-time directory (amd_utils/).
if [[ "$ENGINE" == "atom-disagg" ]]; then
MODELS_YAML="$(pwd)/models_atom.yaml"
elif [[ "$ENGINE" == "tilert" ]]; then
MODELS_YAML="$(pwd)/models_tilert.yaml"
else
MODELS_YAML="$(pwd)/models.yaml"
fi
Expand All @@ -52,16 +50,6 @@ if [[ -z "${DOCKER_IMAGE_NAME:-}" ]]; then
exit 1
fi

if [[ "$ENGINE" == "tilert" && -z "${PREFILL_IMAGE:-}" ]]; then
echo "Error: ENGINE=tilert requires PREFILL_IMAGE (e.g. PREFILL_IMAGE=vllm/vllm-openai-rocm:nightly-<sha> in prefill.additional-settings)."
exit 1
fi
if [[ "$ENGINE" == "tilert" ]]; then
# server_tilert.sh takes the container-creation barrier timeout from the
# recipe (no 300s default as on the SGLang path); fail here, before sbatch
# work is done, rather than inside the container.
check_env_vars CONTAINER_BARRIER_TIMEOUT
fi

# Resolve the models.yaml entry the same way server_sglang.sh does: agentic runs
# (IS_AGENTIC) use the '<model>-AgentX' recipe, non-agentic disaggregated runs use
Expand Down Expand Up @@ -564,41 +552,6 @@ if [[ "$ENGINE" == "atom-disagg" ]]; then
-e EXTRA_SERVER_ARGS=\${EXTRA_SERVER_ARGS:-}
-e IBDEVICES=${IBDEVICES:-}
)
elif [[ "$ENGINE" == "tilert" ]]; then
DOCKER_ENV_ENGINE=(
-e MODEL_PATH=$DOCKER_MODEL_PATH
-e PREFILL_IMAGE=${PREFILL_IMAGE}
-e TILERT_VERSION=${TILERT_VERSION}
-e TILERT_PROFILE=${TILERT_PROFILE}
-e TILERT_MODEL_TYPE=${TILERT_MODEL_TYPE}
-e TILERT_MODEL_PKG=${TILERT_MODEL_PKG}
-e TILERT_MAX_MODEL_LEN=${TILERT_MAX_MODEL_LEN}
-e TILERT_TRANSPORT=${TILERT_TRANSPORT}
-e TILERT_PARSER=${TILERT_PARSER}
-e TILERT_QUEUE_TIMEOUT=${TILERT_QUEUE_TIMEOUT}
-e TILERT_WEIGHTS_DIR=${TILERT_WEIGHTS_DIR}
-e TILERT_RDMA_STRICT=${TILERT_RDMA_STRICT}
-e TILERT_CONVERT_LOCK_WAIT=${TILERT_CONVERT_LOCK_WAIT}
-e TILERT_SIMULATE_ACC_METHOD=${TILERT_SIMULATE_ACC_METHOD}
-e \"TILERT_EXTRA_ENV=${TILERT_EXTRA_ENV:-}\"
-e SERVED_MODEL_NAME=${SERVED_MODEL_NAME}
-e PREFILL_KV_DTYPE=${PREFILL_KV_DTYPE}
-e PREFILL_BLOCK_SIZE=${PREFILL_BLOCK_SIZE}
-e PREFILL_SPEC_TOKENS=${PREFILL_SPEC_TOKENS}
-e DECODE_KV_DTYPE=${DECODE_KV_DTYPE}
-e GPU_MEM_UTIL=${GPU_MEM_UTIL}
-e PREFILL_PORT=${PREFILL_PORT}
-e DECODE_CTRL_PORT=${DECODE_CTRL_PORT}
-e DECODE_HTTP_PORT=${DECODE_HTTP_PORT}
-e DECODE_WAIT=${DECODE_WAIT}
-e PREFILL_WAIT=${PREFILL_WAIT}
-e ROUTER_WAIT=${ROUTER_WAIT}
-e SKIP_CONTAINER_BARRIER=${SKIP_CONTAINER_BARRIER}
# Golden-acceptance selection on agentic MTP runs; unset elsewhere.
-e THINKING_MODE=${THINKING_MODE:-}
-e IBDEVICES=${IBDEVICES:-}
-e PYTHONPYCACHEPREFIX=/tmp/pycache
)
else
DOCKER_ENV_ENGINE=(
-e SGLANG_WS_PATH=${WS_PATH}
Expand Down Expand Up @@ -751,12 +704,6 @@ else
fi
fi # end: if ENGINE == atom-disagg

RANK_IMAGE=
if [[ \"$ENGINE\" == \"tilert\" && \"\$SLURM_PROCID\" -lt \"$xP\" ]]; then
RANK_IMAGE=\"$PREFILL_IMAGE\"
echo \"[tilert] rank \$SLURM_PROCID is a prefill rank; using PREFILL_IMAGE=\$RANK_IMAGE\"
fi

exec \$DOCKER_CMD run \
--init \
--stop-timeout 10 \
Expand Down Expand Up @@ -798,7 +745,7 @@ exec \$DOCKER_CMD run \
${CLIENT_DOCKER_ENV} \
--name \"$DOCKER_CONT_NAME\" \
--entrypoint \"\" \
\"\${RANK_IMAGE:-$DOCKER_IMAGE_NAME}\" bash -lc '
\"$DOCKER_IMAGE_NAME\" bash -lc '
set -o pipefail
mkdir -p /run_logs/slurm_job-'\"\$SLURM_JOB_ID\"'
'"$RUN_FILE_FULL"' 2>&1 | tee /run_logs/slurm_job-'\"\$SLURM_JOB_ID\"'/server_\$(hostname).log
Expand Down

This file was deleted.

3 changes: 0 additions & 3 deletions inferencex-e2e/benchmarks/multi_node/amd_utils/server.sh
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,6 @@ source "$(dirname "${BASH_SOURCE[0]}")/../../benchmark_lib.sh" --validation-only
# Dispatches to the engine-specific server launcher based on ENGINE env var.
# ENGINE=sglang-disagg (default) -> server_sglang.sh (SGLang + MoRI)
# ENGINE=atom-disagg -> server_atom.sh (ATOM + mooncake)
# ENGINE=tilert -> server_tilert.sh (vLLM prefill + TileRT decode)

check_env_vars ENGINE WS_PATH
if [[ -f /config/hicache_mc.env ]]; then
Expand All @@ -20,8 +19,6 @@ echo "[DISPATCHER] ENGINE=$ENGINE WS_PATH=$WS_PATH"
if [[ "$ENGINE" == "atom-disagg" ]]; then
export ATOM_WS_PATH="$WS_PATH"
source "$WS_PATH/server_atom.sh"
elif [[ "$ENGINE" == "tilert" ]]; then
source "$WS_PATH/server_tilert.sh"
else
source "$WS_PATH/server_sglang.sh"
fi
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,6 @@ check_env_vars \

EXTRA_SERVER_ARGS="${EXTRA_SERVER_ARGS:-}"

source $ATOM_WS_PATH/setup_deps.sh
source $ATOM_WS_PATH/env_atom.sh

# lm-eval with high num_concurrent exhausts the default 1024 FD limit.
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,6 @@ BENCH_MAX_CONC_VALUE=$(echo "$BENCH_MAX_CONCURRENCY" | tr 'x' '\n' | sort -n | t
# can resolve formulas like "BENCH_MAX_CONC_VALUE*2" for max_running_requests.
export BENCH_MAX_CONC_VALUE

source $SGLANG_WS_PATH/setup_deps.sh
source $SGLANG_WS_PATH/env.sh

# Install before starting UMBP or serving processes. Early readiness failures must
Expand Down
Loading