Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view

This file was deleted.

Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
#!/usr/bin/env bash
set -eo pipefail
source /infmax-workspace/benchmarks/benchmark_lib.sh --validation-only
check_env_vars TILERT_ROLE

case "$TILERT_ROLE" in
prefill)
python3 -m pip install --quiet --no-cache-dir --no-deps tilert==0.1.5.post3
if ! python3 -c 'import nixl' 2>/dev/null; then
python3 -m pip install --quiet --no-cache-dir nixl==1.3.1
fi
;;
decode|router)
python3 -m pip install --quiet --no-cache-dir tilert==0.1.5.post3
if ! python3 -c 'import uvicorn' 2>/dev/null; then
python3 -m pip install --quiet --no-cache-dir fastapi uvicorn httpx
fi
if ! python3 -c 'import nixl' 2>/dev/null; then
python3 -m pip install --quiet --no-cache-dir nixl==1.3.1
fi
if ! python3 -c 'from importlib.metadata import version; assert int(version("transformers").split(".")[0]) >= 5'; then
python3 -m pip install --quiet --no-cache-dir 'transformers>=5.4.0'
fi
;;
*)
echo "Unknown TileRT setup role: $TILERT_ROLE" >&2
exit 1
;;
esac
Original file line number Diff line number Diff line change
@@ -0,0 +1,88 @@
schema: 2
name: glm5.1-tilert-b200-1k1k-1p1d-tp8-mtp
model:
path: glm5.1-fp8
container: ghcr.io/tile-ai/tilert:0.1.5
precision: fp8
slurm:
time_limit: "00:45:00"
resources:
gpu_type: b200
gpus_per_node: 8
setup_script: tilert-b200-setup.sh
environment:
NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility
roles:
prefill:
engine:
type: vllm
set_visible_devices: true
container: vllm/vllm-openai:v0.26.0
nodes: 1
workers: 1
gpus: 8
env:
TILERT_ROLE: prefill
args:
served-model-name: [glm5, zai-org/GLM-5.1-FP8]
tensor-parallel-size: 8
max-model-len: 2304
enforce-eager: true
trust-remote-code: true
return-tokens-as-token-ids: true
gpu-memory-utilization: 0.75
kv-cache-dtype: fp8_ds_mla
speculative-config: '{"method":"mtp","num_speculative_tokens":1}'
kv-transfer-config: >-
{"kv_connector":"TileRTConnector","kv_connector_module_path":"tilert.pd_vllm.prefill_connector",
"kv_role":"kv_producer","kv_connector_extra_config":{
"tilert_model":"glm5","tilert_max_seq_len":2304,"tilert_transport":"nixl"}}
decode:
engine:
type: tilert
served_model_name: glm5
container: ghcr.io/tile-ai/tilert:0.1.5
nodes: 1
workers: 1
gpus: 8
env:
TILERT_ROLE: decode
PATH: /opt/conda/envs/tilert/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
args:
model: glm5
model-weights-dir: /tilert_weights
max-seq-len: 2304
kv-cache-dtype: fp8
transport: nixl
with-mtp: true
frontend:
type: tilert-router
container_image: ghcr.io/tile-ai/tilert:0.1.5
enable_multiple_frontends: false
env:
TILERT_ROLE: router
PATH: /opt/conda/envs/tilert/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
args:
model-path: /model
parser: none
queue-timeout: 0
srun_options:
container-writable: ""
container-remap-root: ""
health_check:
max_attempts: 720
interval_seconds: 5
benchmark:
type: custom
container_image: vllm/vllm-openai:v0.26.0
concurrencies: [1]
command: bash /infmax-workspace/benchmarks/multi_node/srt_fixed_sequence.sh
env:
ISL: "1024"
OSL: "1024"
CLIENT_BACKEND: openai-chat
BENCHMARK_SERVED_MODEL_NAME: glm5
USE_CHAT_TEMPLATE: "true"
TOKENIZER: /model
NUM_PROMPTS: "16"
Original file line number Diff line number Diff line change
@@ -0,0 +1,98 @@
schema: 2
name: glm5.1-tilert-b200-8k1k-1p1d-tp8-mtp
model:
path: glm5.1-fp8
container: ghcr.io/tile-ai/tilert:0.1.5
precision: fp8
slurm:
time_limit: "01:30:00"
resources:
gpu_type: b200
gpus_per_node: 8
setup_script: tilert-b200-setup.sh
environment:
NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility
roles:
prefill:
engine:
type: vllm
set_visible_devices: true
container: vllm/vllm-openai:v0.26.0
nodes: 1
workers: 1
gpus: 8
env:
TILERT_ROLE: prefill
args:
served-model-name: [glm5, zai-org/GLM-5.1-FP8]
tensor-parallel-size: 8
max-model-len: 9472
enforce-eager: true
trust-remote-code: true
return-tokens-as-token-ids: true
gpu-memory-utilization: 0.75
kv-cache-dtype: fp8_ds_mla
speculative-config: '{"method":"mtp","num_speculative_tokens":1}'
kv-transfer-config: >-
{"kv_connector":"TileRTConnector","kv_connector_module_path":"tilert.pd_vllm.prefill_connector",
"kv_role":"kv_producer","kv_connector_extra_config":{
"tilert_model":"glm5","tilert_max_seq_len":9472,"tilert_transport":"nixl"}}
decode:
engine:
type: tilert
served_model_name: glm5
container: ghcr.io/tile-ai/tilert:0.1.5
nodes: 1
workers: 1
gpus: 8
env:
TILERT_ROLE: decode
PATH: /opt/conda/envs/tilert/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
args:
model: glm5
model-weights-dir: /tilert_weights
max-seq-len: 9472
kv-cache-dtype: fp8
transport: nixl
with-mtp: true
frontend:
type: tilert-router
container_image: ghcr.io/tile-ai/tilert:0.1.5
enable_multiple_frontends: false
env:
TILERT_ROLE: router
PATH: /opt/conda/envs/tilert/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
args:
model-path: /model
parser: none
queue-timeout: 0
srun_options:
container-writable: ""
container-remap-root: ""
health_check:
max_attempts: 720
interval_seconds: 5
telemetry:
enabled: true
collect_interval_ms: 1000
storage_subdir: power
required: true
startup_timeout_seconds: 120
request_timeout_seconds: 2
dcgm_exporter:
container_image: dcgm-exporter
port: 9401
benchmark:
type: custom
container_image: vllm/vllm-openai:v0.26.0
concurrencies: [1]
command: bash /infmax-workspace/benchmarks/multi_node/srt_fixed_sequence.sh
env:
ISL: "8192"
OSL: "1024"
CLIENT_BACKEND: openai-chat
BENCHMARK_SERVED_MODEL_NAME: glm5
USE_CHAT_TEMPLATE: "true"
TOKENIZER: /model
NUM_PROMPTS: "16"
Loading