diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/clean_stale_shm.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/clean_stale_shm.sh new file mode 100644 index 0000000000..64e2e169f9 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/clean_stale_shm.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash + +# Use only with exclusive node allocations. Anchor the cutoff to allocation +# start so a later frontend/worker setup cannot remove this job's segments. +clean_stale_shm() { + python3 -S - "$@" <<'PY' +import os +import socket +import stat +import sys +import time +from pathlib import Path + +directory = Path(sys.argv[1]) +job_start = int(sys.argv[2]) +if job_start <= 600 or job_start > time.time(): + raise ValueError("SLURM_JOB_START_TIME must be a valid allocation start timestamp") +cutoff = job_start - 600 +prefixes = ("vader_segment.", "nccl-", "sem.", "psm3", "fe80::") +entries = list(directory.iterdir()) +removed = 0 +for entry in entries: + if not entry.name.startswith(prefixes): + continue + try: + info = entry.lstat() + if (info.st_uid != os.getuid() or not stat.S_ISREG(info.st_mode) + or info.st_mtime >= cutoff): + continue + entry.unlink() + removed += 1 + except FileNotFoundError: + # Another rank may have already removed the same stale segment. + continue +print(f"[clean_stale_shm] {socket.gethostname()}: removed {removed} stale segments " + f"from {len(entries)} entries; cutoff={cutoff}") +PY +} + +if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then + source /infmax-workspace/benchmarks/benchmark_lib.sh --validation-only || exit 1 + check_env_vars SLURM_JOB_START_TIME || exit 1 + clean_stale_shm /dev/shm "$SLURM_JOB_START_TIME" +fi diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/diagnose-pmix.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/diagnose-pmix.sh new file mode 100644 index 0000000000..23364bf215 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/diagnose-pmix.sh @@ -0,0 +1,61 @@ +#!/usr/bin/env bash + +bash "$(dirname "${BASH_SOURCE[0]}")/clean_stale_shm.sh" || exit 1 + +# Read launch metadata without initializing MPI or changing its configuration. +python3 -S - <<'PY' +import json +import os +import socket +import stat +from pathlib import Path + +names = ( + "SLURM_MPI_TYPE", "SLURM_JOB_ID", "SLURM_STEP_ID", "SLURM_JOB_UID", + "SLURM_PROCID", "SLURM_LOCALID", "PMIX_NAMESPACE", "PMIX_RANK", + "PMIX_SERVER_URI", "PMIX_SERVER_URI2", "PMIX_SERVER_URI21", + "PMIX_SERVER_URI3", "PMIX_SERVER_URI4", "PMIX_SERVER_URI41", + "PMIX_SERVER_TMPDIR", "PMIX_SYSTEM_TMPDIR", "PMIX_SECURITY_MODE", + "PMIX_PTL_MODULE", "PMIX_GDS_MODULE", "PMIX_MCA_psec", + "PMIX_MCA_ptl", "PMIX_MCA_gds", +) +environment = {name: os.environ[name] for name in names if name in os.environ} +paths = {environment[name] for name in ("PMIX_SERVER_TMPDIR", "PMIX_SYSTEM_TMPDIR") + if environment.get(name, "").startswith("/")} +for name, value in environment.items(): + if name.startswith("PMIX_SERVER_URI"): + for component in value.split(";"): + if component.startswith("file:"): + component = component.removeprefix("file:") + if component.startswith("/"): + paths.add(component) +job = environment.get("SLURM_JOB_ID", "") +step = environment.get("SLURM_STEP_ID", "") +uid = environment.get("SLURM_JOB_UID", "") +if job.isdecimal() and step.isdecimal(): + paths.update((f"/var/spool/slurmd/pmix.{job}.{step}", + f"/tmp/spmix_appdir_{job}.{step}")) + if uid.isdecimal(): + paths.add(f"/tmp/spmix_appdir_{uid}_{job}.{step}") +metadata = {} +for name in sorted(paths): + try: + info = Path(name).stat() + metadata[name] = {"uid": info.st_uid, "gid": info.st_gid, + "mode": stat.filemode(info.st_mode)} + except OSError as error: + metadata[name] = {"errno": error.errno} +try: + uid_map = Path("/proc/self/uid_map").read_text().strip() +except OSError as error: + uid_map = f"unavailable: errno={error.errno}" +print("PMIx launch diagnostics: " + json.dumps({ + "hostname": socket.gethostname(), "uid": os.getuid(), "gid": os.getgid(), + "uid_map": uid_map, "environment": environment, "paths": metadata, +}, sort_keys=True)) +PY + +if command -v ompi_info >/dev/null 2>&1; then + timeout 10s ompi_info --version || true + timeout 10s ompi_info --param pmix all || true +fi diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/prepare-dynamo-venv.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/prepare-dynamo-venv.sh new file mode 100644 index 0000000000..3134349590 --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/prepare-dynamo-venv.sh @@ -0,0 +1,57 @@ +#!/usr/bin/env bash + +# Source before the SRT installer so its lock and packages share a writable prefix. +source /infmax-workspace/benchmarks/benchmark_lib.sh --validation-only || return 1 +check_env_vars SLURM_JOB_ID SLURM_STEP_ID || return 1 + +INFX_DYNAMO_VENV=$(python3 - <<'PY' +import fcntl +import os +import re +import venv +from pathlib import Path + +job = os.environ["SLURM_JOB_ID"] +step = os.environ["SLURM_STEP_ID"] +if not re.fullmatch(r"[0-9]+", job) or not re.fullmatch(r"[0-9]+", step): + raise ValueError("Dynamo virtual environment requires numeric Slurm job and step IDs") +root = Path("/tmp") / f"infx-dynamo-{job}.{step}" +root.mkdir(mode=0o700, exist_ok=True) +environment = root / "venv" +with (root / "create.lock").open("w") as lock: + fcntl.flock(lock, fcntl.LOCK_EX) + if not (root / "complete").exists(): + venv.EnvBuilder(system_site_packages=True, with_pip=True, symlinks=True).create(environment) + (root / "complete").touch() +print(environment) +PY +) || return 1 +export VIRTUAL_ENV="$INFX_DYNAMO_VENV" +export PATH="$VIRTUAL_ENV/bin:$PATH" +unset INFX_DYNAMO_VENV + +# Keep transient package download failures inside SRT's existing install lock. +# Other pip invocations retain their original behavior. +pip() { + if [[ $# -ne 7 || "$1" != install || "$2" != --break-system-packages || + "$3" != --quiet || "$4" != --extra-index-url || + "$5" != https://pypi.nvidia.com || "$6" != ai-dynamo-runtime==?* || + "$7" != "ai-dynamo==${6#ai-dynamo-runtime==}" ]]; then + command pip "$@" + return $? + fi + + local attempt install_status + for attempt in 1 2 3; do + if command pip "$@"; then + return 0 + else + install_status=$? + fi + if [[ "$attempt" -lt 3 ]]; then + echo "Dynamo install attempt $attempt failed; retrying unchanged packages." >&2 + sleep 5 || return $? + fi + done + return "$install_status" +} diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml new file mode 100644 index 0000000000..44b24684dd --- /dev/null +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml @@ -0,0 +1,500 @@ +# AgentX GLM-5.2 TensorRT-LLM B300 recipes: shared settings in base, one +# override per benchmark configuration. Select one with +# CONFIG_FILE=recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml:override_. + +schema: 2 +base: + slurm: + time_limit: "06:00:00" + srun_options: + no-container-remap-root: '' + engine: + type: trtllm + numa_cpu_bind: true + publish_events_and_metrics: false + roles: + prefill: + gpus: 4 + env: + DYN_ENGINE_CONV_AFFINITY: '1' + DYN_PUBLISH_KV_EVENTS: '0' + DYN_TOKENIZER: fastokens + DYN_TRTLLM_ENABLE_ATTENTION_DP: '1' + ETCD_LEASE_TTL: '120' + HF_HUB_DISABLE_PROGRESS_BARS: '1' + HF_HUB_OFFLINE: '1' + HOME: /trtllm-jit-cache + MIMALLOC_PURGE_DELAY: '10000' + NCCL_GRAPH_MIXING_SUPPORT: '0' + FI_LOG_LEVEL: warn + NIXL_DISABLE_CUDA_ADDR_WA: '1' + OMP_NUM_THREADS: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: '0.10' + TLLM_EXECUTOR_BATCH_RESP_IN_AWAIT: '1' + TLLM_EXECUTOR_USE_FILE_SOCKET: '1' + TLLM_LOG_LEVEL: INFO + TLLM_PREFIX_TOKEN_CACHE: '1' + TQDM_DISABLE: '1' + TRANSFORMERS_OFFLINE: '1' + TRTLLM_DSA_INDEXER_BF16: '1' + TRTLLM_ENABLE_PDL: '1' + TRTLLM_FUSED_DSA_METADATA: '1' + TRTLLM_KVCACHE_POOL_USE_FABRIC_MEMORY: '0' + TRTLLM_KVCACHE_RECV_BUFFER_COUNT: '1' + TRTLLM_KVCACHE_SEND_BUFFER_COUNT: '1' + TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: '600' + TRTLLM_NIXL_KVCACHE_BACKEND: LIBFABRIC + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_SERVE_ENABLE_MSGSPEC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + args: + attention_dp_config: + enable_kv_cache_aware_routing: false + kv_cache_routing_conversation_affinity: true + kv_cache_routing_max_sessions: 65536 + backend: pytorch + cache_transceiver_config: + backend: NIXL + kv_cache_bounce_size_mb: '0' + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 1048576 + transceiver_runtime: PYTHON + cuda_graph_config: null + enable_attention_dp: true + enable_chunked_prefill: true + gpus_per_node: 8 + kv_cache_config: + use_kv_cache_manager_v2: false + dtype: fp8 + enable_block_reuse: true + event_buffer_max_size: 0 + free_gpu_memory_fraction: 0.75 + tokens_per_block: 64 + max_batch_size: 256 + max_num_tokens: 8192 + max_seq_len: 1048576 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 4 + num_postprocess_workers: 8 + perf_metrics_max_requests: 100000 + pipeline_parallel_size: 1 + print_iter_log: true + return_perf_metrics: true + scheduler_config: + capacity_scheduler_policy: MAX_UTILIZATION + sparse_attention_config: + algorithm: dsa + enable_heuristic_topk: true + speculative_config: + decoding_type: MTP + tensor_parallel_size: 4 + trust_remote_code: true + decode: + env: + DYN_ENGINE_CONV_AFFINITY: '1' + DYN_PUBLISH_KV_EVENTS: '0' + DYN_TOKENIZER: fastokens + DYN_TRTLLM_ENABLE_ATTENTION_DP: '1' + ETCD_LEASE_TTL: '120' + HF_HUB_DISABLE_PROGRESS_BARS: '1' + HF_HUB_OFFLINE: '1' + HOME: /trtllm-jit-cache + MIMALLOC_PURGE_DELAY: '10000' + NCCL_GRAPH_MIXING_SUPPORT: '0' + FI_LOG_LEVEL: warn + NIXL_DISABLE_CUDA_ADDR_WA: '1' + OMP_NUM_THREADS: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + TLLM_ADP_ROUTER_MATCH_RATE_THRESHOLD: '0.10' + TLLM_EXECUTOR_BATCH_RESP_IN_AWAIT: '1' + TLLM_EXECUTOR_USE_FILE_SOCKET: '1' + TLLM_LOG_LEVEL: INFO + TLLM_PREFIX_TOKEN_CACHE: '1' + TQDM_DISABLE: '1' + TRANSFORMERS_OFFLINE: '1' + TRTLLM_DSA_INDEXER_BF16: '1' + TRTLLM_ENABLE_PDL: '1' + TRTLLM_FUSED_DSA_METADATA: '1' + TRTLLM_KVCACHE_POOL_USE_FABRIC_MEMORY: '0' + TRTLLM_KVCACHE_RECV_BUFFER_COUNT: '1' + TRTLLM_KVCACHE_SEND_BUFFER_COUNT: '1' + TRTLLM_KV_CACHE_TRANSFER_TIMEOUT_SEC: '600' + TRTLLM_NIXL_KVCACHE_BACKEND: LIBFABRIC + TRTLLM_SERVER_DISABLE_GC: '1' + TRTLLM_SERVE_ENABLE_MSGSPEC: '1' + TRTLLM_WORKER_DISABLE_GC: '1' + args: + backend: pytorch + cache_transceiver_config: + backend: NIXL + kv_cache_bounce_size_mb: '0' + kv_transfer_timeout_ms: 600000 + max_tokens_in_buffer: 1048576 + transceiver_runtime: PYTHON + cuda_graph_config: + enable_padding: true + gpus_per_node: 8 + kv_cache_config: + use_kv_cache_manager_v2: false + dtype: fp8 + enable_block_reuse: false + event_buffer_max_size: 0 + host_cache_size: 0 + tokens_per_block: 64 + max_num_tokens: 128 + max_seq_len: 1048576 + num_postprocess_workers: 4 + perf_metrics_max_requests: 100000 + pipeline_parallel_size: 1 + print_iter_log: true + return_perf_metrics: true + sparse_attention_config: + algorithm: dsa + enable_heuristic_topk: true + use_cute_dsl_paged_mqa_logits: true + speculative_config: + decoding_type: MTP + stream_interval: 20 + trust_remote_code: true + benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/srt_agentic.sh + env: + AIPERF_APPLY_CHAT_TEMPLATE: 'true' + AIPERF_DATASET_CONFIGURATION_TIMEOUT: '1800' + AIPERF_DATASET_MMAP_CACHE_DIR: /aiperf_mmap_cache + AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES: '0' + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: '3600' + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: 'true' + AIPERF_SERVICE_PROFILE_CONFIGURE_TIMEOUT: '1800' + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: '0' + DURATION: '3600' + FRAMEWORK: dynamo-trt + HF_HUB_CACHE: /hf_hub_cache + HF_HUB_DISABLE_PROGRESS_BARS: '1' + INFMAX_CONTAINER_WORKSPACE: /infmax-workspace + IS_MULTINODE: 'true' + KV_OFFLOADING: none + MAX_MODEL_LEN: '1048576' + MODEL: nvidia/GLM-5.2-NVFP4 + MODEL_PREFIX: glm5.2 + OPENAI_API_KEY: EMPTY + PORT: '8000' + PRECISION: fp4 + RESULT_DIR: /logs/agentic + SERVED_MODEL_NAME: GLM-5.2-NVFP4 + TQDM_DISABLE: '1' + WEKA_LOADER_OVERRIDE: semianalysis_cc_traces_weka_062126 + dynamo: + install: true + source: + pypi: 1.4.0 + request_plane: tcp + frontend: + args: + active-decode-blocks-threshold: None + active-prefill-tokens-threshold: None + active-prefill-tokens-threshold-frac: None + no-kv-events: true + router-mode: kv + env: + DYN_LOG: warn + DYN_ROUTER_QUEUE_THRESHOLD: None + DYN_ROUTER_SESSION_AFFINITY_TTL_SECS: '14400' + DYN_ROUTER_TEMPERATURE: '0' + DYN_TCP_REQUEST_TIMEOUT: '30' + DYN_TOKENIZER: fastokens + DYN_TOKENIZER_CACHE: '1' + DYN_TOKENIZER_CACHE_BYTES: '8000000000' + ETCD_LEASE_TTL: '120' + type: dynamo + health_check: + interval_seconds: 10 + max_attempts: 180 + identity: + model: + repo: nvidia/GLM-5.2-NVFP4 + container: + image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 + frameworks: + dynamo: 1.4.0 + tensorrt_llm: 1.3.0rc26.dev202609040000 + model: + path: nvidia/GLM-5.2-NVFP4 + container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 + precision: fp4 + resources: + spread_workers: true + gpu_type: b300 + gpus_per_node: 8 +override_1p1d_tep8_c1_b1_mtp5: + setup_script: clean_stale_shm.sh + roles: + prefill: + nodes: 1 + workers: 1 + args: + disable_overlap_scheduler: true + kv_cache_config: + host_cache_size: 412316860416 + scheduler_config: + context_chunking_policy: EQUAL_PROGRESS + speculative_config: + num_nextn_predict_layers: 5 + decode: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: + - 1 + enable_attention_dp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 1 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 1 + sparse_attention_config: + use_cute_dsl_topk: true + speculative_config: + max_draft_len: 5 + tensor_parallel_size: 8 + benchmark: + env: + CONC: '1' + placement: + node: first_decode + frontend: + enable_multiple_frontends: false + placement: + node: first_decode + name: dynamo-disagg-b300-1p1d-tep8-compact-c1-b1-mtp5 +override_1p1d_tep8_c20_b5_mtp5: + setup_script: diagnose-pmix.sh + roles: + prefill: + nodes: 1 + workers: 1 + env: + NIXL_LIBFABRIC_NUM_THREADS: '1' + TRTLLM_NIXL_NUM_THREADS: '1' + args: + disable_overlap_scheduler: true + kv_cache_config: + host_cache_size: 412316860416 + scheduler_config: + context_chunking_policy: EQUAL_PROGRESS + speculative_config: + num_nextn_predict_layers: 5 + decode: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 5 + enable_attention_dp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 5 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 1 + speculative_config: + max_draft_len: 5 + tensor_parallel_size: 8 + benchmark: + env: + CONC: '20' + placement: + node: head + frontend: + enable_multiple_frontends: false + placement: + node: head + name: dynamo-disagg-b300-1p1d-tep8-compact-c20-b5-mtp5 +override_1p4d_tep4_c30_b2_mtp5: + setup_script: clean_stale_shm.sh + roles: + prefill: + nodes: 1 + workers: 1 + args: + disable_overlap_scheduler: true + kv_cache_config: + host_cache_size: 274877906944 + scheduler_config: + context_chunking_policy: EQUAL_PROGRESS + speculative_config: + num_nextn_predict_layers: 5 + decode: + nodes: 4 + workers: 4 + gpus: 4 + args: + cuda_graph_config: + batch_sizes: + - 1 + - 2 + enable_attention_dp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 2 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 1 + speculative_config: + max_draft_len: 5 + tensor_parallel_size: 4 + benchmark: + env: + CONC: '30' + placement: + node: head + frontend: + enable_multiple_frontends: false + placement: + node: head + name: dynamo-disagg-b300-1p4d-tep4-compact-c30-b2-mtp5 +override_3p4d_tep4_c60_b5_mtp5: + setup_script: clean_stale_shm.sh + roles: + prefill: + nodes: 3 + workers: 3 + args: + disable_overlap_scheduler: true + kv_cache_config: + host_cache_size: 197568495616 + scheduler_config: + context_chunking_policy: EQUAL_PROGRESS + speculative_config: + num_nextn_predict_layers: 5 + decode: + nodes: 4 + workers: 4 + gpus: 4 + args: + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 5 + enable_attention_dp: false + kv_cache_config: + free_gpu_memory_fraction: 0.8 + max_batch_size: 5 + moe_config: + backend: TRTLLM + moe_expert_parallel_size: 1 + speculative_config: + max_draft_len: 5 + tensor_parallel_size: 4 + benchmark: + env: + CONC: '60' + placement: + node: head + frontend: + enable_multiple_frontends: false + placement: + node: head + name: dynamo-disagg-b300-3p4d-tep4-compact-c60-b5-mtp5 +override_6p1d_dep8_c227_b16_mtp3: + setup_script: clean_stale_shm.sh + roles: + prefill: + nodes: 6 + workers: 6 + args: + disable_overlap_scheduler: true + kv_cache_config: + host_cache_size: 197568495616 + scheduler_config: + context_chunking_policy: EQUAL_PROGRESS + speculative_config: + num_nextn_predict_layers: 3 + decode: + nodes: 1 + workers: 1 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 16 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 8 + speculative_config: + max_draft_len: 3 + tensor_parallel_size: 8 + benchmark: + env: + CONC: '227' + placement: + node: first_decode + frontend: + enable_multiple_frontends: true + name: dynamo-disagg-b300-6p1d-dep8-compact-c227-b16-mtp3 +override_8p2d_dep8_c233_b16_mtp3: + setup_script: clean_stale_shm.sh + roles: + prefill: + nodes: 8 + workers: 8 + args: + disable_overlap_scheduler: false + kv_cache_config: + host_cache_size: 197568495616 + speculative_config: + num_nextn_predict_layers: 3 + decode: + nodes: 2 + workers: 2 + gpus: 8 + args: + cuda_graph_config: + batch_sizes: + - 1 + - 2 + - 4 + - 8 + - 16 + enable_attention_dp: true + enable_lm_head_tp_in_adp: false + kv_cache_config: + free_gpu_memory_fraction: 0.9 + max_batch_size: 16 + moe_config: + backend: CUTEDSL + moe_expert_parallel_size: 8 + speculative_config: + max_draft_len: 3 + tensor_parallel_size: 8 + benchmark: + env: + CONC: '233' + placement: + node: first_decode + frontend: + enable_multiple_frontends: true + name: dynamo-disagg-b300-8p2d-dep8-compact-c233-b16-mtp3 diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index 46976f0f0f..c44c2e484a 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -8924,3 +8924,108 @@ dsv41flash-fp4-gb300-sglang-agentic-dspark: search-space: - { tp: 2, ep: 2, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } - { tp: 4, ep: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 8, 16, 32, 64, 128], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv41flash/sglang/gb300-fp4-mtp/agentic.yaml } + +glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact: + image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc26.dev202609040000 + model: nvidia/GLM-5.2-NVFP4 + model-prefix: glm5.2 + runner: cluster:b300-dsxe + precision: fp4 + framework: dynamo-trt + router: { name: dynamo-router, version: "1.4.0.dev20260807" } + kv-p2p-transfer: nixl + multinode: true + disagg: true + scenarios: + agentic-coding: + - search-space: + - spec-decoding: mtp + conc-list: [1] + kv-offloading: none + prefill: + num-worker: 1 + tp: 4 + ep: 4 + dp-attn: true + additional-settings: + - CONFIG_FILE=recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml:override_1p1d_tep8_c1_b1_mtp5 + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + - spec-decoding: mtp + conc-list: [20] + kv-offloading: none + prefill: + num-worker: 1 + tp: 4 + ep: 4 + dp-attn: true + additional-settings: + - CONFIG_FILE=recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml:override_1p1d_tep8_c20_b5_mtp5 + decode: + num-worker: 1 + tp: 8 + ep: 1 + dp-attn: false + - spec-decoding: mtp + conc-list: [30] + kv-offloading: none + prefill: + num-worker: 1 + tp: 4 + ep: 4 + dp-attn: true + additional-settings: + - CONFIG_FILE=recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml:override_1p4d_tep4_c30_b2_mtp5 + decode: + num-worker: 4 + tp: 4 + ep: 1 + dp-attn: false + - spec-decoding: mtp + conc-list: [60] + kv-offloading: none + prefill: + num-worker: 3 + tp: 4 + ep: 4 + dp-attn: true + additional-settings: + - CONFIG_FILE=recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml:override_3p4d_tep4_c60_b5_mtp5 + decode: + num-worker: 4 + tp: 4 + ep: 1 + dp-attn: false + - spec-decoding: mtp + conc-list: [227] + kv-offloading: none + prefill: + num-worker: 6 + tp: 4 + ep: 4 + dp-attn: true + additional-settings: + - CONFIG_FILE=recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml:override_6p1d_dep8_c227_b16_mtp3 + decode: + num-worker: 1 + tp: 8 + ep: 8 + dp-attn: true + - spec-decoding: mtp + conc-list: [233] + kv-offloading: none + prefill: + num-worker: 8 + tp: 4 + ep: 4 + dp-attn: true + additional-settings: + - CONFIG_FILE=recipes/glm5.2/trtllm/b300-fp4/agentx/disagg-variants.yaml:override_8p2d_dep8_c233_b16_mtp3 + decode: + num-worker: 2 + tp: 8 + ep: 8 + dp-attn: true diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 50986d4c72..7fa4fb9650 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -218,6 +218,10 @@ Mapping source: [`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchm Do not ship one side alone. `srtctl` reads the recipe, while matrix generation reads the master config. Recipe-only changes can mislabel results. Master-only changes do not alter the deployed recipe. +The B300 AgentX launcher forwards the workflow-owned `RESULT_FILENAME` through a native SRT override, and the GLM-5.2 compact recipes leave that name to the caller. It collects the resulting `_concN.json` files from the mounted workspace and requires the Slurm allocation to finish as `COMPLETED` with `ExitCode=0:0`; partial aggregates cannot hide a failed replay or request-error gate. Accounting checks retry briefly for delayed terminal records, and the launcher stages logs and results before returning a failure. + +The GLM-5.2 B300 compact Dynamo recipes prepare a writable virtual environment for workers and frontends. The helper makes up to three attempts to install the exact versioned Dynamo packages inside SRT's existing install lock, preserving package hash checks and the final failure status. Prefill and decode workers retain the `LIBFABRIC` NIXL backend and set `NIXL_DISABLE_CUDA_ADDR_WA=1`. Both roles select `kv_cache_config.use_kv_cache_manager_v2: false` and set `TRTLLM_KVCACHE_POOL_USE_FABRIC_MEMORY=0` to use the legacy KV manager with standard CUDA KV allocations. The GLM-5.2 indexer and MTP settings remain explicit. Both worker roles set `OMP_NUM_THREADS=1`, `MIMALLOC_PURGE_DELAY=10000`, `ETCD_LEASE_TTL=120`, and `FI_LOG_LEVEL=warn`. Each worker occupies a separate node (`resources.spread_workers: true`, with role node counts equal to worker counts). The native `engine.numa_cpu_bind` helper applies each MPI rank's GPU-local NUMA CPU mask before execution, without OpenMP binding variables. The readiness gate requires all workers to register within 30 minutes for both benchmark and eval jobs; the allocation retains its six-hour limit for completed workloads. A setup helper removes matching, same-user stale shared-memory files only when they predate the allocation start by more than ten minutes. This fixed cutoff preserves the current allocation's files when setup runs again for another rank or frontend; concurrency 20 retains its PMIx diagnostics after cleanup. For the single-frontend recipes selected for eval, `frontend.placement.node` and `benchmark.placement.node` are `head`, matching the pinned launcher's loopback eval endpoint. + ## Register an llm-d recipe Sources: [`benchmarks/llm-d/README.md`](../benchmarks/llm-d/README.md), [`benchmarks/multi_node/llm-d/README.md`](../benchmarks/multi_node/llm-d/README.md), and [`llm-d-recipes/`](../benchmarks/multi_node/llm-d-recipes). diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 089763c4e0..a6aefa9355 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -167,6 +167,10 @@ B200 Nscale 的 GLM-5.1 可用 `MODEL_PATH` 指定已有共享权重,覆盖默 不得只提交一侧:`srtctl` 读取配方,而矩阵生成读取主配置。仅改配方可能给结果贴错标签;仅改主配置不会改变实际部署的配方。 +B300 AgentX 启动脚本通过 SRT 原生覆盖参数传递工作流提供的 `RESULT_FILENAME`,GLM-5.2 紧凑型配方由调用方指定该名称。启动脚本从挂载的工作区收集生成的 `_concN.json` 文件,并要求 Slurm 作业以 `COMPLETED` 状态及 `ExitCode=0:0` 结束,避免部分聚合结果掩盖回放失败或请求错误率检查失败。对于延迟出现的最终记账记录,检查会进行有限次数的重试;启动脚本在返回失败前保留日志和结果。 + +GLM-5.2 B300 紧凑型 Dynamo 配方为工作进程和前端准备可写的虚拟环境。辅助脚本在 SRT 现有安装锁内,最多尝试三次安装相同的指定版本 Dynamo 软件包,并保留软件包哈希校验和最终失败状态。prefill 和 decode 工作进程继续使用 `LIBFABRIC` NIXL 后端,并设置 `NIXL_DISABLE_CUDA_ADDR_WA=1`,两类工作进程均选择 `kv_cache_config.use_kv_cache_manager_v2: false` 并设置 `TRTLLM_KVCACHE_POOL_USE_FABRIC_MEMORY=0`,使用旧版 KV 管理器和标准 CUDA KV 内存分配。GLM-5.2 的 indexer 和 MTP 设置保持显式配置,两类工作进程均设置 `OMP_NUM_THREADS=1`、`MIMALLOC_PURGE_DELAY=10000`、`ETCD_LEASE_TTL=120` 和 `FI_LOG_LEVEL=warn`。每个工作进程独占一个节点(`resources.spread_workers: true`,各角色的节点数等于工作进程数)。原生 `engine.numa_cpu_bind` 辅助脚本在执行前,为各 MPI rank 应用其 GPU 所在 NUMA 节点的 CPU 掩码,不设置 OpenMP 绑定变量。基准测试和 eval 的就绪检查都要求全部工作进程在 30 分钟内完成注册;作业仍保留六小时的总时限以完成负载。启动脚本只清理当前用户拥有、名称匹配且修改时间早于作业分配开始时间十分钟以上的共享内存文件。固定的时间界限可避免其他 rank 或前端再次运行启动脚本时删除本次作业的文件;并发 20 的配方在清理后继续记录 PMIx 诊断信息。对于选中进行 eval 的单前端配方,`frontend.placement.node` 和 `benchmark.placement.node` 均设为 `head`,与固定版本启动脚本使用的本地回环 eval 端点保持一致。 + ## 注册 llm-d 配方 来源:[`benchmarks/llm-d/README.md`](../benchmarks/llm-d/README.md)、[`benchmarks/multi_node/llm-d/README.md`](../benchmarks/multi_node/llm-d/README.md)、和 [`llm-d-recipes/`](../benchmarks/multi_node/llm-d-recipes)。 diff --git a/inferencex-e2e/infx/tests/test_clean_stale_shm.py b/inferencex-e2e/infx/tests/test_clean_stale_shm.py new file mode 100644 index 0000000000..18d38e19e4 --- /dev/null +++ b/inferencex-e2e/infx/tests/test_clean_stale_shm.py @@ -0,0 +1,92 @@ +"""Exercise shared-memory cleanup against isolated filesystem fixtures.""" + +import os +import subprocess +import sys +import time +from pathlib import Path + +import pytest + + +SCRIPT = ( + Path(__file__).resolve().parents[2] + / "benchmarks/multi_node/srt-slurm-recipes/configs/clean_stale_shm.sh" +) + + +def run_cleanup(directory: Path, job_start: str) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [ + "bash", + "-c", + 'source "$1"; clean_stale_shm "$2" "$3"', + "bash", + str(SCRIPT), + str(directory), + job_start, + ], + env={**os.environ, "PATH": f"{Path(sys.executable).parent}:{os.environ['PATH']}"}, + capture_output=True, + text=True, + check=False, + ) + + +def segment(path: Path, timestamp: int) -> None: + path.write_text("segment") + os.utime(path, (timestamp, timestamp)) + + +def test_cleanup_removes_only_matching_old_regular_files(tmp_path: Path) -> None: + job_start = int(time.time()) + for name in ("vader_segment.old", "nccl-old", "sem.old", "psm3old", "fe80::old"): + segment(tmp_path / name, job_start - 601) + segment(tmp_path / "nccl-boundary", job_start - 600) + segment(tmp_path / "nccl-current", job_start) + segment(tmp_path / "unrelated", job_start - 3600) + directory = tmp_path / "nccl-directory" + directory.mkdir() + segment(directory / "child", job_start - 3600) + os.utime(directory, (job_start - 3600, job_start - 3600)) + (tmp_path / "nccl-link").symlink_to(tmp_path / "unrelated") + + result = run_cleanup(tmp_path, str(job_start)) + + assert result.returncode == 0, result.stderr + assert "removed 5 stale segments" in result.stdout + assert {path.name for path in tmp_path.iterdir()} == { + "nccl-boundary", + "nccl-current", + "unrelated", + "nccl-directory", + "nccl-link", + } + assert (directory / "child").read_text() == "segment" + assert (tmp_path / "nccl-link").is_symlink() + + +def test_delayed_repeated_setup_keeps_current_allocation_segments(tmp_path: Path) -> None: + job_start = int(time.time()) - 3600 + segment(tmp_path / "nccl-previous", job_start - 601) + segment(tmp_path / "nccl-this-job", job_start + 1) + + first = run_cleanup(tmp_path, str(job_start)) + second = run_cleanup(tmp_path, str(job_start)) + + assert first.returncode == second.returncode == 0, first.stderr + second.stderr + assert "removed 1 stale segments" in first.stdout + assert "removed 0 stale segments" in second.stdout + assert (tmp_path / "nccl-this-job").read_text() == "segment" + assert not (tmp_path / "nccl-previous").exists() + + +@pytest.mark.parametrize("job_start", ["invalid", "0", "9999999999"]) +def test_invalid_allocation_start_leaves_files_untouched(tmp_path: Path, job_start: str) -> None: + segment(tmp_path / "nccl-old", 1) + + result = run_cleanup(tmp_path, job_start) + + assert result.returncode != 0 + assert "ValueError" in result.stderr + assert (tmp_path / "nccl-old").read_text() == "segment" diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index a0ecd2fc71..cf993e8cd4 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9003,3 +9003,75 @@ description: - "Update B300 vLLM AgentX to DSpark6 on a new image with a sampled concurrency grid and per-mode --kv-cache-memory-bytes pins." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3477 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Add six compact GLM-5.2 NVFP4 AgentX recipes on B300 with Dynamo and TensorRT-LLM disaggregated serving." + - "Use TensorRT-LLM 1.3.0rc26, enable the DSA metadata and indexer settings, and provide a persistent JIT cache for the recipe workers." + - "新增六个 B300 GLM-5.2 NVFP4 AgentX 紧凑配方,采用 Dynamo 与 TensorRT-LLM 分离式服务。" + - "使用 TensorRT-LLM 1.3.0rc26,启用 DSA 元数据与索引器设置,并为配方工作进程提供持久化 JIT 缓存。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Record bounded PMIx launch metadata before starting the concurrency-20 TensorRT-LLM workers." + - "在启动并发 20 的 TensorRT-LLM 工作进程前,记录限定范围的 PMIx 启动元数据。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Preserve the Slurm identity for TensorRT-LLM MPI workers and install Dynamo in a separate virtual environment for each job step." + - "为 TensorRT-LLM MPI 工作进程保留 Slurm 身份,并在各作业步骤的独立虚拟环境中安装 Dynamo。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Set an explicit Slurm time limit for server startup, AgentX replay, and result collection." + - "显式设置 Slurm 时限,为服务启动、AgentX 回放和结果收集预留时间。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Use the LIBFABRIC primary CUDA context, set worker OpenMP threads explicitly, colocate single-frontend evaluation endpoints, and retry verified Dynamo installs." + - "使用 LIBFABRIC 主 CUDA 上下文,显式设置工作进程的 OpenMP 线程数,将单前端评估端点与评估进程放在同一节点,并重试保留校验的 Dynamo 安装。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Use the legacy KV manager with standard CUDA KV allocations and enable LIBFABRIC provider warnings for GLM-5.2 B300 prefill and decode workers." + - "为 GLM-5.2 B300 的 prefill 和 decode 工作进程采用旧版 KV 管理器和标准 CUDA KV 内存分配,并启用 LIBFABRIC 提供程序警告。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Forward the workflow result filename into B300 AgentX recipes, collect per-concurrency aggregates from the mounted workspace, and preserve the Slurm producer failure status after staging artifacts." + - "将工作流结果文件名传递给 B300 AgentX 配方,从挂载的工作区收集各并发度的聚合结果,并在保留产物后传递 Slurm 生产作业的失败状态。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Set a single prefill LIBFABRIC posting thread for GLM-5.2 concurrency 20 and select acceptance automatically for the compact recipes." + - "为 GLM-5.2 并发 20 设置单个预填充 LIBFABRIC 提交线程,并为紧凑配方自动选择接受长度。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Spread GLM-5.2 B300 workers across nodes, clean stale shared memory before startup, bind worker CPUs by GPU locality, and bound worker registration waits." + - "将 GLM-5.2 B300 工作进程分散到独立节点,启动前清理过期共享内存,按 GPU 的 NUMA 位置绑定工作进程 CPU,并限制工作进程注册等待时间。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 + +- config-keys: + - glm5.2-fp4-b300-dynamo-trt-agentic-mtp-compact + description: + - "Move the six B300 AgentX configurations into one native srt-slurm base recipe with per-topology overrides, enable chat-template rendering for MTP requests, set worker OpenMP threads to one, and remove submission-side low-precision draft communication." + - "将六个 B300 AgentX 配置迁移为一个原生 srt-slurm 基础配方及按拓扑划分的覆盖项,为 MTP 请求启用聊天模板渲染,将工作进程 OpenMP 线程数设为一,并移除提交侧的低精度草稿通信设置。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2993 diff --git a/inferencex-e2e/runners/launch_b300-dsxe.sh b/inferencex-e2e/runners/launch_b300-dsxe.sh index bd28cec2e9..0e5c6bb274 100755 --- a/inferencex-e2e/runners/launch_b300-dsxe.sh +++ b/inferencex-e2e/runners/launch_b300-dsxe.sh @@ -216,10 +216,80 @@ fi export ISL="$ISL" export OSL="$OSL" +SRT_DEFAULT_BASH_PREAMBLE="" +SRT_CLUSTER_ARGS=() +if [[ "$IS_AGENTIC" == "1" && "$FRAMEWORK" == "dynamo-trt" && "$MODEL_PREFIX" == "glm5.2" ]]; then + # TRT-LLM rc26 ships NIXL v1.4.0 without its LIBFABRIC plugin. Stage the + # matching wheel plugin and EFA userspace runtime in a shared immutable cache. + NIXL_LIBFABRIC_HOST_DIR="/data/home/sa-gha-runner/nixl-libfabric/nixl-1.4.0-efa-1.47.0" + mkdir -p "$(dirname "$NIXL_LIBFABRIC_HOST_DIR")" + ( + exec 9>"${NIXL_LIBFABRIC_HOST_DIR}.lock" + flock -w 1800 9 || exit 1 + + if [[ ! -r "$NIXL_LIBFABRIC_HOST_DIR/nixl/libplugin_LIBFABRIC.so" || + ! -r "$NIXL_LIBFABRIC_HOST_DIR/efa/opt/amazon/efa/lib/libfabric.so.1" || + ! -r "$NIXL_LIBFABRIC_HOST_DIR/efa/usr/lib/x86_64-linux-gnu/libibverbs/libefa-rdmav59.so" ]]; then + if [[ -e "$NIXL_LIBFABRIC_HOST_DIR" ]]; then + echo "Error: incomplete NIXL LIBFABRIC cache: $NIXL_LIBFABRIC_HOST_DIR" >&2 + exit 1 + fi + + nixl_stage=$(mktemp -d "${NIXL_LIBFABRIC_HOST_DIR}.tmp.XXXXXX") + trap 'rm -rf -- "$nixl_stage"' EXIT + mkdir -p "$nixl_stage/runtime/nixl" "$nixl_stage/runtime/efa" \ + "$nixl_stage/installer" + + curl -LfsS --retry 3 -o "$nixl_stage/nixl.whl" \ + "https://files.pythonhosted.org/packages/8b/7c/b79fb09e832233c90f1e9d9b953e88c2b92096d968f2444839c6aa92b645/nixl_cu13-1.4.0-cp312-cp312-manylinux_2_28_x86_64.whl" + echo "3e606fbe80c39ce14899726fad0cb0fec53c6bac9f34168492692c4166b2fabb $nixl_stage/nixl.whl" | sha256sum -c - + unzip -p "$nixl_stage/nixl.whl" \ + nixl_cu13.libs/nixl/libplugin_LIBFABRIC.so \ + > "$nixl_stage/runtime/nixl/libplugin_LIBFABRIC.so" + unzip -p "$nixl_stage/nixl.whl" \ + nixl_cu13.libs/libnuma-3387f5e3.so.1.0.0 \ + > "$nixl_stage/runtime/nixl/libnuma-3387f5e3.so.1.0.0" + + curl -LfsS --retry 3 -o "$nixl_stage/efa.tar.gz" \ + "https://efa-installer.amazonaws.com/aws-efa-installer-1.47.0.tar.gz" + echo "2df4201e046833c7dc8160907bee7f52b76ff80ed147376a2d0ed8a0dd66b2db $nixl_stage/efa.tar.gz" | sha256sum -c - + tar -xzf "$nixl_stage/efa.tar.gz" -C "$nixl_stage/installer" \ + aws-efa-installer/DEBS/UBUNTU2404/x86_64/libfabric1-aws_2.4.0amzn1.0_amd64.deb \ + aws-efa-installer/DEBS/UBUNTU2404/x86_64/rdma-core/ibverbs-providers_61.0-1_amd64.deb \ + aws-efa-installer/DEBS/UBUNTU2404/x86_64/rdma-core/libibverbs1_61.0-1_amd64.deb \ + aws-efa-installer/DEBS/UBUNTU2404/x86_64/rdma-core/librdmacm1_61.0-1_amd64.deb \ + aws-efa-installer/DEBS/UBUNTU2404/x86_64/rdma-core/rdma-core_61.0-1_amd64.deb + while IFS= read -r -d '' efa_deb; do + dpkg-deb -x "$efa_deb" "$nixl_stage/runtime/efa" + done < <(find "$nixl_stage/installer" -name '*.deb' -print0) + mv "$nixl_stage/runtime" "$NIXL_LIBFABRIC_HOST_DIR" + fi + ) || exit 1 + + export MELLANOX_VISIBLE_DEVICES=void + AIPERF_MMAP_CACHE_HOST_PATH="/data/home/sa-gha-runner/aiperf-cache" + HF_HUB_CACHE_HOST_PATH="/data/home/sa-gha-runner/hf-hub-cache" + TRTLLM_JIT_CACHE_HOST_PATH="/data/home/sa-gha-runner/trtllm-jit-cache" + mkdir -p "$AIPERF_MMAP_CACHE_HOST_PATH" "$HF_HUB_CACHE_HOST_PATH" \ + "$TRTLLM_JIT_CACHE_HOST_PATH" + chmod 0777 "$AIPERF_MMAP_CACHE_HOST_PATH" "$HF_HUB_CACHE_HOST_PATH" \ + "$TRTLLM_JIT_CACHE_HOST_PATH" 2>/dev/null || true + + SRT_DEFAULT_BASH_PREAMBLE='if command -v trtllm-llmapi-launch >/dev/null 2>&1; then source /configs/prepare-dynamo-venv.sh || exit; fi; export NIXL_PLUGIN_DIR=/nixl-libfabric/nixl; export LD_LIBRARY_PATH=/nixl-libfabric/efa/opt/amazon/efa/lib:/nixl-libfabric/efa/usr/lib/x86_64-linux-gnu:/nixl-libfabric/nixl:${LD_LIBRARY_PATH:-}; export IBV_DRIVERS_PATH=/nixl-libfabric/efa/usr/lib/x86_64-linux-gnu/libibverbs; export FI_PROVIDER=efa; export FI_EFA_USE_DEVICE_RDMA=1' + SRT_CLUSTER_ARGS=( + --mount "$NIXL_LIBFABRIC_HOST_DIR" /nixl-libfabric + --mount "$AIPERF_MMAP_CACHE_HOST_PATH" /aiperf_mmap_cache + --mount "$HF_HUB_CACHE_HOST_PATH" /hf_hub_cache + --mount "$TRTLLM_JIT_CACHE_HOST_PATH" /trtllm-jit-cache + ) +fi + SRTCTL_ROOT="${GITHUB_WORKSPACE}/${SRT_REPO_DIR}" echo "Creating srtslurm.yaml configuration..." write_srt_cluster_config b300-dsxe srtslurm.yaml "$USES_DCGM_POWER" \ - --var MODEL_ROOT "$MODEL_ROOT" --var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" || exit 1 + --var MODEL_ROOT "$MODEL_ROOT" --var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" \ + --var SRT_DEFAULT_BASH_PREAMBLE "$SRT_DEFAULT_BASH_PREAMBLE" \ + "${SRT_CLUSTER_ARGS[@]}" || exit 1 echo "Generated srtslurm.yaml:" cat srtslurm.yaml @@ -254,6 +324,10 @@ SRTCTL_APPLY_ARGS=( --no-preflight --tags "b300,${MODEL_PREFIX},${PRECISION},${ISL}x${OSL},infmax-$(date +%Y%m%d)" ) +if [[ "$IS_AGENTIC" == "1" && "$EVAL_ONLY" != "true" ]]; then + check_env_vars RESULT_FILENAME + SRTCTL_APPLY_ARGS+=(--set "benchmark.env.RESULT_FILENAME=$RESULT_FILENAME") +fi SRTCTL_OUTPUT=$(apply_srt_recipe "$CONFIG_FILE" "$FRAMEWORK" "${SRTCTL_EVAL_ARGS[@]}" "${SRTCTL_APPLY_ARGS[@]}" 2>&1) echo "$SRTCTL_OUTPUT" @@ -296,8 +370,14 @@ fi cp -r "$LOGS_DIR" "$GITHUB_WORKSPACE/LOGS" tar czf "$GITHUB_WORKSPACE/multinode_server_logs.tar.gz" -C "$LOGS_DIR" . +RESULT_COLLECTION_RC=0 if [[ "${EVAL_ONLY}" != "true" ]]; then - copy_fixed_sequence_results "$LOGS_DIR" "$GITHUB_WORKSPACE" "$RESULT_FILENAME" || exit 1 + if [[ "$IS_AGENTIC" == "1" ]]; then + check_slurm_job_success "$JOB_ID" "$LOGS_DIR" || RESULT_COLLECTION_RC=1 + copy_agentic_results "$GITHUB_WORKSPACE" "$GITHUB_WORKSPACE" "$RESULT_FILENAME" || RESULT_COLLECTION_RC=1 + else + copy_fixed_sequence_results "$LOGS_DIR" "$GITHUB_WORKSPACE" "$RESULT_FILENAME" || RESULT_COLLECTION_RC=1 + fi else echo "EVAL_ONLY=true: Skipping benchmark result collection" fi @@ -328,6 +408,9 @@ for i in 1 2 3 4 5; do done find . -name '.nfs*' -delete 2>/dev/null || true # Preserve diagnostics and eval outputs before propagating a failed allocation. +if [[ "$RESULT_COLLECTION_RC" != "0" ]]; then + exit "$RESULT_COLLECTION_RC" +fi exit "$SRT_JOB_RC" else diff --git a/inferencex-e2e/runners/slurm_utils.sh b/inferencex-e2e/runners/slurm_utils.sh index 50370edfeb..b06c3e0540 100644 --- a/inferencex-e2e/runners/slurm_utils.sh +++ b/inferencex-e2e/runners/slurm_utils.sh @@ -191,6 +191,7 @@ launch_srt_single_node() { --var SRTCTL_ROOT "$SRTCTL_ROOT" --var SQUASH_FILE "$SRT_CONTAINER" \ --var IMAGE "$IMAGE" --var NGINX_SQUASH_FILE nginx:1.27.4 \ --var SRT_DEFAULT_TIME_LIMIT "$SALLOC_TIME_LIMIT" \ + --var SRT_DEFAULT_BASH_PREAMBLE "" \ --model "hf:$MODEL" "$SRT_MODEL_PATH" --container "$IMAGE" "$SRT_CONTAINER" \ --mount "$HF_HUB_CACHE_MOUNT" "$HF_HUB_CACHE" --exclusive "$@" run_srt_setup "ARCH=${SRT_SETUP_ARCH:-x86_64}" @@ -387,6 +388,36 @@ copy_fixed_sequence_results() { echo "All result files processed" } +# Check the allocation's final exit status, not a step or a partial result file. +check_slurm_job_success() { + local job_id="$1" logs_dir="$2" + local attempt + for attempt in 1 2 3; do + echo "$attempt" > "$logs_dir/native-job-status-attempts.txt" || return 1 + if sacct -X -n -P -j "$job_id" --format=JobIDRaw,State,ExitCode \ + > "$logs_dir/native-job-status.txt" \ + 2>> "$logs_dir/native-job-status.stderr"; then + if awk -F'|' -v job="$job_id" ' + $1 == job && $2 !~ /^(PENDING|RUNNING|COMPLETING)$/ { found = 1 } + END { exit !found } + ' "$logs_dir/native-job-status.txt"; then + if awk -F'|' -v job="$job_id" ' + $1 == job { found = 1; if ($2 != "COMPLETED" || $3 != "0:0") failed = 1 } + END { exit (!found || failed) } + ' "$logs_dir/native-job-status.txt"; then + return 0 + fi + echo "ERROR: Slurm job $job_id did not complete successfully" >&2 + return 1 + fi + fi + # Accounting can lag squeue removal; keep the wait bounded. + if [[ "$attempt" != "3" ]]; then sleep 5; fi + done + echo "ERROR: no successful terminal accounting record for Slurm job $job_id" >&2 + return 1 +} + copy_agentic_results() { local source_dir="$1" local workspace="$2" diff --git a/inferencex-e2e/runners/srt-slurm/b300-dsxe.yaml b/inferencex-e2e/runners/srt-slurm/b300-dsxe.yaml index 19c846c11f..f7469a64b5 100644 --- a/inferencex-e2e/runners/srt-slurm/b300-dsxe.yaml +++ b/inferencex-e2e/runners/srt-slurm/b300-dsxe.yaml @@ -31,6 +31,7 @@ containers: "${IMAGE}": ${SQUASH_FILE} nginx-sqsh: ${NGINX_SQUASH_FILE} use_exclusive_sbatch_directive: true +default_bash_preamble: ${SRT_DEFAULT_BASH_PREAMBLE} default_sbatch_directives: cpus-per-task: '192' # gpu-16 retains a foreign 1.63 TB tmpfs allocation; rechecked 2026-09-21.