diff --git a/docker/Dockerfile.multi b/docker/Dockerfile.multi index a6f9fd57185f..5ee9a46eba5f 100644 --- a/docker/Dockerfile.multi +++ b/docker/Dockerfile.multi @@ -15,8 +15,8 @@ # Multi-stage Dockerfile ARG BASE_IMAGE=nvcr.io/nvidia/pytorch ARG TRITON_IMAGE=nvcr.io/nvidia/tritonserver -ARG BASE_TAG=26.08-py3 -ARG TRITON_BASE_TAG=26.08-py3 +ARG BASE_TAG=26.09-py3 +ARG TRITON_BASE_TAG=26.09-py3 ARG DEVEL_IMAGE=devel FROM ${BASE_IMAGE}:${BASE_TAG} AS base diff --git a/docker/Makefile b/docker/Makefile index 4c1042168b92..3278733b3b40 100644 --- a/docker/Makefile +++ b/docker/Makefile @@ -208,21 +208,21 @@ jenkins-rockylinux8_%: PYTHON_VERSION_TAG_ID = $(if $(findstring 3.12,${PYTHON_V jenkins-rockylinux8_%: IMAGE_WITH_TAG = $(shell . ../jenkins/current_image_tags.properties && echo $$LLM_ROCKYLINUX8_${PYTHON_VERSION_TAG_ID}_DOCKER_IMAGE) jenkins-rockylinux8_%: STAGE = tritondevel jenkins-rockylinux8_%: BASE_IMAGE = nvcr.io/nvidia/cuda -jenkins-rockylinux8_%: BASE_TAG = 13.3.1-devel-rockylinux8 +jenkins-rockylinux8_%: BASE_TAG = 13.4.1-devel-rockylinux8 rockylinux8_%: STAGE = tritondevel rockylinux8_%: BASE_IMAGE = nvcr.io/nvidia/cuda -rockylinux8_%: BASE_TAG = 13.3.1-devel-rockylinux8 +rockylinux8_%: BASE_TAG = 13.4.1-devel-rockylinux8 # For x86_64 ubuntu22_%: STAGE = tritondevel ubuntu22_%: BASE_IMAGE = nvcr.io/nvidia/cuda -ubuntu22_%: BASE_TAG = 13.3.1-devel-ubuntu22.04 +ubuntu22_%: BASE_TAG = 13.4.1-devel-ubuntu22.04 # For x86_64 and aarch64 ubuntu24_%: STAGE = tritondevel ubuntu24_%: BASE_IMAGE = nvcr.io/nvidia/cuda -ubuntu24_%: BASE_TAG = 13.3.1-devel-ubuntu24.04 +ubuntu24_%: BASE_TAG = 13.4.1-devel-ubuntu24.04 trtllm_%: STAGE = release trtllm_%: PUSH_TO_STAGING := 0 diff --git a/docker/common/install_cuda_libs.sh b/docker/common/install_cuda_libs.sh index d076d6e583cb..8b0231f3d85e 100644 --- a/docker/common/install_cuda_libs.sh +++ b/docker/common/install_cuda_libs.sh @@ -12,9 +12,9 @@ CUDA_VER="13.4" # the NGC PyTorch image's CUDA_VERSION, major.minor # included, then ends up on a cuDNN anyone can install, and the tests validate that combination # rather than one only NGC can reproduce. The version guard below will not match on NGC PyTorch, # so its cuDNN is deliberately purged and reinstalled at this version. -CUDNN_VER="9.25.1.1-1" -NCCL_VER="2.30.7-1+cuda13.3" -CUBLAS_VER="13.7.0.27-1" +CUDNN_VER="9.26.0.51-1" +NCCL_VER="2.31.2-1+cuda13.4" +CUBLAS_VER="13.8.0.4-1" # Align with the pre-installed CUDA / NVCC / NVRTC versions from # https://docs.nvidia.com/cuda/cuda-toolkit-release-notes/index.html NVRTC_VER="13.4.59-1" diff --git a/docker/common/install_mooncake.sh b/docker/common/install_mooncake.sh index 0935c7f31a2f..5bc4686e1ad9 100644 --- a/docker/common/install_mooncake.sh +++ b/docker/common/install_mooncake.sh @@ -30,7 +30,7 @@ apt-get install -y --no-install-recommends \ mkdir -p /third-party-source -git clone --depth 1 https://github.com/alibaba/yalantinglibs.git +git clone --depth 1 -b 0.5.5 https://github.com/alibaba/yalantinglibs.git tar -czf /third-party-source/yalantinglibs.tar.gz yalantinglibs cd yalantinglibs mkdir build && cd build @@ -53,3 +53,14 @@ cd ../.. rm -rf Mooncake echo "export LD_LIBRARY_PATH=${MOONCAKE_INSTALL_PATH}/lib:\$LD_LIBRARY_PATH" >> "${ENV}" + +# `make install` also emits a `mooncake` package that omits +# libmooncake_store.so, so importing mooncake.store from it fails. It has to go +# before the wheel is installed: CMake writes +# store.cpython-312-x86_64-linux-gnu.so where the wheel writes store.so, and +# importlib prefers the interpreter-tagged suffix, so the broken extension +# would win even after pip reports success. The directory is the one +# mooncake-integration/CMakeLists.txt chose, which this repeats. +MOONCAKE_CMAKE_PACKAGE="$(python3 -c "import sys; print([s for s in sys.path if 'packages' in s][0])")/mooncake" +echo "removing CMake-generated mooncake package: ${MOONCAKE_CMAKE_PACKAGE}" +rm -rf "${MOONCAKE_CMAKE_PACKAGE}" diff --git a/docker/common/install_pytorch.sh b/docker/common/install_pytorch.sh index 87e95de810d6..184d0c94cba2 100644 --- a/docker/common/install_pytorch.sh +++ b/docker/common/install_pytorch.sh @@ -12,7 +12,7 @@ fi # Use latest stable version from https://pypi.org/project/torch/#history # and closest to the version specified in -# https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-26-08.html#rel-26-08 +# https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-26-09.html#rel-26-09 TORCH_VERSION="2.14.0" SYSTEM_ID=$(grep -oP '(?<=^ID=).+' /etc/os-release | tr -d '"') diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index edeabb07295c..eb35d3083ded 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -71,7 +71,7 @@ ARTIFACTORY_DOCKER_HOST = "artifactory.nvidia.com" ARTIFACTORY_CREDENTIALS_ID = "trtllm-artifactory-credentials" // DLFW torch image -DLFW_IMAGE = "urm.nvidia.com/docker/nvidia/pytorch:26.08-py3" +DLFW_IMAGE = "urm.nvidia.com/docker/nvidia/pytorch:26.09-py3" MODEL_EXPRESS_VERSION = "0.5.1" MODEL_EXPRESS_SERVER_IMAGE = "urm.nvidia.com/docker/nvidia/ai-dynamo/modelexpress-server:${MODEL_EXPRESS_VERSION}" @@ -4145,7 +4145,7 @@ def launchTestListCheck(pipeline) trtllm_utils.llmExecStepWithRetry(pipeline, script: "pip3 install -r ${llmSrc}/requirements-dev.txt") // --validate --parity: after --l0/--qa generate the collectable lists, assert every // statically-verified parametrize ID is actually collectable (validate<->collection parity). - sh "NVIDIA_TRITON_SERVER_VERSION=26.08 LLM_ROOT=${llmSrc} LLM_BACKEND_ROOT=${llmSrc}/triton_backend python3 ${llmSrc}/scripts/check_test_list.py --l0 --qa --waive --validate --parity" + sh "NVIDIA_TRITON_SERVER_VERSION=26.09 LLM_ROOT=${llmSrc} LLM_BACKEND_ROOT=${llmSrc}/triton_backend python3 ${llmSrc}/scripts/check_test_list.py --l0 --qa --waive --validate --parity" } catch (InterruptedException e) { throw e } catch (Exception e) { diff --git a/jenkins/current_image_tags.properties b/jenkins/current_image_tags.properties index b615eb04d28a..b43224170147 100644 --- a/jenkins/current_image_tags.properties +++ b/jenkins/current_image_tags.properties @@ -13,8 +13,8 @@ # images are adopted from PostMerge pipelines, the abbreviated commit hash is used instead. IMAGE_NAME=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm -LLM_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:pytorch-26.08-py3-x86_64-ubuntu24.04-skip-tritondevel-202609210610-19345 -LLM_SBSA_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:pytorch-26.08-py3-sbsa-ubuntu24.04-skip-tritondevel-202609210610-19345 -LLM_ROCKYLINUX8_PY310_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:cuda-13.4.1-devel-rocky8-x86_64-rocky8-py310-skip-tritondevel-202609210610-19345 -LLM_ROCKYLINUX8_PY312_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:cuda-13.4.1-devel-rocky8-x86_64-rocky8-py312-skip-tritondevel-202609210610-19345 -LLM_SBSA_WHEEL_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:cuda-13.4.1-devel-ubuntu24.04-sbsa-ubuntu24.04-py312-skip-tritondevel-202609210610-19345 +LLM_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:pytorch-26.09-py3-x86_64-ubuntu24.04-skip-tritondevel-202610050217-19679 +LLM_SBSA_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:pytorch-26.09-py3-sbsa-ubuntu24.04-skip-tritondevel-202610050217-19679 +LLM_ROCKYLINUX8_PY310_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:cuda-13.4.1-devel-rocky8-x86_64-rocky8-py310-skip-tritondevel-202610050217-19679 +LLM_ROCKYLINUX8_PY312_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:cuda-13.4.1-devel-rocky8-x86_64-rocky8-py312-skip-tritondevel-202610050217-19679 +LLM_SBSA_WHEEL_DOCKER_IMAGE=artifactory.nvidia.com/sw-tensorrt-llm-docker-local/tensorrt-llm:cuda-13.4.1-devel-ubuntu24.04-sbsa-ubuntu24.04-py312-skip-tritondevel-202610050217-19679 diff --git a/requirements.txt b/requirements.txt index 63c1972c7c35..811de826fb74 100644 --- a/requirements.txt +++ b/requirements.txt @@ -23,17 +23,17 @@ pandas h5py==3.12.1 StrEnum sentencepiece>=0.1.99 -# https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-26-08.html#rel-26-08 uses 2.14.0a0. +# https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-26-09.html#rel-26-09 uses 2.14.0a0. # The floor is the NGC alpha, which PEP 440 orders below the public release of the same # series; the ceiling is that public release, which is what the resolver picks off the public # index. Ordered comparisons drop the local version, so the container's +nv build stays # admissible against the floor. torch>=2.14.0a0,<=2.14.0 torchvision -# https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-26-08.html#rel-26-08 uses 2.30.7 -# The public torch above pins the same version, so an exact pin covers both it and the NGC -# container's own NCCL; as with triton, carrying no local version keeps the +nv build admissible. -nvidia-nccl-cu13==2.30.7 +# The floor is the exact pin of the public torch above; the ceiling is the NCCL that +# https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-26-09.html#rel-26-09 +# ships, so the resolver lands on the container's own NCCL version there instead of an older one. +nvidia-nccl-cu13>=2.30.7,<=2.31.2 # NCCL-EP uses nccl4py's nccl.ep package. The nccl-extensions wheel is built # from pinned source by CMake so it links against the container's NCCL. nccl4py>=0.4,<=0.5 @@ -79,7 +79,7 @@ meson ninja blake3 soundfile -# https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-26-08.html#rel-26-08 +# https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-26-09.html#rel-26-09 # ships 3.8.0 and the torch pin above requires the same series, so floor and ceiling coincide. # The pin carries no local version, which keeps the container's own +nv build admissible -- it # must not be reinstalled underneath the rest of the stack that was built against it. This is diff --git a/tests/unittest/_torch/thop/parallel/test_locality_domain_utils.py b/tests/unittest/_torch/thop/parallel/test_locality_domain_utils.py index a5b7b5480104..1658ee7c198c 100644 --- a/tests/unittest/_torch/thop/parallel/test_locality_domain_utils.py +++ b/tests/unittest/_torch/thop/parallel/test_locality_domain_utils.py @@ -68,6 +68,17 @@ def check_locality_domain_support(): pytest.skip("LOCALITY_DOMAIN localization is not supported on this system") +@pytest.fixture(scope="module") +def check_locality_domain_enabled(check_locality_domain_support): + """Skip unless locality_domain_device() actually switches the current domain. + + Driver support alone is not enough: the feature is only enabled on SM107, and + elsewhere locality_domain_device() is a no-op. + """ + if not is_locality_domain_enabled(): + pytest.skip("LOCALITY_DOMAIN localization is not enabled on this system") + + class TestLocalityDomainSupport: """Tests for LOCALITY_DOMAIN support detection.""" @@ -666,7 +677,7 @@ def _pool_allocated_bytes(pool): class TestLocalityDomainNestedMemPool: """Nested optional_locality_domain_mem_pool scopes across locality domains.""" - def test_nested_same_domain_does_not_reenter(self, check_locality_domain_support): + def test_nested_same_domain_does_not_reenter(self, check_locality_domain_enabled): """Re-entering the same pool is skipped; use_mem_pool would reject it.""" manager = locality_domain_utils.get_locality_domain_resource_manager() device_id = torch.cuda.current_device() @@ -684,7 +695,7 @@ def test_nested_same_domain_does_not_reenter(self, check_locality_domain_support assert getattr(manager.in_mem_pool_context, "active_pool_key", None) is None - def test_nested_other_domain_restores_outer_key(self, check_locality_domain_support): + def test_nested_other_domain_restores_outer_key(self, check_locality_domain_enabled): """A nested scope for a different domain enters it and unwinds to the outer one.""" manager = locality_domain_utils.get_locality_domain_resource_manager() device_id = torch.cuda.current_device() @@ -704,7 +715,7 @@ def test_nested_other_domain_restores_outer_key(self, check_locality_domain_supp assert getattr(manager.in_mem_pool_context, "active_pool_key", None) is None - def test_nested_other_domain_allocates_from_inner_pool(self, check_locality_domain_support): + def test_nested_other_domain_allocates_from_inner_pool(self, check_locality_domain_enabled): """A nested domain's tensor comes from its own pool, not the enclosing one.""" try: pool0 = get_locality_domain_mempool(0) @@ -728,7 +739,7 @@ def test_nested_other_domain_allocates_from_inner_pool(self, check_locality_doma assert after0 == before0, "nested domain 1 allocation leaked into domain 0's pool" del inner - def test_exception_restores_previous_key(self, check_locality_domain_support): + def test_exception_restores_previous_key(self, check_locality_domain_enabled): """An exception inside a nested scope must still unwind the key.""" manager = locality_domain_utils.get_locality_domain_resource_manager() device_id = torch.cuda.current_device()