diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index e34949c51a88..d2bf119f953a 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -58,6 +58,13 @@ BOLT_OVERLAY_ENABLED = (params.boltOverlayEnabled ?: env.boltOverlayEnabled ?: " // carries profiles). Left false until the enable PR wires it true for the // release/nightly path; premerge/new-branch stays lenient (retag plain build). BOLT_PROFILES_REQUIRED = (params.boltProfilesRequired ?: env.boltProfilesRequired ?: "false").toString() == "true" +// Separate from the two above: those govern the profile BUNDLE baked in as a +// thin layer (Dockerfile.bolt), which documents how to reproduce a BOLTed build +// but does NOT optimize the binaries the image actually installs. This one +// governs the INSTALLED wheel -- when true, the wheel unpacked from the build +// tarball is BOLT-optimized in the image build, before the release stage pip +// installs it. Kept independent so it can be rolled back on its own. +BOLT_OPTIMIZE_WHEEL = (params.boltOptimizeWheel ?: env.boltOptimizeWheel ?: "false").toString() == "true" // The bundle this pipeline pinned, hoisted out of globalVars in launchBuildJobs // because overlayBoltBundle runs well below the scope globalVars is passed into. // Empty means unpinned, i.e. take whatever `latest` is. @@ -281,14 +288,48 @@ def createKubernetesPodConfig(type, arch = "amd64", build_wheel = false) def prepareWheelFromBuildStage(dockerfileStage, arch) { - if (!ENABLE_USE_WHEEL_FROM_BUILD_STAGE) { - echo "useWheelFromBuildStage is false, skip preparing wheel from build stage" - return "" - } + // Whether THIS image has to ship a BOLT-optimized wheel. Answered before the + // gates below, and deliberately not subject to them. + // + // Optimizing the installed wheel is only possible on this path: the wheel is + // BOLTed by get_wheel_from_package.py as it is unpacked from the build + // tarball, so an image that compiles its own wheel in-container has nothing + // to optimize. That makes the two gates below load-bearing for BOLT, and + // both are unreliable for reasons that have nothing to do with BOLT: + // + // useWheelFromBuildStage -- a kill switch added for nvbug 5433581 in Aug + // 2025 and never reverted. It was also read without being declared, so + // it was not merely false but incapable of being true. + // triggerType -- read from env and not a declared parameter of this job, + // so whether it survives the launch is a property of job registration + // rather than of this repo. When it does not, TRIGGER_TYPE is "manual" + // and this returns early no matter what the caller asked for. + // + // boltOptimizeWheel, by contrast, IS declared, so it reliably arrives. Key + // off it and let a BOLT request carry itself past both gates. The bypass is + // deliberately narrow -- one arch, one dockerfile stage, only when BOLT was + // asked for -- so the nvbug's blast radius stays at the SBSA release image + // instead of being reopened for every image this job builds. + def boltWheelRequired = BOLT_OPTIMIZE_WHEEL && arch == "sbsa" && dockerfileStage == "release" + + if (!boltWheelRequired) { + if (!ENABLE_USE_WHEEL_FROM_BUILD_STAGE) { + echo "useWheelFromBuildStage is false, skip preparing wheel from build stage" + return "" + } - if (!(TRIGGER_TYPE in ["post-merge", "nightly-release"])) { - echo "Trigger type does not use the build stage wheel" - return "" + if (!(TRIGGER_TYPE in ["post-merge", "nightly-release"])) { + echo "Trigger type does not use the build stage wheel" + return "" + } + } else if (!ENABLE_USE_WHEEL_FROM_BUILD_STAGE || + !(TRIGGER_TYPE in ["post-merge", "nightly-release"])) { + // Say so rather than doing it quietly: this is the one place the image + // build departs from what its parameters literally asked for. + echo "[BOLT] boltOptimizeWheel is set for the ${arch} release image, so the " + + "build-stage wheel path runs even though useWheelFromBuildStage=" + + "${ENABLE_USE_WHEEL_FROM_BUILD_STAGE} and triggerType=${TRIGGER_TYPE} would " + + "otherwise skip it. Optimizing the installed wheel is not possible any other way." } if (!dockerfileStage || !arch) { @@ -302,10 +343,49 @@ def prepareWheelFromBuildStage(dockerfileStage, arch) { } def wheelScript = 'scripts/get_wheel_from_package.py' - def wheelArgs = "--arch ${arch} --timeout ${WAIT_TIME_FOR_BUILD_STAGE} --artifact_path " + env.uploadPath + // UPLOAD_PATH, not env.uploadPath: they agree whenever the parent passed the + // parameter, but an unset env leaves the raw reference interpolating to + // "null" and the download then polls .../null/ until it times out. + def wheelArgs = "--arch ${arch} --timeout ${WAIT_TIME_FOR_BUILD_STAGE} --artifact_path ${UPLOAD_PATH}" + + // Only aarch64 has a promoted profile bundle, so only the SBSA image has + // anything to apply; x86 installs the wheel as built. + if (BOLT_OPTIMIZE_WHEEL && arch == "sbsa") { + // The branch whose promoted bundle to apply, resolved the same way the + // image overlay resolves it: an explicit override, else this build's own + // branch, else main. Deliberately NOT this run's own profiles -- those + // are not published until BoltProfileGen finishes, hours after this + // build starts, and waiting on them would serialize every release + // behind a multi-hour GPU job. get_wheel_from_package.py tries these in + // order and fails if none has a bundle. + def branches = [params.boltProfileBranch, LLM_BRANCH, "main"] + .collect { it?.toString()?.trim() } + .findAll { it } + .unique() + // Pinned, the candidate list collapses to the branch the pin came from: + // the ref names one immutable bundle under one promote directory, so + // falling through to another branch would optimize the image's wheel + // with different profiles than the release wheel and the tested build. + if (BOLT_PINNED_REF && BOLT_PINNED_BRANCH) { + branches = [BOLT_PINNED_BRANCH] + wheelArgs += " --bolt-profile-ref ${BOLT_PINNED_REF}" + echo "Release image for ${arch} is pinned to BOLT bundle ${BOLT_PINNED_REF} on ${BOLT_PINNED_BRANCH}" + } + echo "Release image for ${arch} will BOLT-optimize its wheel using profiles from: ${branches.join(', ')}" + wheelArgs += " --bolt-branch ${branches.join(',')}" + } return " BUILD_WHEEL_SCRIPT=${wheelScript} BUILD_WHEEL_ARGS='${wheelArgs}'" } +// Whether a docker build that failed WITH the downloaded-wheel args may be +// retried without them. The retry rebuilds the wheel from source in-container, +// which is a fine recovery for an ordinary build but silently defeats the point +// when that build was also responsible for optimizing the wheel -- the retry +// would produce an unoptimized release image that looks identical. +def mayRetryWithoutBuildStageWheel(arch) { + return !(BOLT_OPTIMIZE_WHEEL && arch == "sbsa") +} + // Produce each CANONICAL image from its raw `-noprofiles` build by // overlaying the merged LLVM BOLT profile bundle as a thin layer (docker/ // Dockerfile.bolt via the docker/Makefile `bolt_overlay` target). Canonical is @@ -614,6 +694,11 @@ def buildImage(config, imageKeyToTag, versionOverride) if (buildWheelArgs.trim().isEmpty()) { throw ex } + if (!mayRetryWithoutBuildStageWheel(arch)) { + echo "Build failed with wheel arguments and the BOLT-optimized wheel is required for ${arch}; " + + "NOT retrying from source (that would publish an unoptimized release image)" + throw ex + } echo "Build failed with wheel arguments, retrying without them" buildWheelArgs = "" trtllm_utils.llmExecStepWithRetry(this, script: """ @@ -888,6 +973,16 @@ pipeline { defaultValue: false, description: "When boltOverlayEnabled is true, treat a missing/empty BOLT bundle as a FATAL error instead of retagging the plain build as canonical. Enable for the release/nightly path to guarantee canonical images carry profiles." ) + booleanParam( + name: "useWheelFromBuildStage", + defaultValue: false, + description: "Install the wheel from the build-stage tarball instead of compiling one inside the image. Read since Aug 2025 but never DECLARED, so params.useWheelFromBuildStage was always null and prepareWheelFromBuildStage() returned early on every run -- see nvbug 5433581, whose temporary kill switch was never reverted. Declared here so the flag is at least capable of being set; it stays false by default, and nothing turns it on. boltOptimizeWheel does NOT depend on it: a BOLT request carries itself past this gate for the SBSA release image only." + ) + booleanParam( + name: "boltOptimizeWheel", + defaultValue: false, + description: "BOLT-optimize the wheel the SBSA release image installs, applying the branch's latest promoted profile bundle during the image build, and fail if no bundle can be applied. Independent of boltOverlayEnabled, which only bakes the bundle in as a layer and leaves the installed binaries unoptimized. Uses the last promoted bundle rather than this run's, so the image build never waits on BoltProfileGen. Ignored on x86_64, which has no promoted bundle." + ) string( name: "boltProfileBranch", defaultValue: "", diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index 3e05cb73f888..e175306398fd 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -2617,6 +2617,17 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) // main, so a ref with no promoted bundle still gets profiles. 'boltOverlayEnabled': true, 'boltProfilesRequired': true, + // The overlay above only bakes in the profile bundle; it + // leaves the installed wheel unoptimized. This BOLTs the + // wheel the SBSA release image installs, applying the + // branch's last promoted bundle during the image build. + // Deliberately not this run's profiles: those are not + // published until BoltProfileGen finishes, hours after + // this build starts, so depending on them would serialize + // every release behind a multi-hour GPU job. Inert on + // x86_64 (no promoted bundle) and whenever the wheel is + // built from source rather than downloaded. + 'boltOptimizeWheel': true, ] if (runMode == "nightly_release") { additionalParameters += [ @@ -2676,9 +2687,12 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) 'uploadPath': UPLOAD_PATH, // Must match Build-Docker-Images above: this path pushes the // same tags, so the scanned+registered image has to be the - // BOLTed canonical one rather than a plain build. + // BOLTed canonical one rather than a plain build, built on + // the BOLT-optimized wheel rather than the first tarball to + // appear. 'boltOverlayEnabled': true, 'boltProfilesRequired': true, + 'boltOptimizeWheel': true, ] if (runMode == "nightly_release") { additionalParameters += [ diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index edeabb07295c..cc4fd7277d69 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -5712,7 +5712,8 @@ def checkKitmakerWheelDryRun(pipeline, kitmakerDryRunMetadata) // a missing bundle as a skip; it has to, because the same switch covers x86_64, // where nothing is promoted yet. This path is aarch64-only and the switch is // main-only, so there is always a bundle to find. -def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) +def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, String boltProfileRef = "", + String boltProfileBranch = "") { def llvmArch = (cpu_arch == AARCH64_TRIPLE) ? "ARM64" : "X64" // apply_latest.sh resolves exactly one branch, so try the build's own branch @@ -5723,6 +5724,13 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) .collect { it?.toString()?.trim() } .findAll { it } .unique() + // A pin now supplies its own branch, so nothing is left to guess: the ref + // names one object under that branch's promote directory. Walking candidates + // would only add ways to fetch something other than what was pinned. + if (boltProfileRef && boltProfileBranch) { + branches = [boltProfileBranch] + echo "[bolt-wheel] pinned to BOLT bundle ${boltProfileRef} on ${boltProfileBranch}" + } stage("BOLT release wheel") { sh """ @@ -5748,8 +5756,12 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) for (b in branches) { rc = sh(returnStatus: true, script: """ export PATH="\$PWD/.bolt-llvm/bin:\$PATH" + export BOLT_PROFILE_REF='${boltProfileRef}' + # Quoted: the branch comes from job env and the wheel name from a + # directory listing, so an unquoted expansion would let a shell + # metacharacter in either run before apply_latest.sh starts. bash tensorrt_llm/scripts/bolt/internal/apply_latest.sh \ - ${b} ${cpu_arch} ${wheel} ${wheel}.bolted + '${b}' '${cpu_arch}' '${wheel}' '${wheel}.bolted' """) if (rc != 3) { appliedFrom = b @@ -5757,6 +5769,11 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) } echo "[bolt-wheel] no promoted bundle for ${b}/${cpu_arch}; trying next candidate branch" } + if (rc == 3 && boltProfileRef) { + error("[bolt-wheel] pinned BOLT bundle ${boltProfileRef} (${cpu_arch}) not found under any of " + + "${branches.join(', ')}. This pipeline pinned a bundle the release wheel cannot fetch; " + + "refusing to publish a wheel optimized with anything else.") + } if (rc == 3) { error("[bolt-wheel] no promoted BOLT bundle for any of ${branches.join(', ')} (${cpu_arch}); " + "refusing to upload an unoptimized release wheel (promote a bundle via BoltProfileGen, " + @@ -5766,7 +5783,8 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) error("[bolt-wheel] apply_latest.sh failed (rc=${rc}) for ${appliedFrom}/${cpu_arch}") } sh "mv -f ${wheel}.bolted ${wheel}" - echo "[bolt-wheel] ${wheel} is now BOLTed (profiles from ${appliedFrom}/${cpu_arch})" + echo "[bolt-wheel] ${wheel} is now BOLTed (profiles from ${appliedFrom}/${cpu_arch}" + + (boltProfileRef ? ", bundle ${boltProfileRef})" : ")") } } @@ -5779,7 +5797,9 @@ def runLLMBuild( cpver="cp312", plat_name="", is_dlfw=false, - boltConsume=false) + boltConsume=false, + boltProfileRef="", + boltProfileBranch="") { sh "pwd && ls -alh" sh "env | sort" @@ -5842,7 +5862,8 @@ def runLLMBuild( // inside an already-released image to prove that image can still build from // source; optimizing it would prove nothing and only add a failure mode. if (boltConsume && cpu_arch == AARCH64_TRIPLE && !wheel_path) { - applyLatestBoltToWheel(pipeline, "tensorrt_llm/build/${wheelName}", cpu_arch) + applyLatestBoltToWheel(pipeline, "tensorrt_llm/build/${wheelName}", cpu_arch, + boltProfileRef, boltProfileBranch) } def rootWheelUploadPath = "${cpu_arch}/${wheel_path}" @@ -7018,11 +7039,17 @@ def launchTestJobs(pipeline, testFilter, globalVars) // Same switch the build helpers use for the tarball, so the released // wheel and the released tarball are never optimized differently. def boltConsume = globalVars[BOLT_CONSUME_BUILD]?.toString() == "true" + // The bundle this pipeline pinned, so the released wheel is optimized + // with the same profiles the tested build was. Empty means unpinned: + // apply_latest.sh then takes whatever `latest` is, i.e. today's + // behaviour. + def boltProfileRef = globalVars[BOLT_PROFILE_REF]?.toString() ?: "" + def boltProfileBranch = globalVars[BOLT_PROFILE_BRANCH]?.toString() ?: "" buildRunner("[${toStageName(values[1], key)}] Build") { wheelPath = runLLMBuild( pipeline, cpu_arch, values[3], "", versionOverride, cpver, - values[7], isDlfw, boltConsume) + values[7], isDlfw, boltConsume, boltProfileRef, boltProfileBranch) } // TODO: Re-enable the sanity check after updating GPU testers' driver version. diff --git a/scripts/bolt/internal/apply_latest.sh b/scripts/bolt/internal/apply_latest.sh index 72f8984baf76..ddf9c071a9a4 100755 --- a/scripts/bolt/internal/apply_latest.sh +++ b/scripts/bolt/internal/apply_latest.sh @@ -56,6 +56,15 @@ fi DEST="$(mktemp -d)" trap 'rm -rf "$DEST"' EXIT +# 0) Put llvm-bolt on PATH. No-op when the caller already staged it (the Jenkins +# build pods do), which keeps this free for them and lets callers that cannot +# stage it themselves -- notably the image build, where this runs inside a +# docker layer -- just call apply_latest.sh and get a working toolchain. +if ! . "$HERE/stage_llvm_bolt.sh"; then + echo "[apply_latest] FATAL: could not stage llvm-bolt" >&2 + exit 2 +fi + # 1) Pull the branch `latest` bundle. A missing bundle is fatal here (see header): # consumption was requested but there is nothing promoted to consume. if ! bash "$HERE/artifactory.sh" pull-latest "$BRANCH" "$TRIPLE" "$DEST"; then diff --git a/scripts/bolt/internal/stage_llvm_bolt.sh b/scripts/bolt/internal/stage_llvm_bolt.sh new file mode 100644 index 000000000000..3f36b0d3024f --- /dev/null +++ b/scripts/bolt/internal/stage_llvm_bolt.sh @@ -0,0 +1,79 @@ +#!/bin/bash +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# stage_llvm_bolt.sh - put the pinned llvm-bolt on PATH, downloading it if needed. +# +# No-op when llvm-bolt is already available, so callers can invoke it +# unconditionally. Prints the bin directory it staged (nothing when the toolchain +# was already present), and exports PATH for anything sourcing it. +# +# Usage: +# . scripts/bolt/internal/stage_llvm_bolt.sh # sourced: updates PATH +# bash scripts/bolt/internal/stage_llvm_bolt.sh # executed: prints bin dir +# +# Env: +# BOLT_LLVM_STAGE_DIR where to unpack (default ./.bolt-llvm) +# GITHUB_MIRROR mirror base in place of https://github.com, matching +# docker/common/install_ccache.sh and install_cmake.sh so +# this works inside an image build +# LLVM_BOLT_VERSION overrides the pin in llvm_bolt_version.sh + +_bolt_stage_here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +_bolt_stage_llvm() { + command -v llvm-bolt >/dev/null 2>&1 && return 0 + + local dir="${BOLT_LLVM_STAGE_DIR:-$PWD/.bolt-llvm}" + if [ -x "$dir/bin/llvm-bolt" ]; then + export PATH="$dir/bin:$PATH" + return 0 + fi + + . "$_bolt_stage_here/llvm_bolt_version.sh" + + local arch + case "$(uname -m)" in + aarch64) arch=ARM64 ;; + x86_64) arch=X64 ;; + *) echo "[stage-llvm-bolt] unsupported arch $(uname -m)" >&2; return 1 ;; + esac + + local base="${GITHUB_MIRROR:-https://github.com}" + local tb="LLVM-${LLVM_BOLT_VERSION}-Linux-${arch}.tar.xz" + local url="${base}/llvm/llvm-project/releases/download/llvmorg-${LLVM_BOLT_VERSION}/${tb}" + + # Extract into a staging dir and rename, so a concurrent or interrupted run + # can never leave a half-populated toolchain that the -x check above would + # then accept. + local stage="${dir}.stage.$$" + rm -rf "$stage"; mkdir -p "$stage" + echo "[stage-llvm-bolt] staging llvm-bolt ${LLVM_BOLT_VERSION} -> $dir" >&2 + if ! curl -fSL --retry 10 --retry-all-errors --retry-delay 15 --connect-timeout 60 \ + -o "/tmp/${tb}.$$" "$url"; then + rm -rf "$stage"; rm -f "/tmp/${tb}.$$" + echo "[stage-llvm-bolt] download failed: $url" >&2 + return 1 + fi + tar -xJf "/tmp/${tb}.$$" -C "$stage" --strip-components=1 + rm -f "/tmp/${tb}.$$" + mv -T "$stage" "$dir" 2>/dev/null || rm -rf "$stage" + + [ -x "$dir/bin/llvm-bolt" ] || { echo "[stage-llvm-bolt] no llvm-bolt in $dir" >&2; return 1; } + export PATH="$dir/bin:$PATH" + echo "$dir/bin" +} + +_bolt_stage_llvm diff --git a/scripts/get_wheel_from_package.py b/scripts/get_wheel_from_package.py index f8dc16652361..9fdf1607c8c0 100644 --- a/scripts/get_wheel_from_package.py +++ b/scripts/get_wheel_from_package.py @@ -41,9 +41,87 @@ def add_arguments(parser: ArgumentParser): type=int, default=60, help="Timeout in minutes") - - -def get_wheel_from_package(arch, artifact_path, timeout): + parser.add_argument("--bolt-branch", + "-b", + default=None, + help="Comma-separated branches whose promoted BOLT " + "profile bundle should be applied to the extracted " + "wheel, tried in order. The image installs this wheel, " + "so optimizing it here is what makes the released " + "container carry optimized binaries. Omit to install " + "the wheel as built. Fatal if set and no branch has a " + "usable bundle.") + parser.add_argument("--bolt-profile-ref", + default=None, + help="Pin to one immutable profile bundle by ref, so " + "the image's wheel is optimized with the same profiles " + "as the tested build and the released wheel. Collapses " + "--bolt-branch to its first entry, since a ref names " + "one bundle under one branch. Omit to take whatever is " + "currently promoted as latest.") + + +def bolt_optimize_wheels(build_dir, arch, bolt_branch, bolt_profile_ref=None): + """Apply the latest promoted BOLT bundle to each wheel in `build_dir`. + + Deliberately uses the branch's last promoted bundle rather than one + generated from this commit: the optimized tarball for THIS run is not + published until BoltProfileGen finishes, hours after the image build starts, + and waiting on it would serialize the release behind a multi-hour GPU job. + Profiles are function-name-keyed and applied with -infer-stale-profile, so a + bundle from a nearby commit costs some optimization quality, never + correctness -- the same trade the pre-merge consume path and the image + profile overlay already make. + """ + bolt_internal = get_project_dir() / "scripts" / "bolt" / "internal" + apply_latest = str(bolt_internal / "apply_latest.sh") + triple = "x86_64-linux-gnu" if arch == "x86_64" else "aarch64-linux-gnu" + branches = [b.strip() for b in bolt_branch.split(",") if b.strip()] + env = os.environ.copy() + if bolt_profile_ref: + # The caller already narrowed --bolt-branch to the branch the pin was + # resolved against, so this list is normally a single entry. Setting the + # ref makes apply_latest.sh fetch that exact bundle. + env["BOLT_PROFILE_REF"] = bolt_profile_ref + print(f"Pinned to BOLT bundle {bolt_profile_ref} on {branches}") + + for wheel in sorted(Path(build_dir).glob("tensorrt_llm*.whl")): + bolted = wheel.with_suffix(".whl.bolted") + for branch in branches: + print(f"Applying BOLT profiles from {branch}/{triple} to " + f"{wheel.name}") + cmd = [ + "bash", apply_latest, branch, triple, + str(wheel), + str(bolted) + ] + # 3 = that branch has nothing promoted; anything else is decisive. + rc = subprocess.run(cmd, env=env).returncode + if rc == 0: + os.replace(bolted, wheel) + print(f"BOLT optimized {wheel.name} ({branch}/{triple})") + break + if rc != 3: + raise RuntimeError( + f"BOLT apply failed for {wheel.name} (rc={rc})") + print(f"No promoted bundle for {branch}/{triple}; " + "trying next branch") + else: + if bolt_profile_ref: + raise RuntimeError( + f"Pinned BOLT bundle {bolt_profile_ref} ({triple}) not " + f"found under any of {branches}; refusing to build an " + f"image whose wheel is optimized with anything else") + raise RuntimeError( + f"No promoted BOLT bundle for any of {branches} ({triple}); " + f"refusing to build an unoptimized release image") + + +def get_wheel_from_package(arch, + artifact_path, + timeout, + bolt_branch=None, + bolt_profile_ref=None): if arch == "x86_64": tarfile_name = "TensorRT-LLM.tar.gz" else: @@ -52,7 +130,11 @@ def get_wheel_from_package(arch, artifact_path, timeout): tarfile_link = f"https://urm.nvidia.com/artifactory/{artifact_path}/{tarfile_name}" for attempt in range(timeout): try: - subprocess.run(["wget", "-nv", tarfile_link], check=True) + # -O pins the output name: without it wget falls back to + # .1 when a previous attempt left a partial file behind, + # and the extract below would then read stale bytes. + subprocess.run(["wget", "-nv", "-O", tarfile_name, tarfile_link], + check=True) print(f"Tarfile is available at {tarfile_link}") break except Exception: @@ -88,6 +170,11 @@ def get_wheel_from_package(arch, artifact_path, timeout): if os.path.exists(tarfile_name): os.remove(tarfile_name) + # After the move, before the Dockerfile's release stage pip installs + # whatever is in build/. + if bolt_branch: + bolt_optimize_wheels(build_dir, arch, bolt_branch, bolt_profile_ref) + if __name__ == "__main__": parser = ArgumentParser()