From 6d9a3a2d3bd57c9274998e12d8a39a4350e7f2e7 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Fri, 18 Sep 2026 17:06:31 +0000 Subject: [PATCH 01/20] [TRTLLMINF-336][infra] BOLT the released SBSA wheel The released SBSA manylinux wheel is not optimized today. L0_Test.groovy's runLLMBuild() compiles it separately from the packed tarball and uploads it straight to /, and ENABLE_BOLT_COMPATIBLE=ON only makes those binaries BOLT-able -- it does not apply the optimization. The tarball gets BOLTed elsewhere (Build.groovy pre-merge, BoltProfileGen post-merge), so the wheel was the one release artifact with no path to it at all. apply_bolt.py already knew how to BOLT a wheel and regenerate its RECORD; that code was only reachable for wheels packed inside a release tarball. Add a --wheel mode that reuses process_wheel() on a standalone .whl, and teach apply_latest.sh to pick tarball-vs-wheel mode from the input extension (--manifest stays tarball-only: it verifies ELF hashes against the release layout, which a bare wheel has no tree for). runLLMBuild() then applies the branch's latest promoted bundle in place, before the upload and before the local pip install and the DLFW repack, so every consumer sees the same bytes. Scoped to aarch64 with an empty wheel_path, i.e. the wheel published at the root of / that the release job picks up; an imageTest/ build is a throwaway wheel compiled inside an already-released image to prove that image can still build from source, so optimizing it would prove nothing and only add a failure mode. Failure handling mirrors Build.groovy's applyLatestBolt: a real apply error is always fatal (never upload a half-BOLTed wheel), while a MISSING bundle is the cold-start case and is fatal only for a nightly_release run, whose wheels the release job actually publishes. Otherwise it is a loud skip, so cutting a new branch does not start failing every sanity-check build. Verified locally against a synthetic wheel with a stubbed llvm-bolt: the profiled lib changes, the unprofiled one does not, every RECORD row still matches the archive contents, the input wheel is left untouched, and apply_latest.sh returns 0 / 3 / 2 on apply, missing bundle, and apply error. Signed-off-by: Matt Lefebvre --- jenkins/L0_Test.groovy | 97 ++++++++++++++++++++++++++- scripts/bolt/apply_bolt.py | 55 +++++++++++++-- scripts/bolt/internal/apply_latest.sh | 41 ++++++----- 3 files changed, 172 insertions(+), 21 deletions(-) diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index b8d17a854a22..73388eeabf40 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -5718,6 +5718,77 @@ def checkKitmakerWheelDryRun(pipeline, kitmakerDryRunMetadata) } +// Apply the branch's latest promoted BOLT profile bundle to a freshly built +// wheel, in place. Mirrors Build.groovy's premerge applyLatestBolt (same +// llvm-bolt staging, same apply_latest.sh, same exit-code contract), but for a +// standalone .whl rather than a packed tarball. +// +// Strict on a real apply failure: a half-BOLTed or unverifiable wheel must never +// reach the upload. A MISSING bundle is different -- that is the cold-start case +// (a branch nothing has ever promoted for), and it is fatal only when the caller +// says this wheel is a release artifact. Otherwise it is a loud skip, so an +// ordinary sanity-check build does not start failing the day a new branch is cut. +def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, boolean boltRequired) +{ + def llvmVer = "21.1.5" // keep in sync with scripts/bolt internal/slurm_*.sh + def llvmArch = (cpu_arch == AARCH64_TRIPLE) ? "ARM64" : "X64" + // apply_latest.sh resolves exactly one branch, so try the build's own branch + // and fall back to main. Profiles are function-name-keyed and applied with + // -infer-stale-profile, so a nearby branch's bundle is valid -- the same + // candidate-branch fallback BuildDockerImage.groovy's overlay uses. + def branches = [env.gitlabTargetBranch, env.branch_name, "main"] + .collect { it?.toString()?.trim() } + .findAll { it } + .unique() + + stage("BOLT release wheel") { + sh """ + set -e + export PATH="\$PWD/.bolt-llvm/bin:\$PATH" + if ! command -v llvm-bolt >/dev/null 2>&1; then + echo '[bolt-wheel] staging llvm-bolt ${llvmVer}' + tb=LLVM-${llvmVer}-Linux-${llvmArch}.tar.xz + mkdir -p .bolt-llvm + curl -fSL --retry 10 --retry-all-errors --retry-delay 15 --connect-timeout 60 \ + -o /tmp/\$tb https://github.com/llvm/llvm-project/releases/download/llvmorg-${llvmVer}/\$tb + tar -xJf /tmp/\$tb -C .bolt-llvm --strip-components=1 + rm -f /tmp/\$tb + fi + """ + // Exit codes (apply_latest.sh): 3 = no promoted bundle for branch/triple, + // 2 = apply error, 0 = applied. Only 3 is worth trying the next branch for; + // an apply error means the bundle IS there and did not take, which retrying + // against a different branch would only paper over. + def rc = 3 + def appliedFrom = null + for (b in branches) { + rc = sh(returnStatus: true, script: """ + export PATH="\$PWD/.bolt-llvm/bin:\$PATH" + bash tensorrt_llm/scripts/bolt/internal/apply_latest.sh \ + ${b} ${cpu_arch} ${wheel} ${wheel}.bolted + """) + if (rc != 3) { + appliedFrom = b + break + } + echo "[bolt-wheel] no promoted bundle for ${b}/${cpu_arch}; trying next candidate branch" + } + if (rc == 3) { + if (boltRequired) { + error("[bolt-wheel] no promoted BOLT bundle for any of ${branches.join(', ')} (${cpu_arch}); " + + "refusing to upload an unoptimized release wheel (promote a bundle via BoltProfileGen)") + } + echo "[bolt-wheel] no promoted bundle for any of ${branches.join(', ')} (${cpu_arch}); wheel stays un-BOLTed" + return + } + if (rc != 0) { + error("[bolt-wheel] apply_latest.sh failed (rc=${rc}) for ${appliedFrom}/${cpu_arch}") + } + sh "mv -f ${wheel}.bolted ${wheel}" + echo "[bolt-wheel] ${wheel} is now BOLTed (profiles from ${appliedFrom}/${cpu_arch})" + } +} + def runLLMBuild( pipeline, cpu_arch, @@ -5726,7 +5797,8 @@ def runLLMBuild( version_override="", cpver="cp312", plat_name="", - is_dlfw=false) + is_dlfw=false, + isReleaseWheel=false) { sh "pwd && ls -alh" sh "env | sort" @@ -5771,6 +5843,22 @@ def runLLMBuild( } def wheelName = sh(returnStdout: true, script: 'cd tensorrt_llm/build && ls -1 *.whl').trim() + + // ENABLE_BOLT_COMPATIBLE=ON above only makes the binaries BOLT-able; it does + // not optimize them. The tarball gets the actual optimization elsewhere + // (Build.groovy premerge, BoltProfileGen postmerge), but this wheel is built + // and uploaded on its own, so without this step the released SBSA wheel ships + // unoptimized. Must run BEFORE the upload below, and before the local + // pip install / DLFW repack, so every consumer sees the same bytes. + // + // Scoped to the wheel published at the root of /, which is the one the + // release job picks up. An imageTest/ build is a throwaway wheel compiled + // inside an already-released image to prove that image can still build from + // source; optimizing it would prove nothing and only add a failure mode. + if (cpu_arch == AARCH64_TRIPLE && !wheel_path) { + applyLatestBoltToWheel(pipeline, "tensorrt_llm/build/${wheelName}", cpu_arch, isReleaseWheel) + } + def rootWheelUploadPath = "${cpu_arch}/${wheel_path}" // DLFW publishes the built public-version wheel under its subdirectory. Other // builds continue to publish the built wheel at the original path. @@ -6894,10 +6982,15 @@ def launchTestJobs(pipeline, testFilter, globalVars) pyver = "3.10" } + // A nightly_release run is the one whose wheels the release job + // publishes, so a missing BOLT bundle has to fail it rather than + // quietly ship an unoptimized wheel to PyPI. + def isReleaseWheel = globalVars[RUN_MODE] == "nightly_release" + buildRunner("[${toStageName(values[1], key)}] Build") { wheelPath = runLLMBuild( pipeline, cpu_arch, values[3], "", versionOverride, cpver, - values[7], isDlfw) + values[7], isDlfw, isReleaseWheel) } // TODO: Re-enable the sanity check after updating GPU testers' driver version. diff --git a/scripts/bolt/apply_bolt.py b/scripts/bolt/apply_bolt.py index 024539964b1c..e3245315c42c 100644 --- a/scripts/bolt/apply_bolt.py +++ b/scripts/bolt/apply_bolt.py @@ -30,6 +30,15 @@ --output bolt-TensorRT-LLM-GH200.tar.gz \ [--manifest manifest.json] [--strip] [--dry-run] +`--wheel` is the same operation on a standalone .whl. The released SBSA +manylinux wheel is built and uploaded on its own rather than packed into a +tarball, so it needs an entry point that does not go through the release +layout: + + apply_bolt.py --wheel tensorrt_llm--cp312-cp312-manylinux_2_39_aarch64.whl \ + --profiles /path/to/_merged \ + --output bolted.whl [--strip] [--dry-run] + llvm-bolt must be on PATH (same version used to instrument/merge). The optimize flags mirror scripts/bolt/bolt_lib.sh::optimize_libraries -- keep them in sync. """ @@ -233,6 +242,33 @@ def process_wheel( return bolted +def bolt_standalone_wheel( + wheel: Path, output: Path, profiles_dir: Path, strip: bool, dry_run: bool +) -> int: + """BOLT a .whl that is not packed inside a release tarball. + + process_wheel() rewrites in place, so the copy happens first and the input + is never touched: a failed apply leaves the original wheel uploadable. + """ + log(f"Standalone wheel: {wheel.name}") + if dry_run: + bolted = process_wheel(wheel, profiles_dir, DEFAULT_BOLT_FLAGS, strip, True) + log(f"dry-run: {bolted} member(s) would be bolted; skipping repack.") + return 0 + + output.parent.mkdir(parents=True, exist_ok=True) + if output.resolve() != wheel.resolve(): + shutil.copy2(wheel, output) + bolted = process_wheel(output, profiles_dir, DEFAULT_BOLT_FLAGS, strip, False) + if bolted == 0: + err("no ELF in the wheel matched a profile -- nothing bolted. Check --profiles names.") + if output.resolve() != wheel.resolve(): + output.unlink(missing_ok=True) + return 2 + log(f"Done. Bolted wheel: {output} ({bolted} lib(s) bolted, RECORD updated)") + return 0 + + def _rewrite_record(record_path: Path, wheel_root: Path, changed: list[Path]) -> None: changed_rel = {p.relative_to(wheel_root).as_posix() for p in changed} rows_out: list[list[str]] = [] @@ -311,12 +347,17 @@ def main() -> int: ap = argparse.ArgumentParser( description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) - ap.add_argument( + src = ap.add_mutually_exclusive_group(required=True) + src.add_argument( "--tarball", - required=True, type=Path, help="BOLT-compatible release tarball (TensorRT-LLM*.tar.gz)", ) + src.add_argument( + "--wheel", + type=Path, + help="BOLT-compatible standalone wheel (the released manylinux .whl)", + ) ap.add_argument( "--profiles", required=True, @@ -347,8 +388,9 @@ def main() -> int: if not args.dry_run and shutil.which("llvm-bolt") is None: err("llvm-bolt not on PATH") return 2 - if not args.tarball.is_file(): - err(f"tarball not found: {args.tarball}") + source = args.tarball or args.wheel + if not source.is_file(): + err(f"input not found: {source}") return 2 owns_workdir = args.workdir is None @@ -362,6 +404,11 @@ def main() -> int: f"{len(list(profiles_dir.glob('*.fdata')))} fdata)" ) + if args.wheel: + return bolt_standalone_wheel( + args.wheel, args.output, profiles_dir, args.strip, args.dry_run + ) + # Extract the tarball. extract = workdir / "extract" extract.mkdir(parents=True, exist_ok=True) diff --git a/scripts/bolt/internal/apply_latest.sh b/scripts/bolt/internal/apply_latest.sh index 42474ac5053e..72f8984baf76 100755 --- a/scripts/bolt/internal/apply_latest.sh +++ b/scripts/bolt/internal/apply_latest.sh @@ -17,20 +17,25 @@ # apply_latest.sh - premerge "consume the latest bundle" step. # # Pulls the branch-keyed `latest` BOLT profile bundle promoted by postmerge and -# applies it to a BOLT-compatible tarball, producing a bolted tarball -# (no recompile; apply_bolt.py swaps bolted ELFs into the wheel + tree). +# applies it to a BOLT-compatible artifact, producing a bolted one (no +# recompile; apply_bolt.py swaps bolted ELFs into the wheel + tree). +# +# The artifact is either a release tarball or a standalone .whl -- the released +# SBSA manylinux wheel is built and uploaded outside any tarball, so it has to +# be BOLTed on its own. The mode is picked from the input extension. # # STRICT/FATAL by design. This runs only when the caller has opted into BOLT # consumption (premerge BOLT_CONSUME), i.e. it expects a BOLTed build. Silently # proceeding un-BOLTed would let premerge "pass" while actually testing the wrong # binary, so every failure here is fatal (distinct non-zero codes for triage): -# 1 usage / input tarball missing +# 1 usage / input artifact missing # 2 apply_bolt failed (bundle present but did not apply cleanly) # 3 no bundle promoted for the branch (cold start / new branch): nothing to # consume -- await/trigger a postmerge BoltProfileGen (PROMOTE=true), or turn # BOLT_CONSUME off for this build. # -# Usage: apply_latest.sh +# Usage: apply_latest.sh +# is a release tarball or a .whl. # Requires: llvm-bolt on PATH; urm-artifactory-creds for artifactory.sh pull-latest. set -euo pipefail @@ -38,13 +43,13 @@ set -euo pipefail HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # .../scripts/bolt/internal TOOLKIT="$(dirname "$HERE")" # .../scripts/bolt -BRANCH="${1:?apply_latest: }" +BRANCH="${1:?apply_latest: }" TRIPLE="${2:?triple required}" -IN_TAR="${3:?in_tarball required}" -OUT_TAR="${4:?out_bolted_tarball required}" +IN_ARTIFACT="${3:?in_artifact required}" +OUT_ARTIFACT="${4:?out_bolted_artifact required}" -if [ ! -f "$IN_TAR" ]; then - echo "[apply_latest] FATAL: input tarball not found: $IN_TAR" >&2 +if [ ! -f "$IN_ARTIFACT" ]; then + echo "[apply_latest] FATAL: input artifact not found: $IN_ARTIFACT" >&2 exit 1 fi @@ -59,15 +64,21 @@ if ! bash "$HERE/artifactory.sh" pull-latest "$BRANCH" "$TRIPLE" "$DEST"; then exit 3 fi -# 2) Apply the pulled profiles to the tarball. Fatal on failure -- do NOT fall -# back to the un-BOLTed tarball (that would silently test the wrong binary). +# 2) Apply the pulled profiles. Fatal on failure -- do NOT fall back to the +# un-BOLTed artifact (that would silently test/ship the wrong binary). +# --manifest verifies ELF hashes against the release layout, which only the +# tarball has; a standalone wheel has no tree to walk, so it is omitted there. +case "$IN_ARTIFACT" in + *.whl) APPLY_ARGS=(--wheel "$IN_ARTIFACT") ;; + *) APPLY_ARGS=(--tarball "$IN_ARTIFACT" --manifest "$DEST/manifest.json") ;; +esac + if ! python3 "$TOOLKIT/apply_bolt.py" \ - --tarball "$IN_TAR" \ + "${APPLY_ARGS[@]}" \ --profiles "$DEST" \ - --manifest "$DEST/manifest.json" \ - --output "$OUT_TAR"; then + --output "$OUT_ARTIFACT"; then echo "[apply_latest] FATAL: apply_bolt failed for ${BRANCH}/${TRIPLE}" >&2 exit 2 fi -echo "[apply_latest] bolted tarball -> $OUT_TAR" +echo "[apply_latest] bolted artifact -> $OUT_ARTIFACT" From 174f861dc67b0c199caa6f9bb77d188ae4f8efe9 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Fri, 18 Sep 2026 17:09:14 +0000 Subject: [PATCH 02/20] [TRTLLMINF-336][infra] let the SBSA release image require the BOLTed wheel boltOverlayEnabled and boltProfilesRequired only govern the profile BUNDLE baked in as a thin Dockerfile.bolt layer, which documents how to reproduce a BOLTed build but leaves the binaries the image actually installs unoptimized. Nothing governs the installed wheel, and the tarball it comes from is whichever one exists first -- the canonical, unoptimized one, since Build-Docker-Images runs in parallel with the SBSA branch. Add boltRequireBoltedWheel: when set, the SBSA release image is built from bolted- (get_wheel_from_package.py --bolted) rather than the canonical tarball, waiting for it to be published and failing if it never is. Kept independent of the overlay flags so it can be rolled back on its own, and scoped to sbsa because x86_64 has no promoted bundle to wait for. Two supporting changes. The wait gets its own, much longer clock: the canonical tarball lands when the build stage finishes, the BOLTed one only after BoltProfileGen's SLURM fan-out, whose own stage timeout is 8h -- so 60 minutes is the wrong budget for it. And the retry that rebuilds the wheel from source when a build fails with the download args is disabled for this path: it is a fine recovery for an ordinary build, but here it would silently produce an unoptimized release image that looks identical to an optimized one. Also pins wget's output name in the download loop. Without -O it falls back to .1 when a previous attempt left a partial file behind, and the extract would then read stale bytes -- a latent bug that gets much more reachable now that the loop can run for hours. Default false, so this is inert until a follow-up change turns it on. Signed-off-by: Matt Lefebvre --- jenkins/BuildDockerImage.groovy | 42 ++++++++++++++++++++++++++++++- scripts/get_wheel_from_package.py | 25 +++++++++++++++--- 2 files changed, 62 insertions(+), 5 deletions(-) diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index 3dcaf2ba5597..3608771cc737 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -58,11 +58,24 @@ BOLT_OVERLAY_ENABLED = (params.boltOverlayEnabled ?: env.boltOverlayEnabled ?: " // carries profiles). Left false until the enable PR wires it true for the // release/nightly path; premerge/new-branch stays lenient (retag plain build). BOLT_PROFILES_REQUIRED = (params.boltProfilesRequired ?: env.boltProfilesRequired ?: "false").toString() == "true" +// Separate from the two above: those govern the profile BUNDLE baked in as a +// thin layer (Dockerfile.bolt), which documents how to reproduce a BOLTed build +// but does NOT optimize the binaries the image actually installs. This one +// governs the INSTALLED wheel -- when true, the release image must be built from +// the BOLT-optimized tarball (bolted-) rather than whichever tarball +// happens to exist first. Kept independent so it can be rolled back on its own. +BOLT_REQUIRE_BOLTED_WHEEL = (params.boltRequireBoltedWheel ?: env.boltRequireBoltedWheel ?: "false").toString() == "true" // <<< BOLT profile-bundle overlay <<< ENABLE_USE_WHEEL_FROM_BUILD_STAGE = params.useWheelFromBuildStage ?: false WAIT_TIME_FOR_BUILD_STAGE = 60 // minutes +// The canonical tarball lands as soon as the build stage finishes; the BOLTed +// one only after BoltProfileGen's SLURM fan-out (three perf workloads at up to +// 4h walltime each, plus queue and merge -- its own stage timeout is 8h). So +// requiring the optimized build means waiting on a different, much longer clock +// than "the build stage finished". +WAIT_TIME_FOR_BOLTED_BUILD_STAGE = 480 // minutes BUILD_JOBS = "32" BUILD_JOBS_RELEASE_X86_64 = "32" @@ -286,10 +299,27 @@ def prepareWheelFromBuildStage(dockerfileStage, arch) { } def wheelScript = 'scripts/get_wheel_from_package.py' - def wheelArgs = "--arch ${arch} --timeout ${WAIT_TIME_FOR_BUILD_STAGE} --artifact_path " + env.uploadPath + // Only aarch64 has a promoted profile bundle, so only the SBSA image has an + // optimized tarball to wait for; x86 keeps taking the canonical one. + def requireBolted = BOLT_REQUIRE_BOLTED_WHEEL && arch == "sbsa" + def waitTime = requireBolted ? WAIT_TIME_FOR_BOLTED_BUILD_STAGE : WAIT_TIME_FOR_BUILD_STAGE + def wheelArgs = "--arch ${arch} --timeout ${waitTime} --artifact_path " + env.uploadPath + if (requireBolted) { + echo "Release image for ${arch} requires the BOLT-optimized build; waiting up to ${waitTime} minutes for it" + wheelArgs += " --bolted" + } return " BUILD_WHEEL_SCRIPT=${wheelScript} BUILD_WHEEL_ARGS='${wheelArgs}'" } +// Whether a docker build that failed WITH the downloaded-wheel args may be +// retried without them. The retry rebuilds the wheel from source in-container, +// which is a fine recovery for an ordinary build but silently defeats the point +// when the whole reason for the download was to install BOLT-optimized binaries +// -- the retry would produce an unoptimized release image that looks identical. +def mayRetryWithoutBuildStageWheel(arch) { + return !(BOLT_REQUIRE_BOLTED_WHEEL && arch == "sbsa") +} + // Produce each CANONICAL image from its raw `-noprofiles` build by // overlaying the merged LLVM BOLT profile bundle as a thin layer (docker/ // Dockerfile.bolt via the docker/Makefile `bolt_overlay` target). Canonical is @@ -587,6 +617,11 @@ def buildImage(config, imageKeyToTag, versionOverride) if (buildWheelArgs.trim().isEmpty()) { throw ex } + if (!mayRetryWithoutBuildStageWheel(arch)) { + echo "Build failed with wheel arguments and the BOLT-optimized wheel is required for ${arch}; " + + "NOT retrying from source (that would publish an unoptimized release image)" + throw ex + } echo "Build failed with wheel arguments, retrying without them" buildWheelArgs = "" trtllm_utils.llmExecStepWithRetry(this, script: """ @@ -867,6 +902,11 @@ pipeline { defaultValue: false, description: "When boltOverlayEnabled is true, treat a missing/empty BOLT bundle as a FATAL error instead of retagging the plain build as canonical. Enable for the release/nightly path to guarantee canonical images carry profiles." ) + booleanParam( + name: "boltRequireBoltedWheel", + defaultValue: false, + description: "Build the SBSA release image from the BOLT-optimized tarball (bolted-) instead of the canonical one, waiting for it to be published and failing if it never is. Independent of boltOverlayEnabled, which only bakes in the profile bundle and leaves the installed binaries unoptimized. Enable for the release path; ignored on x86_64, which has no promoted bundle." + ) string( name: "boltProfileBranch", defaultValue: "", diff --git a/scripts/get_wheel_from_package.py b/scripts/get_wheel_from_package.py index f8dc16652361..eaad10edf2ed 100644 --- a/scripts/get_wheel_from_package.py +++ b/scripts/get_wheel_from_package.py @@ -41,18 +41,35 @@ def add_arguments(parser: ArgumentParser): type=int, default=60, help="Timeout in minutes") - - -def get_wheel_from_package(arch, artifact_path, timeout): + parser.add_argument("--bolted", + "-b", + action="store_true", + help="Require the BOLT-optimized build " + "(bolted-) that BoltProfileGen publishes " + "alongside the plain one. The canonical tarfile " + "appears as soon as the build stage finishes, long " + "before BOLT has run, so an image that must ship " + "optimized binaries has to wait on this distinct " + "name instead. Never falls back.") + + +def get_wheel_from_package(arch, artifact_path, timeout, bolted=False): if arch == "x86_64": tarfile_name = "TensorRT-LLM.tar.gz" else: tarfile_name = "TensorRT-LLM-GH200.tar.gz" + if bolted: + tarfile_name = f"bolted-{tarfile_name}" + tarfile_link = f"https://urm.nvidia.com/artifactory/{artifact_path}/{tarfile_name}" for attempt in range(timeout): try: - subprocess.run(["wget", "-nv", tarfile_link], check=True) + # -O pins the output name: without it wget falls back to + # .1 when a previous attempt left a partial file behind, + # and the extract below would then read stale bytes. + subprocess.run(["wget", "-nv", "-O", tarfile_name, tarfile_link], + check=True) print(f"Tarfile is available at {tarfile_link}") break except Exception: From e2364796d594041f23c4c985236e22b21050f4e0 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Fri, 18 Sep 2026 17:09:47 +0000 Subject: [PATCH 03/20] [TRTLLMINF-336][infra] enable the BOLTed wheel for the SBSA release image Pass boltRequireBoltedWheel=true from both BuildDockerImages launches, so the SBSA release image is built from bolted- instead of racing the canonical tarball and permanently baking in the unoptimized wheel. Both launches are set together because they push the same tags: the scanned and registered artifact has to be the same image the post-merge path publishes. The scaffolding landed inert in the previous change, so this only answers "should this pipeline require the optimized build". Operational note: this makes the SBSA release image build wait on BoltProfileGen rather than on Build-SBSA -- up to 8h rather than 60 minutes -- because that is what depending on the optimized artifact costs. The alternative is to decouple and have the image apply the branch's previously promoted bundle itself, the way pre-merge consume and the image overlay already do, which gets optimized images with no wait and no race at the cost of profiles from the prior commit. Rollback: drop boltRequireBoltedWheel from both launches, or set the job parameter to false. Nothing else has to change -- the canonical tarball is untouched by BOLT, so the image falls straight back to it. Signed-off-by: Matt Lefebvre --- jenkins/L0_MergeRequest.groovy | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index 45562b395a6f..f5b0506ca5c0 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -2526,6 +2526,15 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) // main, so a ref with no promoted bundle still gets profiles. 'boltOverlayEnabled': true, 'boltProfilesRequired': true, + // The overlay above only bakes in the profile bundle; it + // leaves the installed wheel unoptimized. This makes the + // SBSA release image install the BOLT-optimized wheel, + // by waiting for BoltProfileGen to publish + // bolted- rather than grabbing whichever + // tarball exists first. Inert on x86_64 (no promoted + // bundle) and whenever the wheel is built from source + // rather than downloaded. + 'boltRequireBoltedWheel': true, ] if (runMode == "nightly_release") { additionalParameters += [ @@ -2585,9 +2594,12 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) 'uploadPath': UPLOAD_PATH, // Must match Build-Docker-Images above: this path pushes the // same tags, so the scanned+registered image has to be the - // BOLTed canonical one rather than a plain build. + // BOLTed canonical one rather than a plain build, built on + // the BOLT-optimized wheel rather than the first tarball to + // appear. 'boltOverlayEnabled': true, 'boltProfilesRequired': true, + 'boltRequireBoltedWheel': true, ] if (runMode == "nightly_release") { additionalParameters += [ From 19899acb34d8234e4dc27ecaa2325988acc8cf44 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Sat, 19 Sep 2026 01:40:01 +0000 Subject: [PATCH 04/20] [TRTLLMINF-336][infra] single source of truth for the llvm-bolt pin Review feedback: the "keep in sync with scripts/bolt internal/slurm_*.sh" comment was a one-way reference with nothing on the other end. The pin had five copies -- slurm_merge.sh, perf_instrument_hook.sh, Build.groovy, BoltProfileGen.groovy, and the one this PR added in L0_Test.groovy. Drift between them is not a build error: instrumentation, merge and apply exchange .fdata/.yaml profiles whose formats are not guaranteed stable across releases, so a mismatched version silently produces a bad or empty profile. A comment asking five call sites to stay in sync is the wrong tool for that. Move the pin to scripts/bolt/internal/llvm_bolt_version.sh and source it everywhere. The Groovy callers read it from the checkout they already have rather than interpolating a literal, so bumping the version is now one line in one file. The conditional assignment is preserved, so LLVM_BOLT_VERSION from the environment still overrides it for a one-off run. Two resolution details. BoltProfileGen builds the tarball name and cache key shell-side now, because its staging step runs after the cluster-side extract and so can source the file from the extracted tree. slurm_merge.sh resolves via TOOLKIT_HOST rather than BASH_SOURCE, because sbatch runs a copy of the script out of the node's spool directory and its own path says nothing about where the toolkit lives -- the same reason that file already took TOOLKIT_HOST from the environment. Signed-off-by: Matt Lefebvre --- jenkins/BoltProfileGen.groovy | 12 ++++-- jenkins/Build.groovy | 8 ++-- jenkins/L0_Test.groovy | 8 ++-- scripts/bolt/internal/llvm_bolt_version.sh | 38 +++++++++++++++++++ scripts/bolt/internal/perf_instrument_hook.sh | 4 +- scripts/bolt/internal/slurm_merge.sh | 6 ++- 6 files changed, 61 insertions(+), 15 deletions(-) create mode 100644 scripts/bolt/internal/llvm_bolt_version.sh diff --git a/jenkins/BoltProfileGen.groovy b/jenkins/BoltProfileGen.groovy index 343276b83cc2..ee1cc95db416 100644 --- a/jenkins/BoltProfileGen.groovy +++ b/jenkins/BoltProfileGen.groovy @@ -407,15 +407,19 @@ def submitProfileGen(pipeline) // then every later run just does a local lustre extract from the cached // .tar.xz (no WAN). The one-time fetch reuses fetch_verified(). def llvmArch = (TARGET_ARCH == AARCH64_TRIPLE) ? "ARM64" : "X64" - def llvmVer = "21.1.5" // keep in sync with internal/slurm_merge.sh LLVM_BOLT_VERSION - def llvmTb = "LLVM-${llvmVer}-Linux-${llvmArch}.tar.xz" // Cache lives OUTSIDE the bolt-ci retention root (purged at depth 4 after 7 // days) so the per-run workspace reaper can't delete it. def llvmCacheDir = "${scratch}/users/svc_tensorrt/bolt-cache/llvm" + // The pin is read from the extracted tree (this stage runs after "Bootstrap: + // extract tarball"), so the collect hook, the merge job and this bootstrap + // all resolve the same llvm-bolt from one file. Tarball name and cache key + // are therefore built shell-side, where the version is known. def llvmStage = """ - LLVM_URL='https://github.com/llvm/llvm-project/releases/download/llvmorg-${llvmVer}/${llvmTb}' + . '${ws}/TensorRT-LLM/src/scripts/bolt/internal/llvm_bolt_version.sh' + LLVM_TB="LLVM-\${LLVM_BOLT_VERSION}-Linux-${llvmArch}.tar.xz" + LLVM_URL="https://github.com/llvm/llvm-project/releases/download/llvmorg-\${LLVM_BOLT_VERSION}/\${LLVM_TB}" LLVM_CACHE_DIR='${llvmCacheDir}' - LLVM_CACHE_TB='${llvmCacheDir}/${llvmTb}' + LLVM_CACHE_TB="${llvmCacheDir}/\${LLVM_TB}" LLVM_DIR='${ws}/builds/llvm' PARTS=16 """.stripIndent() + boltFetchLib + ''' diff --git a/jenkins/Build.groovy b/jenkins/Build.groovy index 1cb2347c9dd1..b75be28ed2d6 100644 --- a/jenkins/Build.groovy +++ b/jenkins/Build.groovy @@ -551,7 +551,6 @@ def applyLatestBolt(pipeline, tarName, is_linux_x86_64, artifacts=null) def branch = env.gitlabTargetBranch ?: env.branch_name ?: "main" def triple = is_linux_x86_64 ? "x86_64-linux-gnu" : "aarch64-linux-gnu" def llvmArch = is_linux_x86_64 ? "X64" : "ARM64" - def llvmVer = "21.1.5" // keep in sync with scripts/bolt internal/slurm_*.sh stage("BOLT consume") { // apply_latest.sh exit codes: 3 = no promoted bundle for branch/triple, // 2 = apply error, 0 = applied. Capture the code so a MISSING bundle (e.g. @@ -561,11 +560,12 @@ def applyLatestBolt(pipeline, tarName, is_linux_x86_64, artifacts=null) set -e export PATH="\$PWD/.bolt-llvm/bin:\$PATH" if ! command -v llvm-bolt >/dev/null 2>&1; then - echo '[bolt-consume] staging llvm-bolt ${llvmVer}' - tb=LLVM-${llvmVer}-Linux-${llvmArch}.tar.xz + . ${LLM_ROOT}/scripts/bolt/internal/llvm_bolt_version.sh + echo "[bolt-consume] staging llvm-bolt \${LLVM_BOLT_VERSION}" + tb=LLVM-\${LLVM_BOLT_VERSION}-Linux-${llvmArch}.tar.xz mkdir -p .bolt-llvm curl -fSL --retry 10 --retry-all-errors --retry-delay 15 --connect-timeout 60 \ - -o /tmp/\$tb https://github.com/llvm/llvm-project/releases/download/llvmorg-${llvmVer}/\$tb + -o /tmp/\$tb "https://github.com/llvm/llvm-project/releases/download/llvmorg-\${LLVM_BOLT_VERSION}/\$tb" tar -xJf /tmp/\$tb -C .bolt-llvm --strip-components=1 rm -f /tmp/\$tb fi diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index 73388eeabf40..18c2724fa827 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -5730,7 +5730,6 @@ def checkKitmakerWheelDryRun(pipeline, kitmakerDryRunMetadata) // ordinary sanity-check build does not start failing the day a new branch is cut. def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, boolean boltRequired) { - def llvmVer = "21.1.5" // keep in sync with scripts/bolt internal/slurm_*.sh def llvmArch = (cpu_arch == AARCH64_TRIPLE) ? "ARM64" : "X64" // apply_latest.sh resolves exactly one branch, so try the build's own branch // and fall back to main. Profiles are function-name-keyed and applied with @@ -5746,11 +5745,12 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, boolean bolt set -e export PATH="\$PWD/.bolt-llvm/bin:\$PATH" if ! command -v llvm-bolt >/dev/null 2>&1; then - echo '[bolt-wheel] staging llvm-bolt ${llvmVer}' - tb=LLVM-${llvmVer}-Linux-${llvmArch}.tar.xz + . tensorrt_llm/scripts/bolt/internal/llvm_bolt_version.sh + echo "[bolt-wheel] staging llvm-bolt \${LLVM_BOLT_VERSION}" + tb=LLVM-\${LLVM_BOLT_VERSION}-Linux-${llvmArch}.tar.xz mkdir -p .bolt-llvm curl -fSL --retry 10 --retry-all-errors --retry-delay 15 --connect-timeout 60 \ - -o /tmp/\$tb https://github.com/llvm/llvm-project/releases/download/llvmorg-${llvmVer}/\$tb + -o /tmp/\$tb "https://github.com/llvm/llvm-project/releases/download/llvmorg-\${LLVM_BOLT_VERSION}/\$tb" tar -xJf /tmp/\$tb -C .bolt-llvm --strip-components=1 rm -f /tmp/\$tb fi diff --git a/scripts/bolt/internal/llvm_bolt_version.sh b/scripts/bolt/internal/llvm_bolt_version.sh new file mode 100644 index 000000000000..2ea766c9f6b1 --- /dev/null +++ b/scripts/bolt/internal/llvm_bolt_version.sh @@ -0,0 +1,38 @@ +#!/bin/bash +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# llvm_bolt_version.sh - the pinned llvm-bolt release, defined once. +# +# Every BOLT step has to run the SAME llvm-bolt. Instrumentation, merge, and +# apply exchange .fdata and .yaml profiles, and those formats are not guaranteed +# stable across releases -- so a version that drifts between steps does not fail +# the build, it silently produces a bad or empty profile. That failure mode is +# why this is a shared file rather than a comment asking five call sites to stay +# in sync. +# +# Sourced by: +# scripts/bolt/internal/slurm_merge.sh (merge job, cluster-side) +# scripts/bolt/internal/perf_instrument_hook.sh (collect job, per node) +# jenkins/Build.groovy (pre-merge consume, build pod) +# jenkins/L0_Test.groovy (release wheel, build pod) +# jenkins/BoltProfileGen.groovy (bootstrap, cluster frontend) +# +# The Groovy callers source this from the checkout they already have rather than +# interpolating a literal, so bumping the pin here is the whole change. +# +# Assignment is conditional so a one-off run can pin a different release from the +# environment without editing the file. +LLVM_BOLT_VERSION="${LLVM_BOLT_VERSION:-21.1.5}" diff --git a/scripts/bolt/internal/perf_instrument_hook.sh b/scripts/bolt/internal/perf_instrument_hook.sh index 8cf9934f98af..3cc3993e2780 100755 --- a/scripts/bolt/internal/perf_instrument_hook.sh +++ b/scripts/bolt/internal/perf_instrument_hook.sh @@ -31,7 +31,7 @@ # BOLT_FDATA_DIR (required) run-level shared dir; this node writes / # BOLT_LLVM_DIR (optional) dir to stage/reuse llvm-bolt (default /tmp/bolt-llvm) # BOLT_WORK_DIR (optional) per-node work dir (default /tmp/bolt_work_) -# LLVM_BOLT_VERSION (optional) default 21.1.5 +# LLVM_BOLT_VERSION (optional) overrides the pin in internal/llvm_bolt_version.sh set -euo pipefail @@ -47,7 +47,7 @@ mkdir -p "$FDATA_OUTPUT_DIR" # Ensure llvm-bolt / merge-fdata on PATH (self-stage if absent). Race-safe # extract-then-atomic-rename in case BOLT_LLVM_DIR is a shared path hit by # multiple nodes. -LLVM_BOLT_VERSION="${LLVM_BOLT_VERSION:-21.1.5}" +. "$HERE/llvm_bolt_version.sh" BOLT_LLVM_DIR="${BOLT_LLVM_DIR:-/tmp/bolt-llvm}" if ! command -v llvm-bolt >/dev/null 2>&1; then if [ ! -x "$BOLT_LLVM_DIR/bin/llvm-bolt" ]; then diff --git a/scripts/bolt/internal/slurm_merge.sh b/scripts/bolt/internal/slurm_merge.sh index f04e7f4afcd2..ac85ecad9fad 100755 --- a/scripts/bolt/internal/slurm_merge.sh +++ b/scripts/bolt/internal/slurm_merge.sh @@ -84,7 +84,11 @@ echo "[INFO] merge: FDATA_ROOT=$FDATA_ROOT REF=$BOLT_REF TRIPLE=$TRIPLE OUT=$ export ENROOT_CACHE_PATH="${ENROOT_CACHE_PATH:-/home/svc_tensorrt/.cache/enroot}" # ---- CI self-staging: llvm-bolt -------------------------------------------- -LLVM_BOLT_VERSION="${LLVM_BOLT_VERSION:-21.1.5}" +# Pin comes from llvm_bolt_version.sh so this job, the collect hook, and the +# Jenkins build pods cannot drift onto different llvm-bolt releases. Resolved via +# TOOLKIT_HOST, not BASH_SOURCE: sbatch runs a COPY of this script out of the +# node's spool dir, so its own path says nothing about where the toolkit lives. +. "$TOOLKIT_HOST/internal/llvm_bolt_version.sh" if [ ! -x "$BUILDS_HOST/llvm/bin/llvm-bolt" ]; then echo "[INFO] Installing llvm-bolt ${LLVM_BOLT_VERSION} -> $BUILDS_HOST/llvm" case "$(uname -m)" in From d2116f77f1b55d43a991d40840dd92985452fb42 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Sat, 19 Sep 2026 01:54:33 +0000 Subject: [PATCH 05/20] [TRTLLMINF-336][infra] gate the release wheel on the BOLT switch, not run mode Review feedback, two parts. nightly_release was the wrong trigger. The wheel the release job publishes is the same artifact whether the run is nightly, weekly or GA, so keying off one run mode meant the others shipped un-BOLTed. The question is "is BOLT on", not "is this a nightly". And the step ignored the pipeline's BOLT switch entirely -- it would have run on any nightly_release regardless of whether the run asked for BOLTed binaries. Gate on globalVars.bolt_consume_build, the same value L0_MergeRequest::resolveBoltConsume hands the build helpers for the tarball, so the released wheel and the released tarball can never be optimized differently. Read with the ?.toString() == "true" idiom Build.groovy already uses, since the value crosses a JSON parameter boundary. With the switch as the trigger, the required/optional split goes away: a missing bundle is now fatal like every other failure. "BOLT is on" has to mean the wheel IS optimized -- degrading to an un-BOLTed wheel with a log line is exactly how an unoptimized artifact reaches PyPI unnoticed. This is stricter than Build.groovy's applyLatestBolt, which skips on a missing bundle; it has to be lenient because the same switch covers x86_64, where nothing is promoted yet, while this path is aarch64-only and the switch is main-only. Known gap worth a look: resolveBoltConsume also disables the switch for post-merge, because the post-merge TARBALL is the un-BOLTed input BoltProfileGen profiles. That reasoning does not apply to this wheel -- it is a separate build that BoltProfileGen never reads -- so post-merge sanity-check wheels stay un-BOLTed as a side effect. Harmless today (those wheels are not released), but say the word if the release path ever draws from a post-merge run and it should get its own switch instead. Signed-off-by: Matt Lefebvre --- jenkins/L0_Test.groovy | 51 +++++++++++++++++++++++++----------------- 1 file changed, 31 insertions(+), 20 deletions(-) diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index 18c2724fa827..bc350274a0ea 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -3152,6 +3152,12 @@ def IMAGE_KEY_TO_TAG = "image_key_to_tag" def TRTLLM_VERSION_OVERRIDE = "trtllm_version_override" @Field def RUN_MODE = "run_mode" +// Whether this pipeline asked for BOLT-optimized binaries. Resolved once by +// L0_MergeRequest.groovy::resolveBoltConsume and propagated here in globalVars, +// so the release wheel below follows the same switch as the build tarball +// instead of inventing a second answer to "is BOLT on". +@Field +def BOLT_CONSUME_BUILD = "bolt_consume_build" def globalVars = [ (GITHUB_PR_API_URL): null, (CACHED_CHANGED_FILE_LIST): null, @@ -3159,6 +3165,7 @@ def globalVars = [ (IMAGE_KEY_TO_TAG): [:], (TRTLLM_VERSION_OVERRIDE): null, (RUN_MODE): null, + (BOLT_CONSUME_BUILD): false, ] class GlobalState { @@ -5723,12 +5730,15 @@ def checkKitmakerWheelDryRun(pipeline, kitmakerDryRunMetadata) // llvm-bolt staging, same apply_latest.sh, same exit-code contract), but for a // standalone .whl rather than a packed tarball. // -// Strict on a real apply failure: a half-BOLTed or unverifiable wheel must never -// reach the upload. A MISSING bundle is different -- that is the cold-start case -// (a branch nothing has ever promoted for), and it is fatal only when the caller -// says this wheel is a release artifact. Otherwise it is a loud skip, so an -// ordinary sanity-check build does not start failing the day a new branch is cut. -def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, boolean boltRequired) +// Every failure is fatal, including a missing bundle. The caller only reaches +// here when the pipeline asked for BOLT, and this wheel is what the release job +// publishes, so "BOLT is on" has to mean the wheel IS optimized -- degrading to +// an un-BOLTed wheel with a log line is how an unoptimized artifact reaches PyPI +// unnoticed. That is stricter than Build.groovy's applyLatestBolt, which treats +// a missing bundle as a skip; it has to, because the same switch covers x86_64, +// where nothing is promoted yet. This path is aarch64-only and the switch is +// main-only, so there is always a bundle to find. +def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) { def llvmArch = (cpu_arch == AARCH64_TRIPLE) ? "ARM64" : "X64" // apply_latest.sh resolves exactly one branch, so try the build's own branch @@ -5774,12 +5784,9 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, boolean bolt echo "[bolt-wheel] no promoted bundle for ${b}/${cpu_arch}; trying next candidate branch" } if (rc == 3) { - if (boltRequired) { - error("[bolt-wheel] no promoted BOLT bundle for any of ${branches.join(', ')} (${cpu_arch}); " + - "refusing to upload an unoptimized release wheel (promote a bundle via BoltProfileGen)") - } - echo "[bolt-wheel] no promoted bundle for any of ${branches.join(', ')} (${cpu_arch}); wheel stays un-BOLTed" - return + error("[bolt-wheel] no promoted BOLT bundle for any of ${branches.join(', ')} (${cpu_arch}); " + + "refusing to upload an unoptimized release wheel (promote a bundle via BoltProfileGen, " + + "or turn BOLT consume off for this run)") } if (rc != 0) { error("[bolt-wheel] apply_latest.sh failed (rc=${rc}) for ${appliedFrom}/${cpu_arch}") @@ -5798,7 +5805,7 @@ def runLLMBuild( cpver="cp312", plat_name="", is_dlfw=false, - isReleaseWheel=false) + boltConsume=false) { sh "pwd && ls -alh" sh "env | sort" @@ -5851,12 +5858,17 @@ def runLLMBuild( // unoptimized. Must run BEFORE the upload below, and before the local // pip install / DLFW repack, so every consumer sees the same bytes. // + // Gated on the pipeline's BOLT switch, not on the run mode: the wheel the + // release job publishes is the same artifact whether the run is nightly, + // weekly or GA, so the trigger has to be "is BOLT on", not "is this a + // nightly". aarch64 only -- x86_64 has no promoted bundle. + // // Scoped to the wheel published at the root of /, which is the one the // release job picks up. An imageTest/ build is a throwaway wheel compiled // inside an already-released image to prove that image can still build from // source; optimizing it would prove nothing and only add a failure mode. - if (cpu_arch == AARCH64_TRIPLE && !wheel_path) { - applyLatestBoltToWheel(pipeline, "tensorrt_llm/build/${wheelName}", cpu_arch, isReleaseWheel) + if (boltConsume && cpu_arch == AARCH64_TRIPLE && !wheel_path) { + applyLatestBoltToWheel(pipeline, "tensorrt_llm/build/${wheelName}", cpu_arch) } def rootWheelUploadPath = "${cpu_arch}/${wheel_path}" @@ -6982,15 +6994,14 @@ def launchTestJobs(pipeline, testFilter, globalVars) pyver = "3.10" } - // A nightly_release run is the one whose wheels the release job - // publishes, so a missing BOLT bundle has to fail it rather than - // quietly ship an unoptimized wheel to PyPI. - def isReleaseWheel = globalVars[RUN_MODE] == "nightly_release" + // Same switch the build helpers use for the tarball, so the released + // wheel and the released tarball are never optimized differently. + def boltConsume = globalVars[BOLT_CONSUME_BUILD]?.toString() == "true" buildRunner("[${toStageName(values[1], key)}] Build") { wheelPath = runLLMBuild( pipeline, cpu_arch, values[3], "", versionOverride, cpver, - values[7], isDlfw, isReleaseWheel) + values[7], isDlfw, boltConsume) } // TODO: Re-enable the sanity check after updating GPU testers' driver version. From 1a2d4f159752f21cfd8ee99a585a64004ba6c752 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Mon, 21 Sep 2026 16:39:31 +0000 Subject: [PATCH 06/20] [TRTLLMINF-336][infra] never emit a partially BOLTed artifact Review feedback, two related holes in the apply path. bolt_elf() returned False when llvm-bolt failed and both callers just skipped the member. With two matching ELFs where one fails, the count stayed positive, the wheel was repacked, and the result was uploaded -- a partially optimized artifact is indistinguishable from a fully optimized one, so nothing downstream would ever notice. bolt_elf() now raises BoltApplyError instead. There is no caller that can sensibly continue, so reporting-and-skipping was the wrong shape. main() turns it into exit 2 after the detailed llvm-bolt output bolt_elf already logged, and bolt_standalone_wheel() deletes the copy it made before re-raising, so no half-written wheel is left where a caller might upload it. The loose-ELF loop in the tarball path had the identical hole and is fixed by the same change. Also reject --output == --wheel. process_wheel() rewrites its argument, so the in-place case silently violated the contract the function documents: the input is copied first precisely so a failed apply leaves it usable. Extracting the body into _apply() is what lets main() own the try/except/finally without re-indenting the whole function under a second try. Verified with a stubbed llvm-bolt that fails for one selected library: - wheel, all libs bolt: repacked, RECORD rows all match, unprofiled lib untouched - wheel, one lib fails: exit 2, no output file left behind - tarball, loose ELF fails: exit 2, no output file left behind - --output == --wheel: rejected, input wheel unmodified Signed-off-by: Matt Lefebvre --- scripts/bolt/apply_bolt.py | 176 ++++++++++++++++++++++--------------- 1 file changed, 103 insertions(+), 73 deletions(-) diff --git a/scripts/bolt/apply_bolt.py b/scripts/bolt/apply_bolt.py index e3245315c42c..e135c2abcf69 100644 --- a/scripts/bolt/apply_bolt.py +++ b/scripts/bolt/apply_bolt.py @@ -119,15 +119,25 @@ def is_elf(path: Path) -> bool: return False +class BoltApplyError(RuntimeError): + """An in-scope ELF did not optimize. + + Raised rather than reported, because there is no caller that can sensibly + carry on: an artifact where some matching ELFs were optimized and others + were skipped is indistinguishable from a fully optimized one, and it is the + version that would get uploaded. + """ + + # --------------------------------------------------------------------------- # BOLT one ELF (in place) # --------------------------------------------------------------------------- -def bolt_elf(elf: Path, profile: Path, flags: list[str], strip: bool, dry_run: bool) -> bool: - """BOLT `elf` in place using `profile`. Returns True if optimized.""" +def bolt_elf(elf: Path, profile: Path, flags: list[str], strip: bool, dry_run: bool) -> None: + """BOLT `elf` in place using `profile`. Raises BoltApplyError if it cannot.""" rel = elf.name if dry_run: log(f" would bolt {rel} <- {profile.name}") - return True + return out = elf.with_suffix(elf.suffix + ".bolted") cmd = ["llvm-bolt", str(elf), "-o", str(out), f"-data={profile}"] + flags @@ -138,7 +148,7 @@ def bolt_elf(elf: Path, profile: Path, flags: list[str], strip: bool, dry_run: b for line in output.splitlines()[-15:]: err(f" {line}") out.unlink(missing_ok=True) - return False + raise BoltApplyError(f"llvm-bolt failed for {rel} (rc={rc})") if strip: rc_s, _ = run(["llvm-strip", "--strip-all", str(out)]) @@ -148,7 +158,6 @@ def bolt_elf(elf: Path, profile: Path, flags: list[str], strip: bool, dry_run: b # Preserve mode, then replace original. out.chmod(elf.stat().st_mode) os.replace(out, elf) - return True # --------------------------------------------------------------------------- @@ -214,9 +223,9 @@ def process_wheel( prof = profile_for(f.name, profiles_dir) if prof is None: continue - if bolt_elf(f, prof, flags, strip, dry_run): - bolted += 1 - changed.append(f) + bolt_elf(f, prof, flags, strip, dry_run) + bolted += 1 + changed.append(f) if bolted == 0: log(" no matching ELFs in wheel; leaving it unchanged") @@ -256,14 +265,26 @@ def bolt_standalone_wheel( log(f"dry-run: {bolted} member(s) would be bolted; skipping repack.") return 0 + # In place would break the contract above: process_wheel() rewrites its + # argument, so a failed apply would leave the caller's input half-written + # with nothing to fall back on. + if output.resolve() == wheel.resolve(): + err("--output must differ from --wheel; the input is preserved so a " + "failed apply leaves it usable") + return 2 + output.parent.mkdir(parents=True, exist_ok=True) - if output.resolve() != wheel.resolve(): - shutil.copy2(wheel, output) - bolted = process_wheel(output, profiles_dir, DEFAULT_BOLT_FLAGS, strip, False) + shutil.copy2(wheel, output) + try: + bolted = process_wheel(output, profiles_dir, DEFAULT_BOLT_FLAGS, strip, False) + except BoltApplyError: + # Partially bolted: drop it rather than leave something uploadable that + # looks finished. + output.unlink(missing_ok=True) + raise if bolted == 0: err("no ELF in the wheel matched a profile -- nothing bolted. Check --profiles names.") - if output.resolve() != wheel.resolve(): - output.unlink(missing_ok=True) + output.unlink(missing_ok=True) return 2 log(f"Done. Bolted wheel: {output} ({bolted} lib(s) bolted, RECORD updated)") return 0 @@ -397,71 +418,80 @@ def main() -> int: workdir = Path(tempfile.mkdtemp(prefix="apply_bolt_")) if owns_workdir else args.workdir workdir.mkdir(parents=True, exist_ok=True) try: - profiles_dir = resolve_profiles(args.profiles, workdir) - log( - f"Profiles: {profiles_dir} " - f"({len(list(profiles_dir.glob('*.yaml')))} yaml, " - f"{len(list(profiles_dir.glob('*.fdata')))} fdata)" + return _apply(args, workdir) + except BoltApplyError as e: + # Already reported in detail by bolt_elf; this is the exit code, and the + # guarantee that a partial result never reaches the output path. + err(f"aborting without producing an output: {e}") + return 2 + finally: + if owns_workdir: + shutil.rmtree(workdir, ignore_errors=True) + + +def _apply(args: argparse.Namespace, workdir: Path) -> int: + profiles_dir = resolve_profiles(args.profiles, workdir) + log( + f"Profiles: {profiles_dir} " + f"({len(list(profiles_dir.glob('*.yaml')))} yaml, " + f"{len(list(profiles_dir.glob('*.fdata')))} fdata)" + ) + + if args.wheel: + return bolt_standalone_wheel( + args.wheel, args.output, profiles_dir, args.strip, args.dry_run ) - if args.wheel: - return bolt_standalone_wheel( - args.wheel, args.output, profiles_dir, args.strip, args.dry_run - ) + # Extract the tarball. + extract = workdir / "extract" + extract.mkdir(parents=True, exist_ok=True) + log(f"Extracting {args.tarball.name}") + rc, out = run(["tar", "-xf", str(args.tarball), "-C", str(extract)]) + if rc != 0: + err(f"failed to extract tarball: {out}") + return 2 + roots = [p for p in extract.iterdir() if p.is_dir()] + tree = roots[0] if len(roots) == 1 else extract + log(f"Tarball root: {tree.name}") + + if args.manifest: + verify_manifest(args.manifest, tree, args.strict) + + total = 0 + # 1) Loose ELFs in the layout (benchmarks/cpp, triton_backend, etc.). + for f in sorted(tree.rglob("*")): + if f.suffix == ".whl" or not f.is_file() or not is_elf(f): + continue + prof = profile_for(f.name, profiles_dir) + if prof is None: + continue + bolt_elf(f, prof, DEFAULT_BOLT_FLAGS, args.strip, args.dry_run) + total += 1 + + # 2) Wheel(s). + for wheel in sorted(tree.rglob("tensorrt_llm-*.whl")): + total += process_wheel( + wheel, profiles_dir, DEFAULT_BOLT_FLAGS, args.strip, args.dry_run + ) - # Extract the tarball. - extract = workdir / "extract" - extract.mkdir(parents=True, exist_ok=True) - log(f"Extracting {args.tarball.name}") - rc, out = run(["tar", "-xf", str(args.tarball), "-C", str(extract)]) - if rc != 0: - err(f"failed to extract tarball: {out}") - return 2 - roots = [p for p in extract.iterdir() if p.is_dir()] - tree = roots[0] if len(roots) == 1 else extract - log(f"Tarball root: {tree.name}") - - if args.manifest: - verify_manifest(args.manifest, tree, args.strict) - - total = 0 - # 1) Loose ELFs in the layout (benchmarks/cpp, triton_backend, etc.). - for f in sorted(tree.rglob("*")): - if f.suffix == ".whl" or not f.is_file() or not is_elf(f): - continue - prof = profile_for(f.name, profiles_dir) - if prof is None: - continue - if bolt_elf(f, prof, DEFAULT_BOLT_FLAGS, args.strip, args.dry_run): - total += 1 - - # 2) Wheel(s). - for wheel in sorted(tree.rglob("tensorrt_llm-*.whl")): - total += process_wheel( - wheel, profiles_dir, DEFAULT_BOLT_FLAGS, args.strip, args.dry_run - ) - - if total == 0: - err("no ELFs matched a profile -- nothing bolted. Check --profiles names.") - return 2 - log(f"Bolted {total} ELF(s) total (loose + wheel).") - - if args.dry_run: - log("dry-run: skipping repack.") - return 0 + if total == 0: + err("no ELFs matched a profile -- nothing bolted. Check --profiles names.") + return 2 + log(f"Bolted {total} ELF(s) total (loose + wheel).") - # Repack the tarball under the new name. - args.output.parent.mkdir(parents=True, exist_ok=True) - log(f"Repacking -> {args.output}") - rc, out = run(["tar", "-C", str(extract), "-czf", str(args.output), tree.name]) - if rc != 0: - err(f"repack failed: {out}") - return 2 - log(f"Done. Bolted tarball: {args.output}") + if args.dry_run: + log("dry-run: skipping repack.") return 0 - finally: - if owns_workdir: - shutil.rmtree(workdir, ignore_errors=True) + + # Repack the tarball under the new name. + args.output.parent.mkdir(parents=True, exist_ok=True) + log(f"Repacking -> {args.output}") + rc, out = run(["tar", "-C", str(extract), "-czf", str(args.output), tree.name]) + if rc != 0: + err(f"repack failed: {out}") + return 2 + log(f"Done. Bolted tarball: {args.output}") + return 0 if __name__ == "__main__": From ec55f29f9cfd41d81e3b78b14a60add276cab7c6 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Mon, 21 Sep 2026 16:46:07 +0000 Subject: [PATCH 07/20] [TRTLLMINF-336][infra] resolve the wheel artifact path via UPLOAD_PATH Review feedback. prepareWheelFromBuildStage built --artifact_path from env.uploadPath directly. The two agree whenever the parent passed the parameter, but an unset env interpolates to the string "null" and the download then polls .../null/. Pre-existing, and previously self-limiting: the poll timed out and the build fell back to compiling the wheel from source. This change removes that fallback for the BOLT-required SBSA path and stretches the wait to 8h, so the same typo now costs most of a day and then fails. UPLOAD_PATH already carries the job-scoped default for exactly this case. Signed-off-by: Matt Lefebvre --- jenkins/BuildDockerImage.groovy | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index 3608771cc737..41a8b6f31fc1 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -303,7 +303,13 @@ def prepareWheelFromBuildStage(dockerfileStage, arch) { // optimized tarball to wait for; x86 keeps taking the canonical one. def requireBolted = BOLT_REQUIRE_BOLTED_WHEEL && arch == "sbsa" def waitTime = requireBolted ? WAIT_TIME_FOR_BOLTED_BUILD_STAGE : WAIT_TIME_FOR_BUILD_STAGE - def wheelArgs = "--arch ${arch} --timeout ${waitTime} --artifact_path " + env.uploadPath + // UPLOAD_PATH, not env.uploadPath: they agree whenever the parent passed the + // parameter, but an unset env leaves the raw reference interpolating to + // "null" and the download polls .../null/. That used to cost a + // wasted wait before the build fell back to compiling from source; with the + // bolted wheel required there is no fallback, so it becomes an 8h wait + // followed by a failure. + def wheelArgs = "--arch ${arch} --timeout ${waitTime} --artifact_path ${UPLOAD_PATH}" if (requireBolted) { echo "Release image for ${arch} requires the BOLT-optimized build; waiting up to ${waitTime} minutes for it" wheelArgs += " --bolted" From 8d98d464442ebcb6f437b2c70293a2b17fe55815 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Mon, 21 Sep 2026 17:52:45 +0000 Subject: [PATCH 08/20] [TRTLLMINF-336][infra] optimize the image's wheel instead of waiting for one Reworks this PR from "wait for this run's bolted- tarball" to "BOLT the wheel the image installs, using the branch's last promoted bundle". The previous shape had the release image build block on BoltProfileGen: up to 8h holding a docker builder pod, and only workable post-merge, because nothing publishes a bolted- tarball in a nightly or GA run (the producer is gated on JOB_NAME matching PostMerge). It also quietly contradicted the decision to treat a producer failure as UNSTABLE -- with the image hard-requiring the artifact, a SLURM hiccup failed post-merge anyway, eight hours later, in a stage that looks unrelated to the cause. The image installs a wheel, so optimizing that wheel is the whole requirement; where the profiles come from is a separate question. Taking the last promoted bundle answers it without an in-run dependency: no wait, no race with Build-Docker-Images, and nightly and GA containers get optimized binaries for the first time. The cost is profile drift -- the bundle is from the last successful promote rather than this commit. That is the same contract Build.groovy's pre-merge consume and the Dockerfile.bolt overlay already ship with, and -infer-stale- profile means drift costs optimization quality, never correctness. Same-commit profiles are still used where they are free: the post-merge SBSA test stages run after BoltProfileGen in the same branch and keep preferring this run's tarball. Mechanically this reuses the wheel support added for the release wheel: get_wheel_from_package.py downloads the canonical tarball as before, then runs apply_latest.sh on the extracted wheel before the release stage pip installs it. Dockerfile.multi's wheel stage already COPYs scripts/, so the toolkit is in context. Branch resolution mirrors the image overlay (explicit override, then LLM_BRANCH, then main) and is fatal if none has a bundle. llvm-bolt staging moves into apply_latest.sh via a new stage_llvm_bolt.sh, since the image build cannot stage it the way the Jenkins pods do. It is a no-op when llvm-bolt is already on PATH, and honors GITHUB_MIRROR like install_ccache.sh and install_cmake.sh, which already pull release tarballs from github.com during the image build. boltRequireBoltedWheel is renamed boltOptimizeWheel to match what it now does, and the 8h wait constant is gone. Verified with a stubbed bundle pull and llvm-bolt: a branch with no bundle falls through to the next candidate, the wheel landing in build/ is optimized, its RECORD rows all match, and unprofiled libraries are untouched. Signed-off-by: Matt Lefebvre --- jenkins/BuildDockerImage.groovy | 55 +++++++++-------- scripts/bolt/internal/apply_latest.sh | 9 +++ scripts/bolt/internal/stage_llvm_bolt.sh | 79 ++++++++++++++++++++++++ scripts/get_wheel_from_package.py | 73 +++++++++++++++++----- 4 files changed, 174 insertions(+), 42 deletions(-) create mode 100644 scripts/bolt/internal/stage_llvm_bolt.sh diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index 41a8b6f31fc1..9bdba3d275d8 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -61,21 +61,15 @@ BOLT_PROFILES_REQUIRED = (params.boltProfilesRequired ?: env.boltProfilesRequire // Separate from the two above: those govern the profile BUNDLE baked in as a // thin layer (Dockerfile.bolt), which documents how to reproduce a BOLTed build // but does NOT optimize the binaries the image actually installs. This one -// governs the INSTALLED wheel -- when true, the release image must be built from -// the BOLT-optimized tarball (bolted-) rather than whichever tarball -// happens to exist first. Kept independent so it can be rolled back on its own. -BOLT_REQUIRE_BOLTED_WHEEL = (params.boltRequireBoltedWheel ?: env.boltRequireBoltedWheel ?: "false").toString() == "true" +// governs the INSTALLED wheel -- when true, the wheel unpacked from the build +// tarball is BOLT-optimized in the image build, before the release stage pip +// installs it. Kept independent so it can be rolled back on its own. +BOLT_OPTIMIZE_WHEEL = (params.boltOptimizeWheel ?: env.boltOptimizeWheel ?: "false").toString() == "true" // <<< BOLT profile-bundle overlay <<< ENABLE_USE_WHEEL_FROM_BUILD_STAGE = params.useWheelFromBuildStage ?: false WAIT_TIME_FOR_BUILD_STAGE = 60 // minutes -// The canonical tarball lands as soon as the build stage finishes; the BOLTed -// one only after BoltProfileGen's SLURM fan-out (three perf workloads at up to -// 4h walltime each, plus queue and merge -- its own stage timeout is 8h). So -// requiring the optimized build means waiting on a different, much longer clock -// than "the build stage finished". -WAIT_TIME_FOR_BOLTED_BUILD_STAGE = 480 // minutes BUILD_JOBS = "32" BUILD_JOBS_RELEASE_X86_64 = "32" @@ -299,20 +293,27 @@ def prepareWheelFromBuildStage(dockerfileStage, arch) { } def wheelScript = 'scripts/get_wheel_from_package.py' - // Only aarch64 has a promoted profile bundle, so only the SBSA image has an - // optimized tarball to wait for; x86 keeps taking the canonical one. - def requireBolted = BOLT_REQUIRE_BOLTED_WHEEL && arch == "sbsa" - def waitTime = requireBolted ? WAIT_TIME_FOR_BOLTED_BUILD_STAGE : WAIT_TIME_FOR_BUILD_STAGE // UPLOAD_PATH, not env.uploadPath: they agree whenever the parent passed the // parameter, but an unset env leaves the raw reference interpolating to - // "null" and the download polls .../null/. That used to cost a - // wasted wait before the build fell back to compiling from source; with the - // bolted wheel required there is no fallback, so it becomes an 8h wait - // followed by a failure. - def wheelArgs = "--arch ${arch} --timeout ${waitTime} --artifact_path ${UPLOAD_PATH}" - if (requireBolted) { - echo "Release image for ${arch} requires the BOLT-optimized build; waiting up to ${waitTime} minutes for it" - wheelArgs += " --bolted" + // "null" and the download then polls .../null/ until it times out. + def wheelArgs = "--arch ${arch} --timeout ${WAIT_TIME_FOR_BUILD_STAGE} --artifact_path ${UPLOAD_PATH}" + + // Only aarch64 has a promoted profile bundle, so only the SBSA image has + // anything to apply; x86 installs the wheel as built. + if (BOLT_OPTIMIZE_WHEEL && arch == "sbsa") { + // The branch whose promoted bundle to apply, resolved the same way the + // image overlay resolves it: an explicit override, else this build's own + // branch, else main. Deliberately NOT this run's own profiles -- those + // are not published until BoltProfileGen finishes, hours after this + // build starts, and waiting on them would serialize every release + // behind a multi-hour GPU job. get_wheel_from_package.py tries these in + // order and fails if none has a bundle. + def branches = [params.boltProfileBranch, LLM_BRANCH, "main"] + .collect { it?.toString()?.trim() } + .findAll { it } + .unique() + echo "Release image for ${arch} will BOLT-optimize its wheel using profiles from: ${branches.join(', ')}" + wheelArgs += " --bolt-branch ${branches.join(',')}" } return " BUILD_WHEEL_SCRIPT=${wheelScript} BUILD_WHEEL_ARGS='${wheelArgs}'" } @@ -320,10 +321,10 @@ def prepareWheelFromBuildStage(dockerfileStage, arch) { // Whether a docker build that failed WITH the downloaded-wheel args may be // retried without them. The retry rebuilds the wheel from source in-container, // which is a fine recovery for an ordinary build but silently defeats the point -// when the whole reason for the download was to install BOLT-optimized binaries -// -- the retry would produce an unoptimized release image that looks identical. +// when that build was also responsible for optimizing the wheel -- the retry +// would produce an unoptimized release image that looks identical. def mayRetryWithoutBuildStageWheel(arch) { - return !(BOLT_REQUIRE_BOLTED_WHEEL && arch == "sbsa") + return !(BOLT_OPTIMIZE_WHEEL && arch == "sbsa") } // Produce each CANONICAL image from its raw `-noprofiles` build by @@ -909,9 +910,9 @@ pipeline { description: "When boltOverlayEnabled is true, treat a missing/empty BOLT bundle as a FATAL error instead of retagging the plain build as canonical. Enable for the release/nightly path to guarantee canonical images carry profiles." ) booleanParam( - name: "boltRequireBoltedWheel", + name: "boltOptimizeWheel", defaultValue: false, - description: "Build the SBSA release image from the BOLT-optimized tarball (bolted-) instead of the canonical one, waiting for it to be published and failing if it never is. Independent of boltOverlayEnabled, which only bakes in the profile bundle and leaves the installed binaries unoptimized. Enable for the release path; ignored on x86_64, which has no promoted bundle." + description: "BOLT-optimize the wheel the SBSA release image installs, applying the branch's latest promoted profile bundle during the image build, and fail if no bundle can be applied. Independent of boltOverlayEnabled, which only bakes the bundle in as a layer and leaves the installed binaries unoptimized. Uses the last promoted bundle rather than this run's, so the image build never waits on BoltProfileGen. Ignored on x86_64, which has no promoted bundle." ) string( name: "boltProfileBranch", diff --git a/scripts/bolt/internal/apply_latest.sh b/scripts/bolt/internal/apply_latest.sh index 72f8984baf76..ddf9c071a9a4 100755 --- a/scripts/bolt/internal/apply_latest.sh +++ b/scripts/bolt/internal/apply_latest.sh @@ -56,6 +56,15 @@ fi DEST="$(mktemp -d)" trap 'rm -rf "$DEST"' EXIT +# 0) Put llvm-bolt on PATH. No-op when the caller already staged it (the Jenkins +# build pods do), which keeps this free for them and lets callers that cannot +# stage it themselves -- notably the image build, where this runs inside a +# docker layer -- just call apply_latest.sh and get a working toolchain. +if ! . "$HERE/stage_llvm_bolt.sh"; then + echo "[apply_latest] FATAL: could not stage llvm-bolt" >&2 + exit 2 +fi + # 1) Pull the branch `latest` bundle. A missing bundle is fatal here (see header): # consumption was requested but there is nothing promoted to consume. if ! bash "$HERE/artifactory.sh" pull-latest "$BRANCH" "$TRIPLE" "$DEST"; then diff --git a/scripts/bolt/internal/stage_llvm_bolt.sh b/scripts/bolt/internal/stage_llvm_bolt.sh new file mode 100644 index 000000000000..3f36b0d3024f --- /dev/null +++ b/scripts/bolt/internal/stage_llvm_bolt.sh @@ -0,0 +1,79 @@ +#!/bin/bash +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# stage_llvm_bolt.sh - put the pinned llvm-bolt on PATH, downloading it if needed. +# +# No-op when llvm-bolt is already available, so callers can invoke it +# unconditionally. Prints the bin directory it staged (nothing when the toolchain +# was already present), and exports PATH for anything sourcing it. +# +# Usage: +# . scripts/bolt/internal/stage_llvm_bolt.sh # sourced: updates PATH +# bash scripts/bolt/internal/stage_llvm_bolt.sh # executed: prints bin dir +# +# Env: +# BOLT_LLVM_STAGE_DIR where to unpack (default ./.bolt-llvm) +# GITHUB_MIRROR mirror base in place of https://github.com, matching +# docker/common/install_ccache.sh and install_cmake.sh so +# this works inside an image build +# LLVM_BOLT_VERSION overrides the pin in llvm_bolt_version.sh + +_bolt_stage_here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +_bolt_stage_llvm() { + command -v llvm-bolt >/dev/null 2>&1 && return 0 + + local dir="${BOLT_LLVM_STAGE_DIR:-$PWD/.bolt-llvm}" + if [ -x "$dir/bin/llvm-bolt" ]; then + export PATH="$dir/bin:$PATH" + return 0 + fi + + . "$_bolt_stage_here/llvm_bolt_version.sh" + + local arch + case "$(uname -m)" in + aarch64) arch=ARM64 ;; + x86_64) arch=X64 ;; + *) echo "[stage-llvm-bolt] unsupported arch $(uname -m)" >&2; return 1 ;; + esac + + local base="${GITHUB_MIRROR:-https://github.com}" + local tb="LLVM-${LLVM_BOLT_VERSION}-Linux-${arch}.tar.xz" + local url="${base}/llvm/llvm-project/releases/download/llvmorg-${LLVM_BOLT_VERSION}/${tb}" + + # Extract into a staging dir and rename, so a concurrent or interrupted run + # can never leave a half-populated toolchain that the -x check above would + # then accept. + local stage="${dir}.stage.$$" + rm -rf "$stage"; mkdir -p "$stage" + echo "[stage-llvm-bolt] staging llvm-bolt ${LLVM_BOLT_VERSION} -> $dir" >&2 + if ! curl -fSL --retry 10 --retry-all-errors --retry-delay 15 --connect-timeout 60 \ + -o "/tmp/${tb}.$$" "$url"; then + rm -rf "$stage"; rm -f "/tmp/${tb}.$$" + echo "[stage-llvm-bolt] download failed: $url" >&2 + return 1 + fi + tar -xJf "/tmp/${tb}.$$" -C "$stage" --strip-components=1 + rm -f "/tmp/${tb}.$$" + mv -T "$stage" "$dir" 2>/dev/null || rm -rf "$stage" + + [ -x "$dir/bin/llvm-bolt" ] || { echo "[stage-llvm-bolt] no llvm-bolt in $dir" >&2; return 1; } + export PATH="$dir/bin:$PATH" + echo "$dir/bin" +} + +_bolt_stage_llvm diff --git a/scripts/get_wheel_from_package.py b/scripts/get_wheel_from_package.py index eaad10edf2ed..181e38371859 100644 --- a/scripts/get_wheel_from_package.py +++ b/scripts/get_wheel_from_package.py @@ -41,27 +41,65 @@ def add_arguments(parser: ArgumentParser): type=int, default=60, help="Timeout in minutes") - parser.add_argument("--bolted", + parser.add_argument("--bolt-branch", "-b", - action="store_true", - help="Require the BOLT-optimized build " - "(bolted-) that BoltProfileGen publishes " - "alongside the plain one. The canonical tarfile " - "appears as soon as the build stage finishes, long " - "before BOLT has run, so an image that must ship " - "optimized binaries has to wait on this distinct " - "name instead. Never falls back.") - - -def get_wheel_from_package(arch, artifact_path, timeout, bolted=False): + default=None, + help="Comma-separated branches whose promoted BOLT " + "profile bundle should be applied to the extracted " + "wheel, tried in order. The image installs this wheel, " + "so optimizing it here is what makes the released " + "container carry optimized binaries. Omit to install " + "the wheel as built. Fatal if set and no branch has a " + "usable bundle.") + + +def bolt_optimize_wheels(build_dir, arch, bolt_branch): + """Apply the latest promoted BOLT bundle to each wheel in `build_dir`. + + Deliberately uses the branch's last promoted bundle rather than one + generated from this commit: the optimized tarball for THIS run is not + published until BoltProfileGen finishes, hours after the image build starts, + and waiting on it would serialize the release behind a multi-hour GPU job. + Profiles are function-name-keyed and applied with -infer-stale-profile, so a + bundle from a nearby commit costs some optimization quality, never + correctness -- the same trade the pre-merge consume path and the image + profile overlay already make. + """ + bolt_internal = get_project_dir() / "scripts" / "bolt" / "internal" + apply_latest = bolt_internal / "apply_latest.sh" + triple = "x86_64-linux-gnu" if arch == "x86_64" else "aarch64-linux-gnu" + branches = [b.strip() for b in bolt_branch.split(",") if b.strip()] + + for wheel in sorted(Path(build_dir).glob("tensorrt_llm*.whl")): + bolted = wheel.with_suffix(".whl.bolted") + for branch in branches: + print(f"Applying BOLT profiles from {branch}/{triple} to " + f"{wheel.name}") + # 3 = that branch has nothing promoted; anything else is decisive. + rc = subprocess.run( + ["bash", str(apply_latest), branch, triple, + str(wheel), str(bolted)]).returncode + if rc == 0: + os.replace(bolted, wheel) + print(f"BOLT optimized {wheel.name} ({branch}/{triple})") + break + if rc != 3: + raise RuntimeError( + f"BOLT apply failed for {wheel.name} (rc={rc})") + print(f"No promoted bundle for {branch}/{triple}; " + "trying next branch") + else: + raise RuntimeError( + f"No promoted BOLT bundle for any of {branches} ({triple}); " + f"refusing to build an unoptimized release image") + + +def get_wheel_from_package(arch, artifact_path, timeout, bolt_branch=None): if arch == "x86_64": tarfile_name = "TensorRT-LLM.tar.gz" else: tarfile_name = "TensorRT-LLM-GH200.tar.gz" - if bolted: - tarfile_name = f"bolted-{tarfile_name}" - tarfile_link = f"https://urm.nvidia.com/artifactory/{artifact_path}/{tarfile_name}" for attempt in range(timeout): try: @@ -105,6 +143,11 @@ def get_wheel_from_package(arch, artifact_path, timeout, bolted=False): if os.path.exists(tarfile_name): os.remove(tarfile_name) + # After the move, before the Dockerfile's release stage pip installs + # whatever is in build/. + if bolt_branch: + bolt_optimize_wheels(build_dir, arch, bolt_branch) + if __name__ == "__main__": parser = ArgumentParser() From b50a3d1c75ae8d14c670f7bc933577dd75709ffa Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Mon, 21 Sep 2026 17:55:47 +0000 Subject: [PATCH 09/20] [TRTLLMINF-336][infra] enable BOLT optimization of the SBSA release image wheel Follows the rework in the previous change: the parameter is boltOptimizeWheel and it BOLTs the wheel during the image build rather than waiting for this run's bolted- tarball. Two consequences worth noting for review. This no longer depends on the post-merge publication toggle (#19209) at all -- it reads the branch's last promoted bundle, so the only prerequisite is that a bundle has been promoted at some point. And the image build no longer waits on BoltProfileGen, so a producer failure stays UNSTABLE as intended instead of failing post-merge eight hours later from an unrelated-looking stage. Both launches are set together because they push the same tags: the scanned and registered artifact has to be the same image the post-merge path publishes. Rollback: drop boltOptimizeWheel from both launches, or set the job parameter to false. The image then installs the wheel as built, exactly as today. Signed-off-by: Matt Lefebvre --- jenkins/L0_MergeRequest.groovy | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index f5b0506ca5c0..cf26df525fd2 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -2527,14 +2527,16 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) 'boltOverlayEnabled': true, 'boltProfilesRequired': true, // The overlay above only bakes in the profile bundle; it - // leaves the installed wheel unoptimized. This makes the - // SBSA release image install the BOLT-optimized wheel, - // by waiting for BoltProfileGen to publish - // bolted- rather than grabbing whichever - // tarball exists first. Inert on x86_64 (no promoted - // bundle) and whenever the wheel is built from source - // rather than downloaded. - 'boltRequireBoltedWheel': true, + // leaves the installed wheel unoptimized. This BOLTs the + // wheel the SBSA release image installs, applying the + // branch's last promoted bundle during the image build. + // Deliberately not this run's profiles: those are not + // published until BoltProfileGen finishes, hours after + // this build starts, so depending on them would serialize + // every release behind a multi-hour GPU job. Inert on + // x86_64 (no promoted bundle) and whenever the wheel is + // built from source rather than downloaded. + 'boltOptimizeWheel': true, ] if (runMode == "nightly_release") { additionalParameters += [ @@ -2599,7 +2601,7 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) // appear. 'boltOverlayEnabled': true, 'boltProfilesRequired': true, - 'boltRequireBoltedWheel': true, + 'boltOptimizeWheel': true, ] if (runMode == "nightly_release") { additionalParameters += [ From ba1aa58c702d5316e95c93596f1ed8450189c0f4 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Mon, 21 Sep 2026 18:06:16 +0000 Subject: [PATCH 10/20] [TRTLLMINF-336][infra] apply ruff-format to apply_bolt.py Pre-commit's ruff-format hook. The dedent into _apply() brought the process_wheel() call under the line limit, so it collapses to one line, and the new err() call wraps differently than I wrote it. Signed-off-by: Matt Lefebvre --- scripts/bolt/apply_bolt.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/scripts/bolt/apply_bolt.py b/scripts/bolt/apply_bolt.py index e135c2abcf69..95693189ab9f 100644 --- a/scripts/bolt/apply_bolt.py +++ b/scripts/bolt/apply_bolt.py @@ -269,8 +269,10 @@ def bolt_standalone_wheel( # argument, so a failed apply would leave the caller's input half-written # with nothing to fall back on. if output.resolve() == wheel.resolve(): - err("--output must differ from --wheel; the input is preserved so a " - "failed apply leaves it usable") + err( + "--output must differ from --wheel; the input is preserved so a " + "failed apply leaves it usable" + ) return 2 output.parent.mkdir(parents=True, exist_ok=True) @@ -470,9 +472,7 @@ def _apply(args: argparse.Namespace, workdir: Path) -> int: # 2) Wheel(s). for wheel in sorted(tree.rglob("tensorrt_llm-*.whl")): - total += process_wheel( - wheel, profiles_dir, DEFAULT_BOLT_FLAGS, args.strip, args.dry_run - ) + total += process_wheel(wheel, profiles_dir, DEFAULT_BOLT_FLAGS, args.strip, args.dry_run) if total == 0: err("no ELFs matched a profile -- nothing bolted. Check --profiles names.") From c893e0637258df4cdadd16d4400da8b4d2982507 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Mon, 21 Sep 2026 18:08:29 +0000 Subject: [PATCH 11/20] [TRTLLMINF-336][infra] apply yapf to get_wheel_from_package.py Pre-commit's yapf hook (this file is yapf-formatted, not ruff). Lifting the argv into a local keeps the result readable -- yapf's split of the inline subprocess.run([...]).returncode call was not. Signed-off-by: Matt Lefebvre --- scripts/get_wheel_from_package.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/scripts/get_wheel_from_package.py b/scripts/get_wheel_from_package.py index 181e38371859..6dfc45c42f90 100644 --- a/scripts/get_wheel_from_package.py +++ b/scripts/get_wheel_from_package.py @@ -66,7 +66,7 @@ def bolt_optimize_wheels(build_dir, arch, bolt_branch): profile overlay already make. """ bolt_internal = get_project_dir() / "scripts" / "bolt" / "internal" - apply_latest = bolt_internal / "apply_latest.sh" + apply_latest = str(bolt_internal / "apply_latest.sh") triple = "x86_64-linux-gnu" if arch == "x86_64" else "aarch64-linux-gnu" branches = [b.strip() for b in bolt_branch.split(",") if b.strip()] @@ -75,10 +75,13 @@ def bolt_optimize_wheels(build_dir, arch, bolt_branch): for branch in branches: print(f"Applying BOLT profiles from {branch}/{triple} to " f"{wheel.name}") + cmd = [ + "bash", apply_latest, branch, triple, + str(wheel), + str(bolted) + ] # 3 = that branch has nothing promoted; anything else is decisive. - rc = subprocess.run( - ["bash", str(apply_latest), branch, triple, - str(wheel), str(bolted)]).returncode + rc = subprocess.run(cmd).returncode if rc == 0: os.replace(bolted, wheel) print(f"BOLT optimized {wheel.name} ({branch}/{triple})") From 35fa28d17a06a5a3219dbe23fe284690df93d921 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Wed, 23 Sep 2026 19:48:48 +0000 Subject: [PATCH 12/20] [TRTLLMINF-336][infra] pin one BOLT profile bundle per pipeline Profile bundles are promoted to a mutable pointer, latest.tar.gz, and every consumer resolves it independently at whatever moment it happens to run. Four do so today: pre-merge tarball consume, the release wheel, the image profile overlay, and the image's installed wheel. Post-merge promotes a new bundle mid-pipeline, so a consumer that pulls before the promote and one that pulls after legitimately get different profiles -- and parallel post-merge pipelines widen the window. Nothing errors; the run is simply green having tested something other than what it shipped. Make the bundle reference a value the pipeline decides once and passes down. Absent means "use latest", so nothing changes for anyone who does not opt in. Every promoted bundle already exists under an immutable versioned name beside the pointer, so there is nothing new to publish: bolt-profile--.tar.gz immutable latest.tar.gz mutable pointer promoteBundle now labels latest.tar.gz with bolt.ref via matrix parameters, in the SAME request that uploads it -- a separate pointer object would leave a window where the label and the bytes disagree, which is the class of bug this change exists to remove. artifactory.sh gains resolve-latest to read that label back, and pull-latest honours BOLT_PROFILE_REF. Threading the pin as an environment variable rather than an argument means no call-site signatures change. L0_MergeRequest resolves once in preparation() -- the first point with both a node and a checkout, since the script-level BOLT setup runs outside any node and cannot shell out -- and stores it in globalVars. Pinning is post-merge only: pre-merge keeps reading latest, whose drift is acceptable, and returning "" leaves it on exactly today's code path. Resolution is best-effort; a branch with nothing promoted, or a bundle predating the label, leaves the pipeline unpinned rather than failing. An unresolvable pin is a reason to keep the old behaviour, not to break the build. BOLT_PROFILE_REF is declared AND pre-seeded in Build, L0_Test and BuildDockerImage. updateMapWithJson only updates keys already present in the target map, so a consumer missing the key would silently run unpinned -- the exact failure mode being fixed. Build.groovy already carries a comment about being bitten by this with bolt_consume_build. Nothing consumes the pin yet, so this is inert beyond a new stage logging the resolved ref. Verified against urm: resolve-latest returns d79ecd55 for main/aarch64-linux-gnu (via the manifest fallback, as no bundle carries the label yet), the versioned object exists at exactly the name pull-latest constructs, and an unpromoted branch yields empty output and exit 0. Signed-off-by: Matt Lefebvre --- jenkins/BoltProfileGen.groovy | 10 +++- jenkins/Build.groovy | 7 +++ jenkins/BuildDockerImage.groovy | 7 +++ jenkins/L0_MergeRequest.groovy | 56 ++++++++++++++++++++++ jenkins/L0_Test.groovy | 7 +++ scripts/bolt/internal/artifactory.sh | 72 +++++++++++++++++++++++++--- 6 files changed, 150 insertions(+), 9 deletions(-) diff --git a/jenkins/BoltProfileGen.groovy b/jenkins/BoltProfileGen.groovy index 77b612e1c138..7ff288e7be1b 100644 --- a/jenkins/BoltProfileGen.groovy +++ b/jenkins/BoltProfileGen.groovy @@ -760,7 +760,13 @@ def promoteBundle(pipeline, remote, String bundle) cat > "${netrc}" chmod 600 "${netrc}" curl -fsS --netrc-file "${netrc}" --retry 5 --retry-all-errors -T "${bundle}" "${base}/${bundleName}" - curl -fsS --netrc-file "${netrc}" --retry 5 --retry-all-errors -T "${bundle}" "${base}/latest.tar.gz" + # Matrix parameters label latest.tar.gz with the ref it now points at, so a + # consumer can resolve "which bundle is current" with a small metadata GET + # instead of downloading the bundle to read its manifest. Set in the SAME + # request as the upload, so the label and the bytes can never disagree -- + # a separate pointer object would leave a window where they do. + curl -fsS --netrc-file "${netrc}" --retry 5 --retry-all-errors -T "${bundle}" \\ + "${base}/latest.tar.gz;bolt.ref=${BOLT_REF};bolt.branch=${BRANCH};bolt.triple=${TRIPLE}" """.stripIndent() pipeline.withCredentials([pipeline.usernamePassword(credentialsId: 'urm-artifactory-creds', usernameVariable: 'ART_USER', passwordVariable: 'ART_PASS')]) { @@ -769,7 +775,7 @@ def promoteBundle(pipeline, remote, String bundle) Utils.exec(pipeline, timeout: false, numRetries: 2, noNVDFEvent: true, script: feed + Utils.sshUserCmd(remote, b64BashRemoteCmdStdin(promote, "${bundle}.promote.sh"))) } - pipeline.echo("Promoted. latest = ${base}/latest.tar.gz") + pipeline.echo("Promoted. latest = ${base}/latest.tar.gz (bolt.ref=${BOLT_REF})") } // --------------------------------------------------------------------------- diff --git a/jenkins/Build.groovy b/jenkins/Build.groovy index cccabe235ccd..df2e8dd7a126 100644 --- a/jenkins/Build.groovy +++ b/jenkins/Build.groovy @@ -143,6 +143,8 @@ def ACTION_INFO = "action_info" def TRTLLM_VERSION_OVERRIDE = "trtllm_version_override" @Field def BOLT_CONSUME_BUILD = "bolt_consume_build" +@Field +def BOLT_PROFILE_REF = "bolt_profile_ref" def globalVars = [ (GITHUB_PR_API_URL): null, (CACHED_CHANGED_FILE_LIST): null, @@ -152,6 +154,11 @@ def globalVars = [ // that helper only updates keys already present in the target map, so a key that // is absent here (like this one previously) is silently dropped during the merge. (BOLT_CONSUME_BUILD): false, + // Pre-declared so updateMapWithJson() populates it from the parent: that + // helper only updates keys already present here, so an absent key is + // silently dropped -- which for this one would mean running unpinned + // without saying so. + (BOLT_PROFILE_REF): "", ] // TODO: Move common variables to an unified location diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index 6c3dbefa350f..535476de9762 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -80,12 +80,19 @@ def ACTION_INFO = "action_info" def IMAGE_KEY_TO_TAG = "image_key_to_tag" @Field def TRTLLM_VERSION_OVERRIDE = "trtllm_version_override" +@Field +def BOLT_PROFILE_REF = "bolt_profile_ref" def globalVars = [ (GITHUB_PR_API_URL): null, (CACHED_CHANGED_FILE_LIST): null, (ACTION_INFO): null, (IMAGE_KEY_TO_TAG): [:], (TRTLLM_VERSION_OVERRIDE): null, + // Pre-declared so updateMapWithJson() populates it from the parent: that + // helper only updates keys already present here, so an absent key is + // silently dropped -- which for this one would mean running unpinned + // without saying so. + (BOLT_PROFILE_REF): "", ] @Field diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index 372e781abaee..77beafc28e6c 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -261,6 +261,13 @@ def RUN_MODE = "run_mode" def BUILD_BRANCH = "build_branch" @Field def BOLT_CONSUME_BUILD = "bolt_consume_build" +// The one BOLT profile bundle every consumer in this pipeline applies. Resolved +// once (see resolveBoltProfileRef) rather than each consumer reading the mutable +// `latest` pointer whenever it happens to run -- which made two stages in the +// same pipeline able to disagree about which profiles they used. Empty means +// unpinned, i.e. today's read-`latest`-per-consumer behaviour. +@Field +def BOLT_PROFILE_REF = "bolt_profile_ref" @Field def RELEASE_TARGET = "release_target" def globalVars = [ @@ -273,6 +280,7 @@ def globalVars = [ (RUN_MODE): runMode, (RELEASE_TARGET): runMode == "nightly_release" ? normalizeReleaseTargets(gitlabParamsFromBot.get(RELEASE_TARGET, null)) : [], + (BOLT_PROFILE_REF): "", ] globalVars[BUILD_BRANCH] = resolveBuildBranch(globalVars) // Compare against "true" rather than relying on Groovy truthiness: the bot phrase @@ -597,6 +605,14 @@ def launchReleaseCheck(pipeline, globalVars) trtllm_utils.checkoutSource(LLM_REPO, env.gitlabCommit, LLM_ROOT, true, true) sh "cd ${LLM_ROOT} && git config --unset-all core.hooksPath" + // Pin the BOLT profile bundle for the whole pipeline, before any consumer + // can resolve `latest` on its own. Resolved here because this is the first + // point with both a node and a checkout -- the script-level BOLT setup + // above runs outside any node and cannot shell out. + stage("Pin BOLT Profile Bundle") { + globalVars[BOLT_PROFILE_REF] = resolveBoltProfileRef(globalVars[BUILD_BRANCH]) + } + // Step 2: Run guardwords scan def isOfficialPostMergeJob = (env.JOB_NAME ==~ /.*PostMerge.*/) if (env.alternativeTRT || isOfficialPostMergeJob) { @@ -1926,6 +1942,46 @@ def resolveBuildBranch(globalVars) // // Skips are announced rather than silent: a run that asked for BOLTed binaries // and did not get them should say so in the log. +// Pick the single BOLT profile bundle this pipeline will use, and pin it. +// +// Without a pin every consumer resolves `latest` independently, at whatever +// moment it happens to run. Post-merge promotes a new bundle mid-pipeline, so a +// stage that pulls before the promote and one that pulls after legitimately get +// different profiles -- and parallel post-merge pipelines make it worse. That is +// invisible: each consumer succeeds, and the run is green having tested +// something other than what it shipped. +// +// Pinning is post-merge only. Pre-merge keeps reading `latest` (requirement: its +// drift is acceptable and its behaviour must not change), and returning "" leaves +// every consumer on exactly today's path. +// +// Best-effort by design. A branch with nothing promoted, or a bundle too old to +// carry the label, yields "" and the pipeline runs unpinned rather than failing +// -- an unresolvable pin is a reason to keep the old behaviour, not to break the +// build. +def resolveBoltProfileRef(String branch, String triple = "aarch64-linux-gnu") +{ + if (!(env.JOB_NAME ==~ /.*PostMerge.*/)) { + return "" + } + def ref = "" + try { + ref = sh(returnStdout: true, script: """ + bash ${LLM_ROOT}/scripts/bolt/internal/artifactory.sh \ + resolve-latest ${branch} ${triple} 2>/dev/null || true + """).trim() + } catch (Exception e) { + echo "BOLT profile pin: resolve failed (${e.message}); running unpinned." + return "" + } + if (!ref) { + echo "BOLT profile pin: nothing promoted for ${branch}/${triple}; running unpinned." + return "" + } + echo "BOLT profile pin: ${branch}/${triple} -> ${ref}. Every consumer in this pipeline uses this bundle." + return ref +} + def resolveBoltConsume(boolean requested, String targetBranch) { if (!requested) { diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index 71b0557898e8..2a68749b283a 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -3155,6 +3155,8 @@ def IMAGE_KEY_TO_TAG = "image_key_to_tag" def TRTLLM_VERSION_OVERRIDE = "trtllm_version_override" @Field def RUN_MODE = "run_mode" +@Field +def BOLT_PROFILE_REF = "bolt_profile_ref" def globalVars = [ (GITHUB_PR_API_URL): null, (CACHED_CHANGED_FILE_LIST): null, @@ -3162,6 +3164,11 @@ def globalVars = [ (IMAGE_KEY_TO_TAG): [:], (TRTLLM_VERSION_OVERRIDE): null, (RUN_MODE): null, + // Pre-declared so updateMapWithJson() populates it from the parent: that + // helper only updates keys already present here, so an absent key is + // silently dropped -- which for this one would mean running unpinned + // without saying so. + (BOLT_PROFILE_REF): "", ] class GlobalState { diff --git a/scripts/bolt/internal/artifactory.sh b/scripts/bolt/internal/artifactory.sh index 84a4152cb2ad..109ee382a840 100755 --- a/scripts/bolt/internal/artifactory.sh +++ b/scripts/bolt/internal/artifactory.sh @@ -20,8 +20,15 @@ # bundle for an OLD ref without repointing latest at it. Used by POSTMERGE. # # pull-latest -# Download + extract latest.tar.gz for / into . -# Used by PREMERGE (and postmerge fallback is intentionally NOT provided). +# Download + extract a bundle for / into . +# Honours BOLT_PROFILE_REF: when set, pulls the immutable +# bolt-profile--.tar.gz instead of the mutable latest.tar.gz, +# so every consumer in a pipeline can be pinned to one bundle. Unset keeps +# today's behaviour. +# +# resolve-latest +# Print the ref latest.tar.gz currently points at, for a caller that wants +# to pin the rest of a pipeline to it. Prints nothing if undeterminable. # # The premerge "override" case does NOT use promote: the gen recipe packages a # bundle locally and apply consumes it directly (run-scoped, never promoted). @@ -116,12 +123,62 @@ cmd_promote() { fi } +# Print the ref that /'s latest.tar.gz currently points at, or +# nothing when it cannot be determined. Read-only, anonymous, cheap. +# +# `promote` labels latest.tar.gz with a bolt.ref property in the same request that +# uploads it, so the label and the bytes can never disagree. Bundles promoted +# before that existed carry no property, hence the manifest fallback: download the +# bundle (~1 MB) and read `ref` out of manifest.json. A caller that gets nothing +# back should stay unpinned rather than guess. +cmd_resolve_latest() { + local branch="${1:?resolve-latest: }" + local triple="${2:?triple required}" + local base="${BOLT_ARTIFACTORY_BASE:-https://urm.nvidia.com/artifactory}" + local path; path="$(promote_dir "$branch" "$triple")/latest.tar.gz" + + local props ref="" + props="$(curl -fsSL --retry 3 --retry-all-errors --connect-timeout 30 \ + "$base/api/storage/$path?properties" 2>/dev/null || true)" + # { "properties": { "bolt.ref": [ "" ] } } -- fixed, tiny shape, so a sed + # extraction avoids depending on python3 or jq, neither of which is guaranteed + # on every pod that might call this. + ref="$(printf '%s' "$props" \ + | tr -d ' \n' \ + | sed -n 's/.*"bolt\.ref":\["\([^"]*\)"\].*/\1/p')" + if [[ -n "$ref" ]]; then + printf '%s\n' "$ref" + return 0 + fi + + log "no bolt.ref property on $path; falling back to the bundle manifest" + local tmp; tmp="$(mktemp -d)" + if curl -fsSL --retry 3 --retry-all-errors --connect-timeout 60 \ + -o "$tmp/latest.tar.gz" "$base/$path" 2>/dev/null \ + && tar -xzf "$tmp/latest.tar.gz" -C "$tmp" manifest.json 2>/dev/null; then + ref="$(tr -d ' \n' < "$tmp/manifest.json" \ + | sed -n 's/.*"ref":"\([^"]*\)".*/\1/p')" + fi + rm -rf "$tmp" + [[ -n "$ref" ]] && printf '%s\n' "$ref" + return 0 +} + cmd_pull_latest() { local branch="${1:?pull-latest: }" local triple="${2:?triple required}" local dest="${3:?dest_dir required}" local base="${BOLT_ARTIFACTORY_BASE:-https://urm.nvidia.com/artifactory}" - local url="$base/$(promote_dir "$branch" "$triple")/latest.tar.gz" + # BOLT_PROFILE_REF pins this pull to one immutable bundle. Threaded as an env + # var rather than an argument so every existing call site keeps working: a + # caller opts into pinning by exporting it, and unset means today's behaviour. + local url + if [[ -n "${BOLT_PROFILE_REF:-}" ]]; then + url="$base/$(promote_dir "$branch" "$triple")/bolt-profile-${BOLT_PROFILE_REF}-${triple}.tar.gz" + log "Pinned to ${BOLT_PROFILE_REF}" + else + url="$base/$(promote_dir "$branch" "$triple")/latest.tar.gz" + fi mkdir -p "$dest" log "Pulling $url -> $dest" # Anonymous download (like jenkins/Build.groovy's tarball from the same @@ -148,8 +205,9 @@ cmd_pull_latest() { } case "${1:-}" in - package) shift; cmd_package "$@" ;; - promote) shift; cmd_promote "$@" ;; - pull-latest) shift; cmd_pull_latest "$@" ;; - *) die "usage: artifactory.sh {package|promote|pull-latest} ..." ;; + package) shift; cmd_package "$@" ;; + promote) shift; cmd_promote "$@" ;; + pull-latest) shift; cmd_pull_latest "$@" ;; + resolve-latest) shift; cmd_resolve_latest "$@" ;; + *) die "usage: artifactory.sh {package|promote|pull-latest|resolve-latest} ..." ;; esac From ada11f40b7667125a5a18fff774d86e07d4e3f49 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Wed, 23 Sep 2026 20:54:48 +0000 Subject: [PATCH 13/20] [TRTLLMINF-336][infra] Release wheel and image wheel honour the pinned bundle The pipeline pins one profile bundle, and the build and test stages already use it, but the two release consumers did not: both resolved a branch's mutable `latest` independently. A promotion landing mid-pipeline was therefore enough to ship a wheel optimized with different profiles than the build that was tested -- the misalignment the pin exists to remove. Both now take BOLT_PROFILE_REF from globalVars and pass it to apply_latest.sh, and both fail loudly when the pinned bundle cannot be fetched rather than falling back to latest, since a silent fallback here is how an unexpectedly optimized wheel reaches PyPI or a released container. The candidate-branch walk is kept under a pin. The pin is resolved upstream against BUILD_BRANCH, which is not the expression seeding either consumer's candidate list, so narrowing to one entry would search the wrong promote directory. Walking is safe because the ref is a commit SHA: the same ref under another branch's directory is the bundle built from that same commit. Unpinned runs are unchanged -- an empty ref means apply_latest.sh resolves latest exactly as before, which is what pre-merge does. Signed-off-by: Matt Lefebvre --- jenkins/BuildDockerImage.groovy | 13 ++++++++++++ jenkins/L0_Test.groovy | 33 +++++++++++++++++++++++++----- scripts/get_wheel_from_package.py | 34 +++++++++++++++++++++++++++---- 3 files changed, 71 insertions(+), 9 deletions(-) diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index 0a6f66c64c73..f03847241a81 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -65,6 +65,10 @@ BOLT_PROFILES_REQUIRED = (params.boltProfilesRequired ?: env.boltProfilesRequire // tarball is BOLT-optimized in the image build, before the release stage pip // installs it. Kept independent so it can be rolled back on its own. BOLT_OPTIMIZE_WHEEL = (params.boltOptimizeWheel ?: env.boltOptimizeWheel ?: "false").toString() == "true" +// The bundle this pipeline pinned, hoisted out of globalVars in launchBuildJobs +// because prepareWheelFromBuildStage runs well below the scope globalVars is +// passed into. Empty means unpinned, i.e. take whatever `latest` is. +BOLT_PINNED_REF = "" // <<< BOLT profile-bundle overlay <<< ENABLE_USE_WHEEL_FROM_BUILD_STAGE = params.useWheelFromBuildStage ?: false @@ -321,6 +325,14 @@ def prepareWheelFromBuildStage(dockerfileStage, arch) { .unique() echo "Release image for ${arch} will BOLT-optimize its wheel using profiles from: ${branches.join(', ')}" wheelArgs += " --bolt-branch ${branches.join(',')}" + // With a pin the candidate list collapses to its first entry: the ref + // names one immutable bundle under one branch's promote directory, so + // falling through to another branch would optimize the image's wheel + // with different profiles than the release wheel and the tested build. + if (BOLT_PINNED_REF) { + echo "Release image for ${arch} is pinned to BOLT bundle ${BOLT_PINNED_REF}" + wheelArgs += " --bolt-profile-ref ${BOLT_PINNED_REF}" + } } return " BUILD_WHEEL_SCRIPT=${wheelScript} BUILD_WHEEL_ARGS='${wheelArgs}'" } @@ -694,6 +706,7 @@ def buildImage(config, imageKeyToTag, versionOverride) def launchBuildJobs(pipeline, globalVars, imageKeyToTag) { def versionOverride = globalVars[TRTLLM_VERSION_OVERRIDE] ?: "" + BOLT_PINNED_REF = globalVars[BOLT_PROFILE_REF]?.toString() ?: "" def defaultBuildConfig = [ target: "tritondevel", action: params.action, diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index eb5f195b9985..38b99be8a801 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -5720,7 +5720,7 @@ def checkKitmakerWheelDryRun(pipeline, kitmakerDryRunMetadata) // a missing bundle as a skip; it has to, because the same switch covers x86_64, // where nothing is promoted yet. This path is aarch64-only and the switch is // main-only, so there is always a bundle to find. -def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) +def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, String boltProfileRef = "") { def llvmArch = (cpu_arch == AARCH64_TRIPLE) ? "ARM64" : "X64" // apply_latest.sh resolves exactly one branch, so try the build's own branch @@ -5731,6 +5731,16 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) .collect { it?.toString()?.trim() } .findAll { it } .unique() + // The candidate walk stays as-is under a pin. The pin is resolved upstream + // against BUILD_BRANCH (env.gitlabBranch), which is not the expression that + // seeds this list, so narrowing to one entry here would look in the wrong + // promote directory. Walking is safe precisely because the ref is a commit + // SHA: the same ref under another branch's directory is the bundle built + // from that same commit, so whichever candidate resolves it, the profiles + // are the ones this pipeline pinned. + if (boltProfileRef) { + echo "[bolt-wheel] pinned to BOLT bundle ${boltProfileRef}" + } stage("BOLT release wheel") { sh """ @@ -5756,6 +5766,7 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) for (b in branches) { rc = sh(returnStatus: true, script: """ export PATH="\$PWD/.bolt-llvm/bin:\$PATH" + export BOLT_PROFILE_REF='${boltProfileRef}' bash tensorrt_llm/scripts/bolt/internal/apply_latest.sh \ ${b} ${cpu_arch} ${wheel} ${wheel}.bolted """) @@ -5765,6 +5776,11 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) } echo "[bolt-wheel] no promoted bundle for ${b}/${cpu_arch}; trying next candidate branch" } + if (rc == 3 && boltProfileRef) { + error("[bolt-wheel] pinned BOLT bundle ${boltProfileRef} (${cpu_arch}) not found under any of " + + "${branches.join(', ')}. This pipeline pinned a bundle the release wheel cannot fetch; " + + "refusing to publish a wheel optimized with anything else.") + } if (rc == 3) { error("[bolt-wheel] no promoted BOLT bundle for any of ${branches.join(', ')} (${cpu_arch}); " + "refusing to upload an unoptimized release wheel (promote a bundle via BoltProfileGen, " + @@ -5774,7 +5790,8 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch) error("[bolt-wheel] apply_latest.sh failed (rc=${rc}) for ${appliedFrom}/${cpu_arch}") } sh "mv -f ${wheel}.bolted ${wheel}" - echo "[bolt-wheel] ${wheel} is now BOLTed (profiles from ${appliedFrom}/${cpu_arch})" + echo "[bolt-wheel] ${wheel} is now BOLTed (profiles from ${appliedFrom}/${cpu_arch}" + + (boltProfileRef ? ", bundle ${boltProfileRef})" : ")") } } @@ -5787,7 +5804,8 @@ def runLLMBuild( cpver="cp312", plat_name="", is_dlfw=false, - boltConsume=false) + boltConsume=false, + boltProfileRef="") { sh "pwd && ls -alh" sh "env | sort" @@ -5850,7 +5868,7 @@ def runLLMBuild( // inside an already-released image to prove that image can still build from // source; optimizing it would prove nothing and only add a failure mode. if (boltConsume && cpu_arch == AARCH64_TRIPLE && !wheel_path) { - applyLatestBoltToWheel(pipeline, "tensorrt_llm/build/${wheelName}", cpu_arch) + applyLatestBoltToWheel(pipeline, "tensorrt_llm/build/${wheelName}", cpu_arch, boltProfileRef) } def rootWheelUploadPath = "${cpu_arch}/${wheel_path}" @@ -7026,11 +7044,16 @@ def launchTestJobs(pipeline, testFilter, globalVars) // Same switch the build helpers use for the tarball, so the released // wheel and the released tarball are never optimized differently. def boltConsume = globalVars[BOLT_CONSUME_BUILD]?.toString() == "true" + // The bundle this pipeline pinned, so the released wheel is optimized + // with the same profiles the tested build was. Empty means unpinned: + // apply_latest.sh then takes whatever `latest` is, i.e. today's + // behaviour. + def boltProfileRef = globalVars[BOLT_PROFILE_REF]?.toString() ?: "" buildRunner("[${toStageName(values[1], key)}] Build") { wheelPath = runLLMBuild( pipeline, cpu_arch, values[3], "", versionOverride, cpver, - values[7], isDlfw, boltConsume) + values[7], isDlfw, boltConsume, boltProfileRef) } // TODO: Re-enable the sanity check after updating GPU testers' driver version. diff --git a/scripts/get_wheel_from_package.py b/scripts/get_wheel_from_package.py index 6dfc45c42f90..0f3d673e2951 100644 --- a/scripts/get_wheel_from_package.py +++ b/scripts/get_wheel_from_package.py @@ -51,9 +51,17 @@ def add_arguments(parser: ArgumentParser): "container carry optimized binaries. Omit to install " "the wheel as built. Fatal if set and no branch has a " "usable bundle.") + parser.add_argument("--bolt-profile-ref", + default=None, + help="Pin to one immutable profile bundle by ref, so " + "the image's wheel is optimized with the same profiles " + "as the tested build and the released wheel. Collapses " + "--bolt-branch to its first entry, since a ref names " + "one bundle under one branch. Omit to take whatever is " + "currently promoted as latest.") -def bolt_optimize_wheels(build_dir, arch, bolt_branch): +def bolt_optimize_wheels(build_dir, arch, bolt_branch, bolt_profile_ref=None): """Apply the latest promoted BOLT bundle to each wheel in `build_dir`. Deliberately uses the branch's last promoted bundle rather than one @@ -69,6 +77,15 @@ def bolt_optimize_wheels(build_dir, arch, bolt_branch): apply_latest = str(bolt_internal / "apply_latest.sh") triple = "x86_64-linux-gnu" if arch == "x86_64" else "aarch64-linux-gnu" branches = [b.strip() for b in bolt_branch.split(",") if b.strip()] + env = os.environ.copy() + if bolt_profile_ref: + # The branch walk is kept under a pin. The ref is a commit SHA, so the + # same ref under another branch's promote directory is the bundle built + # from that same commit -- whichever candidate resolves it, the profiles + # are the pinned ones. Narrowing the list would instead risk looking in + # a directory the pin was never resolved against. + env["BOLT_PROFILE_REF"] = bolt_profile_ref + print(f"Pinned to BOLT bundle {bolt_profile_ref}") for wheel in sorted(Path(build_dir).glob("tensorrt_llm*.whl")): bolted = wheel.with_suffix(".whl.bolted") @@ -81,7 +98,7 @@ def bolt_optimize_wheels(build_dir, arch, bolt_branch): str(bolted) ] # 3 = that branch has nothing promoted; anything else is decisive. - rc = subprocess.run(cmd).returncode + rc = subprocess.run(cmd, env=env).returncode if rc == 0: os.replace(bolted, wheel) print(f"BOLT optimized {wheel.name} ({branch}/{triple})") @@ -92,12 +109,21 @@ def bolt_optimize_wheels(build_dir, arch, bolt_branch): print(f"No promoted bundle for {branch}/{triple}; " "trying next branch") else: + if bolt_profile_ref: + raise RuntimeError( + f"Pinned BOLT bundle {bolt_profile_ref} ({triple}) not " + f"found under any of {branches}; refusing to build an " + f"image whose wheel is optimized with anything else") raise RuntimeError( f"No promoted BOLT bundle for any of {branches} ({triple}); " f"refusing to build an unoptimized release image") -def get_wheel_from_package(arch, artifact_path, timeout, bolt_branch=None): +def get_wheel_from_package(arch, + artifact_path, + timeout, + bolt_branch=None, + bolt_profile_ref=None): if arch == "x86_64": tarfile_name = "TensorRT-LLM.tar.gz" else: @@ -149,7 +175,7 @@ def get_wheel_from_package(arch, artifact_path, timeout, bolt_branch=None): # After the move, before the Dockerfile's release stage pip installs # whatever is in build/. if bolt_branch: - bolt_optimize_wheels(build_dir, arch, bolt_branch) + bolt_optimize_wheels(build_dir, arch, bolt_branch, bolt_profile_ref) if __name__ == "__main__": From a24893198dc16faa0a3e51e10af32f395a3c75a8 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Wed, 23 Sep 2026 21:00:25 +0000 Subject: [PATCH 14/20] [TRTLLMINF-336][infra] Resolve the BOLT pin before the pipeline forks The pin was resolved inside launchReleaseCheck, which is one parallel branch of launchStages. launchJob serializes globalVars per child job, so build jobs in the sibling branches race the assignment: whichever serializes first receives an empty pin and resolves `latest` on its own. That is the drift the pin exists to remove, reintroduced with the added property of being nondeterministic. Release Check is also skippable, which would leave the whole pipeline unpinned. Resolve in preparation() instead, which completes before launchStages forks anything. It is the earliest correct point: the script-level BOLT setup runs outside any node and cannot shell out, while Setup Environment has already checked the repo out into LLM_ROOT, where the resolver's script lives. Also corrects the pin log line. It claimed every consumer in the pipeline uses the bundle, which is not yet true -- the image profile-bundle overlay still resolves latest independently. Signed-off-by: Matt Lefebvre --- jenkins/L0_MergeRequest.groovy | 27 ++++++++++++++++++--------- 1 file changed, 18 insertions(+), 9 deletions(-) diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index 77beafc28e6c..3ad1b1e404fd 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -569,6 +569,22 @@ def preparation(pipeline, testFilter, globalVars) stage("Setup Environment") { setupPipelineEnvironment(pipeline, testFilter, globalVars) } + // Must resolve BEFORE launchStages, not inside any branch of it. + // launchStages forks Release-Check, SBSA and x86_64 in parallel and + // launchJob serializes globalVars per child job, so a pin assigned in + // one branch races the jobs in the others: whichever serializes first + // receives an empty pin and resolves `latest` on its own -- the drift + // this exists to remove, now with the added property of being + // nondeterministic. Release Check is also skippable, which would leave + // the whole pipeline unpinned. + // + // preparation() is the earliest correct home: the script-level BOLT + // setup runs outside any node and cannot shell out, while the stage + // above has already checked the repo out into ${LLM_ROOT}, which is + // where the resolver's script lives. + stage("Pin BOLT Profile Bundle") { + globalVars[BOLT_PROFILE_REF] = resolveBoltProfileRef(globalVars[BUILD_BRANCH]) + } stage("Upload Build Info") { try { def branch = globalVars[BUILD_BRANCH] @@ -605,14 +621,6 @@ def launchReleaseCheck(pipeline, globalVars) trtllm_utils.checkoutSource(LLM_REPO, env.gitlabCommit, LLM_ROOT, true, true) sh "cd ${LLM_ROOT} && git config --unset-all core.hooksPath" - // Pin the BOLT profile bundle for the whole pipeline, before any consumer - // can resolve `latest` on its own. Resolved here because this is the first - // point with both a node and a checkout -- the script-level BOLT setup - // above runs outside any node and cannot shell out. - stage("Pin BOLT Profile Bundle") { - globalVars[BOLT_PROFILE_REF] = resolveBoltProfileRef(globalVars[BUILD_BRANCH]) - } - // Step 2: Run guardwords scan def isOfficialPostMergeJob = (env.JOB_NAME ==~ /.*PostMerge.*/) if (env.alternativeTRT || isOfficialPostMergeJob) { @@ -1978,7 +1986,8 @@ def resolveBoltProfileRef(String branch, String triple = "aarch64-linux-gnu") echo "BOLT profile pin: nothing promoted for ${branch}/${triple}; running unpinned." return "" } - echo "BOLT profile pin: ${branch}/${triple} -> ${ref}. Every consumer in this pipeline uses this bundle." + echo "BOLT profile pin: ${branch}/${triple} -> ${ref}. Consumers honouring the pin use this bundle; " + + "the image profile-bundle overlay still resolves latest independently." return ref } From e4ab2481ba607cbd4e845e7cb94cb10cffde0bc7 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Wed, 23 Sep 2026 21:03:03 +0000 Subject: [PATCH 15/20] [TRTLLMINF-336][infra] Image profile-bundle overlay honours the pinned bundle The overlay bakes a profile bundle into the released image as a thin layer, and it was the last consumer still resolving `latest` at whatever moment it ran. A released image could therefore document a bundle that nothing else in the run was built from. Export BOLT_PROFILE_REF on the pull-latest step. With all four consumers now pinned -- build tarball, release wheel, installed image wheel and this overlay -- the pin log line's claim that every consumer uses the bundle is true again, so the caveat added alongside the resolver move is removed. The candidate-branch walk is kept, for the same reason as the other consumers: the ref is a commit SHA, so the same ref under another branch's promote directory is the bundle built from that same commit. Found by CodeRabbit on #19603. Signed-off-by: Matt Lefebvre --- jenkins/BuildDockerImage.groovy | 11 +++++++++++ jenkins/L0_MergeRequest.groovy | 3 +-- 2 files changed, 12 insertions(+), 2 deletions(-) diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index f03847241a81..9d885d00186c 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -395,10 +395,21 @@ def overlayBoltBundle(pairs, arch, action) { // (manifest + >=1 profile) so "pulled but empty" is not accepted. First hit wins. def haveBundle = false def branch = null + // The overlay is a consumer like any other, so it takes the pipeline's pin + // rather than resolving `latest` when it happens to run. Without this the + // released image could carry a profile bundle that no other artifact in the + // run was built from. Empty means unpinned and pull-latest behaves exactly + // as before. The candidate walk is kept: the ref is a commit SHA, so the + // same ref under another branch's promote directory is the bundle built + // from that same commit. + if (BOLT_PINNED_REF) { + echo "[BOLT] overlay pinned to bundle ${BOLT_PINNED_REF}" + } for (cand in candidates) { for (int attempt = 1; attempt <= 3 && !haveBundle; attempt++) { def rc = sh(script: """ rm -rf ${ctxDir} && mkdir -p ${ctxDir}/${bundleSub} && \ + export BOLT_PROFILE_REF='${BOLT_PINNED_REF}' && \ cd ${LLM_ROOT} && bash scripts/bolt/internal/artifactory.sh pull-latest ${cand} ${triple} ${ctxDir}/${bundleSub} """, returnStatus: true) if (rc == 0) { diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index 6ecc2d2d669a..c4f9df0f6859 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -1986,8 +1986,7 @@ def resolveBoltProfileRef(String branch, String triple = "aarch64-linux-gnu") echo "BOLT profile pin: nothing promoted for ${branch}/${triple}; running unpinned." return "" } - echo "BOLT profile pin: ${branch}/${triple} -> ${ref}. Consumers honouring the pin use this bundle; " + - "the image profile-bundle overlay still resolves latest independently." + echo "BOLT profile pin: ${branch}/${triple} -> ${ref}. Every consumer in this pipeline uses this bundle." return ref } From 5f02c6443ca0d5f1dee8fafa44e57db571614558 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Thu, 24 Sep 2026 03:35:40 +0000 Subject: [PATCH 16/20] [TRTLLMINF-336][infra] Carry the pin's branch with its ref; harden the resolver Review findings on #19602 and #19603. A ref does not address an object on its own: the path is //bolt-profile--.tar.gz. Consumers were supplying the branch themselves, from four expressions that agree on post-merge main and nowhere guaranteed else -- resolveBuildBranch here, gitlabTargetBranch in Build.groovy, and two different candidate walks in the wheel and overlay paths. A disagreement fetches nothing, and apply_latest.sh reports a failed pull as "nothing promoted", which is indistinguishable from a cold start and therefore degrades silently to unpinned. Publish the branch alongside the ref so a pinned consumer never guesses; it is left empty unless a ref resolved, so a consumer cannot half-honour a pin. resolveBoltProfileRef caught bare Exception. Aborts and timeouts arrive as FlowInterruptedException, which extends InterruptedException, so cancelling a pipeline was logged as a failed lookup and the run continued unpinned and launched every build job. preparation() already rethrows interrupts a few lines away; follow it. The resolver's curls had --connect-timeout but no --max-time, so a connected but stalled transfer had no bound. This runs before any job launches, so that stalls the whole pipeline instead of falling back to unpinned. Also separates resolveBoltConsume's doc comment from resolveBoltProfileRef's; they had fused into one block that read as documentation for the wrong function. Signed-off-by: Matt Lefebvre --- jenkins/Build.groovy | 3 +++ jenkins/BuildDockerImage.groovy | 3 +++ jenkins/L0_MergeRequest.groovy | 25 ++++++++++++++++++++++++- jenkins/L0_Test.groovy | 3 +++ scripts/bolt/internal/artifactory.sh | 8 ++++++-- 5 files changed, 39 insertions(+), 3 deletions(-) diff --git a/jenkins/Build.groovy b/jenkins/Build.groovy index df2e8dd7a126..33134aceb52b 100644 --- a/jenkins/Build.groovy +++ b/jenkins/Build.groovy @@ -145,6 +145,8 @@ def TRTLLM_VERSION_OVERRIDE = "trtllm_version_override" def BOLT_CONSUME_BUILD = "bolt_consume_build" @Field def BOLT_PROFILE_REF = "bolt_profile_ref" +@Field +def BOLT_PROFILE_BRANCH = "bolt_profile_branch" def globalVars = [ (GITHUB_PR_API_URL): null, (CACHED_CHANGED_FILE_LIST): null, @@ -159,6 +161,7 @@ def globalVars = [ // silently dropped -- which for this one would mean running unpinned // without saying so. (BOLT_PROFILE_REF): "", + (BOLT_PROFILE_BRANCH): "", ] // TODO: Move common variables to an unified location diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index 535476de9762..762ad5d3261a 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -82,6 +82,8 @@ def IMAGE_KEY_TO_TAG = "image_key_to_tag" def TRTLLM_VERSION_OVERRIDE = "trtllm_version_override" @Field def BOLT_PROFILE_REF = "bolt_profile_ref" +@Field +def BOLT_PROFILE_BRANCH = "bolt_profile_branch" def globalVars = [ (GITHUB_PR_API_URL): null, (CACHED_CHANGED_FILE_LIST): null, @@ -93,6 +95,7 @@ def globalVars = [ // silently dropped -- which for this one would mean running unpinned // without saying so. (BOLT_PROFILE_REF): "", + (BOLT_PROFILE_BRANCH): "", ] @Field diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index 3ad1b1e404fd..6cffdf2683e3 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -268,6 +268,15 @@ def BOLT_CONSUME_BUILD = "bolt_consume_build" // unpinned, i.e. today's read-`latest`-per-consumer behaviour. @Field def BOLT_PROFILE_REF = "bolt_profile_ref" +// The branch whose promote directory holds that bundle. A ref alone does not +// address an object -- the path is //bolt-profile-- +// -- and consumers used to supply the branch themselves from four different +// expressions. They agree on post-merge main and nowhere guaranteed else, and a +// disagreement is a 404, which apply_latest.sh reports as "nothing promoted": +// indistinguishable from a cold start, so it degrades silently. Carrying the +// branch with the ref means a pinned consumer never has to guess. +@Field +def BOLT_PROFILE_BRANCH = "bolt_profile_branch" @Field def RELEASE_TARGET = "release_target" def globalVars = [ @@ -281,6 +290,7 @@ def globalVars = [ (RELEASE_TARGET): runMode == "nightly_release" ? normalizeReleaseTargets(gitlabParamsFromBot.get(RELEASE_TARGET, null)) : [], (BOLT_PROFILE_REF): "", + (BOLT_PROFILE_BRANCH): "", ] globalVars[BUILD_BRANCH] = resolveBuildBranch(globalVars) // Compare against "true" rather than relying on Groovy truthiness: the bot phrase @@ -583,7 +593,12 @@ def preparation(pipeline, testFilter, globalVars) // above has already checked the repo out into ${LLM_ROOT}, which is // where the resolver's script lives. stage("Pin BOLT Profile Bundle") { - globalVars[BOLT_PROFILE_REF] = resolveBoltProfileRef(globalVars[BUILD_BRANCH]) + def pinBranch = globalVars[BUILD_BRANCH] + def pinRef = resolveBoltProfileRef(pinBranch) + globalVars[BOLT_PROFILE_REF] = pinRef + // Only meaningful alongside a ref, so it is left empty when the pin + // does not resolve -- a consumer cannot then half-honour a pin. + globalVars[BOLT_PROFILE_BRANCH] = pinRef ? pinBranch : "" } stage("Upload Build Info") { try { @@ -1950,6 +1965,8 @@ def resolveBuildBranch(globalVars) // // Skips are announced rather than silent: a run that asked for BOLTed binaries // and did not get them should say so in the log. +// (Everything above documents resolveBoltConsume, defined further down.) + // Pick the single BOLT profile bundle this pipeline will use, and pin it. // // Without a pin every consumer resolves `latest` independently, at whatever @@ -1978,6 +1995,12 @@ def resolveBoltProfileRef(String branch, String triple = "aarch64-linux-gnu") bash ${LLM_ROOT}/scripts/bolt/internal/artifactory.sh \ resolve-latest ${branch} ${triple} 2>/dev/null || true """).trim() + } catch (InterruptedException e) { + // Aborts and timeouts reach here as FlowInterruptedException, which + // extends InterruptedException. Letting the catch below swallow one + // would turn a cancelled pipeline into an unpinned pipeline that + // carries on and launches every build job. + throw e } catch (Exception e) { echo "BOLT profile pin: resolve failed (${e.message}); running unpinned." return "" diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index 2a68749b283a..da4bedb4608b 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -3157,6 +3157,8 @@ def TRTLLM_VERSION_OVERRIDE = "trtllm_version_override" def RUN_MODE = "run_mode" @Field def BOLT_PROFILE_REF = "bolt_profile_ref" +@Field +def BOLT_PROFILE_BRANCH = "bolt_profile_branch" def globalVars = [ (GITHUB_PR_API_URL): null, (CACHED_CHANGED_FILE_LIST): null, @@ -3169,6 +3171,7 @@ def globalVars = [ // silently dropped -- which for this one would mean running unpinned // without saying so. (BOLT_PROFILE_REF): "", + (BOLT_PROFILE_BRANCH): "", ] class GlobalState { diff --git a/scripts/bolt/internal/artifactory.sh b/scripts/bolt/internal/artifactory.sh index 109ee382a840..903d6ac06ffa 100755 --- a/scripts/bolt/internal/artifactory.sh +++ b/scripts/bolt/internal/artifactory.sh @@ -138,7 +138,11 @@ cmd_resolve_latest() { local path; path="$(promote_dir "$branch" "$triple")/latest.tar.gz" local props ref="" - props="$(curl -fsSL --retry 3 --retry-all-errors --connect-timeout 30 \ + # --max-time as well as --connect-timeout: the caller runs this in + # preparation(), before any job launches, so a stalled transfer (connected, + # then silent) would hold the entire pipeline rather than fall back to + # unpinned. Bound the whole request, retries included. + props="$(curl -fsSL --retry 3 --retry-all-errors --connect-timeout 30 --max-time 60 \ "$base/api/storage/$path?properties" 2>/dev/null || true)" # { "properties": { "bolt.ref": [ "" ] } } -- fixed, tiny shape, so a sed # extraction avoids depending on python3 or jq, neither of which is guaranteed @@ -153,7 +157,7 @@ cmd_resolve_latest() { log "no bolt.ref property on $path; falling back to the bundle manifest" local tmp; tmp="$(mktemp -d)" - if curl -fsSL --retry 3 --retry-all-errors --connect-timeout 60 \ + if curl -fsSL --retry 3 --retry-all-errors --connect-timeout 60 --max-time 300 \ -o "$tmp/latest.tar.gz" "$base/$path" 2>/dev/null \ && tar -xzf "$tmp/latest.tar.gz" -C "$tmp" manifest.json 2>/dev/null; then ref="$(tr -d ' \n' < "$tmp/manifest.json" \ From 6e241f9391e40557c6e76e53b29db0270f16e4ea Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Thu, 24 Sep 2026 03:41:09 +0000 Subject: [PATCH 17/20] [TRTLLMINF-336][infra] Consumers take the branch from the pin; quote apply args The release wheel, the image profile overlay and the image's installed wheel each walked their own candidate branch list to locate the pinned bundle. A ref does not address an object on its own, so that walk was three more chances to fetch something other than what the pipeline pinned. The pin now carries its branch, so when one is set the walk collapses to it. Also quotes the arguments to apply_latest.sh in the release-wheel step. The branch comes from job environment and the wheel name from a directory listing, and both were interpolated bare, so a shell metacharacter in either would run before apply_latest.sh started. Signed-off-by: Matt Lefebvre --- jenkins/BuildDockerImage.groovy | 26 ++++++++++++++----------- jenkins/L0_Test.groovy | 32 +++++++++++++++++-------------- scripts/get_wheel_from_package.py | 10 ++++------ 3 files changed, 37 insertions(+), 31 deletions(-) diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index 5bb91120881d..af38c1d74244 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -69,6 +69,8 @@ BOLT_OPTIMIZE_WHEEL = (params.boltOptimizeWheel ?: env.boltOptimizeWheel ?: "fal // because prepareWheelFromBuildStage runs well below the scope globalVars is // passed into. Empty means unpinned, i.e. take whatever `latest` is. BOLT_PINNED_REF = "" +// The branch that pin lives under. Set only together with the ref. +BOLT_PINNED_BRANCH = "" // <<< BOLT profile-bundle overlay <<< ENABLE_USE_WHEEL_FROM_BUILD_STAGE = params.useWheelFromBuildStage ?: false @@ -326,16 +328,17 @@ def prepareWheelFromBuildStage(dockerfileStage, arch) { .collect { it?.toString()?.trim() } .findAll { it } .unique() - echo "Release image for ${arch} will BOLT-optimize its wheel using profiles from: ${branches.join(', ')}" - wheelArgs += " --bolt-branch ${branches.join(',')}" - // With a pin the candidate list collapses to its first entry: the ref - // names one immutable bundle under one branch's promote directory, so + // Pinned, the candidate list collapses to the branch the pin came from: + // the ref names one immutable bundle under one promote directory, so // falling through to another branch would optimize the image's wheel // with different profiles than the release wheel and the tested build. - if (BOLT_PINNED_REF) { - echo "Release image for ${arch} is pinned to BOLT bundle ${BOLT_PINNED_REF}" + if (BOLT_PINNED_REF && BOLT_PINNED_BRANCH) { + branches = [BOLT_PINNED_BRANCH] wheelArgs += " --bolt-profile-ref ${BOLT_PINNED_REF}" + echo "Release image for ${arch} is pinned to BOLT bundle ${BOLT_PINNED_REF} on ${BOLT_PINNED_BRANCH}" } + echo "Release image for ${arch} will BOLT-optimize its wheel using profiles from: ${branches.join(', ')}" + wheelArgs += " --bolt-branch ${branches.join(',')}" } return " BUILD_WHEEL_SCRIPT=${wheelScript} BUILD_WHEEL_ARGS='${wheelArgs}'" } @@ -402,11 +405,11 @@ def overlayBoltBundle(pairs, arch, action) { // rather than resolving `latest` when it happens to run. Without this the // released image could carry a profile bundle that no other artifact in the // run was built from. Empty means unpinned and pull-latest behaves exactly - // as before. The candidate walk is kept: the ref is a commit SHA, so the - // same ref under another branch's promote directory is the bundle built - // from that same commit. - if (BOLT_PINNED_REF) { - echo "[BOLT] overlay pinned to bundle ${BOLT_PINNED_REF}" + // as before; pinned, the pin supplies its own branch and the candidate walk + // collapses to it, because the ref names one object under one directory. + if (BOLT_PINNED_REF && BOLT_PINNED_BRANCH) { + candidates = [BOLT_PINNED_BRANCH] + echo "[BOLT] overlay pinned to bundle ${BOLT_PINNED_REF} on ${BOLT_PINNED_BRANCH}" } for (cand in candidates) { for (int attempt = 1; attempt <= 3 && !haveBundle; attempt++) { @@ -721,6 +724,7 @@ def buildImage(config, imageKeyToTag, versionOverride) def launchBuildJobs(pipeline, globalVars, imageKeyToTag) { def versionOverride = globalVars[TRTLLM_VERSION_OVERRIDE] ?: "" BOLT_PINNED_REF = globalVars[BOLT_PROFILE_REF]?.toString() ?: "" + BOLT_PINNED_BRANCH = globalVars[BOLT_PROFILE_BRANCH]?.toString() ?: "" def defaultBuildConfig = [ target: "tritondevel", action: params.action, diff --git a/jenkins/L0_Test.groovy b/jenkins/L0_Test.groovy index 4f10757128e0..22e5d0c4bd24 100644 --- a/jenkins/L0_Test.groovy +++ b/jenkins/L0_Test.groovy @@ -5723,7 +5723,8 @@ def checkKitmakerWheelDryRun(pipeline, kitmakerDryRunMetadata) // a missing bundle as a skip; it has to, because the same switch covers x86_64, // where nothing is promoted yet. This path is aarch64-only and the switch is // main-only, so there is always a bundle to find. -def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, String boltProfileRef = "") +def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, String boltProfileRef = "", + String boltProfileBranch = "") { def llvmArch = (cpu_arch == AARCH64_TRIPLE) ? "ARM64" : "X64" // apply_latest.sh resolves exactly one branch, so try the build's own branch @@ -5734,15 +5735,12 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, String boltP .collect { it?.toString()?.trim() } .findAll { it } .unique() - // The candidate walk stays as-is under a pin. The pin is resolved upstream - // against BUILD_BRANCH (env.gitlabBranch), which is not the expression that - // seeds this list, so narrowing to one entry here would look in the wrong - // promote directory. Walking is safe precisely because the ref is a commit - // SHA: the same ref under another branch's directory is the bundle built - // from that same commit, so whichever candidate resolves it, the profiles - // are the ones this pipeline pinned. - if (boltProfileRef) { - echo "[bolt-wheel] pinned to BOLT bundle ${boltProfileRef}" + // A pin now supplies its own branch, so nothing is left to guess: the ref + // names one object under that branch's promote directory. Walking candidates + // would only add ways to fetch something other than what was pinned. + if (boltProfileRef && boltProfileBranch) { + branches = [boltProfileBranch] + echo "[bolt-wheel] pinned to BOLT bundle ${boltProfileRef} on ${boltProfileBranch}" } stage("BOLT release wheel") { @@ -5770,8 +5768,11 @@ def applyLatestBoltToWheel(pipeline, String wheel, String cpu_arch, String boltP rc = sh(returnStatus: true, script: """ export PATH="\$PWD/.bolt-llvm/bin:\$PATH" export BOLT_PROFILE_REF='${boltProfileRef}' + # Quoted: the branch comes from job env and the wheel name from a + # directory listing, so an unquoted expansion would let a shell + # metacharacter in either run before apply_latest.sh starts. bash tensorrt_llm/scripts/bolt/internal/apply_latest.sh \ - ${b} ${cpu_arch} ${wheel} ${wheel}.bolted + '${b}' '${cpu_arch}' '${wheel}' '${wheel}.bolted' """) if (rc != 3) { appliedFrom = b @@ -5808,7 +5809,8 @@ def runLLMBuild( plat_name="", is_dlfw=false, boltConsume=false, - boltProfileRef="") + boltProfileRef="", + boltProfileBranch="") { sh "pwd && ls -alh" sh "env | sort" @@ -5871,7 +5873,8 @@ def runLLMBuild( // inside an already-released image to prove that image can still build from // source; optimizing it would prove nothing and only add a failure mode. if (boltConsume && cpu_arch == AARCH64_TRIPLE && !wheel_path) { - applyLatestBoltToWheel(pipeline, "tensorrt_llm/build/${wheelName}", cpu_arch, boltProfileRef) + applyLatestBoltToWheel(pipeline, "tensorrt_llm/build/${wheelName}", cpu_arch, + boltProfileRef, boltProfileBranch) } def rootWheelUploadPath = "${cpu_arch}/${wheel_path}" @@ -7052,11 +7055,12 @@ def launchTestJobs(pipeline, testFilter, globalVars) // apply_latest.sh then takes whatever `latest` is, i.e. today's // behaviour. def boltProfileRef = globalVars[BOLT_PROFILE_REF]?.toString() ?: "" + def boltProfileBranch = globalVars[BOLT_PROFILE_BRANCH]?.toString() ?: "" buildRunner("[${toStageName(values[1], key)}] Build") { wheelPath = runLLMBuild( pipeline, cpu_arch, values[3], "", versionOverride, cpver, - values[7], isDlfw, boltConsume, boltProfileRef) + values[7], isDlfw, boltConsume, boltProfileRef, boltProfileBranch) } // TODO: Re-enable the sanity check after updating GPU testers' driver version. diff --git a/scripts/get_wheel_from_package.py b/scripts/get_wheel_from_package.py index 0f3d673e2951..9fdf1607c8c0 100644 --- a/scripts/get_wheel_from_package.py +++ b/scripts/get_wheel_from_package.py @@ -79,13 +79,11 @@ def bolt_optimize_wheels(build_dir, arch, bolt_branch, bolt_profile_ref=None): branches = [b.strip() for b in bolt_branch.split(",") if b.strip()] env = os.environ.copy() if bolt_profile_ref: - # The branch walk is kept under a pin. The ref is a commit SHA, so the - # same ref under another branch's promote directory is the bundle built - # from that same commit -- whichever candidate resolves it, the profiles - # are the pinned ones. Narrowing the list would instead risk looking in - # a directory the pin was never resolved against. + # The caller already narrowed --bolt-branch to the branch the pin was + # resolved against, so this list is normally a single entry. Setting the + # ref makes apply_latest.sh fetch that exact bundle. env["BOLT_PROFILE_REF"] = bolt_profile_ref - print(f"Pinned to BOLT bundle {bolt_profile_ref}") + print(f"Pinned to BOLT bundle {bolt_profile_ref} on {branches}") for wheel in sorted(Path(build_dir).glob("tensorrt_llm*.whl")): bolted = wheel.with_suffix(".whl.bolted") From 125ae012b496ed75f5ec474114d1268884f3405c Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Thu, 24 Sep 2026 03:42:21 +0000 Subject: [PATCH 18/20] [TRTLLMINF-336][infra] Un-inert the release-image wheel path This PR enables boltOptimizeWheel, but that flag is read inside prepareWheelFromBuildStage(), whose first statement returns early unless ENABLE_USE_WHEEL_FROM_BUILD_STAGE is set. That reads params.useWheelFromBuildStage, which has never been declared in the parameters{} block, so it has always been null and the function has returned "" on every run since 6a9b4b11be (Aug 2025) -- a temporary kill switch for nvbug 5433581 that was never reverted. So as it stood this PR turned on a switch that does nothing: the nightly_release build would pass boltOptimizeWheel: true, log nothing unusual, and install an unoptimized wheel. Declare the parameter and set it explicitly at both launch sites, so the answer lives in the repo rather than in Jenkins job config. CAUTION FOR REVIEW: this re-enables a path disabled for nvbug 5433581. That bug must be confirmed resolved before this merges. If it is not, the alternative is to re-anchor the BOLT step outside prepareWheelFromBuildStage() so it stops inheriting an unrelated gate. Signed-off-by: Matt Lefebvre --- jenkins/BuildDockerImage.groovy | 5 +++++ jenkins/L0_MergeRequest.groovy | 12 +++++++++++- 2 files changed, 16 insertions(+), 1 deletion(-) diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index af38c1d74244..836b39af6b60 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -939,6 +939,11 @@ pipeline { defaultValue: false, description: "When boltOverlayEnabled is true, treat a missing/empty BOLT bundle as a FATAL error instead of retagging the plain build as canonical. Enable for the release/nightly path to guarantee canonical images carry profiles." ) + booleanParam( + name: "useWheelFromBuildStage", + defaultValue: false, + description: "Install the wheel from the build-stage tarball instead of compiling one inside the image. Read since Aug 2025 but never DECLARED, so params.useWheelFromBuildStage was always null and prepareWheelFromBuildStage() returned early on every run -- see nvbug 5433581, whose temporary kill switch was never reverted. Declaring it puts the decision in the repo. boltOptimizeWheel depends on this path running." + ) booleanParam( name: "boltOptimizeWheel", defaultValue: false, diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index 5156cec74398..cd57a688d4f1 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -2628,13 +2628,20 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) // x86_64 (no promoted bundle) and whenever the wheel is // built from source rather than downloaded. 'boltOptimizeWheel': true, + // boltOptimizeWheel is read inside + // prepareWheelFromBuildStage(), which returns early + // unless this is set -- so without it the image + // silently installs an unoptimized wheel and logs + // nothing unusual. The parameter was read but never + // declared from Aug 2025 (nvbug 5433581) until it was + // declared alongside this line. + 'useWheelFromBuildStage': true, ] if (runMode == "nightly_release") { additionalParameters += [ 'buildInternalRelease': false, 'buildCiImage': false, 'buildNgcRelease': true, - 'useWheelFromBuildStage': false, 'wait_success_seconds': "", ] } @@ -2693,6 +2700,9 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) 'boltOverlayEnabled': true, 'boltProfilesRequired': true, 'boltOptimizeWheel': true, + // Same reason as the launch above: boltOptimizeWheel + // does nothing unless the build-stage wheel path runs. + 'useWheelFromBuildStage': true, ] if (runMode == "nightly_release") { additionalParameters += [ From 439e9ac07beae71fc2d160ffa4f002140e28db4e Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Thu, 24 Sep 2026 03:49:37 +0000 Subject: [PATCH 19/20] [TRTLLMINF-336][infra] Route the BOLTed image wheel around two dead gates The previous commit enabled useWheelFromBuildStage to un-inert this path. That is not enough, and relying on it is not safe. Optimizing the wheel an image installs is only possible on this path, because get_wheel_from_package.py BOLTs the wheel as it unpacks it from the build tarball -- an image that compiles its own wheel in-container has nothing to optimize. So prepareWheelFromBuildStage's two early returns are load-bearing for BOLT, and both are unreliable for reasons unrelated to it: useWheelFromBuildStage is a kill switch added for nvbug 5433581 in Aug 2025 and never reverted. It was read without being declared, so it was not merely false, it was incapable of being true. triggerType is read from env and is NOT a declared parameter of this job, so whether it survives the launch depends on job registration rather than on anything in this repo. When it does not, TRIGGER_TYPE is "manual" and the second gate returns early whatever the caller asked for. The same is true of defaultTag and runSanityCheck, both also read-but-undeclared, which is why this is not a safe thing to depend on. boltOptimizeWheel IS declared, so it reliably arrives. Decide the BOLT requirement from it before either gate and let the request carry itself past both. The bypass is narrow on purpose -- one arch, one dockerfile stage, only when BOLT was asked for -- so nvbug 5433581 is reopened for the SBSA release image alone rather than for every image this job builds, and it announces itself in the log instead of departing quietly from the parameters it was given. useWheelFromBuildStage stays declared, stays false, and nothing sets it: the global re-enable from the previous commit is reverted. mayRetryWithoutBuildStageWheel already refuses the source-rebuild retry when BOLT is required, so the bypass cannot be undone by a retry. Signed-off-by: Matt Lefebvre --- jenkins/BuildDockerImage.groovy | 50 +++++++++++++++++++++++++++------ jenkins/L0_MergeRequest.groovy | 12 +------- 2 files changed, 43 insertions(+), 19 deletions(-) diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index 836b39af6b60..8ab3fd2fad06 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -288,14 +288,48 @@ def createKubernetesPodConfig(type, arch = "amd64", build_wheel = false) def prepareWheelFromBuildStage(dockerfileStage, arch) { - if (!ENABLE_USE_WHEEL_FROM_BUILD_STAGE) { - echo "useWheelFromBuildStage is false, skip preparing wheel from build stage" - return "" - } + // Whether THIS image has to ship a BOLT-optimized wheel. Answered before the + // gates below, and deliberately not subject to them. + // + // Optimizing the installed wheel is only possible on this path: the wheel is + // BOLTed by get_wheel_from_package.py as it is unpacked from the build + // tarball, so an image that compiles its own wheel in-container has nothing + // to optimize. That makes the two gates below load-bearing for BOLT, and + // both are unreliable for reasons that have nothing to do with BOLT: + // + // useWheelFromBuildStage -- a kill switch added for nvbug 5433581 in Aug + // 2025 and never reverted. It was also read without being declared, so + // it was not merely false but incapable of being true. + // triggerType -- read from env and not a declared parameter of this job, + // so whether it survives the launch is a property of job registration + // rather than of this repo. When it does not, TRIGGER_TYPE is "manual" + // and this returns early no matter what the caller asked for. + // + // boltOptimizeWheel, by contrast, IS declared, so it reliably arrives. Key + // off it and let a BOLT request carry itself past both gates. The bypass is + // deliberately narrow -- one arch, one dockerfile stage, only when BOLT was + // asked for -- so the nvbug's blast radius stays at the SBSA release image + // instead of being reopened for every image this job builds. + def boltWheelRequired = BOLT_OPTIMIZE_WHEEL && arch == "sbsa" && dockerfileStage == "release" + + if (!boltWheelRequired) { + if (!ENABLE_USE_WHEEL_FROM_BUILD_STAGE) { + echo "useWheelFromBuildStage is false, skip preparing wheel from build stage" + return "" + } - if (!(TRIGGER_TYPE in ["post-merge", "nightly-release"])) { - echo "Trigger type does not use the build stage wheel" - return "" + if (!(TRIGGER_TYPE in ["post-merge", "nightly-release"])) { + echo "Trigger type does not use the build stage wheel" + return "" + } + } else if (!ENABLE_USE_WHEEL_FROM_BUILD_STAGE || + !(TRIGGER_TYPE in ["post-merge", "nightly-release"])) { + // Say so rather than doing it quietly: this is the one place the image + // build departs from what its parameters literally asked for. + echo "[BOLT] boltOptimizeWheel is set for the ${arch} release image, so the " + + "build-stage wheel path runs even though useWheelFromBuildStage=" + + "${ENABLE_USE_WHEEL_FROM_BUILD_STAGE} and triggerType=${TRIGGER_TYPE} would " + + "otherwise skip it. Optimizing the installed wheel is not possible any other way." } if (!dockerfileStage || !arch) { @@ -942,7 +976,7 @@ pipeline { booleanParam( name: "useWheelFromBuildStage", defaultValue: false, - description: "Install the wheel from the build-stage tarball instead of compiling one inside the image. Read since Aug 2025 but never DECLARED, so params.useWheelFromBuildStage was always null and prepareWheelFromBuildStage() returned early on every run -- see nvbug 5433581, whose temporary kill switch was never reverted. Declaring it puts the decision in the repo. boltOptimizeWheel depends on this path running." + description: "Install the wheel from the build-stage tarball instead of compiling one inside the image. Read since Aug 2025 but never DECLARED, so params.useWheelFromBuildStage was always null and prepareWheelFromBuildStage() returned early on every run -- see nvbug 5433581, whose temporary kill switch was never reverted. Declared here so the flag is at least capable of being set; it stays false by default, and nothing turns it on. boltOptimizeWheel does NOT depend on it: a BOLT request carries itself past this gate for the SBSA release image only." ) booleanParam( name: "boltOptimizeWheel", diff --git a/jenkins/L0_MergeRequest.groovy b/jenkins/L0_MergeRequest.groovy index cd57a688d4f1..5156cec74398 100644 --- a/jenkins/L0_MergeRequest.groovy +++ b/jenkins/L0_MergeRequest.groovy @@ -2628,20 +2628,13 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) // x86_64 (no promoted bundle) and whenever the wheel is // built from source rather than downloaded. 'boltOptimizeWheel': true, - // boltOptimizeWheel is read inside - // prepareWheelFromBuildStage(), which returns early - // unless this is set -- so without it the image - // silently installs an unoptimized wheel and logs - // nothing unusual. The parameter was read but never - // declared from Aug 2025 (nvbug 5433581) until it was - // declared alongside this line. - 'useWheelFromBuildStage': true, ] if (runMode == "nightly_release") { additionalParameters += [ 'buildInternalRelease': false, 'buildCiImage': false, 'buildNgcRelease': true, + 'useWheelFromBuildStage': false, 'wait_success_seconds': "", ] } @@ -2700,9 +2693,6 @@ def launchStages(pipeline, reuseBuild, testFilter, enableFailFast, globalVars) 'boltOverlayEnabled': true, 'boltProfilesRequired': true, 'boltOptimizeWheel': true, - // Same reason as the launch above: boltOptimizeWheel - // does nothing unless the build-stage wheel path runs. - 'useWheelFromBuildStage': true, ] if (runMode == "nightly_release") { additionalParameters += [ From e3c1c72d7ad364fcaa2ea3a3925ede6e3661c1a7 Mon Sep 17 00:00:00 2001 From: Matt Lefebvre Date: Thu, 1 Oct 2026 07:06:47 +0000 Subject: [PATCH 20/20] [TRTLLMINF-336][infra] Name the overlay in the pin hoist comment The hoist exists for every consumer below launchBuildJobs' scope, and the overlay is the one that predates this PR, so naming it keeps the comment true on the pin PR this work is stacked behind as well. Signed-off-by: Matt Lefebvre --- jenkins/BuildDockerImage.groovy | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/jenkins/BuildDockerImage.groovy b/jenkins/BuildDockerImage.groovy index af38c1d74244..1f7bde8d65a7 100644 --- a/jenkins/BuildDockerImage.groovy +++ b/jenkins/BuildDockerImage.groovy @@ -66,8 +66,8 @@ BOLT_PROFILES_REQUIRED = (params.boltProfilesRequired ?: env.boltProfilesRequire // installs it. Kept independent so it can be rolled back on its own. BOLT_OPTIMIZE_WHEEL = (params.boltOptimizeWheel ?: env.boltOptimizeWheel ?: "false").toString() == "true" // The bundle this pipeline pinned, hoisted out of globalVars in launchBuildJobs -// because prepareWheelFromBuildStage runs well below the scope globalVars is -// passed into. Empty means unpinned, i.e. take whatever `latest` is. +// because overlayBoltBundle runs well below the scope globalVars is passed into. +// Empty means unpinned, i.e. take whatever `latest` is. BOLT_PINNED_REF = "" // The branch that pin lives under. Set only together with the ref. BOLT_PINNED_BRANCH = ""