From 329118e1789a745d0658dbf64887dab7f741b880 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?R=C3=A9mi=20Vezy?= Date: Mon, 21 Sep 2026 10:46:39 +0200 Subject: [PATCH 1/5] Release unused benchmark sample results before the next run --- .github/workflows/FullPerformance.yml | 4 ++- benchmark/performance_regression.jl | 20 +++++++----- benchmark/test/runtests.jl | 47 +++++++++++++++++++++++++-- 3 files changed, 60 insertions(+), 11 deletions(-) diff --git a/.github/workflows/FullPerformance.yml b/.github/workflows/FullPerformance.yml index 6f7169e5b..58f131a76 100644 --- a/.github/workflows/FullPerformance.yml +++ b/.github/workflows/FullPerformance.yml @@ -116,10 +116,12 @@ jobs: "PlantBiophysics benchmark (API smoke|performance)" - name: Run the complete XPalm performance and correctness matrix + # Leave time to upload checkpoint CSVs if the complete matrix stalls. + timeout-minutes: 35 run: >- julia --project=benchmark --color=yes benchmark/test/runtests.jl - "XPalm staged performance profile full" + "XPalm (performance measurement lifetime smoke|staged performance profile full)" - name: Persist measurements and the resolved environment if: always() diff --git a/benchmark/performance_regression.jl b/benchmark/performance_regression.jl index 64d0305d3..5e6400e52 100644 --- a/benchmark/performance_regression.jl +++ b/benchmark/performance_regression.jl @@ -235,19 +235,23 @@ function _measure_performance_stage!( _checkpoint_performance_records(checkpoint_path, records) rethrow() end - measurements = Any[measurement] + # Keep the first result for correctness checks, but retain only statistics + # from later samples. A full output history can be much larger than the + # model itself and must be collectible before the next sample starts. + times = [measurement.time] + memories = [measurement.bytes] + allocations = [_performance_allocation_count(measurement)] for _ in 2:samples sample_operation = isnothing(sample_factory) ? operation : sample_factory() - push!( - measurements, - _timed_performance_operation(sample_operation), - ) + sample_measurement = _timed_performance_operation(sample_operation) + push!(times, sample_measurement.time) + push!(memories, sample_measurement.bytes) + push!(allocations, _performance_allocation_count(sample_measurement)) + sample_measurement = nothing + sample_operation = nothing end - times = getproperty.(measurements, :time) - memories = getproperty.(measurements, :bytes) - allocations = _performance_allocation_count.(measurements) _performance_record!( records, metadata, diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index f285e380c..50e56553f 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -1071,9 +1071,51 @@ if benchmark_test_enabled("XPalm benchmark API smoke") end end +if benchmark_test_enabled("XPalm performance measurement lifetime smoke") + @testset "XPalm performance measurement lifetime smoke" begin + isdefined(@__MODULE__, :_measure_performance_stage!) || + include(joinpath(@__DIR__, "..", "performance_regression.jl")) + calls = Ref(0) + first_result = Ref{Any}(nothing) + sample_results = WeakRef[] + second_released_before_third = Ref(false) + operation = () -> begin + calls[] += 1 + if calls[] == 3 + GC.gc(true) + second_released_before_third[] = sample_results[2].value === nothing + end + result = Ref(calls[]) + if calls[] == 1 + first_result[] = result + end + push!(sample_results, WeakRef(result)) + return result + end + records = NamedTuple[] + result = _measure_performance_stage!( + operation, records, NamedTuple(), :smoke, :result_lifetime; + samples=3, + ) + @test result === first_result[] + @test result[] == 1 + @test calls[] == length(sample_results) == 3 + @test second_released_before_third[] + metrics = Dict(row.metric => row.value for row in records) + @test metrics["samples"] == 3 + @test metrics["minimum_time"] <= metrics["median_time"] + @test metrics["minimum_time"] <= metrics["wall_time"] + @test metrics["minimum_memory"] <= metrics["median_memory"] + @test metrics["minimum_memory"] <= metrics["allocated"] + @test metrics["minimum_allocations"] <= metrics["median_allocations"] + @test metrics["minimum_allocations"] <= metrics["allocations"] + end +end + if benchmark_test_enabled("XPalm staged performance profile smoke") @testset "XPalm staged performance profile smoke" begin - include(joinpath(@__DIR__, "..", "performance_regression.jl")) + isdefined(@__MODULE__, :_measure_performance_stage!) || + include(joinpath(@__DIR__, "..", "performance_regression.jl")) metadata = _performance_metadata(; warmup_policy="metadata smoke") @test length(metadata.manifest_hash) == 64 @test length(metadata.fixture_hash) == 64 @@ -1230,7 +1272,8 @@ end if !isnothing(BENCHMARK_TEST_PATTERN) && benchmark_test_enabled("XPalm staged performance profile full") @testset "XPalm staged performance profile full" begin - include(joinpath(@__DIR__, "..", "performance_regression.jl")) + isdefined(@__MODULE__, :_measure_performance_stage!) || + include(joinpath(@__DIR__, "..", "performance_regression.jl")) output_path = joinpath( @__DIR__, "..", From b595762b3208acdc3ee10dc9d3d9f24d83b1b7f3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?R=C3=A9mi=20Vezy?= Date: Mon, 21 Sep 2026 10:56:01 +0200 Subject: [PATCH 2/5] Validate current XPalm benchmarks against their committed oracle --- .github/workflows/FullPerformance.yml | 2 +- benchmark/Project.toml | 1 + benchmark/README.md | 8 +++ benchmark/performance_regression.jl | 13 +++-- benchmark/test-xpalm.jl | 51 +++++++++++++++++-- benchmark/test/runtests.jl | 72 +++++++++++++++++++++++++-- 6 files changed, 132 insertions(+), 15 deletions(-) diff --git a/.github/workflows/FullPerformance.yml b/.github/workflows/FullPerformance.yml index 58f131a76..57717f43f 100644 --- a/.github/workflows/FullPerformance.yml +++ b/.github/workflows/FullPerformance.yml @@ -121,7 +121,7 @@ jobs: run: >- julia --project=benchmark --color=yes benchmark/test/runtests.jl - "XPalm (performance measurement lifetime smoke|staged performance profile full)" + "XPalm (reference oracle smoke|performance measurement lifetime smoke|staged performance profile full)" - name: Persist measurements and the resolved environment if: always() diff --git a/benchmark/Project.toml b/benchmark/Project.toml index 097e0be4d..a4e4c0e29 100644 --- a/benchmark/Project.toml +++ b/benchmark/Project.toml @@ -10,6 +10,7 @@ SHA = "ea8e919c-243c-51af-8825-aaa63cd721ce" Sockets = "6462fe0b-24de-5631-8697-dd941f90decc" Statistics = "10745b16-79ce-11e8-11f9-7d13ad32a3b2" Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" +TOML = "fa267f1f-6049-4f14-aa54-33bafae1ed76" [sources] PlantSimEngine = {path = ".."} diff --git a/benchmark/README.md b/benchmark/README.md index 364a66532..98d6139e3 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -25,3 +25,11 @@ The flag enables the downstream suite; it does not install its dependencies. `benchmark/test/runtests.jl` uses the same default. Passing a test-name pattern explicitly selects those tests, as the dedicated downstream CI jobs do. The pinned release comparisons in `release_baselines/` use separate projects. + +The current XPalm full-cycle benchmark uses the committed `v0.7.0-dev` reference +from the checked-out XPalm package. It validates the reference's source metadata, +meteorology hash, dates, and 4,160-day horizon before measuring the scenario, then +reads its expected final state from `summary.csv`. The same numerical tolerances +apply to every output-retention mode and the high-level end-to-end run. Recorded +fixture hashes include this active reference. The isolated `release_baselines/` +projects retain the historical XPalm 0.6.1 / PlantSimEngine 0.14.1 comparison. diff --git a/benchmark/performance_regression.jl b/benchmark/performance_regression.jl index 5e6400e52..cac4fe72f 100644 --- a/benchmark/performance_regression.jl +++ b/benchmark/performance_regression.jl @@ -58,7 +58,7 @@ function _performance_fixture_hash(xpalm_root) "test", "references", "regression", - "v0.6.1", + XPALM_REFERENCE_BASELINE, ), ) entries = String[ @@ -100,6 +100,7 @@ function _performance_metadata(; warmup_policy) xpalm_revision=_performance_git_revision(xpalm_root), manifest_hash=_performance_path_hash(manifest_path), fixture_hash=_performance_fixture_hash(xpalm_root), + reference_baseline=XPALM_REFERENCE_BASELINE, warmup_policy=warmup_policy, ) end @@ -452,6 +453,8 @@ function run_xpalm_performance_profile(; ) normalized_profile = Symbol(profile) nsteps = _performance_steps(normalized_profile) + expected_state = normalized_profile == :full ? + xpalm_reference_full_cycle_expected_state() : nothing warmup_policy = "unmeasured outputs=:none prefix ($(min(nsteps, PERFORMANCE_SHORT_STEPS)) steps) " * "plus requested-output smoke ($(PERFORMANCE_SMOKE_STEPS) steps)" @@ -752,8 +755,8 @@ function run_xpalm_performance_profile(; ) if normalized_profile == :full - xpalm_reference_state_matches(reference_state) || error( - "XPalm full-cycle performance fixture does not match the committed v0.6.1 final state: ", + xpalm_reference_state_matches(reference_state, expected_state) || error( + "XPalm full-cycle performance fixture does not match the committed $(XPALM_REFERENCE_BASELINE) final state: ", "$(reference_state).", ) high_level_outputs = _measure_performance_stage!( @@ -766,9 +769,9 @@ function run_xpalm_performance_profile(; ) do xpalm_reference_end_to_end(; nsteps=nsteps) end - xpalm_reference_high_level_state_matches(high_level_outputs) || error( + xpalm_reference_high_level_state_matches(high_level_outputs; expected=expected_state) || error( "XPalm historical end-to-end performance fixture does not match the committed ", - "v0.6.1 final state.", + "$(XPALM_REFERENCE_BASELINE) final state.", ) end diff --git a/benchmark/test-xpalm.jl b/benchmark/test-xpalm.jl index e06317c2a..d2e0b162f 100644 --- a/benchmark/test-xpalm.jl +++ b/benchmark/test-xpalm.jl @@ -3,8 +3,13 @@ using CSV using DataFrames using Dates using PlantSimEngine +using SHA +using TOML using XPalm +const XPALM_REFERENCE_BASELINE = "v0.7.0-dev" +const XPALM_REFERENCE_SOURCE_COMMIT = "d4ded8891d03f85bb5691b89572a4e003ac6eb7a" + const XPALM_REFERENCE_BENCHMARK_VARIABLES = Dict{Symbol,Any}( :Scene => (:lai, :leaf_area, :aPPFD), :Plant => ( @@ -183,12 +188,48 @@ function xpalm_reference_param_run( ) end -function xpalm_reference_full_cycle_expected_state() +function xpalm_reference_full_cycle_expected_state(; + xpalm_root=dirname(dirname(pathof(XPalm))), +) + reference_dir = joinpath( + xpalm_root, "test", "references", "regression", XPALM_REFERENCE_BASELINE, + ) + metadata = TOML.parsefile(joinpath(reference_dir, "metadata.toml")) + source = metadata["source"] + inputs = metadata["inputs"] + source["baseline_id"] == XPALM_REFERENCE_BASELINE || + error("XPalm benchmark reference baseline does not match $(XPALM_REFERENCE_BASELINE).") + source["xpalm_commit"] == XPALM_REFERENCE_SOURCE_COMMIT || + error("XPalm benchmark reference source commit has changed; review its provenance.") + source["parameter_source"] == "XPalm.default_parameters()" || + error("XPalm benchmark reference uses a different parameter scenario.") + inputs["meteo_file"] == "0-data/meteo.csv" || + error("XPalm benchmark reference uses a different meteorology file.") + inputs["nsteps"] == 4160 || + error("XPalm benchmark reference must cover the complete 4,160-day scenario.") + + meteo_path = joinpath(xpalm_root, "0-data", "meteo.csv") + bytes2hex(SHA.sha256(read(meteo_path))) == inputs["meteo_sha256"] || + error("XPalm benchmark meteorology does not match the committed reference hash.") + meteo = CSV.File(meteo_path) + length(meteo) == inputs["nsteps"] || + error("XPalm benchmark meteorology length does not match the reference.") + first(meteo).date == Date(inputs["start_date"]) && + last(meteo).date == Date(inputs["end_date"]) || + error("XPalm benchmark meteorology dates do not match the reference.") + + summary = only(CSV.File(joinpath(reference_dir, "summary.csv"))) + summary.nsteps == inputs["nsteps"] && + summary.start_date == Date(inputs["start_date"]) && + summary.end_date == Date(inputs["end_date"]) || + error("XPalm benchmark summary does not describe the reference meteorology.") + isfinite(summary.final_lai) && isfinite(summary.final_ftsw) || + error("XPalm benchmark reference final state must be finite.") return ( - current_step=4160, - phytomer_count=344, - lai=5.0587602356164405, - ftsw=0.7991179101191216, + current_step=Int(summary.nsteps), + phytomer_count=Int(summary.final_phytomer_count), + lai=summary.final_lai, + ftsw=summary.final_ftsw, ) end diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index 50e56553f..74cdbbe22 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -1071,6 +1071,70 @@ if benchmark_test_enabled("XPalm benchmark API smoke") end end +if benchmark_test_enabled("XPalm reference oracle smoke") + @testset "XPalm reference oracle smoke" begin + isdefined(@__MODULE__, :xpalm_reference_param_create) || + include(joinpath(@__DIR__, "..", "test-xpalm.jl")) + xpalm_root = dirname(dirname(pathof(XPalm))) + reference_relpath = joinpath( + "test", "references", "regression", XPALM_REFERENCE_BASELINE, + ) + expected = xpalm_reference_full_cycle_expected_state() + summary = only(CSV.File(joinpath(xpalm_root, reference_relpath, "summary.csv"))) + @test expected.current_step == summary.nsteps == 4160 + @test expected.phytomer_count == summary.final_phytomer_count + @test expected.lai == summary.final_lai + @test expected.ftsw == summary.final_ftsw + historical_summary = only(CSV.File(joinpath( + xpalm_root, "test", "references", "regression", "v0.6.1", "summary.csv", + ))) + @test !xpalm_reference_state_matches( + merge(expected, (lai=historical_summary.final_lai,)), expected, + ) + + mktempdir() do fixture_root + for relative_path in ( + joinpath("0-data", "meteo.csv"), + joinpath(reference_relpath, "metadata.toml"), + joinpath(reference_relpath, "summary.csv"), + ) + target = joinpath(fixture_root, relative_path) + mkpath(dirname(target)) + cp(joinpath(xpalm_root, relative_path), target) + end + @test xpalm_reference_full_cycle_expected_state(; xpalm_root=fixture_root) == expected + metadata_path = joinpath(fixture_root, reference_relpath, "metadata.toml") + metadata = TOML.parsefile(metadata_path) + for (section, key, wrong_value) in ( + ("source", "baseline_id", "v0.6.1"), + ("source", "xpalm_commit", "unknown"), + ("source", "parameter_source", "historical_parameters()"), + ("inputs", "nsteps", 1000), + ("inputs", "meteo_sha256", repeat("0", 64)), + ) + changed_metadata = deepcopy(metadata) + changed_metadata[section][key] = wrong_value + open(metadata_path, "w") do io + TOML.print(io, changed_metadata) + end + @test_throws ErrorException xpalm_reference_full_cycle_expected_state(; + xpalm_root=fixture_root, + ) + end + open(metadata_path, "w") do io + TOML.print(io, metadata) + end + summary_path = joinpath(fixture_root, reference_relpath, "summary.csv") + changed_summary = CSV.read(summary_path, DataFrame) + changed_summary.nsteps[1] -= 1 + CSV.write(summary_path, changed_summary) + @test_throws ErrorException xpalm_reference_full_cycle_expected_state(; + xpalm_root=fixture_root, + ) + end + end +end + if benchmark_test_enabled("XPalm performance measurement lifetime smoke") @testset "XPalm performance measurement lifetime smoke" begin isdefined(@__MODULE__, :_measure_performance_stage!) || @@ -1085,12 +1149,12 @@ if benchmark_test_enabled("XPalm performance measurement lifetime smoke") GC.gc(true) second_released_before_third[] = sample_results[2].value === nothing end - result = Ref(calls[]) + local payload = Ref(calls[]) if calls[] == 1 - first_result[] = result + first_result[] = payload end - push!(sample_results, WeakRef(result)) - return result + push!(sample_results, WeakRef(payload)) + return payload end records = NamedTuple[] result = _measure_performance_stage!( From 39b54f548dcb18f3f3e9e0e009c80e0a4ea4faf5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?R=C3=A9mi=20Vezy?= Date: Mon, 21 Sep 2026 11:27:31 +0200 Subject: [PATCH 3/5] Add diagnostics for interrupted downstream performance checks --- .github/workflows/FullPerformance.yml | 42 +++++++++++++++++++++++---- benchmark/README.md | 7 +++++ benchmark/performance_regression.jl | 8 ++++- benchmark/test/runtests.jl | 7 +++-- 4 files changed, 55 insertions(+), 9 deletions(-) diff --git a/.github/workflows/FullPerformance.yml b/.github/workflows/FullPerformance.yml index 57717f43f..3fbd5bd56 100644 --- a/.github/workflows/FullPerformance.yml +++ b/.github/workflows/FullPerformance.yml @@ -18,6 +18,12 @@ on: required: true default: main type: string + xpalm_profile: + description: Full matrix or the existing warmed 4160-day no-output gate + required: true + default: full + type: choice + options: [full, warmed-no-output] workflow_call: inputs: xpalm_ref: @@ -35,6 +41,10 @@ on: required: false default: main type: string + xpalm_profile: + required: false + default: full + type: string permissions: contents: read @@ -115,22 +125,42 @@ jobs: benchmark/test/runtests.jl "PlantBiophysics benchmark (API smoke|performance)" - - name: Run the complete XPalm performance and correctness matrix + - name: Run XPalm performance and correctness checks # Leave time to upload checkpoint CSVs if the complete matrix stalls. timeout-minutes: 35 - run: >- - julia --project=benchmark --color=yes - benchmark/test/runtests.jl - "XPalm (reference oracle smoke|performance measurement lifetime smoke|staged performance profile full)" + env: + XPALM_PROFILE: ${{ inputs.xpalm_profile || 'full' }} + run: | + case "$XPALM_PROFILE" in + full) xpalm_case='staged performance profile full' ;; + warmed-no-output) xpalm_case='full warmed no-output performance' ;; + *) exit 2 ;; + esac + mkdir -p benchmark/results + julia --project=benchmark --color=yes benchmark/test/runtests.jl \ + "XPalm (reference oracle smoke|performance measurement lifetime smoke|${xpalm_case})" & + pse_benchmark_pid=$! + ( + while kill -0 "$pse_benchmark_pid" 2>/dev/null; do + date -u '+%Y-%m-%dT%H:%M:%SZ' | tee -a benchmark/results/xpalm-resources.log + ps -p "$pse_benchmark_pid" -o pid,etime,pcpu,rss,vsz | tee -a benchmark/results/xpalm-resources.log + sleep 15 + done + ) & + pse_monitor_pid=$! + trap 'kill "$pse_monitor_pid" 2>/dev/null || true' EXIT + wait "$pse_benchmark_pid" - name: Persist measurements and the resolved environment if: always() + timeout-minutes: 3 uses: actions/upload-artifact@v7 with: name: downstream-full-performance-${{ github.run_id }}-${{ github.run_attempt }} path: | benchmark/results/plantbiophysics-full-latest.csv - benchmark/results/xpalm-full-latest.csv + benchmark/results/xpalm-*.csv + benchmark/results/xpalm-resources.log benchmark/Manifest.toml if-no-files-found: warn retention-days: 90 diff --git a/benchmark/README.md b/benchmark/README.md index 98d6139e3..24b300fd8 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -33,3 +33,10 @@ reads its expected final state from `summary.csv`. The same numerical tolerances apply to every output-retention mode and the high-level end-to-end run. Recorded fixture hashes include this active reference. The isolated `release_baselines/` projects retain the historical XPalm 0.6.1 / PlantSimEngine 0.14.1 comparison. + +The full downstream workflow defaults to the complete output-retention matrix. +For diagnosis, its `warmed-no-output` option selects the existing 4,160-day +warmup and measured no-output run, including the unchanged 20-second gate. +This diagnostic does not replace acceptance of the complete matrix. Both modes +log stage/sample progress outside measured operations and record process memory +alongside the checkpoint CSVs, so interrupted runs can be investigated. diff --git a/benchmark/performance_regression.jl b/benchmark/performance_regression.jl index cac4fe72f..05da2aa61 100644 --- a/benchmark/performance_regression.jl +++ b/benchmark/performance_regression.jl @@ -211,6 +211,7 @@ function _measure_performance_stage!( sample_factory=nothing, ) samples >= 1 || error("Performance stage samples must be positive.") + @info "Performance sample starting" profile stage sample=1 samples peak_rss_bytes=Sys.maxrss() started_at = time_ns() measurement = try _timed_performance_operation(operation) @@ -236,17 +237,20 @@ function _measure_performance_stage!( _checkpoint_performance_records(checkpoint_path, records) rethrow() end + @info "Performance sample completed" profile stage sample=1 seconds=measurement.time allocated_bytes=measurement.bytes peak_rss_bytes=Sys.maxrss() # Keep the first result for correctness checks, but retain only statistics # from later samples. A full output history can be much larger than the # model itself and must be collectible before the next sample starts. times = [measurement.time] memories = [measurement.bytes] allocations = [_performance_allocation_count(measurement)] - for _ in 2:samples + for sample in 2:samples + @info "Performance sample starting" profile stage sample samples peak_rss_bytes=Sys.maxrss() sample_operation = isnothing(sample_factory) ? operation : sample_factory() sample_measurement = _timed_performance_operation(sample_operation) + @info "Performance sample completed" profile stage sample seconds=sample_measurement.time allocated_bytes=sample_measurement.bytes peak_rss_bytes=Sys.maxrss() push!(times, sample_measurement.time) push!(memories, sample_measurement.bytes) push!(allocations, _performance_allocation_count(sample_measurement)) @@ -427,6 +431,7 @@ end function _warmup_xpalm_performance!(profile_steps) lifecycle_steps = min(profile_steps, PERFORMANCE_SHORT_STEPS) + @info "XPalm profile warmup starting" lifecycle_steps retained_steps=PERFORMANCE_SMOKE_STEPS no_output_model, no_output_steps = xpalm_reference_model_create(; nsteps=lifecycle_steps) xpalm_reference_param_run( @@ -444,6 +449,7 @@ function _warmup_xpalm_performance!(profile_steps) reference_steps, ) xpalm_default_param_collect_outputs(reference_simulation) + @info "XPalm profile warmup completed" peak_rss_bytes=Sys.maxrss() return nothing end diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index 74cdbbe22..1a7aba247 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -1411,9 +1411,11 @@ if !isnothing(BENCHMARK_TEST_PATTERN) && end if !isnothing(BENCHMARK_TEST_PATTERN) && - benchmark_test_enabled("XPalm full warmed no-output performance") + benchmark_test_enabled("XPalm full warmed no-output performance") @testset "XPalm full warmed no-output performance" begin - include(joinpath(@__DIR__, "..", "performance_regression.jl")) + isdefined(@__MODULE__, :_measure_performance_stage!) || + include(joinpath(@__DIR__, "..", "performance_regression.jl")) + @info "XPalm complete lifecycle warmup starting" steps=PERFORMANCE_FULL_STEPS warmup_model, warmup_steps = xpalm_reference_model_create(; nsteps=PERFORMANCE_FULL_STEPS) xpalm_reference_param_run( @@ -1422,6 +1424,7 @@ if !isnothing(BENCHMARK_TEST_PATTERN) && warmup_steps; outputs=:none, ) + @info "XPalm complete lifecycle warmup completed" peak_rss_bytes=Sys.maxrss() model, nsteps = xpalm_reference_model_create(; nsteps=PERFORMANCE_FULL_STEPS) metadata = _performance_metadata(; From ddaab19e081abdfbc71a3bc6a7ffb92fe6134a6c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?R=C3=A9mi=20Vezy?= Date: Mon, 21 Sep 2026 11:38:54 +0200 Subject: [PATCH 4/5] End repeated benchmark sample frames before collecting results --- benchmark/performance_regression.jl | 28 +++++++---- benchmark/test/runtests.jl | 77 +++++++++++++++++------------ 2 files changed, 64 insertions(+), 41 deletions(-) diff --git a/benchmark/performance_regression.jl b/benchmark/performance_regression.jl index 05da2aa61..1507b103e 100644 --- a/benchmark/performance_regression.jl +++ b/benchmark/performance_regression.jl @@ -199,6 +199,19 @@ function _timed_performance_operation(operation) return @timed operation() end +# End the sample's stack frame before the caller starts the next repetition. +# Assigning `nothing` inside the caller is insufficient: compiler temporaries +# (including those introduced by logging) can keep the result or closure live. +@noinline function _additional_performance_sample(operation, sample_factory) + sample_operation = isnothing(sample_factory) ? operation : sample_factory() + measurement = _timed_performance_operation(sample_operation) + return ( + time=measurement.time, + bytes=measurement.bytes, + allocations=_performance_allocation_count(measurement), + ) +end + function _measure_performance_stage!( operation, records, @@ -246,16 +259,11 @@ function _measure_performance_stage!( allocations = [_performance_allocation_count(measurement)] for sample in 2:samples @info "Performance sample starting" profile stage sample samples peak_rss_bytes=Sys.maxrss() - sample_operation = isnothing(sample_factory) ? - operation : - sample_factory() - sample_measurement = _timed_performance_operation(sample_operation) - @info "Performance sample completed" profile stage sample seconds=sample_measurement.time allocated_bytes=sample_measurement.bytes peak_rss_bytes=Sys.maxrss() - push!(times, sample_measurement.time) - push!(memories, sample_measurement.bytes) - push!(allocations, _performance_allocation_count(sample_measurement)) - sample_measurement = nothing - sample_operation = nothing + sample_statistics = _additional_performance_sample(operation, sample_factory) + @info "Performance sample completed" profile stage sample seconds=sample_statistics.time allocated_bytes=sample_statistics.bytes peak_rss_bytes=Sys.maxrss() + push!(times, sample_statistics.time) + push!(memories, sample_statistics.bytes) + push!(allocations, sample_statistics.allocations) end _performance_record!( records, diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index 1a7aba247..3872df05c 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -1139,40 +1139,55 @@ if benchmark_test_enabled("XPalm performance measurement lifetime smoke") @testset "XPalm performance measurement lifetime smoke" begin isdefined(@__MODULE__, :_measure_performance_stage!) || include(joinpath(@__DIR__, "..", "performance_regression.jl")) - calls = Ref(0) - first_result = Ref{Any}(nothing) - sample_results = WeakRef[] - second_released_before_third = Ref(false) - operation = () -> begin - calls[] += 1 - if calls[] == 3 - GC.gc(true) - second_released_before_third[] = sample_results[2].value === nothing + for use_factory in (false, true) + calls = Ref(0) + first_result = Ref{Any}(nothing) + sample_results = WeakRef[] + second_released_before_third = Ref(false) + factory_payloads = WeakRef[] + factory_released_before_third = Ref(false) + operation = () -> begin + calls[] += 1 + if calls[] == 3 + GC.gc(true) + second_released_before_third[] = sample_results[2].value === nothing + factory_released_before_third[] = !use_factory || + factory_payloads[1].value === nothing + end + local payload = Ref(calls[]) + if calls[] == 1 + first_result[] = payload + end + push!(sample_results, WeakRef(payload)) + return payload end - local payload = Ref(calls[]) - if calls[] == 1 - first_result[] = payload + factory = () -> begin + local held = Ref(0) + push!(factory_payloads, WeakRef(held)) + return () -> begin + held[] += 1 + operation() + end end - push!(sample_results, WeakRef(payload)) - return payload + records = NamedTuple[] + result = _measure_performance_stage!( + operation, records, NamedTuple(), :smoke, :result_lifetime; + samples=3, sample_factory=use_factory ? factory : nothing, + ) + @test result === first_result[] + @test result[] == 1 + @test calls[] == length(sample_results) == 3 + @test second_released_before_third[] + @test factory_released_before_third[] + metrics = Dict(row.metric => row.value for row in records) + @test metrics["samples"] == 3 + @test metrics["minimum_time"] <= metrics["median_time"] + @test metrics["minimum_time"] <= metrics["wall_time"] + @test metrics["minimum_memory"] <= metrics["median_memory"] + @test metrics["minimum_memory"] <= metrics["allocated"] + @test metrics["minimum_allocations"] <= metrics["median_allocations"] + @test metrics["minimum_allocations"] <= metrics["allocations"] end - records = NamedTuple[] - result = _measure_performance_stage!( - operation, records, NamedTuple(), :smoke, :result_lifetime; - samples=3, - ) - @test result === first_result[] - @test result[] == 1 - @test calls[] == length(sample_results) == 3 - @test second_released_before_third[] - metrics = Dict(row.metric => row.value for row in records) - @test metrics["samples"] == 3 - @test metrics["minimum_time"] <= metrics["median_time"] - @test metrics["minimum_time"] <= metrics["wall_time"] - @test metrics["minimum_memory"] <= metrics["median_memory"] - @test metrics["minimum_memory"] <= metrics["allocated"] - @test metrics["minimum_allocations"] <= metrics["median_allocations"] - @test metrics["minimum_allocations"] <= metrics["allocations"] end end From f69d08713ab7a717488f8abb1e0f7e9d9cef7454 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?R=C3=A9mi=20Vezy?= Date: Mon, 21 Sep 2026 12:17:50 +0200 Subject: [PATCH 5/5] Release summarized output history before repeated benchmarks --- benchmark/README.md | 7 +++++ benchmark/performance_regression.jl | 44 +++++++++++++++++++---------- benchmark/test/runtests.jl | 28 ++++++++++++++++++ 3 files changed, 64 insertions(+), 15 deletions(-) diff --git a/benchmark/README.md b/benchmark/README.md index 24b300fd8..2bcb97473 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -40,3 +40,10 @@ warmup and measured no-output run, including the unchanged 20-second gate. This diagnostic does not replace acceptance of the complete matrix. Both modes log stage/sample progress outside measured operations and record process memory alongside the checkpoint CSVs, so interrupted runs can be investigated. + +Repeated samples retain scalar measurements, and retain the first result only +when later validation needs it. For `outputs=:all`, the harness extracts that +first run's final state and runtime counters outside the timed operation, then +releases its complete history before the second sample. Each of the three +samples still retains all outputs for the full 4,160 days while it executes. +This avoids overlapping two large histories solely for benchmark bookkeeping. diff --git a/benchmark/performance_regression.jl b/benchmark/performance_regression.jl index 1507b103e..5a59c9307 100644 --- a/benchmark/performance_regression.jl +++ b/benchmark/performance_regression.jl @@ -199,6 +199,19 @@ function _timed_performance_operation(operation) return @timed operation() end +# Result extraction is outside the measured operation. The caller can retain a +# small scientific summary when the complete history is not needed afterward. +@noinline function _first_performance_sample(operation, result_transform) + measurement = _timed_performance_operation(operation) + return ( + value=result_transform(measurement.value), + time=measurement.time, + bytes=measurement.bytes, + gctime=measurement.gctime, + allocations=_performance_allocation_count(measurement), + ) +end + # End the sample's stack frame before the caller starts the next repetition. # Assigning `nothing` inside the caller is insufficient: compiler temporaries # (including those introduced by logging) can keep the result or closure live. @@ -222,12 +235,13 @@ function _measure_performance_stage!( ; samples::Int=1, sample_factory=nothing, + result_transform=identity, ) samples >= 1 || error("Performance stage samples must be positive.") @info "Performance sample starting" profile stage sample=1 samples peak_rss_bytes=Sys.maxrss() started_at = time_ns() measurement = try - _timed_performance_operation(operation) + _first_performance_sample(operation, result_transform) catch _performance_record!( records, @@ -251,12 +265,11 @@ function _measure_performance_stage!( rethrow() end @info "Performance sample completed" profile stage sample=1 seconds=measurement.time allocated_bytes=measurement.bytes peak_rss_bytes=Sys.maxrss() - # Keep the first result for correctness checks, but retain only statistics - # from later samples. A full output history can be much larger than the - # model itself and must be collectible before the next sample starts. + # Keep the first result (or its requested summary) for correctness checks, + # but retain only statistics from later samples. times = [measurement.time] memories = [measurement.bytes] - allocations = [_performance_allocation_count(measurement)] + allocations = [measurement.allocations] for sample in 2:samples @info "Performance sample starting" profile stage sample samples peak_rss_bytes=Sys.maxrss() sample_statistics = _additional_performance_sample(operation, sample_factory) @@ -691,13 +704,23 @@ function run_xpalm_performance_profile(; xpalm_reference_model_create(; nsteps=nsteps) end all_output_model, all_output_steps = all_output_setup - all_output_simulation = _measure_performance_stage!( + all_output_state = _measure_performance_stage!( records, metadata, normalized_profile, :simulation_all_outputs, checkpoint_path, samples=PERFORMANCE_STATISTICAL_SAMPLES, + result_transform=simulation -> begin + _record_runtime_performance!( + records, + metadata, + normalized_profile, + :simulation_all_outputs, + simulation, + ) + xpalm_reference_final_state(simulation) + end, sample_factory=() -> begin sample_model, sample_steps = xpalm_reference_model_create(; nsteps=nsteps) @@ -718,18 +741,9 @@ function run_xpalm_performance_profile(; performance=true, ) end - _record_runtime_performance!( - records, - metadata, - normalized_profile, - :simulation_all_outputs, - all_output_simulation, - ) - no_output_state = xpalm_reference_final_state(no_output_simulation) small_state = xpalm_reference_final_state(small_simulation) reference_state = xpalm_reference_final_state(reference_simulation) - all_output_state = xpalm_reference_final_state(all_output_simulation) _record_xpalm_state!( records, metadata, diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index 3872df05c..96d1f8c59 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -1188,6 +1188,34 @@ if benchmark_test_enabled("XPalm performance measurement lifetime smoke") @test metrics["minimum_allocations"] <= metrics["median_allocations"] @test metrics["minimum_allocations"] <= metrics["allocations"] end + + summarized_calls = Ref(0) + transform_calls = Ref(0) + raw_results = WeakRef[] + prior_results_released = Bool[] + summarized_operation = () -> begin + GC.gc(true) + push!(prior_results_released, all(ref -> ref.value === nothing, raw_results)) + summarized_calls[] += 1 + local payload = Ref(summarized_calls[]) + push!(raw_results, WeakRef(payload)) + return payload + end + summarized_records = NamedTuple[] + summary = _measure_performance_stage!( + summarized_operation, summarized_records, NamedTuple(), :smoke, :summary_lifetime; + samples=3, + result_transform=payload -> begin + transform_calls[] += 1 + payload[] + end, + ) + @test summary == 1 + @test summarized_calls[] == 3 + @test transform_calls[] == 1 + @test all(prior_results_released) + @test length(raw_results) == 3 + @test only(row.value for row in summarized_records if row.metric == "samples") == 3 end end