Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
44 changes: 38 additions & 6 deletions .github/workflows/FullPerformance.yml
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,12 @@ on:
required: true
default: main
type: string
xpalm_profile:
description: Full matrix or the existing warmed 4160-day no-output gate
required: true
default: full
type: choice
options: [full, warmed-no-output]
workflow_call:
inputs:
xpalm_ref:
Expand All @@ -35,6 +41,10 @@ on:
required: false
default: main
type: string
xpalm_profile:
required: false
default: full
type: string

permissions:
contents: read
Expand Down Expand Up @@ -115,20 +125,42 @@ jobs:
benchmark/test/runtests.jl
"PlantBiophysics benchmark (API smoke|performance)"

- name: Run the complete XPalm performance and correctness matrix
run: >-
julia --project=benchmark --color=yes
benchmark/test/runtests.jl
"XPalm staged performance profile full"
- name: Run XPalm performance and correctness checks
# Leave time to upload checkpoint CSVs if the complete matrix stalls.
timeout-minutes: 35
env:
XPALM_PROFILE: ${{ inputs.xpalm_profile || 'full' }}
run: |
case "$XPALM_PROFILE" in
full) xpalm_case='staged performance profile full' ;;
warmed-no-output) xpalm_case='full warmed no-output performance' ;;
*) exit 2 ;;
esac
mkdir -p benchmark/results
julia --project=benchmark --color=yes benchmark/test/runtests.jl \
"XPalm (reference oracle smoke|performance measurement lifetime smoke|${xpalm_case})" &
pse_benchmark_pid=$!
(
while kill -0 "$pse_benchmark_pid" 2>/dev/null; do
date -u '+%Y-%m-%dT%H:%M:%SZ' | tee -a benchmark/results/xpalm-resources.log
ps -p "$pse_benchmark_pid" -o pid,etime,pcpu,rss,vsz | tee -a benchmark/results/xpalm-resources.log
sleep 15
done
) &
pse_monitor_pid=$!
trap 'kill "$pse_monitor_pid" 2>/dev/null || true' EXIT
wait "$pse_benchmark_pid"

- name: Persist measurements and the resolved environment
if: always()
timeout-minutes: 3
uses: actions/upload-artifact@v7
with:
name: downstream-full-performance-${{ github.run_id }}-${{ github.run_attempt }}
path: |
benchmark/results/plantbiophysics-full-latest.csv
benchmark/results/xpalm-full-latest.csv
benchmark/results/xpalm-*.csv
benchmark/results/xpalm-resources.log
benchmark/Manifest.toml
if-no-files-found: warn
retention-days: 90
1 change: 1 addition & 0 deletions benchmark/Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@ SHA = "ea8e919c-243c-51af-8825-aaa63cd721ce"
Sockets = "6462fe0b-24de-5631-8697-dd941f90decc"
Statistics = "10745b16-79ce-11e8-11f9-7d13ad32a3b2"
Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40"
TOML = "fa267f1f-6049-4f14-aa54-33bafae1ed76"

[sources]
PlantSimEngine = {path = ".."}
22 changes: 22 additions & 0 deletions benchmark/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -25,3 +25,25 @@ The flag enables the downstream suite; it does not install its dependencies.
`benchmark/test/runtests.jl` uses the same default. Passing a test-name pattern explicitly
selects those tests, as the dedicated downstream CI jobs do.
The pinned release comparisons in `release_baselines/` use separate projects.

The current XPalm full-cycle benchmark uses the committed `v0.7.0-dev` reference
from the checked-out XPalm package. It validates the reference's source metadata,
meteorology hash, dates, and 4,160-day horizon before measuring the scenario, then
reads its expected final state from `summary.csv`. The same numerical tolerances
apply to every output-retention mode and the high-level end-to-end run. Recorded
fixture hashes include this active reference. The isolated `release_baselines/`
projects retain the historical XPalm 0.6.1 / PlantSimEngine 0.14.1 comparison.

The full downstream workflow defaults to the complete output-retention matrix.
For diagnosis, its `warmed-no-output` option selects the existing 4,160-day
warmup and measured no-output run, including the unchanged 20-second gate.
This diagnostic does not replace acceptance of the complete matrix. Both modes
log stage/sample progress outside measured operations and record process memory
alongside the checkpoint CSVs, so interrupted runs can be investigated.

Repeated samples retain scalar measurements, and retain the first result only
when later validation needs it. For `outputs=:all`, the harness extracts that
first run's final state and runtime counters outside the timed operation, then
releases its complete history before the second sample. Each of the three
samples still retains all outputs for the full 4,160 days while it executes.
This avoids overlapping two large histories solely for benchmark bookkeeping.
91 changes: 63 additions & 28 deletions benchmark/performance_regression.jl
Original file line number Diff line number Diff line change
Expand Up @@ -58,7 +58,7 @@ function _performance_fixture_hash(xpalm_root)
"test",
"references",
"regression",
"v0.6.1",
XPALM_REFERENCE_BASELINE,
),
)
entries = String[
Expand Down Expand Up @@ -100,6 +100,7 @@ function _performance_metadata(; warmup_policy)
xpalm_revision=_performance_git_revision(xpalm_root),
manifest_hash=_performance_path_hash(manifest_path),
fixture_hash=_performance_fixture_hash(xpalm_root),
reference_baseline=XPALM_REFERENCE_BASELINE,
warmup_policy=warmup_policy,
)
end
Expand Down Expand Up @@ -198,6 +199,32 @@ function _timed_performance_operation(operation)
return @timed operation()
end

# Result extraction is outside the measured operation. The caller can retain a
# small scientific summary when the complete history is not needed afterward.
@noinline function _first_performance_sample(operation, result_transform)
measurement = _timed_performance_operation(operation)
return (
value=result_transform(measurement.value),
time=measurement.time,
bytes=measurement.bytes,
gctime=measurement.gctime,
allocations=_performance_allocation_count(measurement),
)
end

# End the sample's stack frame before the caller starts the next repetition.
# Assigning `nothing` inside the caller is insufficient: compiler temporaries
# (including those introduced by logging) can keep the result or closure live.
@noinline function _additional_performance_sample(operation, sample_factory)
sample_operation = isnothing(sample_factory) ? operation : sample_factory()
measurement = _timed_performance_operation(sample_operation)
return (
time=measurement.time,
bytes=measurement.bytes,
allocations=_performance_allocation_count(measurement),
)
end

function _measure_performance_stage!(
operation,
records,
Expand All @@ -208,11 +235,13 @@ function _measure_performance_stage!(
;
samples::Int=1,
sample_factory=nothing,
result_transform=identity,
)
samples >= 1 || error("Performance stage samples must be positive.")
@info "Performance sample starting" profile stage sample=1 samples peak_rss_bytes=Sys.maxrss()
started_at = time_ns()
measurement = try
_timed_performance_operation(operation)
_first_performance_sample(operation, result_transform)
catch
_performance_record!(
records,
Expand All @@ -235,19 +264,20 @@ function _measure_performance_stage!(
_checkpoint_performance_records(checkpoint_path, records)
rethrow()
end
measurements = Any[measurement]
for _ in 2:samples
sample_operation = isnothing(sample_factory) ?
operation :
sample_factory()
push!(
measurements,
_timed_performance_operation(sample_operation),
)
@info "Performance sample completed" profile stage sample=1 seconds=measurement.time allocated_bytes=measurement.bytes peak_rss_bytes=Sys.maxrss()
# Keep the first result (or its requested summary) for correctness checks,
# but retain only statistics from later samples.
times = [measurement.time]
memories = [measurement.bytes]
allocations = [measurement.allocations]
for sample in 2:samples
@info "Performance sample starting" profile stage sample samples peak_rss_bytes=Sys.maxrss()
sample_statistics = _additional_performance_sample(operation, sample_factory)
@info "Performance sample completed" profile stage sample seconds=sample_statistics.time allocated_bytes=sample_statistics.bytes peak_rss_bytes=Sys.maxrss()
push!(times, sample_statistics.time)
push!(memories, sample_statistics.bytes)
push!(allocations, sample_statistics.allocations)
end
times = getproperty.(measurements, :time)
memories = getproperty.(measurements, :bytes)
allocations = _performance_allocation_count.(measurements)
_performance_record!(
records,
metadata,
Expand Down Expand Up @@ -422,6 +452,7 @@ end

function _warmup_xpalm_performance!(profile_steps)
lifecycle_steps = min(profile_steps, PERFORMANCE_SHORT_STEPS)
@info "XPalm profile warmup starting" lifecycle_steps retained_steps=PERFORMANCE_SMOKE_STEPS
no_output_model, no_output_steps =
xpalm_reference_model_create(; nsteps=lifecycle_steps)
xpalm_reference_param_run(
Expand All @@ -439,6 +470,7 @@ function _warmup_xpalm_performance!(profile_steps)
reference_steps,
)
xpalm_default_param_collect_outputs(reference_simulation)
@info "XPalm profile warmup completed" peak_rss_bytes=Sys.maxrss()
return nothing
end

Expand All @@ -448,6 +480,8 @@ function run_xpalm_performance_profile(;
)
normalized_profile = Symbol(profile)
nsteps = _performance_steps(normalized_profile)
expected_state = normalized_profile == :full ?
xpalm_reference_full_cycle_expected_state() : nothing
warmup_policy =
"unmeasured outputs=:none prefix ($(min(nsteps, PERFORMANCE_SHORT_STEPS)) steps) " *
"plus requested-output smoke ($(PERFORMANCE_SMOKE_STEPS) steps)"
Expand Down Expand Up @@ -670,13 +704,23 @@ function run_xpalm_performance_profile(;
xpalm_reference_model_create(; nsteps=nsteps)
end
all_output_model, all_output_steps = all_output_setup
all_output_simulation = _measure_performance_stage!(
all_output_state = _measure_performance_stage!(
records,
metadata,
normalized_profile,
:simulation_all_outputs,
checkpoint_path,
samples=PERFORMANCE_STATISTICAL_SAMPLES,
result_transform=simulation -> begin
_record_runtime_performance!(
records,
metadata,
normalized_profile,
:simulation_all_outputs,
simulation,
)
xpalm_reference_final_state(simulation)
end,
sample_factory=() -> begin
sample_model, sample_steps =
xpalm_reference_model_create(; nsteps=nsteps)
Expand All @@ -697,18 +741,9 @@ function run_xpalm_performance_profile(;
performance=true,
)
end
_record_runtime_performance!(
records,
metadata,
normalized_profile,
:simulation_all_outputs,
all_output_simulation,
)

no_output_state = xpalm_reference_final_state(no_output_simulation)
small_state = xpalm_reference_final_state(small_simulation)
reference_state = xpalm_reference_final_state(reference_simulation)
all_output_state = xpalm_reference_final_state(all_output_simulation)
_record_xpalm_state!(
records,
metadata,
Expand Down Expand Up @@ -748,8 +783,8 @@ function run_xpalm_performance_profile(;
)

if normalized_profile == :full
xpalm_reference_state_matches(reference_state) || error(
"XPalm full-cycle performance fixture does not match the committed v0.6.1 final state: ",
xpalm_reference_state_matches(reference_state, expected_state) || error(
"XPalm full-cycle performance fixture does not match the committed $(XPALM_REFERENCE_BASELINE) final state: ",
"$(reference_state).",
)
high_level_outputs = _measure_performance_stage!(
Expand All @@ -762,9 +797,9 @@ function run_xpalm_performance_profile(;
) do
xpalm_reference_end_to_end(; nsteps=nsteps)
end
xpalm_reference_high_level_state_matches(high_level_outputs) || error(
xpalm_reference_high_level_state_matches(high_level_outputs; expected=expected_state) || error(
"XPalm historical end-to-end performance fixture does not match the committed ",
"v0.6.1 final state.",
"$(XPALM_REFERENCE_BASELINE) final state.",
)
end

Expand Down
51 changes: 46 additions & 5 deletions benchmark/test-xpalm.jl
Original file line number Diff line number Diff line change
Expand Up @@ -3,8 +3,13 @@ using CSV
using DataFrames
using Dates
using PlantSimEngine
using SHA
using TOML
using XPalm

const XPALM_REFERENCE_BASELINE = "v0.7.0-dev"
const XPALM_REFERENCE_SOURCE_COMMIT = "d4ded8891d03f85bb5691b89572a4e003ac6eb7a"

const XPALM_REFERENCE_BENCHMARK_VARIABLES = Dict{Symbol,Any}(
:Scene => (:lai, :leaf_area, :aPPFD),
:Plant => (
Expand Down Expand Up @@ -183,12 +188,48 @@ function xpalm_reference_param_run(
)
end

function xpalm_reference_full_cycle_expected_state()
function xpalm_reference_full_cycle_expected_state(;
xpalm_root=dirname(dirname(pathof(XPalm))),
)
reference_dir = joinpath(
xpalm_root, "test", "references", "regression", XPALM_REFERENCE_BASELINE,
)
metadata = TOML.parsefile(joinpath(reference_dir, "metadata.toml"))
source = metadata["source"]
inputs = metadata["inputs"]
source["baseline_id"] == XPALM_REFERENCE_BASELINE ||
error("XPalm benchmark reference baseline does not match $(XPALM_REFERENCE_BASELINE).")
source["xpalm_commit"] == XPALM_REFERENCE_SOURCE_COMMIT ||
error("XPalm benchmark reference source commit has changed; review its provenance.")
source["parameter_source"] == "XPalm.default_parameters()" ||
error("XPalm benchmark reference uses a different parameter scenario.")
inputs["meteo_file"] == "0-data/meteo.csv" ||
error("XPalm benchmark reference uses a different meteorology file.")
inputs["nsteps"] == 4160 ||
error("XPalm benchmark reference must cover the complete 4,160-day scenario.")

meteo_path = joinpath(xpalm_root, "0-data", "meteo.csv")
bytes2hex(SHA.sha256(read(meteo_path))) == inputs["meteo_sha256"] ||
error("XPalm benchmark meteorology does not match the committed reference hash.")
meteo = CSV.File(meteo_path)
length(meteo) == inputs["nsteps"] ||
error("XPalm benchmark meteorology length does not match the reference.")
first(meteo).date == Date(inputs["start_date"]) &&
last(meteo).date == Date(inputs["end_date"]) ||
error("XPalm benchmark meteorology dates do not match the reference.")

summary = only(CSV.File(joinpath(reference_dir, "summary.csv")))
summary.nsteps == inputs["nsteps"] &&
summary.start_date == Date(inputs["start_date"]) &&
summary.end_date == Date(inputs["end_date"]) ||
error("XPalm benchmark summary does not describe the reference meteorology.")
isfinite(summary.final_lai) && isfinite(summary.final_ftsw) ||
error("XPalm benchmark reference final state must be finite.")
return (
current_step=4160,
phytomer_count=344,
lai=5.0587602356164405,
ftsw=0.7991179101191216,
current_step=Int(summary.nsteps),
phytomer_count=Int(summary.final_phytomer_count),
lai=summary.final_lai,
ftsw=summary.final_ftsw,
)
end

Expand Down
Loading
Loading