Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 24 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,30 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added

- **`gpjax.xarray` input transforms.** `from_xarray(..., transforms=[...])`
turns the named inputs into the columns of `X`, and the `GridSpec` applies the
same fitted transforms to every new grid. `Standardise` scales inputs with the
training mean and standard deviation. `UnitSphere` maps latitude and longitude
to a point on the unit sphere, so a stationary kernel is valid on the whole
globe. `Cyclic` encodes a periodic input, such as the seasonal cycle, as a
point on a circle. `GridSpec.columns` names the resulting columns.
- **`GridSpec.predict`: chunked prediction on large grids.**
`spec.predict(lambda x: posterior(x, covariance="diagonal"), grid)` gives the
predictive mean and variance as an `xr.Dataset`, with at most `chunk_size`
cells in memory at once and one compilation. A Dask-backed grid gives a lazy
result that is predicted block by block.
- **New example: *Infilling Global Surface Temperature*.** It reads a netCDF
file of the 2024 temperature anomaly from the NCEP-NCAR Reanalysis 1, which is
complete, and removes cells to test the infill against the truth. Cells removed
at random are filled well, and joint samples give a global mean whose interval
holds the truth. Cells removed because they are warm show how a GP is biased,
and overconfident, when data are missing not at random.
- **The xarray introduction is now *Working with Gridded Data*, under Getting
started.** It uses `Standardise` and `GridSpec.predict`, and shows
`UnitSphere` and `Cyclic`. Its URL is unchanged.

### Changed

- Allow JAX and JAXlib 0.11 in downstream environments by removing the
Expand Down
90 changes: 88 additions & 2 deletions docs/examples/data/_pull_reference_datasets.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,8 +10,9 @@
uv run --extra docs python docs/examples/data/_pull_reference_datasets.py

The ``--extra docs`` is only needed for the UCI Auto MPG pull, which uses
``ucimlrepo``; the other three pulls need nothing beyond ``pandas`` and
``requests``.
``ucimlrepo``. The NCEP reanalysis pull reads netCDF4, so it also needs
``--with h5netcdf --with h5py``. The other pulls need nothing beyond ``pandas``
and ``requests``.

Data sources
------------
Expand All @@ -32,11 +33,23 @@
https://archive.ics.uci.edu/dataset/9/auto-mpg (fetched via ``ucimlrepo``).
Licence: CC BY 4.0. The features and the target are concatenated into a single
frame so the notebook can split them back out without ``ucimlrepo``.
- NCEP-NCAR Reanalysis 1 temperature anomaly (``ncep_air_anomaly_2024.nc``):
the 2024 annual-mean anomaly of near-surface (0.995 sigma) air temperature,
relative to the 1991-2020 mean of each grid cell, on the native 2.5-degree
grid. A reanalysis has a value in every cell, so the field is complete.
Source: https://psl.noaa.gov/data/gridded/data.ncep.reanalysis.html (monthly
means, ``air.mon.mean.nc``). Provider: NOAA Physical Sciences Laboratory
(Kalnay et al., 1996, https://doi.org/10.1175/1520-0477(1996)077<0437:TNYRP>2.0.CO;2).
Work of the US Government, so public domain; PSL ask for the acknowledgement
"NCEP-NCAR Reanalysis 1 data provided by the NOAA PSL, Boulder, Colorado, USA,
from their website at https://psl.noaa.gov". Written as netCDF3, which xarray
reads with SciPy, so the notebook needs no netCDF4 library.
"""

from __future__ import annotations

from pathlib import Path
import tempfile
import time

import pandas as pd
Expand Down Expand Up @@ -104,8 +117,81 @@ def pull_auto_mpg() -> None:
_save(pd.concat([features, targets], axis=1), "auto_mpg.csv")


# --------------------------------------------------------------------------- #
# xarray_workflow — NCEP-NCAR Reanalysis 1 temperature anomaly for 2024. #
# --------------------------------------------------------------------------- #
NCEP_URL = (
"https://psl.noaa.gov/thredds/fileServer/Datasets/ncep.reanalysis/"
"Monthlies/surface/air.mon.mean.nc"
)


def _download_resumable(url: str, path: Path, attempts: int = 8) -> None:
"""Download ``url`` to ``path``, resuming when the server cuts it short."""
for _ in range(attempts):
start = path.stat().st_size if path.exists() else 0
headers = {"User-Agent": "gpjax-docs", "Range": f"bytes={start}-"}
try:
with requests.get(url, headers=headers, stream=True, timeout=120) as resp:
if resp.status_code == 416: # nothing left to fetch
return
resp.raise_for_status()
total = start + int(resp.headers["Content-Length"])
mode = "ab" if resp.status_code == 206 else "wb"
with path.open(mode) as file:
for chunk in resp.iter_content(1 << 20):
file.write(chunk)
if path.stat().st_size >= total:
return
except requests.RequestException as err:
print(f" retrying after: {err}")
time.sleep(2)
raise RuntimeError(f"Failed to fetch {url} in {attempts} attempts")


def pull_ncep_anomaly(nc_path: Path | None = None) -> None:
"""Save the 2024 NCEP anomaly; pass ``nc_path`` to reuse a download."""
print("xarray_workflow: NCEP-NCAR Reanalysis 1 temperature anomaly, 2024")
import xarray as xr

with tempfile.TemporaryDirectory() as tmp:
if nc_path is None:
nc_path = Path(tmp) / "air.mon.mean.nc"
_download_resumable(NCEP_URL, nc_path)
with xr.open_dataset(nc_path, engine="h5netcdf") as monthly:
annual = monthly["air"].resample(time="YS").mean().load()
climatology = annual.sel(time=slice("1991", "2020")).mean("time")
anomaly = annual.sel(time="2024").squeeze("time", drop=True) - climatology
# Longitude from 0..357.5 to -180..177.5, so maps are centred on Greenwich.
anomaly = anomaly.assign_coords(lon=(anomaly["lon"] + 180.0) % 360.0 - 180.0)
anomaly = anomaly.sortby(["lat", "lon"]).astype("float32")
anomaly.attrs = {
"long_name": "Near-surface air temperature anomaly",
"units": "K",
"cell_methods": "time: mean (2024, anomaly relative to 1991-2020)",
}
anomaly["lat"].attrs = {"standard_name": "latitude", "units": "degrees_north"}
anomaly["lon"].attrs = {"standard_name": "longitude", "units": "degrees_east"}
dataset = anomaly.to_dataset(name="tas_anomaly")
dataset.attrs = {
"title": "2024 annual-mean near-surface air temperature anomaly",
"source": "NCEP-NCAR Reanalysis 1, monthly air.sig995 (air.mon.mean.nc)",
"references": "Kalnay et al. (1996), doi:10.1175/1520-0477(1996)077<0437:TNYRP>2.0.CO;2",
"acknowledgement": (
"NCEP-NCAR Reanalysis 1 data provided by the NOAA PSL, Boulder, "
"Colorado, USA, from their website at https://psl.noaa.gov"
),
"history": "Annual means of the monthly means; 2024 minus the 1991-2020 mean.",
"Conventions": "CF-1.8",
}
path = HERE / "ncep_air_anomaly_2024.nc"
dataset.to_netcdf(path, engine="scipy", format="NETCDF3_64BIT")
print(f" wrote {path.name}: {dict(dataset.sizes)}")


if __name__ == "__main__":
pull_mauna_loa_co2()
pull_gulf_velocities()
pull_auto_mpg()
pull_ncep_anomaly()
print("\nDone.")
Binary file added docs/examples/data/ncep_air_anomaly_2024.nc
Binary file not shown.
Loading
Loading