Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions .github/workflows/test.yml
Original file line number Diff line number Diff line change
Expand Up @@ -45,3 +45,16 @@ jobs:

- name: Test
run: ./scripts/test.sh --cu --clean

hip:
name: HIP
if: ${{ vars.XPU_HIP_CI == 'true' && github.event_name != 'pull_request' }}
runs-on: [self-hosted, linux, x64, rocm]
timeout-minutes: 30

steps:
- name: Checkout
uses: actions/checkout@v6

- name: Test
run: ./scripts/test.sh --hip --clean
103 changes: 102 additions & 1 deletion CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@ endif()
project(xpu LANGUAGES CXX)

option(XPU_ENABLE_CUDA "Build the CUDA backend if a CUDA compiler is available" ON)
option(XPU_ENABLE_HIP "Build the HIP backend if a HIP compiler is available and CUDA is not used" ON)
option(XPU_ENABLE_LINALG "Build the optional linear-algebra component" OFF)
option(XPU_ARCH_NATIVE "Tune for the host architecture (-march=native)" ON)
option(XPU_BUILD_TESTS "Build the xpu tests" ${PROJECT_IS_TOP_LEVEL})
Expand Down Expand Up @@ -64,10 +65,89 @@ if(XPU_ENABLE_CUDA)
"xpu: CUDA ${CMAKE_CUDA_COMPILER_VERSION}, arch ${CMAKE_CUDA_ARCHITECTURES}, "
"host ${CMAKE_CXX_COMPILER_ID} ${CMAKE_CXX_COMPILER_VERSION}")
else()
message(STATUS "xpu: no CUDA compiler found, building the CPU-only path")
message(STATUS "xpu: no CUDA compiler found")
endif()
endif()

# HIP
set(XPU_HIP OFF)

if(XPU_ENABLE_HIP AND NOT XPU_CUDA)
include(CheckLanguage)
check_language(HIP)

if(CMAKE_HIP_COMPILER)
enable_language(HIP)

if(DEFINED CMAKE_HIP_PLATFORM AND NOT CMAKE_HIP_PLATFORM STREQUAL "amd")
message(FATAL_ERROR
"xpu's HIP backend targets AMD GPUs, but CMAKE_HIP_PLATFORM is ${CMAKE_HIP_PLATFORM}. "
"Use the CUDA backend on NVIDIA, or configure with -DXPU_ENABLE_HIP=OFF.")
endif()

set(CMAKE_HIP_EXTENSIONS OFF)

# Let the ROCm packages, and the dependencies they look up, find ROCm
list(APPEND CMAKE_PREFIX_PATH ${CMAKE_HIP_COMPILER_ROCM_ROOT} $ENV{ROCM_PATH} /opt/rocm)

# hip-config probes for GPUs unless GPU_TARGETS is set, so reuse CMake's list
if(NOT DEFINED GPU_TARGETS)
set(GPU_TARGETS ${CMAKE_HIP_ARCHITECTURES})
endif()

foreach(package IN ITEMS hip rocprim hipcub hiprand)
find_package(${package} CONFIG)

if(NOT ${package}_FOUND)
message(FATAL_ERROR
"xpu: found a HIP compiler but not the ROCm ${package} package. Point ROCM_PATH or "
"CMAKE_PREFIX_PATH at ROCm, or configure with -DXPU_ENABLE_HIP=OFF to build the CPU-only path.")
endif()
endforeach()

if(hip_VERSION VERSION_LESS 7.0)
message(FATAL_ERROR
"xpu needs ROCm 7.0 or newer for the HIP backend (found HIP ${hip_VERSION}). "
"Configure with -DXPU_ENABLE_HIP=OFF to build the CPU-only path.")
endif()

# hipCUB's C++23 path uses std::extents, which needs <mdspan> from the standard library
include(CheckSourceCompiles)
block()
# try_compile adds a -std of its own from the CMAKE_HIP_* standard settings after
# CMAKE_REQUIRED_FLAGS, so pin that one to C++23 as well
set(CMAKE_HIP_STANDARD 23)
set(CMAKE_REQUIRED_FLAGS -std=c++23)
set(CMAKE_TRY_COMPILE_TARGET_TYPE STATIC_LIBRARY)
check_source_compiles(HIP [=[
#include <mdspan>
int main() { return static_cast<int>(std::extents<int, 1>::rank()) - 1; }
]=] XPU_HIP_HAS_MDSPAN)
endblock()

if(NOT XPU_HIP_HAS_MDSPAN)
message(FATAL_ERROR
"The HIP compiler's C++ standard library has no std::extents in <mdspan>, which hipCUB "
"needs under C++23. Install GCC 15 (ROCm's clang uses the newest GCC it finds) or use "
"libc++, then reconfigure in a fresh build directory.")
endif()

set(XPU_HIP ON)
message(STATUS
"xpu: HIP ${hip_VERSION}, arch ${CMAKE_HIP_ARCHITECTURES}, "
"compiler ${CMAKE_HIP_COMPILER_ID} ${CMAKE_HIP_COMPILER_VERSION}")
else()
message(STATUS "xpu: no HIP compiler found")
endif()
endif()

if(XPU_CUDA OR XPU_HIP)
set(XPU_GPU ON)
else()
set(XPU_GPU OFF)
message(STATUS "xpu: building the CPU-only path")
endif()

# Targets
add_library(xpu INTERFACE)
add_library(xpu::xpu ALIAS xpu)
Expand All @@ -84,6 +164,16 @@ if(XPU_CUDA)
target_link_libraries(xpu INTERFACE CUDA::cudart)

target_compile_options(xpu INTERFACE $<$<COMPILE_LANGUAGE:CUDA>:-std=c++23>)
elseif(XPU_HIP)
target_compile_definitions(xpu INTERFACE XPU_HIP)
# hip::hipcub pulls in hip::device, whose -x hip and --offload-arch would land on every CXX
# source of a consumer, so take the hipCUB and rocPRIM headers without it
target_link_libraries(xpu INTERFACE hip::host roc::rocprim hip::hiprand)
target_include_directories(xpu SYSTEM INTERFACE
$<TARGET_PROPERTY:hip::hipcub,INTERFACE_INCLUDE_DIRECTORIES>
)

target_compile_options(xpu INTERFACE $<$<COMPILE_LANGUAGE:HIP>:-std=c++23>)
else()
find_package(OpenMP QUIET COMPONENTS CXX)
if(OpenMP_CXX_FOUND)
Expand All @@ -103,6 +193,16 @@ if(XPU_ENABLE_LINALG)
CUDA::cusolver
CUDA::cublas
)
elseif(XPU_HIP)
find_package(hipsolver CONFIG)

if(NOT hipsolver_FOUND)
message(FATAL_ERROR
"xpu: XPU_ENABLE_LINALG on the HIP backend needs the ROCm hipsolver package. "
"Install it, or configure with -DXPU_ENABLE_LINALG=OFF.")
endif()

target_link_libraries(xpu_linalg INTERFACE roc::hipsolver)
else()
find_package(LAPACK REQUIRED)
find_path(XPU_LAPACKE_INCLUDE_DIR NAMES lapacke.h REQUIRED)
Expand All @@ -122,6 +222,7 @@ if(XPU_ARCH_NATIVE AND NOT MSVC)
target_compile_options(xpu INTERFACE
$<$<COMPILE_LANGUAGE:CXX>:-march=native>
$<$<COMPILE_LANGUAGE:CUDA>:-Xcompiler=-march=native>
$<$<COMPILE_LANGUAGE:HIP>:-march=native>
)
endif()

Expand Down
96 changes: 70 additions & 26 deletions README.md
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
# xpu

`xpu` is a small, header-only C++ library for code that runs on a CPU or NVIDIA
CUDA. It provides backend-aware allocation, contiguous buffers,
structure-of-arrays storage, math helpers, and basic CUDA launch utilities.
`xpu` is a small, header-only C++ library for code that runs on a CPU, NVIDIA
CUDA, or AMD HIP. It provides backend-aware allocation, contiguous buffers,
structure-of-arrays storage, math helpers, and basic GPU launch utilities.

The project is in early development. The API may change.

Expand All @@ -12,8 +12,13 @@ The project is in early development. The API may change.
- CMake 3.25 or newer
- CUDA 13.3 or newer for the CUDA backend
- A compatible CUDA host compiler, with GCC 15 or newer when GCC is used
- ROCm 7.0 or newer for the HIP backend, with hipCUB, rocPRIM and hipRAND
- A C++ standard library with `<mdspan>` for HIP builds: libstdc++ from GCC 15
or newer, or libc++
- LAPACKE for the optional CPU linear-algebra component
- hipSOLVER for the optional HIP linear-algebra component
- Linux, or Windows through WSL2, for CUDA builds
- Linux for HIP builds

The base CPU-only library has no external dependencies.
CPU builds use OpenMP when CMake finds it; otherwise execution is serial.
Expand Down Expand Up @@ -45,12 +50,30 @@ cmake -S . -B build -G Ninja \
cmake --build build
```

HIP is enabled by default when CMake finds ROCm's `clang++` and CUDA is not in
use. CUDA takes precedence when both toolchains are available. Point CMake at
ROCm's compiler and set the target GPU architecture explicitly:

```bash
cmake -S . -B build-hip -G Ninja \
-DCMAKE_BUILD_TYPE=Release \
-DCMAKE_HIP_COMPILER=/opt/rocm/llvm/bin/clang++ \
-DCMAKE_HIP_ARCHITECTURES=gfx942 \
-DXPU_ENABLE_CUDA=OFF

cmake --build build-hip
```

Without `CMAKE_HIP_ARCHITECTURES`, CMake targets the GPUs that
`rocm_agent_enumerator` reports, or the compiler's default when it finds none.

For a CPU-only build:

```bash
cmake -S . -B build-cpu -G Ninja \
-DCMAKE_BUILD_TYPE=Release \
-DXPU_ENABLE_CUDA=OFF
-DXPU_ENABLE_CUDA=OFF \
-DXPU_ENABLE_HIP=OFF

cmake --build build-cpu
```
Expand All @@ -63,6 +86,7 @@ sudo apt install liblapacke-dev
cmake -S . -B build-cpu-linalg -G Ninja \
-DCMAKE_BUILD_TYPE=Release \
-DXPU_ENABLE_CUDA=OFF \
-DXPU_ENABLE_HIP=OFF \
-DXPU_ENABLE_LINALG=ON

cmake --build build-cpu-linalg
Expand All @@ -77,8 +101,8 @@ target_link_libraries(my_target PRIVATE xpu::xpu)

Link `xpu::linalg` instead when using `<xpu/linear_algebra.hpp>`.

Set `XPU_ENABLE_CUDA` and `XPU_ENABLE_LINALG` before `add_subdirectory` when
you need to select them explicitly.
Set `XPU_ENABLE_CUDA`, `XPU_ENABLE_HIP` and `XPU_ENABLE_LINALG` before
`add_subdirectory` when you need to select them explicitly.

## Buffers

Expand All @@ -98,8 +122,8 @@ const auto count{values.count()};
const auto capacity{values.capacity()};
```

`count()` is the requested number of elements. `capacity()` includes any CPU
padding.
`count()` is the requested number of elements. `capacity()` includes any
alignment padding.

## Structure of arrays

Expand Down Expand Up @@ -150,31 +174,40 @@ is valid only while the original `soa` owns the allocation.

## Backend and memory model

`XPU_CUDA` selects the allocation backend:
`XPU_CUDA` or `XPU_HIP` selects the allocation backend:

| Backend | Allocation |
|---|---|
| CPU | aligned `operator new` |
| CUDA | `cudaMalloc` |
| HIP | `hipMalloc` |

The `xpu::xpu` CMake target sets `XPU_CUDA` for CUDA builds. Do not set it on
individual source files. It must have the same value in every translation unit
linked into a program.
The `xpu::xpu` CMake target sets `XPU_CUDA` for CUDA builds and `XPU_HIP` for
HIP builds. Do not set them on individual source files. They must have the same
value in every translation unit linked into a program. `<xpu/config.hpp>`
defines `XPU_GPU` for either GPU backend, and `xpu::xpu_cuda`, `xpu::xpu_hip`
and `xpu::xpu_gpu` expose the same choice as constants.

When CUDA is enabled, every translation unit that includes an xpu header must be
compiled by nvcc. These files normally use the `.cu` extension.

CUDA allocations are device memory. Pointers returned by `buffer` and `soa`
cannot be dereferenced by host code. The library does not currently wrap memory
transfers, so use the CUDA runtime directly when transfers are required.
When HIP is enabled, every translation unit that includes an xpu header must be
compiled as HIP. Use the `.hip` extension or set the `LANGUAGE HIP` source file
property. The test suite does the latter to reuse its `.cu` sources.

GPU allocations are device memory. Pointers returned by `buffer` and `soa`
cannot be dereferenced by host code. Use `xpu::copy_n` or `xpu::memcpy` to move
data between host and device memory; the runtime infers the direction.

Allocation failure terminates the process with `std::abort`.

## Padding

CPU allocations are aligned to at least `xpu::simd_bytes`. Capacities and SoA
strides are padded to SIMD-lane multiples when the element type is smaller than
the SIMD width. CUDA uses a tight layout with no row padding.
CPU allocations are aligned to at least `xpu::simd_bytes`. GPU builds use
`xpu::cuda_align_bytes`, which is 128 bytes, instead. When the element type is
smaller than that alignment, capacities and SoA strides are padded to a multiple
of `alignment / sizeof(T)` elements (integer division), which fills whole
alignment blocks when `sizeof(T)` is a power of two.

The default SIMD width is 64 bytes with AVX-512, 32 bytes with AVX or AVX2, and
16 bytes otherwise. Pin it when layout must remain stable across machines:
Expand Down Expand Up @@ -203,32 +236,43 @@ if (

Use `solve()` for linear systems. Use `invert()` only when the inverse itself
is required. Matrices use row-major layout, and strides are measured in
elements.
elements. The component uses LAPACKE on the CPU, cuSOLVER on CUDA, and
hipSOLVER on HIP.

## Testing

```bash
./scripts/test.sh # CPU and CUDA test suites
./scripts/test.sh # every available suite: CPU, CUDA, then HIP
./scripts/test.sh --sanitize # CUDA Compute Sanitizer
./scripts/test.sh --cpp # CPU test suite only
./scripts/test.sh --cu # CUDA test suite only
./scripts/test.sh --hip # HIP test suite only
```

Behavioral tests are grouped by component under `tests/`. CPU and CUDA entry
points are kept separate, while backend-neutral cases live beside them in a
The script finds ROCm's `clang++` through `HIPCXX` or
`$ROCM_PATH/llvm/bin/clang++`, where `ROCM_PATH` defaults to `/opt/rocm`. It
skips a GPU suite whose compiler is missing unless that suite was requested
explicitly. `--sanitize` applies to the CUDA suite only.

Behavioral tests are grouped by component under `tests/`. CPU and GPU entry
points are kept separate in `cpu.cpp` and `cuda.cu`, and HIP builds compile the
`cuda.cu` entry points as HIP. Backend-neutral cases live beside them in a
shared `cases.hpp`. Common test support lives in `tests/support`, umbrella-header
coverage lives in `tests/integration`, and every exported header also gets a
compile-only self-containment check.

GitHub Actions runs the CPU suite on Ubuntu 26.04 with GCC 15. CUDA runtime
testing is enabled when the repository variable `XPU_CUDA_CI` is `true` and a
self-hosted Linux x64 runner with the `gpu` label is available.
self-hosted Linux x64 runner with the `gpu` label is available. HIP runtime
testing works the same way with the `XPU_HIP_CI` variable and a runner with the
`rocm` label.

## Current limitations

- NVIDIA CUDA is the only GPU backend.
- CPU and CUDA backends cannot be mixed in one linked program.
- CUDA memory transfers are not wrapped.
- A build uses one backend. CPU, CUDA and HIP cannot be mixed in one linked
program.
- Random sequences are backend-specific. The same seed need not produce the
same values on the CPU, CUDA and HIP.
- Accessors do not perform bounds checking.
- Multidimensional launch configuration is still a work in progress.
- Installed CMake package metadata is incomplete.
Expand Down
Loading