From 7f4d062ea7bc493c66528b5bdfab0a8cf1e5e719 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 15:04:25 +0200 Subject: [PATCH 01/29] Added cupy-cuda12x to new [gpu] optional dependency --- pyproject.toml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/pyproject.toml b/pyproject.toml index 71b7e2a64..ffcc885d1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -59,6 +59,9 @@ file = "LICENSE" mpi = [ "mpi4py<=4.1.1", ] +gpu = [ + "cupy-cuda12x", +] phys = [ "gvec>=1.1.0, <=1.5.0", "desc-opt<=0.17.1", From 7f85df71ca746001174b7c2523913f00e5611e3e Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 15:21:06 +0200 Subject: [PATCH 02/29] Added CudaKernel, Kernel and KernelCatalog classes --- src/struphy/utils/kernel_backends.py | 174 +++++++++++++++++++++++++++ 1 file changed, 174 insertions(+) create mode 100644 src/struphy/utils/kernel_backends.py diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py new file mode 100644 index 000000000..2e9b8283b --- /dev/null +++ b/src/struphy/utils/kernel_backends.py @@ -0,0 +1,174 @@ +"""Pairs of pyccel and CUDA kernels, selected at runtime from the cunumpy backend. + +Each Struphy kernel has a pyccel version (wrapped in :class:`cunumpy.PyccelKernel`) and, +optionally, a CUDA version (:class:`CudaKernel`) with a 1:1 corresponding signature. +:class:`Kernel` holds both and dispatches to the CUDA kernel when the cunumpy backend is +``"cupy"``, see :func:`is_cuda_backend`. The pyccelized argument classes are transformed +once (at setup) into their CUDA counterparts by :func:`~struphy.utils.kernel_transform.transform`. + +Example +------- +>>> kernel = Kernel( +... pyccel_kernel=PyccelKernel(demo_kernels.push_eta_linear), +... cuda_kernel=CudaKernel(PUSH_ETA_LINEAR_SRC, "push_eta_linear"), +... ) +>>> catalog.register(kernel) +>>> catalog.get("push_eta_linear")(dt, stage, args_markers, args_domain) # NumPy backend +>>> # CuPy backend, transform once at setup: +>>> cuda_markers, cuda_domain = transform(args_markers), transform(args_domain) +>>> catalog.get("push_eta_linear")(dt, stage, cuda_markers, cuda_domain) +""" + +import math + +import numpy as np +from cunumpy import PyccelKernel +from cunumpy.xp import array_backend + +from struphy.utils.kernel_transform import CudaArguments + + +def is_cuda_backend() -> bool: + """Whether the active cunumpy backend is CuPy.""" + return array_backend.backend == "cupy" + + +class CudaKernel: + """Call a ``cupy.RawKernel`` with the 1:1 corresponding arguments of its pyccel counterpart. + + The pyccelized argument classes (``MarkerArguments`` etc.) must be transformed beforehand + (once, at setup) with :func:`~struphy.utils.kernel_transform.transform`; no arrays are + converted or copied at call time. Arrays must be CuPy arrays (``cupy`` raises otherwise). + + Parameters + ---------- + source : str + CUDA C source code containing an ``extern "C" __global__`` function ``name``. + + name : str + Name of the kernel function in ``source``. + + block_size : int + Number of threads per block. + """ + + def __init__(self, source: str, name: str, block_size: int = 128): + self._source = source + self._name = name + self._block_size = block_size + self._raw_kernel = None + + def __repr__(self): + return f"CudaKernel(name={self.name!r}, block_size={self._block_size})" + + @property + def name(self) -> str: + """Name of the CUDA kernel.""" + return self._name + + @property + def raw_kernel(self): + """The compiled ``cupy.RawKernel`` (compiled lazily on first access).""" + if self._raw_kernel is None: + import cupy as cp + + self._raw_kernel = cp.RawKernel(self._source, self._name) + return self._raw_kernel + + def __call__(self, *args): + values = [] + n_threads = None + for arg in args: + if isinstance(arg, CudaArguments): + values += arg.values + if n_threads is None: + n_threads = arg.n_threads + else: + values.append(_cuda_scalar(arg)) + + if n_threads is None: + raise ValueError(f"{self.name}: no argument defines the number of CUDA threads (e.g. CudaMarkerArguments).") + + grid = (max(1, math.ceil(n_threads / self._block_size)),) + self.raw_kernel(grid, (self._block_size,), tuple(values)) + + +def _cuda_scalar(arg): + """Python scalars as NumPy scalars with the C type of the kernel signature (int -> int, float -> double).""" + if isinstance(arg, bool): + return np.bool_(arg) + if isinstance(arg, int): + return np.int32(arg) + if isinstance(arg, float): + return np.float64(arg) + return arg + + +class Kernel: + """A pyccel kernel and its 1:1 corresponding CUDA kernel. + + Parameters + ---------- + pyccel_kernel : PyccelKernel + The pyccel kernel, used on the NumPy backend (and on CuPy if there is no CUDA kernel). + + cuda_kernel : CudaKernel | None + The CUDA kernel, used on the CuPy backend. + """ + + def __init__(self, pyccel_kernel: PyccelKernel, cuda_kernel: CudaKernel | None = None): + assert isinstance(pyccel_kernel, PyccelKernel), f"{pyccel_kernel} is not of type PyccelKernel" + assert cuda_kernel is None or isinstance(cuda_kernel, CudaKernel), f"{cuda_kernel} is not of type CudaKernel" + self._pyccel_kernel = pyccel_kernel + self._cuda_kernel = cuda_kernel + + def __repr__(self): + return f"Kernel(pyccel_kernel={self.pyccel_kernel!r}, cuda_kernel={self.cuda_kernel!r})" + + @property + def pyccel_kernel(self) -> PyccelKernel: + return self._pyccel_kernel + + @property + def cuda_kernel(self) -> CudaKernel | None: + return self._cuda_kernel + + @property + def name(self) -> str: + return self.pyccel_kernel.name + + def get_kernel(self) -> PyccelKernel | CudaKernel: + """The kernel for the active cunumpy backend.""" + if is_cuda_backend() and self.cuda_kernel is not None: + return self.cuda_kernel + return self.pyccel_kernel + + def __call__(self, *args, **kwargs): + return self.get_kernel()(*args, **kwargs) + + +class KernelCatalog: + """Registry of :class:`Kernel` objects by name.""" + + def __init__(self): + self._kernels: dict[str, Kernel] = {} + + def register(self, kernel: Kernel, name: str | None = None) -> Kernel: + """Register ``kernel`` under ``name`` (default: name of the pyccel kernel).""" + name = kernel.name if name is None else name + assert name not in self._kernels, f"Kernel {name!r} is already registered." + self._kernels[name] = kernel + return kernel + + def get(self, name: str) -> Kernel: + return self._kernels[name] + + def __contains__(self, name: str) -> bool: + return name in self._kernels + + @property + def names(self) -> list[str]: + return list(self._kernels) + + +catalog = KernelCatalog() From 865019ddf8492a3fc6ce52ec556196fee8a9ef9c Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 15:21:33 +0200 Subject: [PATCH 03/29] Added Cuda versions of the argument classes --- src/struphy/utils/kernel_transform.py | 155 ++++++++++++++++++++++++++ 1 file changed, 155 insertions(+) create mode 100644 src/struphy/utils/kernel_transform.py diff --git a/src/struphy/utils/kernel_transform.py b/src/struphy/utils/kernel_transform.py new file mode 100644 index 000000000..334a32d15 --- /dev/null +++ b/src/struphy/utils/kernel_transform.py @@ -0,0 +1,155 @@ +"""Transform the pyccelized kernel argument classes into arguments for CUDA kernels. + +Pyccel kernels take the argument classes of :mod:`struphy.kernel_arguments.pusher_args_kernels` +as arguments. A ``cupy.RawKernel`` can only take pointers and scalars, hence :func:`transform` +collects the attributes of such a class into a :class:`CudaArguments` object holding CuPy arrays. + +The transform is meant to be done **once** at setup (not at every kernel call). Arrays that are +already CuPy arrays are used as they are (no copy); NumPy arrays are copied to the device once, +after which the device copy is the one updated by CUDA kernels. + +The field order of each :class:`CudaArguments` subclass defines the corresponding part +of the CUDA kernel signature (``double*`` for float arrays, ``long long*`` for int arrays, +``bool*`` for bool arrays, ``int`` for int scalars). +""" + +from dataclasses import dataclass, fields +from functools import singledispatch +from typing import Any + +import numpy as np + +from struphy.kernel_arguments.pusher_args_kernels import DerhamArguments, DomainArguments, MarkerArguments + + +@dataclass(frozen=True) +class CudaArguments: + """Base class for CUDA kernel arguments; :attr:`values` is the flat tuple passed to the kernel.""" + + def __post_init__(self): + object.__setattr__(self, "_values", tuple(getattr(self, f.name) for f in fields(self))) + + @property + def values(self) -> tuple: + """Flat CUDA kernel arguments, in field order.""" + return self._values + + @property + def n_threads(self) -> int | None: + """Number of CUDA threads needed for this argument, None if no preference.""" + return None + + +@dataclass(frozen=True) +class CudaMarkerArguments(CudaArguments): + """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.MarkerArguments`.""" + + markers: Any + valid_mks: Any + n_markers: np.int32 + n_cols: np.int32 + Np: np.int32 + vdim: np.int32 + weight_idx: np.int32 + first_diagnostics_idx: np.int32 + first_init_idx: np.int32 + first_shift_idx: np.int32 + residual_idx: np.int32 + first_free_idx: np.int32 + mu_idx: np.int32 + bc_type: Any + + @property + def n_threads(self) -> int: + return int(self.n_markers) + + +@dataclass(frozen=True) +class CudaDomainArguments(CudaArguments): + """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments`.""" + + kind_map: np.int32 + params: Any + degree: Any + t1: Any + t2: Any + t3: Any + ind1: Any + ind2: Any + ind3: Any + cx: Any + cy: Any + cz: Any + + +@dataclass(frozen=True) +class CudaDerhamArguments(CudaArguments): + """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DerhamArguments`.""" + + pn: Any + tn1: Any + tn2: Any + tn3: Any + starts: Any + + +def _cupy(arr, dtype): + """C-contiguous CuPy array of ``arr`` (no copy if it already is one).""" + import cupy as cp + + return cp.ascontiguousarray(cp.asarray(arr, dtype=dtype)) + + +@singledispatch +def transform(args: Any) -> CudaArguments: + """Transform a pyccelized kernel argument class into its :class:`CudaArguments` counterpart.""" + raise TypeError(f"No CUDA transform for arguments of type {type(args)}.") + + +@transform.register +def _(args: MarkerArguments) -> CudaMarkerArguments: + return CudaMarkerArguments( + markers=_cupy(args.markers, np.float64), + valid_mks=_cupy(args.valid_mks, np.bool_), + n_markers=np.int32(args.n_markers), + n_cols=np.int32(args.markers.shape[1]), + Np=np.int32(args.Np), + vdim=np.int32(args.vdim), + weight_idx=np.int32(args.weight_idx), + first_diagnostics_idx=np.int32(args.first_diagnostics_idx), + first_init_idx=np.int32(args.first_init_idx), + first_shift_idx=np.int32(args.first_shift_idx), + residual_idx=np.int32(args.residual_idx), + first_free_idx=np.int32(args.first_free_idx), + mu_idx=np.int32(args.mu_idx), + bc_type=_cupy(args.bc_type, np.int64), + ) + + +@transform.register +def _(args: DomainArguments) -> CudaDomainArguments: + return CudaDomainArguments( + kind_map=np.int32(args.kind_map), + params=_cupy(args.params, np.float64), + degree=_cupy(args.degree, np.int64), + t1=_cupy(args.t1, np.float64), + t2=_cupy(args.t2, np.float64), + t3=_cupy(args.t3, np.float64), + ind1=_cupy(args.ind1, np.int64), + ind2=_cupy(args.ind2, np.int64), + ind3=_cupy(args.ind3, np.int64), + cx=_cupy(args.cx, np.float64), + cy=_cupy(args.cy, np.float64), + cz=_cupy(args.cz, np.float64), + ) + + +@transform.register +def _(args: DerhamArguments) -> CudaDerhamArguments: + return CudaDerhamArguments( + pn=_cupy(args.pn, np.int64), + tn1=_cupy(args.tn1, np.float64), + tn2=_cupy(args.tn2, np.float64), + tn3=_cupy(args.tn3, np.float64), + starts=_cupy(args.starts, np.int64), + ) From b2ca9287080b0788f27d349d4da5bed438c2583c Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 15:21:55 +0200 Subject: [PATCH 04/29] Added tests for the cuda kernel class --- src/struphy/pic/tests/test_kernel_backends.py | 192 ++++++++++++++++++ 1 file changed, 192 insertions(+) create mode 100644 src/struphy/pic/tests/test_kernel_backends.py diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py new file mode 100644 index 000000000..44abaa628 --- /dev/null +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -0,0 +1,192 @@ +"""Dispatch between pyccel and CUDA kernels, and transformation of kernel arguments.""" + +import cunumpy +import numpy as np +import pytest +from cunumpy import PyccelKernel + +from struphy.geometry.domains import Cuboid +from struphy.kernel_arguments.pusher_args_kernels import DerhamArguments, MarkerArguments +from struphy.pic.pushing.demo_cuda import push_eta_linear +from struphy.utils.kernel_backends import Kernel, KernelCatalog, catalog, is_cuda_backend +from struphy.utils.kernel_transform import CudaMarkerArguments, _cupy, transform + +requires_cupy = pytest.mark.skipif(not cunumpy.cupy_available(), reason="CuPy/GPU not available") + +BACKENDS = ["numpy", pytest.param("cupy", marks=requires_cupy)] + +N_MARKERS = 1000 +N_COLS = 25 + + +def make_marker_args(seed=0): + rng = np.random.default_rng(seed) + markers = rng.random((N_MARKERS, N_COLS)) + valid_mks = rng.random(N_MARKERS) > 0.1 # some holes/ghosts + return MarkerArguments(markers, valid_mks, N_MARKERS, 3, 6, 7, 8, 14, 17, 18, 4, np.zeros(3, dtype=int)) + + +@pytest.fixture +def domain_args(): + return Cuboid().args_domain + + +def expected_push(markers, valid_mks, dt): + out = markers.copy() + out[valid_mks, 0:3] += dt * markers[valid_mks, 3:6] + return out + + +@pytest.mark.parametrize("backend", BACKENDS) +def test_kernel_dispatch(backend): + with cunumpy.use_backend(backend): + assert is_cuda_backend() == (backend == "cupy") + kernel = push_eta_linear.get_kernel() + if backend == "cupy": + assert kernel is push_eta_linear.cuda_kernel + else: + assert kernel is push_eta_linear.pyccel_kernel + + +def test_kernel_without_cuda_falls_back_to_pyccel(): + kernel = Kernel(push_eta_linear.pyccel_kernel) + for backend in ("numpy", "cupy"): + with cunumpy.use_backend(backend): + assert kernel.get_kernel() is kernel.pyccel_kernel + + +def test_catalog(): + assert catalog.get("push_eta_linear") is push_eta_linear + + local = KernelCatalog() + local.register(push_eta_linear, name="foo") + assert "foo" in local and local.names == ["foo"] + with pytest.raises(AssertionError): + local.register(push_eta_linear, name="foo") + + +def test_push_eta_linear_pyccel(domain_args): + dt = 0.1 + args_markers = make_marker_args() + expected = expected_push(args_markers.markers, args_markers.valid_mks, dt) + + with cunumpy.use_backend("numpy"): + push_eta_linear(dt, 0, args_markers, domain_args) + + assert np.allclose(args_markers.markers, expected, rtol=1e-14, atol=0.0) + + +@requires_cupy +def test_push_eta_linear_cuda(domain_args): + dt = 0.1 + args_markers = make_marker_args() + expected = expected_push(args_markers.markers, args_markers.valid_mks, dt) + + # transform once at setup, the kernel works on the device arrays only + cuda_markers = transform(args_markers) + cuda_domain = transform(domain_args) + + with cunumpy.use_backend("cupy"): + push_eta_linear(dt, 0, cuda_markers, cuda_domain) + + assert np.allclose(cuda_markers.markers.get(), expected, rtol=1e-14, atol=0.0) + + +@requires_cupy +def test_pyccel_cuda_agree(domain_args): + dt = 0.05 + args_pyccel = make_marker_args(seed=1) + cuda_markers = transform(make_marker_args(seed=1)) + cuda_domain = transform(domain_args) + + for _ in range(3): + with cunumpy.use_backend("numpy"): + push_eta_linear(dt, 0, args_pyccel, domain_args) + with cunumpy.use_backend("cupy"): + push_eta_linear(dt, 0, cuda_markers, cuda_domain) + + assert np.allclose(args_pyccel.markers, cuda_markers.markers.get(), rtol=1e-14, atol=0.0) + + +@requires_cupy +def test_cuda_kernel_rejects_untransformed_args(domain_args): + """Host arrays are never moved to the device at call time.""" + with cunumpy.use_backend("cupy"): + with pytest.raises(ValueError, match="number of CUDA threads"): + push_eta_linear(0.1, 0, make_marker_args(), domain_args) + with pytest.raises(TypeError): + push_eta_linear.cuda_kernel(0.1, 0, transform(make_marker_args()), domain_args) + + +@requires_cupy +def test_transform_marker_args(): + import cupy as cp + + args_markers = make_marker_args() + out = transform(args_markers) + + assert isinstance(out, CudaMarkerArguments) + assert len(out.values) == 14 + assert out.n_threads == N_MARKERS + assert isinstance(out.markers, cp.ndarray) and out.markers.flags.c_contiguous + assert out.markers.dtype == np.float64 and out.valid_mks.dtype == np.bool_ + assert np.array_equal(out.markers.get(), args_markers.markers) + assert all(isinstance(v, np.int32) for v in out.values[2:13]) + assert out.n_markers == N_MARKERS and out.n_cols == N_COLS + assert out.bc_type.dtype == np.int64 + with pytest.raises(AttributeError): + out.markers = None # frozen, values stay consistent + + +@requires_cupy +def test_transform_does_not_copy_device_arrays(): + import cupy as cp + + x = cp.zeros((4, 3)) + assert _cupy(x, np.float64) is x + + +@requires_cupy +def test_transform_domain_and_derham_args(domain_args): + import cupy as cp + + out = transform(domain_args) + assert len(out.values) == 12 and out.n_threads is None + assert isinstance(out.kind_map, np.int32) and out.kind_map == domain_args.kind_map + assert all(isinstance(v, cp.ndarray) for v in out.values[1:]) + + knots = np.array([0.0, 0.0, 1.0, 1.0]) + args_derham = DerhamArguments(np.ones(3, dtype=int), knots, knots, knots, np.zeros(3, dtype=int)) + out = transform(args_derham) + assert len(out.values) == 5 + assert out.pn.dtype == np.int64 and out.tn1.dtype == np.float64 + + +def test_transform_unknown_type(): + with pytest.raises(TypeError): + transform(np.zeros(3)) + + +def test_kernel_type_checks(): + with pytest.raises(AssertionError): + Kernel(lambda: None) + with pytest.raises(AssertionError): + Kernel(push_eta_linear.pyccel_kernel, cuda_kernel=PyccelKernel(lambda: None)) + + +@requires_cupy +def test_demo_run_push_eta_linear(): + from struphy.pic.pushing.demo_cuda import make_demo_arguments, run_push_eta_linear + + dt, n_steps = 0.01, 5 + args_markers, args_domain = make_demo_arguments(N_MARKERS, seed=2) + expected = args_markers.markers.copy() + for _ in range(n_steps): + expected = expected_push(expected, args_markers.valid_mks, dt) + + markers_pyccel, _ = run_push_eta_linear("numpy", args_markers, args_domain, dt, n_steps) + args_markers, args_domain = make_demo_arguments(N_MARKERS, seed=2) + markers_cuda, _ = run_push_eta_linear("cupy", args_markers, args_domain, dt, n_steps) + + assert np.allclose(markers_pyccel, expected, rtol=1e-13, atol=0.0) + assert np.allclose(markers_cuda, expected, rtol=1e-13, atol=0.0) From 0fed27a6fceff3e4f1576d1d17d46e39ed37895c Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 15:22:28 +0200 Subject: [PATCH 05/29] Added demo files [skip ci] --- src/struphy/pic/pushing/demo_cuda.py | 142 ++++++++++++++++++++++++ src/struphy/pic/pushing/demo_kernels.py | 35 ++++++ 2 files changed, 177 insertions(+) create mode 100644 src/struphy/pic/pushing/demo_cuda.py create mode 100644 src/struphy/pic/pushing/demo_kernels.py diff --git a/src/struphy/pic/pushing/demo_cuda.py b/src/struphy/pic/pushing/demo_cuda.py new file mode 100644 index 000000000..e26d983dd --- /dev/null +++ b/src/struphy/pic/pushing/demo_cuda.py @@ -0,0 +1,142 @@ +"""CUDA counterpart of :mod:`struphy.pic.pushing.demo_kernels` and the corresponding +:class:`~struphy.utils.kernel_backends.Kernel` registered in the kernel catalog. + +Run as a script to push markers with both backends and compare:: + + python -m struphy.pic.pushing.demo_cuda --n-markers 1000000 --n-steps 100 +""" + +import argparse +import time + +import cunumpy +import numpy as np +from cunumpy import PyccelKernel + +from struphy.geometry.domains import Cuboid +from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments +from struphy.pic.pushing import demo_kernels +from struphy.utils.kernel_backends import CudaKernel, Kernel, catalog, is_cuda_backend +from struphy.utils.kernel_transform import transform + +# Argument order = (dt, stage, CudaMarkerArguments, CudaDomainArguments), +# see struphy.utils.kernel_transform. +PUSH_ETA_LINEAR_SRC = r""" +extern "C" __global__ +void push_eta_linear( + double dt, int stage, + double* markers, bool* valid_mks, int n_markers, int n_cols, + int Np, int vdim, int weight_idx, int first_diagnostics_idx, int first_init_idx, + int first_shift_idx, int residual_idx, int first_free_idx, int mu_idx, long long* bc_type, + int kind_map, double* params, long long* degree, + double* t1, double* t2, double* t3, + long long* ind1, long long* ind2, long long* ind3, + double* cx, double* cy, double* cz) +{ + int ip = blockDim.x * blockIdx.x + threadIdx.x; + + // only do something if particle is valid (i.e. not a hole or ghost) + if (ip >= n_markers || !valid_mks[ip]) return; + + double* mk = markers + (long long)ip * n_cols; + mk[0] += dt * mk[3]; + mk[1] += dt * mk[4]; + mk[2] += dt * mk[5]; +} +""" + +push_eta_linear = catalog.register( + Kernel( + pyccel_kernel=PyccelKernel(demo_kernels.push_eta_linear), + cuda_kernel=CudaKernel(PUSH_ETA_LINEAR_SRC, "push_eta_linear"), + ), +) + + +def run_push_eta_linear( + backend: str, + args_markers: "MarkerArguments", + args_domain: "DomainArguments", + dt: float, + n_steps: int, +): + """Push markers ``n_steps`` times with :data:`push_eta_linear` on the given cunumpy backend. + + On the CuPy backend the argument classes are transformed **once** before the time loop; + inside the loop only the kernel is called (no host-device transfers). + + Returns + ------- + markers : numpy.ndarray + Markers after the last step (copied to the host once, at the end). + + time_per_step : float + Wall-clock time per step in seconds. + """ + with cunumpy.use_backend(backend): + if is_cuda_backend(): + args_markers = transform(args_markers) + args_domain = transform(args_domain) + + # warm-up (compiles the CUDA kernel on first call) + push_eta_linear(0.0, 0, args_markers, args_domain) + cunumpy.synchronize() + + t0 = time.perf_counter() + for _ in range(n_steps): + push_eta_linear(dt, 0, args_markers, args_domain) + cunumpy.synchronize() + time_per_step = (time.perf_counter() - t0) / n_steps + + return cunumpy.to_numpy(args_markers.markers), time_per_step + + +def make_demo_arguments(n_markers: int, seed: int = 0): + """Random markers (positions, velocities, some holes) and a Cuboid domain.""" + rng = np.random.default_rng(seed) + markers = rng.random((n_markers, 25)) + valid_mks = rng.random(n_markers) > 0.1 + args_markers = MarkerArguments( + markers, + valid_mks, + n_markers, + 3, + 6, + 7, + 8, + 14, + 17, + 18, + 4, + np.zeros(3, dtype=int), + ) + return args_markers, Cuboid().args_domain + + +def main(): + parser = argparse.ArgumentParser( + description="Push markers with the pyccel and the CUDA version of push_eta_linear." + ) + parser.add_argument("--n-markers", type=int, default=1_000_000) + parser.add_argument("--n-steps", type=int, default=100) + parser.add_argument("--dt", type=float, default=1e-3) + args = parser.parse_args() + + backends = ["numpy"] + (["cupy"] if cunumpy.cupy_available() else []) + results = {} + for backend in backends: + args_markers, args_domain = make_demo_arguments(args.n_markers) + results[backend] = run_push_eta_linear(backend, args_markers, args_domain, args.dt, args.n_steps) + print( + f"{backend:>5}: {results[backend][1] * 1e3:8.3f} ms/step ({args.n_markers} markers, {args.n_steps} steps)" + ) + + if "cupy" in results: + max_diff = np.max(np.abs(results["numpy"][0] - results["cupy"][0])) + print(f"max |pyccel - cuda| = {max_diff:.2e}, speed-up = {results['numpy'][1] / results['cupy'][1]:.1f}x") + else: + print("CuPy/GPU not available, only the pyccel kernel was run.") + + +if __name__ == "__main__": + main() diff --git a/src/struphy/pic/pushing/demo_kernels.py b/src/struphy/pic/pushing/demo_kernels.py new file mode 100644 index 000000000..419a558e6 --- /dev/null +++ b/src/struphy/pic/pushing/demo_kernels.py @@ -0,0 +1,35 @@ +"""Minimal pusher kernel with a 1:1 CUDA counterpart in :mod:`struphy.pic.pushing.demo_cuda`, +used to test the pyccel/CUDA kernel dispatch in :mod:`struphy.utils.kernel_backends`.""" + +# do not remove; needed to identify dependencies +import struphy.kernel_arguments.pusher_args_kernels as pusher_args_kernels +from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments + + +def push_eta_linear( + dt: float, + stage: int, + args_markers: "MarkerArguments", + args_domain: "DomainArguments", +): + r"""Explicit Euler step of + + .. math:: + + \frac{\textnormal d \boldsymbol \eta_p(t)}{\textnormal d t} = \mathbf v_p + + for each valid marker :math:`p`, where :math:`\mathbf v_p` is constant (no mapping, no boundary conditions). + """ + + markers = args_markers.markers + n_markers = args_markers.n_markers + valid_mks = args_markers.valid_mks + + for ip in range(n_markers): + # only do something if particle is valid (i.e. not a hole or ghost) + if not valid_mks[ip]: + continue + + markers[ip, 0] += dt * markers[ip, 3] + markers[ip, 1] += dt * markers[ip, 4] + markers[ip, 2] += dt * markers[ip, 5] From 606f926492ad15840833fd2b6ff085ea1a4fb23d Mon Sep 17 00:00:00 2001 From: Max Date: Wed, 30 Sep 2026 15:26:47 +0200 Subject: [PATCH 06/29] Make SPH linear kernel gradients vanish at r=0 (#475) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The SPH linear smoothing kernels returned a non-zero gradient at zero separation, so every particle pushed on itself. This PR makes all SPH kernel gradients vanish at r=0, which matches the symmetric value the trigonometric and gaussian kernels already give. Closes #437 ### What was wrong - `grad_linear_uni(0, h)` returned `+1/h**2`, the left derivative. That value carries into `grad_linear_1d`, `grad_linear_2d_{1,2}` and `grad_linear_3d_{1,2,3}`, and it applies whenever the component's own coordinate is 0, not only at the origin. - `grad_linear_isotropic_3d_{1,2,3}` returned `-1/h / (C h^3)` at `r == 0`. - As a result the SPH gradient of a lattice-loaded constant density was not zero: it came out at about 3.5 for 1d linear, 3.8 for 3d tensor linear and -0.58 for isotropic in the check below. ### What changed - `src/struphy/pic/sph_smoothing_kernels.py`: at the cusp the gradient now returns 0 (`x == 0` in `grad_linear_uni`, `r == 0` in the isotropic variants). Values away from 0 are unchanged. The trigonometric and gaussian gradients were checked too and already give 0 there. - `src/struphy/pic/tests/test_kernel_setup.py`: - New test `test_sph_kernel_gradients_vanish_at_zero` covers every gradient type in `smoothing_kernel`, at the origin and on the coordinate planes. - `test_sph_tensor_destinations` relied on the old self-gradient `1/h**2`. It now uses a second neighbouring particle so the viscosity-tensor destinations still get non-trivial values. ### Verification - A standalone check calls all 21 gradient kernel types at the origin and on the coordinate planes. Before the fix, 9 types were non-zero at the origin and all 4 plane cases were non-zero. After the fix all of them are 0. At random points away from 0 the values are bit-identical before and after. - The gradient of a lattice-loaded constant density is now ~1e-16 in 1d linear, 3d tensor linear and 3d isotropic. - `pytest src/struphy/pic/tests/test_kernel_setup.py`: 74 passed. The new test fails without the fix. - A subset of `test_sph.py::test_sph_evaluation_1d` (linear_1d, periodic) passes. ### Risks / follow-ups - SPH results that use the linear kernels change slightly, because the self-contribution is gone. That removal is the intended correction. - #436 (viscosity strain without DF^-T) is in the same area and is not addressed here. - This branch includes the commit bumping `feectools` to 0.1.11 so that CI can run. ๐Ÿค– Generated with [Claude Code](https://claude.com/claude-code) --------- Co-authored-by: Claude Opus 5.5 --- src/struphy/pic/sph_smoothing_kernels.py | 14 +++++++--- src/struphy/pic/tests/test_kernel_setup.py | 30 +++++++++++++++++++--- 2 files changed, 36 insertions(+), 8 deletions(-) diff --git a/src/struphy/pic/sph_smoothing_kernels.py b/src/struphy/pic/sph_smoothing_kernels.py index 28d7ad0af..3cdb3acb8 100644 --- a/src/struphy/pic/sph_smoothing_kernels.py +++ b/src/struphy/pic/sph_smoothing_kernels.py @@ -67,8 +67,11 @@ def grad_linear_uni( if abs(x / h) <= 1.0: if x > 0.0: return -(1 / h**2) - else: + elif x < 0.0: return 1 / h**2 + else: + # kink at x=0: use the symmetric value 0 (no self-force) + return 0.0 else: return 0.0 @@ -443,7 +446,8 @@ def grad_linear_isotropic_3d_1( if r / h > 1.0: return 0.0 elif r == 0.0: - return -1 / h / (1.0471975512 * h**3) + # cusp at r=0: use the symmetric value 0 (no self-force) + return 0.0 else: return -r1 / (r * h) / (1.0471975512 * h**3) @@ -465,7 +469,8 @@ def grad_linear_isotropic_3d_2( if r / h > 1.0: return 0.0 elif r == 0.0: - return -1 / h / (1.0471975512 * h**3) + # cusp at r=0: use the symmetric value 0 (no self-force) + return 0.0 else: return -r2 / (r * h) / (1.0471975512 * h**3) @@ -487,7 +492,8 @@ def grad_linear_isotropic_3d_3( if r / h > 1.0: return 0.0 elif r == 0.0: - return -1 / h / (1.0471975512 * h**3) + # cusp at r=0: use the symmetric value 0 (no self-force) + return 0.0 else: return -r3 / (r * h) / (1.0471975512 * h**3) diff --git a/src/struphy/pic/tests/test_kernel_setup.py b/src/struphy/pic/tests/test_kernel_setup.py index 82315ad6e..43b573ff5 100644 --- a/src/struphy/pic/tests/test_kernel_setup.py +++ b/src/struphy/pic/tests/test_kernel_setup.py @@ -8,6 +8,7 @@ from struphy.geometry.domains import Cuboid from struphy.kernel_arguments.pusher_args_kernels import DerhamArguments, MarkerArguments +from struphy.pic import sph_smoothing_kernels from struphy.pic.pushing import eval_kernels_gc, eval_kernels_sph from struphy.pic.pushing.kernel_setup import KernelSetup from struphy.pic.pushing.pusher import Pusher @@ -88,12 +89,17 @@ def test_sph_vector_destinations(marker_args, kernel): def test_sph_tensor_destinations(marker_args): - marker_args.markers[0, 18:21] = (1.0, 2.0, 3.0) - boxes = np.array([[0, -1], [-1, -1]]) + # Second particle at eta1 + 0.2 (inside h = 0.5) with the same velocity coefficients. + marker_args.markers[1] = marker_args.markers[0] + marker_args.markers[1, 0] = 0.6 + marker_args.markers[:, 18:21] = (1.0, 2.0, 3.0) + marker_args.valid_mks[1] = True + boxes = np.array([[0, 1, -1], [-1, -1, -1]]) neighbours = np.ones((1, 27), dtype=int) neighbours[0, 0] = 0 - # The linear 1D kernel uses its right derivative at zero: 1 / h**2. + # Self-gradient of the linear 1D kernel vanishes; the neighbour contributes -+1 / h**2 = -+4. gradient = np.array([[4.0, 0.0, 0.0], [8.0, 0.0, 0.0], [12.0, 0.0, 0.0]]) + density = 2.0 * (2.0 + 1.2) # weight * (W(0) + W(0.2)) indices = (30, 29, 28, 27, None, 25, 24, 23, 22) setup = KernelSetup( kernel=eval_kernels_sph.sph_viscosity_tensor, @@ -103,14 +109,30 @@ def test_sph_tensor_destinations(marker_args): before = marker_args.markers.copy() setup.evaluate(marker_args, Cuboid().args_domain) symmetric = (gradient + gradient.T) / 2.0 - tensor = -3.0 * (symmetric - np.eye(3) * np.trace(symmetric) / 3.0) + tensor = -2.0 * 3.0 * (2.0 / density) * (symmetric - np.eye(3) * np.trace(symmetric) / 3.0) reference = before.copy() for value, index in zip(tensor.flat, indices): if index is not None: reference[0, index] = value + reference[1, index] = -value np.testing.assert_allclose(marker_args.markers, reference) +# Each gradient component must vanish where its own coordinate is zero (no self-force, issue #437). +ZERO_GRADIENT_CASES = [ + (n + k, point) + for n in (100, 110, 120, 340, 350, 360, 670, 680, 690, 700) + for k in (1, 2, 3) + for point in ((0.0, 0.0, 0.0), (0.0, 0.1, 0.05), (0.1, 0.0, 0.05), (0.1, 0.05, 0.0)) + if point[k - 1] == 0.0 and (n != 690 or not any(point)) +] + + +@pytest.mark.parametrize("kernel_type,point", ZERO_GRADIENT_CASES) +def test_sph_kernel_gradients_vanish_at_zero(kernel_type, point): + assert sph_smoothing_kernels.smoothing_kernel(kernel_type, *point, 0.3, 0.25, 0.2) == 0.0 + + @pytest.mark.parametrize("indices", [(), (1, 2), (-1,), (True,), (1.5,), (None,), (1, 1, 2)]) def test_invalid_destinations(indices): with pytest.raises(ValueError): From 3f86e938cdb8412033f486901a38303b3dcd3cb3 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 15:41:12 +0200 Subject: [PATCH 07/29] Removed transform() --- src/struphy/geometry/base.py | 26 +- src/struphy/pic/base.py | 24 +- src/struphy/pic/pushing/demo_cuda.py | 99 ++++--- src/struphy/pic/tests/test_kernel_backends.py | 257 ++++++++++-------- src/struphy/utils/cuda_arguments.py | 157 +++++++++++ src/struphy/utils/kernel_backends.py | 20 +- src/struphy/utils/kernel_transform.py | 155 ----------- 7 files changed, 412 insertions(+), 326 deletions(-) create mode 100644 src/struphy/utils/cuda_arguments.py delete mode 100644 src/struphy/utils/kernel_transform.py diff --git a/src/struphy/geometry/base.py b/src/struphy/geometry/base.py index d162c2f11..b1a355313 100644 --- a/src/struphy/geometry/base.py +++ b/src/struphy/geometry/base.py @@ -16,6 +16,7 @@ from struphy.geometry import evaluation_kernels, transform_kernels from struphy.kernel_arguments.pusher_args_kernels import DomainArguments from struphy.linear_algebra import linalg_kron +from struphy.utils.cuda_arguments import CudaDomainArguments from struphy.utils.docstring_converter import rst_to_html, rst_to_latex, rst_to_markdown from struphy.utils.ipython_compat import HTML, display from struphy.utils.utils import __class_with_params_repr_no_defaults__, all_class_params_are_default, all_subclasses @@ -238,7 +239,7 @@ def _build_args_domain(self): """Build runtime mapping arguments used by compiled evaluation kernels.""" return DomainArguments( self.kind_map, - self.params_numpy, + _to_numpy_for_kernel(self.params_numpy), _to_numpy_for_kernel(xp.array(self.degree)), _to_numpy_for_kernel(self.T[0]), _to_numpy_for_kernel(self.T[1]), @@ -275,7 +276,7 @@ def __deepcopy__(self, memo): memo[id(self)] = result for key, value in self.__dict__.items(): - if key == "_args_domain": + if key in ("_args_domain", "_cuda_args_domain"): continue setattr(result, key, copy.deepcopy(value, memo)) @@ -285,6 +286,7 @@ def __deepcopy__(self, memo): def __getstate__(self): state = self.__dict__.copy() state.pop("_args_domain", None) + state.pop("_cuda_args_domain", None) return state def __setstate__(self, state): @@ -475,6 +477,26 @@ def args_domain(self): return self._args_domain + @property + def cuda_args_domain(self) -> CudaDomainArguments: + """CUDA version of :attr:`args_domain`, referencing the device arrays of the domain (CuPy backend only).""" + if getattr(self, "_cuda_args_domain", None) is None: + self._cuda_args_domain = CudaDomainArguments( + self.kind_map, + self.params_numpy, + xp.array(self.degree), + self.T[0], + self.T[1], + self.T[2], + self.indN[0], + self.indN[1], + self.indN[2], + xp.ascontiguousarray(self.cx), + xp.ascontiguousarray(self.cy), + xp.ascontiguousarray(self.cz), + ) + return self._cuda_args_domain + @property def dict_transformations(self): """Dictionary of str->int for pull, push and transformation functions.""" diff --git a/src/struphy/pic/base.py b/src/struphy/pic/base.py index da7177c1d..3d2531939 100644 --- a/src/struphy/pic/base.py +++ b/src/struphy/pic/base.py @@ -56,6 +56,7 @@ ) from struphy.utils import utils from struphy.utils.clone_config import CloneConfig +from struphy.utils.cuda_arguments import CudaMarkerArguments if TYPE_CHECKING: # importing mpi4py.MPI initializes MPI, which is slow; only needed for annotations from mpi4py.MPI import Intracomm @@ -940,6 +941,26 @@ def args_markers(self) -> MarkerArguments: """Collection of mandatory arguments for pusher kernels.""" return self._args_markers + @property + def cuda_args_markers(self) -> CudaMarkerArguments: + """CUDA version of :attr:`args_markers`, referencing the device arrays of the markers (CuPy backend only).""" + if getattr(self, "_cuda_args_markers", None) is None: + self._cuda_args_markers = CudaMarkerArguments( + self.markers, + self.valid_mks, + self.Np, + self.vdim, + self.index["weights"], + self.first_diagnostics_idx, + self.first_pusher_idx, + self.first_shift_idx, + self.residual_idx, + self.first_free_idx, + self.mu_idx, + self._bc_type, + ) + return self._cuda_args_markers + # ------------------------------------------- # Initial condition and background -> weights # ------------------------------------------- @@ -2527,7 +2548,8 @@ def _allocate_marker_array(self, dry_run: bool = False): self._n_lost_markers = 0 self._lost_markers = xp.zeros((int(self.n_rows * 0.5), 10), dtype=float) - # arguments for kernels + # arguments for kernels (the CUDA version is built on first access, see cuda_args_markers) + self._cuda_args_markers = None self._args_markers = MarkerArguments( _to_numpy_for_kernel(self.markers), _to_numpy_for_kernel(self.valid_mks), diff --git a/src/struphy/pic/pushing/demo_cuda.py b/src/struphy/pic/pushing/demo_cuda.py index e26d983dd..a02d37439 100644 --- a/src/struphy/pic/pushing/demo_cuda.py +++ b/src/struphy/pic/pushing/demo_cuda.py @@ -14,13 +14,13 @@ from cunumpy import PyccelKernel from struphy.geometry.domains import Cuboid -from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments +from struphy.kernel_arguments.pusher_args_kernels import MarkerArguments from struphy.pic.pushing import demo_kernels +from struphy.utils.cuda_arguments import CudaMarkerArguments from struphy.utils.kernel_backends import CudaKernel, Kernel, catalog, is_cuda_backend -from struphy.utils.kernel_transform import transform -# Argument order = (dt, stage, CudaMarkerArguments, CudaDomainArguments), -# see struphy.utils.kernel_transform. +# Argument order = (dt, stage, CudaMarkerArguments.values, CudaDomainArguments.values), +# see struphy.utils.cuda_arguments. PUSH_ETA_LINEAR_SRC = r""" extern "C" __global__ void push_eta_linear( @@ -53,17 +53,40 @@ ) -def run_push_eta_linear( - backend: str, - args_markers: "MarkerArguments", - args_domain: "DomainArguments", - dt: float, - n_steps: int, -): - """Push markers ``n_steps`` times with :data:`push_eta_linear` on the given cunumpy backend. +def make_demo_arguments(n_markers: int, seed: int = 0): + """Random markers (positions, velocities, some holes) and a Cuboid domain on the active cunumpy backend. - On the CuPy backend the argument classes are transformed **once** before the time loop; - inside the loop only the kernel is called (no host-device transfers). + The marker arrays are created on the active backend (on the device for CuPy), as ``Particles`` does; + the kernel arguments reference them without copies. + + Returns + ------- + args_markers : MarkerArguments | CudaMarkerArguments + Marker arguments for the kernel of the active backend. + + args_domain : DomainArguments | CudaDomainArguments + Domain arguments for the kernel of the active backend. + """ + # same random numbers on both backends, such that results can be compared + rng = np.random.default_rng(seed) + markers = cunumpy.asarray(rng.random((n_markers, 25))) + valid_mks = cunumpy.asarray(rng.random(n_markers) > 0.1) + bc_type = cunumpy.zeros(3, dtype=int) + + domain = Cuboid() + if is_cuda_backend(): + args_markers = CudaMarkerArguments(markers, valid_mks, n_markers, 3, 6, 7, 8, 14, 17, 18, 4, bc_type) + args_domain = domain.cuda_args_domain + else: + args_markers = MarkerArguments(markers, valid_mks, n_markers, 3, 6, 7, 8, 14, 17, 18, 4, bc_type) + args_domain = domain.args_domain + return args_markers, args_domain + + +def run_push_eta_linear(args_markers, args_domain, dt: float, n_steps: int): + """Push markers ``n_steps`` times with :data:`push_eta_linear` on the active cunumpy backend. + + Inside the time loop only the kernel is called; no arrays are converted or copied. Returns ------- @@ -73,46 +96,19 @@ def run_push_eta_linear( time_per_step : float Wall-clock time per step in seconds. """ - with cunumpy.use_backend(backend): - if is_cuda_backend(): - args_markers = transform(args_markers) - args_domain = transform(args_domain) + # warm-up (compiles the CUDA kernel on first call) + push_eta_linear(0.0, 0, args_markers, args_domain) + cunumpy.synchronize() - # warm-up (compiles the CUDA kernel on first call) - push_eta_linear(0.0, 0, args_markers, args_domain) - cunumpy.synchronize() - - t0 = time.perf_counter() - for _ in range(n_steps): - push_eta_linear(dt, 0, args_markers, args_domain) - cunumpy.synchronize() - time_per_step = (time.perf_counter() - t0) / n_steps + t0 = time.perf_counter() + for _ in range(n_steps): + push_eta_linear(dt, 0, args_markers, args_domain) + cunumpy.synchronize() + time_per_step = (time.perf_counter() - t0) / n_steps return cunumpy.to_numpy(args_markers.markers), time_per_step -def make_demo_arguments(n_markers: int, seed: int = 0): - """Random markers (positions, velocities, some holes) and a Cuboid domain.""" - rng = np.random.default_rng(seed) - markers = rng.random((n_markers, 25)) - valid_mks = rng.random(n_markers) > 0.1 - args_markers = MarkerArguments( - markers, - valid_mks, - n_markers, - 3, - 6, - 7, - 8, - 14, - 17, - 18, - 4, - np.zeros(3, dtype=int), - ) - return args_markers, Cuboid().args_domain - - def main(): parser = argparse.ArgumentParser( description="Push markers with the pyccel and the CUDA version of push_eta_linear." @@ -125,8 +121,9 @@ def main(): backends = ["numpy"] + (["cupy"] if cunumpy.cupy_available() else []) results = {} for backend in backends: - args_markers, args_domain = make_demo_arguments(args.n_markers) - results[backend] = run_push_eta_linear(backend, args_markers, args_domain, args.dt, args.n_steps) + with cunumpy.use_backend(backend): + args_markers, args_domain = make_demo_arguments(args.n_markers) + results[backend] = run_push_eta_linear(args_markers, args_domain, args.dt, args.n_steps) print( f"{backend:>5}: {results[backend][1] * 1e3:8.3f} ms/step ({args.n_markers} markers, {args.n_steps} steps)" ) diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index 44abaa628..0b928055f 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -1,4 +1,6 @@ -"""Dispatch between pyccel and CUDA kernels, and transformation of kernel arguments.""" +"""Dispatch between pyccel and CUDA kernels, and the CUDA versions of the kernel argument classes.""" + +import copy import cunumpy import numpy as np @@ -6,10 +8,9 @@ from cunumpy import PyccelKernel from struphy.geometry.domains import Cuboid -from struphy.kernel_arguments.pusher_args_kernels import DerhamArguments, MarkerArguments -from struphy.pic.pushing.demo_cuda import push_eta_linear +from struphy.pic.pushing.demo_cuda import make_demo_arguments, push_eta_linear, run_push_eta_linear +from struphy.utils.cuda_arguments import CudaDerhamArguments, CudaMarkerArguments from struphy.utils.kernel_backends import Kernel, KernelCatalog, catalog, is_cuda_backend -from struphy.utils.kernel_transform import CudaMarkerArguments, _cupy, transform requires_cupy = pytest.mark.skipif(not cunumpy.cupy_available(), reason="CuPy/GPU not available") @@ -17,24 +18,25 @@ N_MARKERS = 1000 N_COLS = 25 +MARKER_INDICES = (N_MARKERS, 3, 6, 7, 8, 14, 17, 18, 4) # Np, vdim, weight_idx, ..., mu_idx -def make_marker_args(seed=0): - rng = np.random.default_rng(seed) - markers = rng.random((N_MARKERS, N_COLS)) - valid_mks = rng.random(N_MARKERS) > 0.1 # some holes/ghosts - return MarkerArguments(markers, valid_mks, N_MARKERS, 3, 6, 7, 8, 14, 17, 18, 4, np.zeros(3, dtype=int)) +def expected_push(markers, valid_mks, dt, n_steps=1): + out = markers.copy() + for _ in range(n_steps): + out[valid_mks, 0:3] += dt * out[valid_mks, 3:6] + return out -@pytest.fixture -def domain_args(): - return Cuboid().args_domain +def device_marker_arrays(): + import cupy as cp + return cp.random.random((N_MARKERS, N_COLS)), cp.ones(N_MARKERS, dtype=bool), cp.zeros(3, dtype=int) -def expected_push(markers, valid_mks, dt): - out = markers.copy() - out[valid_mks, 0:3] += dt * markers[valid_mks, 3:6] - return out + +# --------------------------- +# kernel dispatch and catalog +# --------------------------- @pytest.mark.parametrize("backend", BACKENDS) @@ -65,128 +67,169 @@ def test_catalog(): local.register(push_eta_linear, name="foo") -def test_push_eta_linear_pyccel(domain_args): - dt = 0.1 - args_markers = make_marker_args() - expected = expected_push(args_markers.markers, args_markers.valid_mks, dt) +def test_kernel_type_checks(): + with pytest.raises(AssertionError): + Kernel(lambda: None) + with pytest.raises(AssertionError): + Kernel(push_eta_linear.pyccel_kernel, cuda_kernel=PyccelKernel(lambda: None)) - with cunumpy.use_backend("numpy"): - push_eta_linear(dt, 0, args_markers, domain_args) - assert np.allclose(args_markers.markers, expected, rtol=1e-14, atol=0.0) +# ------------------------ +# CUDA argument classes +# ------------------------ @requires_cupy -def test_push_eta_linear_cuda(domain_args): - dt = 0.1 - args_markers = make_marker_args() - expected = expected_push(args_markers.markers, args_markers.valid_mks, dt) +def test_cuda_marker_args_reference_device_arrays(): + """The CUDA arguments hold the very same device arrays (no copies).""" + markers, valid_mks, bc_type = device_marker_arrays() + args = CudaMarkerArguments(markers, valid_mks, *MARKER_INDICES, bc_type) - # transform once at setup, the kernel works on the device arrays only - cuda_markers = transform(args_markers) - cuda_domain = transform(domain_args) + assert args.markers is markers and args.valid_mks is valid_mks and args.bc_type is bc_type + assert args.n_threads == N_MARKERS + assert len(args.values) == 14 + assert args.values[0] is markers + assert all(isinstance(v, np.int32) for v in args.values[2:13]) + assert args.n_markers == N_MARKERS and args.n_cols == N_COLS + assert args.first_init_idx == 8 - with cunumpy.use_backend("cupy"): - push_eta_linear(dt, 0, cuda_markers, cuda_domain) - - assert np.allclose(cuda_markers.markers.get(), expected, rtol=1e-14, atol=0.0) + with pytest.raises(AttributeError, match="read-only"): + args.markers = markers @requires_cupy -def test_pyccel_cuda_agree(domain_args): - dt = 0.05 - args_pyccel = make_marker_args(seed=1) - cuda_markers = transform(make_marker_args(seed=1)) - cuda_domain = transform(domain_args) - - for _ in range(3): - with cunumpy.use_backend("numpy"): - push_eta_linear(dt, 0, args_pyccel, domain_args) - with cunumpy.use_backend("cupy"): - push_eta_linear(dt, 0, cuda_markers, cuda_domain) +def test_cuda_marker_args_reject_host_and_bad_arrays(): + """Host arrays are never copied to the device; wrong dtypes or layouts are not converted.""" + markers, valid_mks, bc_type = device_marker_arrays() - assert np.allclose(args_pyccel.markers, cuda_markers.markers.get(), rtol=1e-14, atol=0.0) + with pytest.raises(TypeError, match="must be a CuPy array"): + CudaMarkerArguments(markers.get(), valid_mks, *MARKER_INDICES, bc_type) + with pytest.raises(TypeError, match="must be a CuPy array"): + CudaMarkerArguments(markers, valid_mks, *MARKER_INDICES, bc_type.get()) + with pytest.raises(TypeError, match="dtype"): + CudaMarkerArguments(markers.astype(np.float32), valid_mks, *MARKER_INDICES, bc_type) + with pytest.raises(ValueError, match="C-contiguous"): + CudaMarkerArguments(markers[:, ::2], valid_mks, *MARKER_INDICES, bc_type) @requires_cupy -def test_cuda_kernel_rejects_untransformed_args(domain_args): - """Host arrays are never moved to the device at call time.""" - with cunumpy.use_backend("cupy"): - with pytest.raises(ValueError, match="number of CUDA threads"): - push_eta_linear(0.1, 0, make_marker_args(), domain_args) - with pytest.raises(TypeError): - push_eta_linear.cuda_kernel(0.1, 0, transform(make_marker_args()), domain_args) - - -@requires_cupy -def test_transform_marker_args(): +def test_cuda_derham_args(): import cupy as cp - args_markers = make_marker_args() - out = transform(args_markers) - - assert isinstance(out, CudaMarkerArguments) - assert len(out.values) == 14 - assert out.n_threads == N_MARKERS - assert isinstance(out.markers, cp.ndarray) and out.markers.flags.c_contiguous - assert out.markers.dtype == np.float64 and out.valid_mks.dtype == np.bool_ - assert np.array_equal(out.markers.get(), args_markers.markers) - assert all(isinstance(v, np.int32) for v in out.values[2:13]) - assert out.n_markers == N_MARKERS and out.n_cols == N_COLS - assert out.bc_type.dtype == np.int64 - with pytest.raises(AttributeError): - out.markers = None # frozen, values stay consistent + knots = cp.array([0.0, 0.0, 1.0, 1.0]) + args = CudaDerhamArguments(cp.ones(3, dtype=int), knots, knots, knots, cp.zeros(3, dtype=int)) + assert len(args.values) == 5 and args.tn1 is knots @requires_cupy -def test_transform_does_not_copy_device_arrays(): - import cupy as cp +def test_domain_cuda_args(): + """Domain.cuda_args_domain references the device arrays of the domain.""" + with cunumpy.use_backend("cupy"): + domain = Cuboid(l1=0.5, r1=2.0) + args = domain.cuda_args_domain - x = cp.zeros((4, 3)) - assert _cupy(x, np.float64) is x + assert domain.cuda_args_domain is args # cached + assert args.t1 is domain.T[0] and args.ind3 is domain.indN[2] + assert args.params is domain.params_numpy + assert args.kind_map == domain.kind_map + assert len(args.values) == 12 + # not deep-copied/pickled along with the domain, rebuilt from the copy's own arrays + domain_copy = copy.deepcopy(domain) + assert domain_copy.cuda_args_domain.t1 is domain_copy.T[0] -@requires_cupy -def test_transform_domain_and_derham_args(domain_args): - import cupy as cp + # a domain created on the NumPy backend has host arrays + with pytest.raises(TypeError, match="must be a CuPy array"): + Cuboid().cuda_args_domain - out = transform(domain_args) - assert len(out.values) == 12 and out.n_threads is None - assert isinstance(out.kind_map, np.int32) and out.kind_map == domain_args.kind_map - assert all(isinstance(v, cp.ndarray) for v in out.values[1:]) - knots = np.array([0.0, 0.0, 1.0, 1.0]) - args_derham = DerhamArguments(np.ones(3, dtype=int), knots, knots, knots, np.zeros(3, dtype=int)) - out = transform(args_derham) - assert len(out.values) == 5 - assert out.pn.dtype == np.int64 and out.tn1.dtype == np.float64 +@requires_cupy +def test_particles_cuda_args_markers(): + """Particles.cuda_args_markers references the marker arrays of the particles. + Particles cannot be created on the CuPy backend yet, hence the arrays of a NumPy-backend + instance are replaced by device arrays here. + """ + import cupy as cp + from feectools.ddm.mpi import mpi as MPI + + from struphy import LoadingParameters + from struphy.pic.particles import Particles6D + + loading_params = LoadingParameters(Np=100, seed=1234, moments=(0.0, 0.0, 0.0, 1.0, 1.0, 1.0), spatial="uniform") + particles = Particles6D(comm_world=MPI.COMM_WORLD, loading_params=loading_params, domain=Cuboid()) + particles.draw_markers() + + host_args = particles.args_markers + particles._markers = cp.asarray(particles.markers) + particles._valid_mks = cp.asarray(particles.valid_mks) + particles._bc_type = cp.asarray(particles._bc_type) + particles._cuda_args_markers = None + + args = particles.cuda_args_markers + assert particles.cuda_args_markers is args # cached + assert args.markers is particles.markers + assert args.valid_mks is particles.valid_mks + assert args.bc_type is particles._bc_type + for name in ( + "n_markers", + "Np", + "vdim", + "weight_idx", + "first_diagnostics_idx", + "first_init_idx", + "first_shift_idx", + "residual_idx", + "first_free_idx", + "mu_idx", + ): + assert getattr(args, name) == getattr(host_args, name), name + + +# ------------------------ +# pushing +# ------------------------ -def test_transform_unknown_type(): - with pytest.raises(TypeError): - transform(np.zeros(3)) +@pytest.mark.parametrize("backend", BACKENDS) +def test_push_eta_linear(backend): + """The kernel updates the owner's marker array in place.""" + dt = 0.1 + with cunumpy.use_backend(backend): + args_markers, args_domain = make_demo_arguments(N_MARKERS) + markers = args_markers.markers + expected = expected_push(cunumpy.to_numpy(markers), cunumpy.to_numpy(args_markers.valid_mks), dt) -def test_kernel_type_checks(): - with pytest.raises(AssertionError): - Kernel(lambda: None) - with pytest.raises(AssertionError): - Kernel(push_eta_linear.pyccel_kernel, cuda_kernel=PyccelKernel(lambda: None)) + push_eta_linear(dt, 0, args_markers, args_domain) + + if backend == "cupy": + assert args_markers.markers is markers + assert np.allclose(cunumpy.to_numpy(markers), expected, rtol=1e-14, atol=0.0) @requires_cupy -def test_demo_run_push_eta_linear(): - from struphy.pic.pushing.demo_cuda import make_demo_arguments, run_push_eta_linear - +def test_pyccel_cuda_agree(): dt, n_steps = 0.01, 5 - args_markers, args_domain = make_demo_arguments(N_MARKERS, seed=2) - expected = args_markers.markers.copy() - for _ in range(n_steps): - expected = expected_push(expected, args_markers.valid_mks, dt) + results = {} + for backend in ("numpy", "cupy"): + with cunumpy.use_backend(backend): + args_markers, args_domain = make_demo_arguments(N_MARKERS, seed=2) + if backend == "numpy": + expected = expected_push(args_markers.markers, args_markers.valid_mks, dt, n_steps) + results[backend], _ = run_push_eta_linear(args_markers, args_domain, dt, n_steps) - markers_pyccel, _ = run_push_eta_linear("numpy", args_markers, args_domain, dt, n_steps) - args_markers, args_domain = make_demo_arguments(N_MARKERS, seed=2) - markers_cuda, _ = run_push_eta_linear("cupy", args_markers, args_domain, dt, n_steps) + assert np.allclose(results["numpy"], expected, rtol=1e-13, atol=0.0) + assert np.allclose(results["cupy"], results["numpy"], rtol=1e-14, atol=0.0) - assert np.allclose(markers_pyccel, expected, rtol=1e-13, atol=0.0) - assert np.allclose(markers_cuda, expected, rtol=1e-13, atol=0.0) + +@requires_cupy +def test_cuda_kernel_rejects_pyccel_args(): + """Passing the pyccel argument classes to the CUDA kernel fails instead of copying.""" + with cunumpy.use_backend("numpy"): + host_markers, host_domain = make_demo_arguments(N_MARKERS) + with cunumpy.use_backend("cupy"): + cuda_markers, cuda_domain = make_demo_arguments(N_MARKERS) + with pytest.raises(ValueError, match="number of CUDA threads"): + push_eta_linear(0.1, 0, host_markers, cuda_domain) + with pytest.raises(TypeError): + push_eta_linear(0.1, 0, cuda_markers, host_domain) diff --git a/src/struphy/utils/cuda_arguments.py b/src/struphy/utils/cuda_arguments.py new file mode 100644 index 000000000..942381b15 --- /dev/null +++ b/src/struphy/utils/cuda_arguments.py @@ -0,0 +1,157 @@ +"""CUDA counterparts of the pyccelized kernel argument classes. + +The classes in :mod:`struphy.kernel_arguments.pusher_args_kernels` hold references to the arrays +of their owner (e.g. ``Particles.markers``), but, once compiled with pyccel, they only accept +NumPy arrays. The classes here take the same constructor arguments, hold references to the +owner's **CuPy** arrays (no copies are made, non-CuPy arrays raise) and flatten them into the +arguments of a ``cupy.RawKernel``, see :attr:`CudaArguments.values`. + +The order of :attr:`CudaArguments.values` defines the corresponding part of the CUDA kernel +signature (``double*`` for float arrays, ``long long*`` for int arrays, ``bool*`` for bool +arrays, ``int`` for int scalars): + +* :class:`CudaMarkerArguments` -> ``double* markers, bool* valid_mks, int n_markers, int n_cols, + int Np, int vdim, int weight_idx, int first_diagnostics_idx, int first_init_idx, + int first_shift_idx, int residual_idx, int first_free_idx, int mu_idx, long long* bc_type`` +* :class:`CudaDomainArguments` -> ``int kind_map, double* params, long long* degree, + double* t1, double* t2, double* t3, long long* ind1, long long* ind2, long long* ind3, + double* cx, double* cy, double* cz`` +* :class:`CudaDerhamArguments` -> ``long long* pn, double* tn1, double* tn2, double* tn3, + long long* starts`` +""" + +import numpy as np + + +def _cupy_array(name: str, arr, dtype): + """Return ``arr`` itself after checking it is a C-contiguous CuPy array of ``dtype`` (never copies).""" + if not hasattr(arr, "__cuda_array_interface__"): + raise TypeError( + f"{name} must be a CuPy array (device memory), got {type(arr)}; no host-device copies are made." + ) + if arr.dtype != dtype: + raise TypeError(f"{name} must have dtype {np.dtype(dtype)}, got {arr.dtype}.") + if not arr.flags.c_contiguous: + raise ValueError(f"{name} must be C-contiguous.") + return arr + + +class CudaArguments: + """Base class; :attr:`values` is the flat tuple passed to the CUDA kernel. + + Instances are read-only after construction, such that :attr:`values` always matches the attributes. + """ + + _names: tuple[str, ...] = () + + def _freeze(self): + object.__setattr__(self, "_values", tuple(getattr(self, name) for name in self._names)) + + def __setattr__(self, name, value): + if hasattr(self, "_values"): + raise AttributeError(f"{type(self).__name__} is read-only.") + object.__setattr__(self, name, value) + + @property + def values(self) -> tuple: + """Flat CUDA kernel arguments.""" + return self._values + + @property + def n_threads(self) -> int | None: + """Number of CUDA threads needed for this argument, None if no preference.""" + return None + + +class CudaMarkerArguments(CudaArguments): + """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.MarkerArguments` (same parameters).""" + + _names = ( + "markers", + "valid_mks", + "n_markers", + "n_cols", + "Np", + "vdim", + "weight_idx", + "first_diagnostics_idx", + "first_init_idx", + "first_shift_idx", + "residual_idx", + "first_free_idx", + "mu_idx", + "bc_type", + ) + + def __init__( + self, + markers, + valid_mks, + Np: int, + vdim: int, + weight_idx: int, + first_diagnostics_idx: int, + first_pusher_idx: int, + first_shift_idx: int, + residual_idx: int, + first_free_idx: int, + mu_idx: int, + bc_type, + ): + self.markers = _cupy_array("markers", markers, np.float64) + self.valid_mks = _cupy_array("valid_mks", valid_mks, np.bool_) + self.n_markers = np.int32(markers.shape[0]) + self.n_cols = np.int32(markers.shape[1]) + self.Np = np.int32(Np) + self.vdim = np.int32(vdim) + self.weight_idx = np.int32(weight_idx) + self.first_diagnostics_idx = np.int32(first_diagnostics_idx) + self.first_init_idx = np.int32(first_pusher_idx) + self.first_shift_idx = np.int32(first_shift_idx) + self.residual_idx = np.int32(residual_idx) + self.first_free_idx = np.int32(first_free_idx) + self.mu_idx = np.int32(mu_idx) + self.bc_type = _cupy_array("bc_type", bc_type, np.int64) + self._freeze() + + @property + def n_threads(self) -> int: + return int(self.n_markers) + + +class CudaDomainArguments(CudaArguments): + """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments` (same parameters).""" + + _names = ("kind_map", "params", "degree", "t1", "t2", "t3", "ind1", "ind2", "ind3", "cx", "cy", "cz") + + def __init__(self, kind_map: int, params, degree, t1, t2, t3, ind1, ind2, ind3, cx, cy, cz): + self.kind_map = np.int32(kind_map) + self.params = _cupy_array("params", params, np.float64) + self.degree = _cupy_array("degree", degree, np.int64) + self.t1 = _cupy_array("t1", t1, np.float64) + self.t2 = _cupy_array("t2", t2, np.float64) + self.t3 = _cupy_array("t3", t3, np.float64) + self.ind1 = _cupy_array("ind1", ind1, np.int64) + self.ind2 = _cupy_array("ind2", ind2, np.int64) + self.ind3 = _cupy_array("ind3", ind3, np.int64) + self.cx = _cupy_array("cx", cx, np.float64) + self.cy = _cupy_array("cy", cy, np.float64) + self.cz = _cupy_array("cz", cz, np.float64) + self._freeze() + + +class CudaDerhamArguments(CudaArguments): + """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DerhamArguments` (same parameters). + + The scratch arrays ``bn1, ..., bd3`` of the pyccel version are not needed; in CUDA they are per-thread local arrays. + """ + + _names = ("pn", "tn1", "tn2", "tn3", "starts") + + def __init__(self, pn, tn1, tn2, tn3, starts): + self.pn = _cupy_array("pn", pn, np.int64) + self.tn1 = _cupy_array("tn1", tn1, np.float64) + self.tn2 = _cupy_array("tn2", tn2, np.float64) + self.tn3 = _cupy_array("tn3", tn3, np.float64) + self.starts = _cupy_array("starts", starts, np.int64) + self._freeze() diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index 2e9b8283b..9e1ddd47b 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -3,8 +3,9 @@ Each Struphy kernel has a pyccel version (wrapped in :class:`cunumpy.PyccelKernel`) and, optionally, a CUDA version (:class:`CudaKernel`) with a 1:1 corresponding signature. :class:`Kernel` holds both and dispatches to the CUDA kernel when the cunumpy backend is -``"cupy"``, see :func:`is_cuda_backend`. The pyccelized argument classes are transformed -once (at setup) into their CUDA counterparts by :func:`~struphy.utils.kernel_transform.transform`. +``"cupy"``, see :func:`is_cuda_backend`. CUDA kernels take the CUDA counterparts of the +pyccelized argument classes, see :mod:`struphy.utils.cuda_arguments`, which reference the +owner's device arrays (e.g. ``Particles.cuda_args_markers``, ``Domain.cuda_args_domain``). Example ------- @@ -13,10 +14,9 @@ ... cuda_kernel=CudaKernel(PUSH_ETA_LINEAR_SRC, "push_eta_linear"), ... ) >>> catalog.register(kernel) ->>> catalog.get("push_eta_linear")(dt, stage, args_markers, args_domain) # NumPy backend ->>> # CuPy backend, transform once at setup: ->>> cuda_markers, cuda_domain = transform(args_markers), transform(args_domain) ->>> catalog.get("push_eta_linear")(dt, stage, cuda_markers, cuda_domain) +>>> push = catalog.get("push_eta_linear") +>>> push(dt, stage, particles.args_markers, domain.args_domain) # NumPy backend +>>> push(dt, stage, particles.cuda_args_markers, domain.cuda_args_domain) # CuPy backend """ import math @@ -25,7 +25,7 @@ from cunumpy import PyccelKernel from cunumpy.xp import array_backend -from struphy.utils.kernel_transform import CudaArguments +from struphy.utils.cuda_arguments import CudaArguments def is_cuda_backend() -> bool: @@ -36,9 +36,9 @@ def is_cuda_backend() -> bool: class CudaKernel: """Call a ``cupy.RawKernel`` with the 1:1 corresponding arguments of its pyccel counterpart. - The pyccelized argument classes (``MarkerArguments`` etc.) must be transformed beforehand - (once, at setup) with :func:`~struphy.utils.kernel_transform.transform`; no arrays are - converted or copied at call time. Arrays must be CuPy arrays (``cupy`` raises otherwise). + Takes the CUDA counterparts of the pyccelized argument classes + (:mod:`struphy.utils.cuda_arguments`); no arrays are converted or copied at call time. + Arrays must be CuPy arrays (``cupy`` raises otherwise). Parameters ---------- diff --git a/src/struphy/utils/kernel_transform.py b/src/struphy/utils/kernel_transform.py deleted file mode 100644 index 334a32d15..000000000 --- a/src/struphy/utils/kernel_transform.py +++ /dev/null @@ -1,155 +0,0 @@ -"""Transform the pyccelized kernel argument classes into arguments for CUDA kernels. - -Pyccel kernels take the argument classes of :mod:`struphy.kernel_arguments.pusher_args_kernels` -as arguments. A ``cupy.RawKernel`` can only take pointers and scalars, hence :func:`transform` -collects the attributes of such a class into a :class:`CudaArguments` object holding CuPy arrays. - -The transform is meant to be done **once** at setup (not at every kernel call). Arrays that are -already CuPy arrays are used as they are (no copy); NumPy arrays are copied to the device once, -after which the device copy is the one updated by CUDA kernels. - -The field order of each :class:`CudaArguments` subclass defines the corresponding part -of the CUDA kernel signature (``double*`` for float arrays, ``long long*`` for int arrays, -``bool*`` for bool arrays, ``int`` for int scalars). -""" - -from dataclasses import dataclass, fields -from functools import singledispatch -from typing import Any - -import numpy as np - -from struphy.kernel_arguments.pusher_args_kernels import DerhamArguments, DomainArguments, MarkerArguments - - -@dataclass(frozen=True) -class CudaArguments: - """Base class for CUDA kernel arguments; :attr:`values` is the flat tuple passed to the kernel.""" - - def __post_init__(self): - object.__setattr__(self, "_values", tuple(getattr(self, f.name) for f in fields(self))) - - @property - def values(self) -> tuple: - """Flat CUDA kernel arguments, in field order.""" - return self._values - - @property - def n_threads(self) -> int | None: - """Number of CUDA threads needed for this argument, None if no preference.""" - return None - - -@dataclass(frozen=True) -class CudaMarkerArguments(CudaArguments): - """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.MarkerArguments`.""" - - markers: Any - valid_mks: Any - n_markers: np.int32 - n_cols: np.int32 - Np: np.int32 - vdim: np.int32 - weight_idx: np.int32 - first_diagnostics_idx: np.int32 - first_init_idx: np.int32 - first_shift_idx: np.int32 - residual_idx: np.int32 - first_free_idx: np.int32 - mu_idx: np.int32 - bc_type: Any - - @property - def n_threads(self) -> int: - return int(self.n_markers) - - -@dataclass(frozen=True) -class CudaDomainArguments(CudaArguments): - """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments`.""" - - kind_map: np.int32 - params: Any - degree: Any - t1: Any - t2: Any - t3: Any - ind1: Any - ind2: Any - ind3: Any - cx: Any - cy: Any - cz: Any - - -@dataclass(frozen=True) -class CudaDerhamArguments(CudaArguments): - """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DerhamArguments`.""" - - pn: Any - tn1: Any - tn2: Any - tn3: Any - starts: Any - - -def _cupy(arr, dtype): - """C-contiguous CuPy array of ``arr`` (no copy if it already is one).""" - import cupy as cp - - return cp.ascontiguousarray(cp.asarray(arr, dtype=dtype)) - - -@singledispatch -def transform(args: Any) -> CudaArguments: - """Transform a pyccelized kernel argument class into its :class:`CudaArguments` counterpart.""" - raise TypeError(f"No CUDA transform for arguments of type {type(args)}.") - - -@transform.register -def _(args: MarkerArguments) -> CudaMarkerArguments: - return CudaMarkerArguments( - markers=_cupy(args.markers, np.float64), - valid_mks=_cupy(args.valid_mks, np.bool_), - n_markers=np.int32(args.n_markers), - n_cols=np.int32(args.markers.shape[1]), - Np=np.int32(args.Np), - vdim=np.int32(args.vdim), - weight_idx=np.int32(args.weight_idx), - first_diagnostics_idx=np.int32(args.first_diagnostics_idx), - first_init_idx=np.int32(args.first_init_idx), - first_shift_idx=np.int32(args.first_shift_idx), - residual_idx=np.int32(args.residual_idx), - first_free_idx=np.int32(args.first_free_idx), - mu_idx=np.int32(args.mu_idx), - bc_type=_cupy(args.bc_type, np.int64), - ) - - -@transform.register -def _(args: DomainArguments) -> CudaDomainArguments: - return CudaDomainArguments( - kind_map=np.int32(args.kind_map), - params=_cupy(args.params, np.float64), - degree=_cupy(args.degree, np.int64), - t1=_cupy(args.t1, np.float64), - t2=_cupy(args.t2, np.float64), - t3=_cupy(args.t3, np.float64), - ind1=_cupy(args.ind1, np.int64), - ind2=_cupy(args.ind2, np.int64), - ind3=_cupy(args.ind3, np.int64), - cx=_cupy(args.cx, np.float64), - cy=_cupy(args.cy, np.float64), - cz=_cupy(args.cz, np.float64), - ) - - -@transform.register -def _(args: DerhamArguments) -> CudaDerhamArguments: - return CudaDerhamArguments( - pn=_cupy(args.pn, np.int64), - tn1=_cupy(args.tn1, np.float64), - tn2=_cupy(args.tn2, np.float64), - tn3=_cupy(args.tn3, np.float64), - starts=_cupy(args.starts, np.int64), - ) From cea5a960070e0bd02188a251e32a196dd9f06b51 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 15:47:41 +0200 Subject: [PATCH 08/29] Cleaned up the changes in the core of the code since this branch should just be a proof of concept --- src/struphy/geometry/base.py | 26 +- src/struphy/pic/base.py | 24 +- src/struphy/pic/pushing/demo_cuda.py | 115 +++------ src/struphy/pic/tests/test_kernel_backends.py | 229 ++---------------- src/struphy/utils/cuda_arguments.py | 168 ++++--------- src/struphy/utils/kernel_backends.py | 159 ++---------- 6 files changed, 124 insertions(+), 597 deletions(-) diff --git a/src/struphy/geometry/base.py b/src/struphy/geometry/base.py index b1a355313..d162c2f11 100644 --- a/src/struphy/geometry/base.py +++ b/src/struphy/geometry/base.py @@ -16,7 +16,6 @@ from struphy.geometry import evaluation_kernels, transform_kernels from struphy.kernel_arguments.pusher_args_kernels import DomainArguments from struphy.linear_algebra import linalg_kron -from struphy.utils.cuda_arguments import CudaDomainArguments from struphy.utils.docstring_converter import rst_to_html, rst_to_latex, rst_to_markdown from struphy.utils.ipython_compat import HTML, display from struphy.utils.utils import __class_with_params_repr_no_defaults__, all_class_params_are_default, all_subclasses @@ -239,7 +238,7 @@ def _build_args_domain(self): """Build runtime mapping arguments used by compiled evaluation kernels.""" return DomainArguments( self.kind_map, - _to_numpy_for_kernel(self.params_numpy), + self.params_numpy, _to_numpy_for_kernel(xp.array(self.degree)), _to_numpy_for_kernel(self.T[0]), _to_numpy_for_kernel(self.T[1]), @@ -276,7 +275,7 @@ def __deepcopy__(self, memo): memo[id(self)] = result for key, value in self.__dict__.items(): - if key in ("_args_domain", "_cuda_args_domain"): + if key == "_args_domain": continue setattr(result, key, copy.deepcopy(value, memo)) @@ -286,7 +285,6 @@ def __deepcopy__(self, memo): def __getstate__(self): state = self.__dict__.copy() state.pop("_args_domain", None) - state.pop("_cuda_args_domain", None) return state def __setstate__(self, state): @@ -477,26 +475,6 @@ def args_domain(self): return self._args_domain - @property - def cuda_args_domain(self) -> CudaDomainArguments: - """CUDA version of :attr:`args_domain`, referencing the device arrays of the domain (CuPy backend only).""" - if getattr(self, "_cuda_args_domain", None) is None: - self._cuda_args_domain = CudaDomainArguments( - self.kind_map, - self.params_numpy, - xp.array(self.degree), - self.T[0], - self.T[1], - self.T[2], - self.indN[0], - self.indN[1], - self.indN[2], - xp.ascontiguousarray(self.cx), - xp.ascontiguousarray(self.cy), - xp.ascontiguousarray(self.cz), - ) - return self._cuda_args_domain - @property def dict_transformations(self): """Dictionary of str->int for pull, push and transformation functions.""" diff --git a/src/struphy/pic/base.py b/src/struphy/pic/base.py index 3d2531939..da7177c1d 100644 --- a/src/struphy/pic/base.py +++ b/src/struphy/pic/base.py @@ -56,7 +56,6 @@ ) from struphy.utils import utils from struphy.utils.clone_config import CloneConfig -from struphy.utils.cuda_arguments import CudaMarkerArguments if TYPE_CHECKING: # importing mpi4py.MPI initializes MPI, which is slow; only needed for annotations from mpi4py.MPI import Intracomm @@ -941,26 +940,6 @@ def args_markers(self) -> MarkerArguments: """Collection of mandatory arguments for pusher kernels.""" return self._args_markers - @property - def cuda_args_markers(self) -> CudaMarkerArguments: - """CUDA version of :attr:`args_markers`, referencing the device arrays of the markers (CuPy backend only).""" - if getattr(self, "_cuda_args_markers", None) is None: - self._cuda_args_markers = CudaMarkerArguments( - self.markers, - self.valid_mks, - self.Np, - self.vdim, - self.index["weights"], - self.first_diagnostics_idx, - self.first_pusher_idx, - self.first_shift_idx, - self.residual_idx, - self.first_free_idx, - self.mu_idx, - self._bc_type, - ) - return self._cuda_args_markers - # ------------------------------------------- # Initial condition and background -> weights # ------------------------------------------- @@ -2548,8 +2527,7 @@ def _allocate_marker_array(self, dry_run: bool = False): self._n_lost_markers = 0 self._lost_markers = xp.zeros((int(self.n_rows * 0.5), 10), dtype=float) - # arguments for kernels (the CUDA version is built on first access, see cuda_args_markers) - self._cuda_args_markers = None + # arguments for kernels self._args_markers = MarkerArguments( _to_numpy_for_kernel(self.markers), _to_numpy_for_kernel(self.valid_mks), diff --git a/src/struphy/pic/pushing/demo_cuda.py b/src/struphy/pic/pushing/demo_cuda.py index a02d37439..ca292ef41 100644 --- a/src/struphy/pic/pushing/demo_cuda.py +++ b/src/struphy/pic/pushing/demo_cuda.py @@ -1,14 +1,8 @@ -"""CUDA counterpart of :mod:`struphy.pic.pushing.demo_kernels` and the corresponding -:class:`~struphy.utils.kernel_backends.Kernel` registered in the kernel catalog. +"""CUDA counterpart of :mod:`struphy.pic.pushing.demo_kernels`. -Run as a script to push markers with both backends and compare:: - - python -m struphy.pic.pushing.demo_cuda --n-markers 1000000 --n-steps 100 +Run ``python -m struphy.pic.pushing.demo_cuda`` to push markers with both backends and compare. """ -import argparse -import time - import cunumpy import numpy as np from cunumpy import PyccelKernel @@ -16,11 +10,10 @@ from struphy.geometry.domains import Cuboid from struphy.kernel_arguments.pusher_args_kernels import MarkerArguments from struphy.pic.pushing import demo_kernels -from struphy.utils.cuda_arguments import CudaMarkerArguments -from struphy.utils.kernel_backends import CudaKernel, Kernel, catalog, is_cuda_backend +from struphy.utils.cuda_arguments import CudaDomainArguments, CudaMarkerArguments +from struphy.utils.kernel_backends import CudaKernel, Kernel, is_cuda_backend -# Argument order = (dt, stage, CudaMarkerArguments.values, CudaDomainArguments.values), -# see struphy.utils.cuda_arguments. +# Arguments: (dt, stage, CudaMarkerArguments, CudaDomainArguments), see struphy.utils.cuda_arguments. PUSH_ETA_LINEAR_SRC = r""" extern "C" __global__ void push_eta_linear( @@ -45,94 +38,50 @@ } """ -push_eta_linear = catalog.register( - Kernel( - pyccel_kernel=PyccelKernel(demo_kernels.push_eta_linear), - cuda_kernel=CudaKernel(PUSH_ETA_LINEAR_SRC, "push_eta_linear"), - ), +push_eta_linear = Kernel( + pyccel_kernel=PyccelKernel(demo_kernels.push_eta_linear), + cuda_kernel=CudaKernel(PUSH_ETA_LINEAR_SRC, "push_eta_linear"), ) def make_demo_arguments(n_markers: int, seed: int = 0): - """Random markers (positions, velocities, some holes) and a Cuboid domain on the active cunumpy backend. - - The marker arrays are created on the active backend (on the device for CuPy), as ``Particles`` does; - the kernel arguments reference them without copies. - - Returns - ------- - args_markers : MarkerArguments | CudaMarkerArguments - Marker arguments for the kernel of the active backend. + """Random markers and a Cuboid domain, as kernel arguments for the active cunumpy backend. - args_domain : DomainArguments | CudaDomainArguments - Domain arguments for the kernel of the active backend. + The arrays are created on the active backend (on the device for CuPy); the arguments reference them without copies. """ - # same random numbers on both backends, such that results can be compared rng = np.random.default_rng(seed) markers = cunumpy.asarray(rng.random((n_markers, 25))) valid_mks = cunumpy.asarray(rng.random(n_markers) > 0.1) bc_type = cunumpy.zeros(3, dtype=int) domain = Cuboid() - if is_cuda_backend(): - args_markers = CudaMarkerArguments(markers, valid_mks, n_markers, 3, 6, 7, 8, 14, 17, 18, 4, bc_type) - args_domain = domain.cuda_args_domain - else: - args_markers = MarkerArguments(markers, valid_mks, n_markers, 3, 6, 7, 8, 14, 17, 18, 4, bc_type) - args_domain = domain.args_domain + if not is_cuda_backend(): + return MarkerArguments(markers, valid_mks, n_markers, 3, 6, 7, 8, 14, 17, 18, 4, bc_type), domain.args_domain + + args_markers = CudaMarkerArguments(markers, valid_mks, n_markers, 3, 6, 7, 8, 14, 17, 18, 4, bc_type) + args_domain = CudaDomainArguments( + domain.kind_map, + domain.params_numpy, + cunumpy.asarray(domain.degree), + *domain.T, + *domain.indN, + domain.cx, + domain.cy, + domain.cz, + ) return args_markers, args_domain -def run_push_eta_linear(args_markers, args_domain, dt: float, n_steps: int): - """Push markers ``n_steps`` times with :data:`push_eta_linear` on the active cunumpy backend. - - Inside the time loop only the kernel is called; no arrays are converted or copied. - - Returns - ------- - markers : numpy.ndarray - Markers after the last step (copied to the host once, at the end). - - time_per_step : float - Wall-clock time per step in seconds. - """ - # warm-up (compiles the CUDA kernel on first call) - push_eta_linear(0.0, 0, args_markers, args_domain) - cunumpy.synchronize() - - t0 = time.perf_counter() - for _ in range(n_steps): - push_eta_linear(dt, 0, args_markers, args_domain) - cunumpy.synchronize() - time_per_step = (time.perf_counter() - t0) / n_steps - - return cunumpy.to_numpy(args_markers.markers), time_per_step - - -def main(): - parser = argparse.ArgumentParser( - description="Push markers with the pyccel and the CUDA version of push_eta_linear." - ) - parser.add_argument("--n-markers", type=int, default=1_000_000) - parser.add_argument("--n-steps", type=int, default=100) - parser.add_argument("--dt", type=float, default=1e-3) - args = parser.parse_args() - - backends = ["numpy"] + (["cupy"] if cunumpy.cupy_available() else []) +def main(n_markers: int = 1_000_000, n_steps: int = 100, dt: float = 1e-3): results = {} - for backend in backends: + for backend in ("numpy", "cupy"): with cunumpy.use_backend(backend): - args_markers, args_domain = make_demo_arguments(args.n_markers) - results[backend] = run_push_eta_linear(args_markers, args_domain, args.dt, args.n_steps) - print( - f"{backend:>5}: {results[backend][1] * 1e3:8.3f} ms/step ({args.n_markers} markers, {args.n_steps} steps)" - ) - - if "cupy" in results: - max_diff = np.max(np.abs(results["numpy"][0] - results["cupy"][0])) - print(f"max |pyccel - cuda| = {max_diff:.2e}, speed-up = {results['numpy'][1] / results['cupy'][1]:.1f}x") - else: - print("CuPy/GPU not available, only the pyccel kernel was run.") + args_markers, args_domain = make_demo_arguments(n_markers) + for _ in range(n_steps): + push_eta_linear(dt, 0, args_markers, args_domain) + results[backend] = cunumpy.to_numpy(args_markers.markers) + + print(f"max |pyccel - cuda| = {np.max(np.abs(results['numpy'] - results['cupy'])):.2e}") if __name__ == "__main__": diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index 0b928055f..2f62fe7cd 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -1,235 +1,36 @@ -"""Dispatch between pyccel and CUDA kernels, and the CUDA versions of the kernel argument classes.""" - -import copy +"""Proof of concept: 1:1 pyccel and CUDA kernels, dispatched by the cunumpy backend.""" import cunumpy import numpy as np import pytest -from cunumpy import PyccelKernel -from struphy.geometry.domains import Cuboid -from struphy.pic.pushing.demo_cuda import make_demo_arguments, push_eta_linear, run_push_eta_linear -from struphy.utils.cuda_arguments import CudaDerhamArguments, CudaMarkerArguments -from struphy.utils.kernel_backends import Kernel, KernelCatalog, catalog, is_cuda_backend +from struphy.pic.pushing.demo_cuda import make_demo_arguments, push_eta_linear requires_cupy = pytest.mark.skipif(not cunumpy.cupy_available(), reason="CuPy/GPU not available") -BACKENDS = ["numpy", pytest.param("cupy", marks=requires_cupy)] - -N_MARKERS = 1000 -N_COLS = 25 -MARKER_INDICES = (N_MARKERS, 3, 6, 7, 8, 14, 17, 18, 4) # Np, vdim, weight_idx, ..., mu_idx - - -def expected_push(markers, valid_mks, dt, n_steps=1): - out = markers.copy() - for _ in range(n_steps): - out[valid_mks, 0:3] += dt * out[valid_mks, 3:6] - return out - - -def device_marker_arrays(): - import cupy as cp - - return cp.random.random((N_MARKERS, N_COLS)), cp.ones(N_MARKERS, dtype=bool), cp.zeros(3, dtype=int) - - -# --------------------------- -# kernel dispatch and catalog -# --------------------------- - - -@pytest.mark.parametrize("backend", BACKENDS) -def test_kernel_dispatch(backend): - with cunumpy.use_backend(backend): - assert is_cuda_backend() == (backend == "cupy") - kernel = push_eta_linear.get_kernel() - if backend == "cupy": - assert kernel is push_eta_linear.cuda_kernel - else: - assert kernel is push_eta_linear.pyccel_kernel - - -def test_kernel_without_cuda_falls_back_to_pyccel(): - kernel = Kernel(push_eta_linear.pyccel_kernel) - for backend in ("numpy", "cupy"): - with cunumpy.use_backend(backend): - assert kernel.get_kernel() is kernel.pyccel_kernel - - -def test_catalog(): - assert catalog.get("push_eta_linear") is push_eta_linear - - local = KernelCatalog() - local.register(push_eta_linear, name="foo") - assert "foo" in local and local.names == ["foo"] - with pytest.raises(AssertionError): - local.register(push_eta_linear, name="foo") - - -def test_kernel_type_checks(): - with pytest.raises(AssertionError): - Kernel(lambda: None) - with pytest.raises(AssertionError): - Kernel(push_eta_linear.pyccel_kernel, cuda_kernel=PyccelKernel(lambda: None)) - - -# ------------------------ -# CUDA argument classes -# ------------------------ - - -@requires_cupy -def test_cuda_marker_args_reference_device_arrays(): - """The CUDA arguments hold the very same device arrays (no copies).""" - markers, valid_mks, bc_type = device_marker_arrays() - args = CudaMarkerArguments(markers, valid_mks, *MARKER_INDICES, bc_type) - - assert args.markers is markers and args.valid_mks is valid_mks and args.bc_type is bc_type - assert args.n_threads == N_MARKERS - assert len(args.values) == 14 - assert args.values[0] is markers - assert all(isinstance(v, np.int32) for v in args.values[2:13]) - assert args.n_markers == N_MARKERS and args.n_cols == N_COLS - assert args.first_init_idx == 8 - - with pytest.raises(AttributeError, match="read-only"): - args.markers = markers - - -@requires_cupy -def test_cuda_marker_args_reject_host_and_bad_arrays(): - """Host arrays are never copied to the device; wrong dtypes or layouts are not converted.""" - markers, valid_mks, bc_type = device_marker_arrays() - - with pytest.raises(TypeError, match="must be a CuPy array"): - CudaMarkerArguments(markers.get(), valid_mks, *MARKER_INDICES, bc_type) - with pytest.raises(TypeError, match="must be a CuPy array"): - CudaMarkerArguments(markers, valid_mks, *MARKER_INDICES, bc_type.get()) - with pytest.raises(TypeError, match="dtype"): - CudaMarkerArguments(markers.astype(np.float32), valid_mks, *MARKER_INDICES, bc_type) - with pytest.raises(ValueError, match="C-contiguous"): - CudaMarkerArguments(markers[:, ::2], valid_mks, *MARKER_INDICES, bc_type) - - -@requires_cupy -def test_cuda_derham_args(): - import cupy as cp - - knots = cp.array([0.0, 0.0, 1.0, 1.0]) - args = CudaDerhamArguments(cp.ones(3, dtype=int), knots, knots, knots, cp.zeros(3, dtype=int)) - assert len(args.values) == 5 and args.tn1 is knots - - -@requires_cupy -def test_domain_cuda_args(): - """Domain.cuda_args_domain references the device arrays of the domain.""" - with cunumpy.use_backend("cupy"): - domain = Cuboid(l1=0.5, r1=2.0) - args = domain.cuda_args_domain - - assert domain.cuda_args_domain is args # cached - assert args.t1 is domain.T[0] and args.ind3 is domain.indN[2] - assert args.params is domain.params_numpy - assert args.kind_map == domain.kind_map - assert len(args.values) == 12 - - # not deep-copied/pickled along with the domain, rebuilt from the copy's own arrays - domain_copy = copy.deepcopy(domain) - assert domain_copy.cuda_args_domain.t1 is domain_copy.T[0] - - # a domain created on the NumPy backend has host arrays - with pytest.raises(TypeError, match="must be a CuPy array"): - Cuboid().cuda_args_domain - - -@requires_cupy -def test_particles_cuda_args_markers(): - """Particles.cuda_args_markers references the marker arrays of the particles. - - Particles cannot be created on the CuPy backend yet, hence the arrays of a NumPy-backend - instance are replaced by device arrays here. - """ - import cupy as cp - from feectools.ddm.mpi import mpi as MPI - - from struphy import LoadingParameters - from struphy.pic.particles import Particles6D - - loading_params = LoadingParameters(Np=100, seed=1234, moments=(0.0, 0.0, 0.0, 1.0, 1.0, 1.0), spatial="uniform") - particles = Particles6D(comm_world=MPI.COMM_WORLD, loading_params=loading_params, domain=Cuboid()) - particles.draw_markers() - - host_args = particles.args_markers - particles._markers = cp.asarray(particles.markers) - particles._valid_mks = cp.asarray(particles.valid_mks) - particles._bc_type = cp.asarray(particles._bc_type) - particles._cuda_args_markers = None - - args = particles.cuda_args_markers - assert particles.cuda_args_markers is args # cached - assert args.markers is particles.markers - assert args.valid_mks is particles.valid_mks - assert args.bc_type is particles._bc_type - for name in ( - "n_markers", - "Np", - "vdim", - "weight_idx", - "first_diagnostics_idx", - "first_init_idx", - "first_shift_idx", - "residual_idx", - "first_free_idx", - "mu_idx", - ): - assert getattr(args, name) == getattr(host_args, name), name - -# ------------------------ -# pushing -# ------------------------ - - -@pytest.mark.parametrize("backend", BACKENDS) +@pytest.mark.parametrize("backend", ["numpy", pytest.param("cupy", marks=requires_cupy)]) def test_push_eta_linear(backend): - """The kernel updates the owner's marker array in place.""" + """The kernel of the active backend updates the marker array in place.""" dt = 0.1 with cunumpy.use_backend(backend): - args_markers, args_domain = make_demo_arguments(N_MARKERS) + args_markers, args_domain = make_demo_arguments(1000) markers = args_markers.markers - expected = expected_push(cunumpy.to_numpy(markers), cunumpy.to_numpy(args_markers.valid_mks), dt) + valid = cunumpy.to_numpy(args_markers.valid_mks) + expected = cunumpy.to_numpy(markers).copy() + expected[valid, 0:3] += dt * expected[valid, 3:6] push_eta_linear(dt, 0, args_markers, args_domain) - if backend == "cupy": - assert args_markers.markers is markers assert np.allclose(cunumpy.to_numpy(markers), expected, rtol=1e-14, atol=0.0) @requires_cupy -def test_pyccel_cuda_agree(): - dt, n_steps = 0.01, 5 - results = {} - for backend in ("numpy", "cupy"): - with cunumpy.use_backend(backend): - args_markers, args_domain = make_demo_arguments(N_MARKERS, seed=2) - if backend == "numpy": - expected = expected_push(args_markers.markers, args_markers.valid_mks, dt, n_steps) - results[backend], _ = run_push_eta_linear(args_markers, args_domain, dt, n_steps) - - assert np.allclose(results["numpy"], expected, rtol=1e-13, atol=0.0) - assert np.allclose(results["cupy"], results["numpy"], rtol=1e-14, atol=0.0) - - -@requires_cupy -def test_cuda_kernel_rejects_pyccel_args(): - """Passing the pyccel argument classes to the CUDA kernel fails instead of copying.""" +def test_cuda_arguments_reject_host_arrays(): + """Host arrays are never copied to the device.""" with cunumpy.use_backend("numpy"): - host_markers, host_domain = make_demo_arguments(N_MARKERS) - with cunumpy.use_backend("cupy"): - cuda_markers, cuda_domain = make_demo_arguments(N_MARKERS) - with pytest.raises(ValueError, match="number of CUDA threads"): - push_eta_linear(0.1, 0, host_markers, cuda_domain) - with pytest.raises(TypeError): - push_eta_linear(0.1, 0, cuda_markers, host_domain) + host_markers, _ = make_demo_arguments(10) + with cunumpy.use_backend("cupy"), pytest.raises(TypeError, match="CuPy array"): + from struphy.utils.cuda_arguments import CudaMarkerArguments + + CudaMarkerArguments(host_markers.markers, host_markers.valid_mks, 10, 3, 6, 7, 8, 14, 17, 18, 4, np.zeros(3)) diff --git a/src/struphy/utils/cuda_arguments.py b/src/struphy/utils/cuda_arguments.py index 942381b15..b34bc19b3 100644 --- a/src/struphy/utils/cuda_arguments.py +++ b/src/struphy/utils/cuda_arguments.py @@ -1,88 +1,31 @@ """CUDA counterparts of the pyccelized kernel argument classes. -The classes in :mod:`struphy.kernel_arguments.pusher_args_kernels` hold references to the arrays -of their owner (e.g. ``Particles.markers``), but, once compiled with pyccel, they only accept -NumPy arrays. The classes here take the same constructor arguments, hold references to the -owner's **CuPy** arrays (no copies are made, non-CuPy arrays raise) and flatten them into the -arguments of a ``cupy.RawKernel``, see :attr:`CudaArguments.values`. - -The order of :attr:`CudaArguments.values` defines the corresponding part of the CUDA kernel -signature (``double*`` for float arrays, ``long long*`` for int arrays, ``bool*`` for bool -arrays, ``int`` for int scalars): - -* :class:`CudaMarkerArguments` -> ``double* markers, bool* valid_mks, int n_markers, int n_cols, - int Np, int vdim, int weight_idx, int first_diagnostics_idx, int first_init_idx, - int first_shift_idx, int residual_idx, int first_free_idx, int mu_idx, long long* bc_type`` -* :class:`CudaDomainArguments` -> ``int kind_map, double* params, long long* degree, - double* t1, double* t2, double* t3, long long* ind1, long long* ind2, long long* ind3, - double* cx, double* cy, double* cz`` -* :class:`CudaDerhamArguments` -> ``long long* pn, double* tn1, double* tn2, double* tn3, - long long* starts`` +The compiled classes in :mod:`struphy.kernel_arguments.pusher_args_kernels` only accept NumPy +arrays. The classes here take the same constructor arguments, keep references to **CuPy** arrays +(no copies, other arrays raise) and flatten them into the arguments of a ``cupy.RawKernel``. +The order of :attr:`values` is the corresponding part of the CUDA kernel signature. """ import numpy as np def _cupy_array(name: str, arr, dtype): - """Return ``arr`` itself after checking it is a C-contiguous CuPy array of ``dtype`` (never copies).""" + """Return ``arr`` itself after checking it is a C-contiguous CuPy array of ``dtype``.""" if not hasattr(arr, "__cuda_array_interface__"): - raise TypeError( - f"{name} must be a CuPy array (device memory), got {type(arr)}; no host-device copies are made." - ) - if arr.dtype != dtype: - raise TypeError(f"{name} must have dtype {np.dtype(dtype)}, got {arr.dtype}.") - if not arr.flags.c_contiguous: - raise ValueError(f"{name} must be C-contiguous.") + raise TypeError(f"{name} must be a CuPy array, got {type(arr)}.") + if arr.dtype != dtype or not arr.flags.c_contiguous: + raise TypeError(f"{name} must be a C-contiguous array of dtype {np.dtype(dtype)}.") return arr -class CudaArguments: - """Base class; :attr:`values` is the flat tuple passed to the CUDA kernel. +class CudaMarkerArguments: + """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.MarkerArguments`. - Instances are read-only after construction, such that :attr:`values` always matches the attributes. + CUDA signature: ``double* markers, bool* valid_mks, int n_markers, int n_cols, int Np, int vdim, + int weight_idx, int first_diagnostics_idx, int first_init_idx, int first_shift_idx, + int residual_idx, int first_free_idx, int mu_idx, long long* bc_type`` """ - _names: tuple[str, ...] = () - - def _freeze(self): - object.__setattr__(self, "_values", tuple(getattr(self, name) for name in self._names)) - - def __setattr__(self, name, value): - if hasattr(self, "_values"): - raise AttributeError(f"{type(self).__name__} is read-only.") - object.__setattr__(self, name, value) - - @property - def values(self) -> tuple: - """Flat CUDA kernel arguments.""" - return self._values - - @property - def n_threads(self) -> int | None: - """Number of CUDA threads needed for this argument, None if no preference.""" - return None - - -class CudaMarkerArguments(CudaArguments): - """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.MarkerArguments` (same parameters).""" - - _names = ( - "markers", - "valid_mks", - "n_markers", - "n_cols", - "Np", - "vdim", - "weight_idx", - "first_diagnostics_idx", - "first_init_idx", - "first_shift_idx", - "residual_idx", - "first_free_idx", - "mu_idx", - "bc_type", - ) - def __init__( self, markers, @@ -100,58 +43,43 @@ def __init__( ): self.markers = _cupy_array("markers", markers, np.float64) self.valid_mks = _cupy_array("valid_mks", valid_mks, np.bool_) - self.n_markers = np.int32(markers.shape[0]) - self.n_cols = np.int32(markers.shape[1]) - self.Np = np.int32(Np) - self.vdim = np.int32(vdim) - self.weight_idx = np.int32(weight_idx) - self.first_diagnostics_idx = np.int32(first_diagnostics_idx) - self.first_init_idx = np.int32(first_pusher_idx) - self.first_shift_idx = np.int32(first_shift_idx) - self.residual_idx = np.int32(residual_idx) - self.first_free_idx = np.int32(first_free_idx) - self.mu_idx = np.int32(mu_idx) - self.bc_type = _cupy_array("bc_type", bc_type, np.int64) - self._freeze() - - @property - def n_threads(self) -> int: - return int(self.n_markers) - - -class CudaDomainArguments(CudaArguments): - """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments` (same parameters).""" - - _names = ("kind_map", "params", "degree", "t1", "t2", "t3", "ind1", "ind2", "ind3", "cx", "cy", "cz") - - def __init__(self, kind_map: int, params, degree, t1, t2, t3, ind1, ind2, ind3, cx, cy, cz): - self.kind_map = np.int32(kind_map) - self.params = _cupy_array("params", params, np.float64) - self.degree = _cupy_array("degree", degree, np.int64) - self.t1 = _cupy_array("t1", t1, np.float64) - self.t2 = _cupy_array("t2", t2, np.float64) - self.t3 = _cupy_array("t3", t3, np.float64) - self.ind1 = _cupy_array("ind1", ind1, np.int64) - self.ind2 = _cupy_array("ind2", ind2, np.int64) - self.ind3 = _cupy_array("ind3", ind3, np.int64) - self.cx = _cupy_array("cx", cx, np.float64) - self.cy = _cupy_array("cy", cy, np.float64) - self.cz = _cupy_array("cz", cz, np.float64) - self._freeze() + self.n_markers = markers.shape[0] + self.values = ( + self.markers, + self.valid_mks, + *( + np.int32(i) + for i in ( + markers.shape[0], + markers.shape[1], + Np, + vdim, + weight_idx, + first_diagnostics_idx, + first_pusher_idx, + first_shift_idx, + residual_idx, + first_free_idx, + mu_idx, + ) + ), + _cupy_array("bc_type", bc_type, np.int64), + ) -class CudaDerhamArguments(CudaArguments): - """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DerhamArguments` (same parameters). +class CudaDomainArguments: + """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments`. - The scratch arrays ``bn1, ..., bd3`` of the pyccel version are not needed; in CUDA they are per-thread local arrays. + CUDA signature: ``int kind_map, double* params, long long* degree, double* t1, double* t2, double* t3, + long long* ind1, long long* ind2, long long* ind3, double* cx, double* cy, double* cz`` """ - _names = ("pn", "tn1", "tn2", "tn3", "starts") - - def __init__(self, pn, tn1, tn2, tn3, starts): - self.pn = _cupy_array("pn", pn, np.int64) - self.tn1 = _cupy_array("tn1", tn1, np.float64) - self.tn2 = _cupy_array("tn2", tn2, np.float64) - self.tn3 = _cupy_array("tn3", tn3, np.float64) - self.starts = _cupy_array("starts", starts, np.int64) - self._freeze() + def __init__(self, kind_map: int, params, degree, t1, t2, t3, ind1, ind2, ind3, cx, cy, cz): + self.values = ( + np.int32(kind_map), + _cupy_array("params", params, np.float64), + _cupy_array("degree", degree, np.int64), + *(_cupy_array("t", t, np.float64) for t in (t1, t2, t3)), + *(_cupy_array("ind", ind, np.int64) for ind in (ind1, ind2, ind3)), + *(_cupy_array("c", c, np.float64) for c in (cx, cy, cz)), + ) diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index 9e1ddd47b..cbf5b1b98 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -1,23 +1,4 @@ -"""Pairs of pyccel and CUDA kernels, selected at runtime from the cunumpy backend. - -Each Struphy kernel has a pyccel version (wrapped in :class:`cunumpy.PyccelKernel`) and, -optionally, a CUDA version (:class:`CudaKernel`) with a 1:1 corresponding signature. -:class:`Kernel` holds both and dispatches to the CUDA kernel when the cunumpy backend is -``"cupy"``, see :func:`is_cuda_backend`. CUDA kernels take the CUDA counterparts of the -pyccelized argument classes, see :mod:`struphy.utils.cuda_arguments`, which reference the -owner's device arrays (e.g. ``Particles.cuda_args_markers``, ``Domain.cuda_args_domain``). - -Example -------- ->>> kernel = Kernel( -... pyccel_kernel=PyccelKernel(demo_kernels.push_eta_linear), -... cuda_kernel=CudaKernel(PUSH_ETA_LINEAR_SRC, "push_eta_linear"), -... ) ->>> catalog.register(kernel) ->>> push = catalog.get("push_eta_linear") ->>> push(dt, stage, particles.args_markers, domain.args_domain) # NumPy backend ->>> push(dt, stage, particles.cuda_args_markers, domain.cuda_args_domain) # CuPy backend -""" +"""Pairs of pyccel and CUDA kernels, selected at runtime from the cunumpy backend.""" import math @@ -25,8 +6,6 @@ from cunumpy import PyccelKernel from cunumpy.xp import array_backend -from struphy.utils.cuda_arguments import CudaArguments - def is_cuda_backend() -> bool: """Whether the active cunumpy backend is CuPy.""" @@ -34,141 +13,55 @@ def is_cuda_backend() -> bool: class CudaKernel: - """Call a ``cupy.RawKernel`` with the 1:1 corresponding arguments of its pyccel counterpart. - - Takes the CUDA counterparts of the pyccelized argument classes - (:mod:`struphy.utils.cuda_arguments`); no arrays are converted or copied at call time. - Arrays must be CuPy arrays (``cupy`` raises otherwise). - - Parameters - ---------- - source : str - CUDA C source code containing an ``extern "C" __global__`` function ``name``. - - name : str - Name of the kernel function in ``source``. + """A ``cupy.RawKernel`` called with the 1:1 corresponding arguments of its pyccel counterpart. - block_size : int - Number of threads per block. + Argument classes must already be the CUDA versions (:mod:`struphy.utils.cuda_arguments`); + no arrays are converted or copied at call time. One thread is launched per marker. """ def __init__(self, source: str, name: str, block_size: int = 128): + self.name = name self._source = source - self._name = name self._block_size = block_size self._raw_kernel = None - def __repr__(self): - return f"CudaKernel(name={self.name!r}, block_size={self._block_size})" - - @property - def name(self) -> str: - """Name of the CUDA kernel.""" - return self._name - - @property - def raw_kernel(self): - """The compiled ``cupy.RawKernel`` (compiled lazily on first access).""" + def __call__(self, *args): if self._raw_kernel is None: import cupy as cp - self._raw_kernel = cp.RawKernel(self._source, self._name) - return self._raw_kernel + self._raw_kernel = cp.RawKernel(self._source, self.name) - def __call__(self, *args): values = [] n_threads = None for arg in args: - if isinstance(arg, CudaArguments): + if hasattr(arg, "values"): values += arg.values - if n_threads is None: - n_threads = arg.n_threads + n_threads = n_threads or getattr(arg, "n_markers", None) + elif isinstance(arg, bool): + values.append(np.bool_(arg)) + elif isinstance(arg, int): + values.append(np.int32(arg)) + elif isinstance(arg, float): + values.append(np.float64(arg)) else: - values.append(_cuda_scalar(arg)) + values.append(arg) if n_threads is None: - raise ValueError(f"{self.name}: no argument defines the number of CUDA threads (e.g. CudaMarkerArguments).") - - grid = (max(1, math.ceil(n_threads / self._block_size)),) - self.raw_kernel(grid, (self._block_size,), tuple(values)) + raise ValueError(f"{self.name}: no CudaMarkerArguments passed, cannot set the number of threads.") - -def _cuda_scalar(arg): - """Python scalars as NumPy scalars with the C type of the kernel signature (int -> int, float -> double).""" - if isinstance(arg, bool): - return np.bool_(arg) - if isinstance(arg, int): - return np.int32(arg) - if isinstance(arg, float): - return np.float64(arg) - return arg + grid = (math.ceil(n_threads / self._block_size),) + self._raw_kernel(grid, (self._block_size,), tuple(values)) class Kernel: - """A pyccel kernel and its 1:1 corresponding CUDA kernel. - - Parameters - ---------- - pyccel_kernel : PyccelKernel - The pyccel kernel, used on the NumPy backend (and on CuPy if there is no CUDA kernel). - - cuda_kernel : CudaKernel | None - The CUDA kernel, used on the CuPy backend. - """ - - def __init__(self, pyccel_kernel: PyccelKernel, cuda_kernel: CudaKernel | None = None): - assert isinstance(pyccel_kernel, PyccelKernel), f"{pyccel_kernel} is not of type PyccelKernel" - assert cuda_kernel is None or isinstance(cuda_kernel, CudaKernel), f"{cuda_kernel} is not of type CudaKernel" - self._pyccel_kernel = pyccel_kernel - self._cuda_kernel = cuda_kernel + """A pyccel kernel and its CUDA counterpart; calls the one matching the cunumpy backend.""" - def __repr__(self): - return f"Kernel(pyccel_kernel={self.pyccel_kernel!r}, cuda_kernel={self.cuda_kernel!r})" - - @property - def pyccel_kernel(self) -> PyccelKernel: - return self._pyccel_kernel - - @property - def cuda_kernel(self) -> CudaKernel | None: - return self._cuda_kernel - - @property - def name(self) -> str: - return self.pyccel_kernel.name + def __init__(self, pyccel_kernel: PyccelKernel, cuda_kernel: CudaKernel): + self.pyccel_kernel = pyccel_kernel + self.cuda_kernel = cuda_kernel def get_kernel(self) -> PyccelKernel | CudaKernel: - """The kernel for the active cunumpy backend.""" - if is_cuda_backend() and self.cuda_kernel is not None: - return self.cuda_kernel - return self.pyccel_kernel - - def __call__(self, *args, **kwargs): - return self.get_kernel()(*args, **kwargs) - - -class KernelCatalog: - """Registry of :class:`Kernel` objects by name.""" - - def __init__(self): - self._kernels: dict[str, Kernel] = {} + return self.cuda_kernel if is_cuda_backend() else self.pyccel_kernel - def register(self, kernel: Kernel, name: str | None = None) -> Kernel: - """Register ``kernel`` under ``name`` (default: name of the pyccel kernel).""" - name = kernel.name if name is None else name - assert name not in self._kernels, f"Kernel {name!r} is already registered." - self._kernels[name] = kernel - return kernel - - def get(self, name: str) -> Kernel: - return self._kernels[name] - - def __contains__(self, name: str) -> bool: - return name in self._kernels - - @property - def names(self) -> list[str]: - return list(self._kernels) - - -catalog = KernelCatalog() + def __call__(self, *args): + return self.get_kernel()(*args) From 21865b6a1b222b06c6fa21afa8c1649ecbd49113 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 15:49:46 +0200 Subject: [PATCH 09/29] Added CUDA_STRATEGY.md [skip ci] --- CUDA_STRATEGY.md | 187 +++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 187 insertions(+) create mode 100644 CUDA_STRATEGY.md diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md new file mode 100644 index 000000000..155137cbe --- /dev/null +++ b/CUDA_STRATEGY.md @@ -0,0 +1,187 @@ +# CUDA strategy for Struphy kernels + +Plan for running Struphy's compute kernels on NVIDIA GPUs, next to the existing pyccel (CPU) kernels. +The work is split into small PRs that can be reviewed and merged one at a time. Nothing here has to be done in one go. + +## PR checklist + +- [ ] **PR 1: Proof of concept** (branch `cuda-kernel-proof-of-concept`) + `Kernel` (pyccel/CUDA pair), `CudaKernel`, `CudaMarkerArguments`/`CudaDomainArguments`, one demo kernel pair and a test on both backends. Also adds this document and the `gpu` optional dependency. +- [ ] **PR 2: CUDA source files** โ€” `CudaKernel` loads CUDA source from a `_cuda.cu` file next to the pyccel file; `.cu`/`.cuh` files are shipped as package data. +- [ ] **PR 3: Kernel catalog** โ€” kernels are defined once in the `__init__.py` of the folder that contains them; a missing CUDA kernel raises an error on the GPU backend. +- [ ] **PR 4: `Pusher` and propagators accept `Kernel`** โ€” replace `PyccelKernel(...)` in the propagators by catalog lookups (no CUDA kernels yet, so no behaviour change on CPU). +- [ ] **PR 5: `Domain` on the GPU** โ€” `Domain.cuda_args_domain`, plus the separate fix for `Domain` deepcopy on the CuPy backend (`_build_args_domain` passes `params_numpy` without `_to_numpy_for_kernel`). +- [ ] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend (e.g. `xp.prod` on Python lists in `pic/base.py`), plus `Particles.cuda_args_markers`. +- [ ] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. +- [ ] **PR 8: Shared CUDA headers for the argument classes** โ€” one `.cuh` per argument class instead of long flat kernel signatures. +- [ ] **PR 9: One folder per kernel, starting with `pic/pushing`** โ€” pure refactor, no behaviour change. +- [ ] **PR 10: Device versions of helper kernels** โ€” B-spline evaluation, mapping evaluation (per domain), small linear algebra, as `__device__` functions in `.cuh` headers. +- [ ] **PR 11: First real CUDA kernel** โ€” `push_eta_stage` with a pyccel/CUDA parity test and an end-to-end run on the GPU. +- [ ] **PR 12+: Port kernels one by one**, in the order they are needed by the models we want on the GPU (see [Porting order](#porting-order)). +- [ ] **CI**: a GPU runner that runs the CUDA tests (can happen any time after PR 1). + +Unrelated bugs found along the way go into their own PRs, not into these ones. + +## Goal + +A developer who adds a new kernel (e.g. for a new model) should only have to: + +1. write the pyccel kernel `_kernels.py` as today, and +2. optionally write `_cuda.cu` **in the same folder**. + +Everything else (loading, dispatch, argument passing, tests for agreement between the two versions) is done by the infrastructure. +CUDA kernels can be added one by one. If the code runs on the GPU and needs a kernel that has no CUDA version yet, it raises a clear error instead of silently falling back to the CPU. + +## Principles + +- **1:1 correspondence.** Each CUDA kernel has the same name and the same arguments (in the same order) as its pyccel kernel. The call site does not know which one runs. +- **The backend decides.** The cunumpy backend (`ARRAY_BACKEND=cupy` or `cunumpy.set_backend("cupy")`) selects the CUDA kernels; with NumPy the pyccel kernels run as today. +- **No conversions at call time.** When a kernel is called, its arguments are already in the right format. There are no host/device copies per kernel call. +- **Data already lives on the GPU.** On the CuPy backend, `xp` is `cupy`, so markers, spline coefficients etc. are CuPy arrays from the start. The CUDA argument objects only collect *references* to these arrays and raise if they get host arrays. +- **No silent CPU fallback on the GPU.** A kernel without a CUDA version raises an error on the GPU backend. Falling back would mean copying data to the host and back at every call. +- **Small steps.** Every PR keeps the CPU code path working and tested. + +## Current state (PR 1) + +| File | Content | +|---|---| +| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily, one thread per marker), `Kernel` (pyccel/CUDA pair, `get_kernel()` picks by backend) | +| `src/struphy/utils/cuda_arguments.py` | `CudaMarkerArguments`, `CudaDomainArguments`: same constructor arguments as the pyccel classes, hold CuPy arrays, flatten them into the CUDA kernel arguments | +| `src/struphy/pic/pushing/demo_kernels.py` | pyccel `push_eta_linear` | +| `src/struphy/pic/pushing/demo_cuda.py` | CUDA `push_eta_linear` (source string), the `Kernel` pair and a demo `main()` comparing both backends | +| `src/struphy/pic/tests/test_kernel_backends.py` | tests on both backends | + +Things we learned in the proof of concept: + +- The pyccel-compiled argument classes (`MarkerArguments`, `DomainArguments`, `DerhamArguments`) hold references to their owner's arrays, but only accept **NumPy** arrays. Hence the CUDA counterparts in `cuda_arguments.py`. +- Today, `Particles` builds `args_markers` from `_to_numpy_for_kernel(self.markers)`, i.e. from a **host copy** when the backend is CuPy. The same holds for `Domain` and `Derham`. +- `Particles6D` and `Derham` cannot be created on the CuPy backend yet (PR 6, PR 7). `Domain` (e.g. `Cuboid`) can, and all its arrays are already CuPy arrays. +- `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Python `int`s are passed as 64-bit integers, so scalars are cast explicitly (`int` โ†’ `np.int32`, `float` โ†’ `np.float64`). +- `struphy compile` compiles every `.py` file whose name contains `kernels`. Non-pyccel modules must not contain `kernels` in their name; `.cu` files are ignored by it. +- On an H100, the demo kernel pushes 10โถ markers in about 0.13 ms per step. + +## Target layout + +Each kernel gets its own folder with the pyccel and the CUDA version side by side. The `__init__.py` of the parent folder defines the catalog of the kernels in it: + +``` +src/struphy/pic/pushing/kernels/ +โ”œโ”€โ”€ __init__.py # catalog = KernelCatalog.from_package(__name__) +โ”œโ”€โ”€ push_eta_stage/ +โ”‚ โ”œโ”€โ”€ __init__.py +โ”‚ โ”œโ”€โ”€ push_eta_stage_kernels.py # pyccel (compiled by `struphy compile`) +โ”‚ โ””โ”€โ”€ push_eta_stage_cuda.cu # CUDA (compiled at runtime by CuPy/NVRTC) +โ”œโ”€โ”€ push_vxb_analytic/ +โ”‚ โ”œโ”€โ”€ __init__.py +โ”‚ โ””โ”€โ”€ push_vxb_analytic_kernels.py # no CUDA version yet -> error on the GPU +โ””โ”€โ”€ ... +``` + +Conventions: + +- The folder name, the pyccel function name, the CUDA `extern "C" __global__` function name and the catalog key are all the same. +- The pyccel file is `_kernels.py` (so `struphy compile` picks it up); the CUDA file is `_cuda.cu`. +- Shared CUDA code (device helper functions, argument structs) lives in `.cuh` headers next to the pyccel module it mirrors, e.g. `bsplines/bsplines_kernels.cuh` for `bsplines/bsplines_kernels.py`. + +The geometry domains already follow a similar layout (`geometry/domains/cuboid/cuboid_kernels.py`), which can be extended with `cuboid_cuda.cuh` for the device version of the mapping. + +Usage at a call site (e.g. in a propagator): + +```python +from struphy.pic.pushing.kernels import catalog + +kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the backend +``` + +## Details per PR + +### PR 2: CUDA source files + +- `CudaKernel.from_file(path, name)` reads `_cuda.cu`. The kernel is compiled lazily on first call. CuPy caches compiled kernels on disk (`~/.cupy/kernel_cache`), so the compile cost is paid once per machine. +- Headers are found through NVRTC include paths (`cupy.RawModule(code=..., options=("-I",))`), so a `.cu` file can `#include "struphy/bsplines/bsplines_kernels.cuh"`. +- Add `"**/*.cu"` and `"**/*.cuh"` to `[tool.setuptools.package-data]` in `pyproject.toml`. +- Move the demo kernel's CUDA source from the string in `demo_cuda.py` into `demo_cuda.cu`. + +### PR 3: Kernel catalog + +- `KernelCatalog.from_package(package)` scans the subfolders of a package. For every `/_kernels.py` it creates a `Kernel` with `PyccelKernel(_kernels.)` and, if `/_cuda.cu` exists, a `CudaKernel`. This is what keeps the work for developers minimal: they add files, not registration code. +- `Kernel` without a CUDA version: on the CuPy backend, `get_kernel()` raises + + ``` + NotImplementedError: No CUDA version of kernel 'push_vxb_analytic' (expected .../push_vxb_analytic/push_vxb_analytic_cuda.cu). + ``` + +- The error should come as early as possible: propagators/pushers call `get_kernel()` when they are set up, not only at the first time step. Then a GPU run fails right away instead of after the initialization. +- A small overview, e.g. `struphy compile --status` also printing "CUDA kernels: 3 of 60", helps to see what is left to port. + +### PR 4: `Pusher` and propagators use `Kernel` + +- `Pusher` currently asserts `isinstance(kernel, PyccelKernel)` (`pic/pushing/pusher.py`). Allow `Kernel` there and use `kernel.name` for profiling as today. +- On the GPU, the pusher passes the CUDA argument objects (`particles.cuda_args_markers`, `domain.cuda_args_domain`, ...) instead of the pyccel ones. The choice is made once when the pusher is set up, together with the kernel. This avoids calling a CUDA kernel with pyccel arguments or vice versa. + +### PR 5โ€“7: Owners build their CUDA arguments + +- `Particles`, `Domain` and `Derham` own the arrays, so they build the CUDA argument objects from their own `xp` arrays (`cuda_args_markers`, `cuda_args_domain`, `cuda_args_derham`), in the same place where the pyccel argument objects are built today. The pyccel `args_*` stay as they are: they are used by 30+ modules through pyccel kernels. +- The CUDA argument objects hold references. If an owner reallocates an array (today the markers are allocated once), it must rebuild its CUDA arguments at the same place, exactly like for the pyccel arguments. +- First these classes must be creatable on the CuPy backend at all: + - `Particles`: `xp.prod`/`xp.sum` on Python lists and similar (`pic/base.py`), probably more. + - `Derham`: NumPy arrays from feectools reach `cupy.ascontiguousarray`. + - `Domain`: deepcopy on CuPy fails (see PR 5 above). + +### PR 8: Argument structs in shared headers + +Today every CUDA kernel repeats the full flat signature (26 parameters for markers and domain alone). CuPy does not check it, so adding a field to `MarkerArguments` would shift all following arguments of all CUDA kernels **without an error**. + +- Define `struct MarkerArgs { double* markers; bool* valid_mks; int n_markers; ... };` etc. in `kernel_arguments/pusher_args.cuh`, and pass one struct per argument class. +- To check first: how to pass a struct to a `cupy.RawKernel` (e.g. as a NumPy structured scalar with pointer fields). If this does not work well, keep the flat signature, but generate it from the Python class so that it is defined in one place. +- A test compares the struct layout (field names, types, order) with the Python class. + +### PR 9: One folder per kernel + +- Start with `pic/pushing` (`pusher_kernels.py`: 19 kernels, `pusher_kernels_gc.py`: 15, `pusher_kernels_sph.py`: 3, `eval_kernels_gc.py`: 5), later `pic/accumulation` (`accum_kernels.py`: 8, `accum_kernels_gc.py`: 8), then the remaining modules as needed. +- Pure refactor: move each kernel into `/_kernels.py`, update the imports at the call sites and in tests. No CUDA code in this PR. +- Things to check: + - pyccel dependencies between kernel modules are found through imports (`# do not remove; needed to identify dependencies`); the new modules must keep these imports. + - Many small pyccel modules instead of a few large ones: compile time with `struphy compile -j N`, and the import time of many `.so` files. + - A re-export module for the old import paths must not have `kernels` in its name, otherwise `struphy compile` tries to compile it. + +### PR 10: Device helper functions + +The pusher kernels call helpers from other pyccel modules: B-spline evaluation (`bsplines_kernels`, `evaluation_kernels_3d`), mapping evaluation (`geometry/evaluation_kernels`, one module per domain), small linear algebra (`linalg_kernels`), boundary conditions (`pusher_utilities_kernels`). Each needs a `__device__` version in a `.cuh` header before the kernels using it can be ported. + +- Port only what the next kernel needs, not whole modules at once. +- Scratch arrays that the pyccel classes allocate once (e.g. `DerhamArguments.bn1`, ..., `bd3`) become per-thread local arrays in CUDA (fixed maximum spline degree, or template parameter). +- Each device helper gets a test against its pyccel version through a small test kernel. + +### PR 11: First real kernel: `push_eta_stage` + +- Write `push_eta_stage_cuda.cu`, using the device helpers from PR 10. +- Parity test: same markers, both backends, results agree to round-off (`rtol ~ 1e-13`). +- End-to-end: run a propagator that only needs this kernel with `ARRAY_BACKEND=cupy`, and check that no host/device transfers happen inside the time loop (e.g. with `nsys` or by counting CuPy memory copies). + +### PR 12+: Port kernels one by one + +For each kernel: add `_cuda.cu`, a parity test is added automatically by the catalog (every kernel with a CUDA version is run on both backends with the same inputs), and the kernel is removed from the "missing" list. + +## Porting order + +Port the kernels in the order the target models need them, so that complete models can run on the GPU as early as possible. Proposed: + +1. `push_eta_stage` (PR 11) and the helpers it needs. +2. The remaining 6D full-orbit pushers (`pusher_kernels.py`), e.g. `push_vxb_analytic`, `push_v_with_efield`. +3. The accumulation kernels these models need (`accum_kernels.py`). Note: accumulation writes to shared grid arrays from many threads, so it needs atomics or a sort-then-reduce strategy. This is a design question of its own. +4. Guiding-center pushers and evaluations (`pusher_kernels_gc.py`, `eval_kernels_gc.py`, `accum_kernels_gc.py`). +5. SPH kernels. + +## Testing + +- Every kernel with a CUDA version has a parity test (pyccel vs. CUDA, same inputs), generated from the catalog. +- CUDA tests are skipped when no GPU is available (`cunumpy.cupy_available()`), so the normal CI keeps working. A GPU runner in CI runs them. +- Regression tests on the CPU (`pic/tests/test_pushers.py`, `pic/tests/test_kernel_setup.py`, ...) must pass in every PR. + +## Open questions + +- **Marker layout.** The markers array is row-major (`n_markers ร— n_cols`). With one thread per marker, the memory accesses are strided. This is fine for now (each thread reads a few neighbouring columns), but a column-major or struct-of-arrays layout may be faster later. This would affect the CPU code too, so it is out of scope here. +- **MPI + GPUs.** One GPU per MPI rank (`cunumpy.set_device(rank % n_gpus)`), and GPU-aware MPI for the marker exchange, so markers do not go through the host. +- **Single-source alternatives.** Before porting a large number of kernels by hand, it may be worth checking whether some of them can be generated from the Python source (e.g. `cupyx.jit` or numba-cuda) instead of written twice. Hand-written CUDA stays the default. +- **Kernel launch configuration.** One thread per marker with a fixed block size for now. Kernels over grid points (accumulation, FEEC) will need their own launch sizes, so `CudaKernel` will need a way to set them. From de851a7aa6d974f3ab6bb8dd2f71d44957712728 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 16:06:34 +0200 Subject: [PATCH 10/29] Combined demo files and test file --- CUDA_STRATEGY.md | 8 +- src/struphy/pic/pushing/demo_cuda.py | 88 ------ src/struphy/pic/pushing/demo_kernels.py | 35 --- src/struphy/pic/tests/test_kernel_backends.py | 284 ++++++++++++++++-- 4 files changed, 270 insertions(+), 145 deletions(-) delete mode 100644 src/struphy/pic/pushing/demo_cuda.py delete mode 100644 src/struphy/pic/pushing/demo_kernels.py diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index 155137cbe..01e98612d 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -6,7 +6,7 @@ The work is split into small PRs that can be reviewed and merged one at a time. ## PR checklist - [ ] **PR 1: Proof of concept** (branch `cuda-kernel-proof-of-concept`) - `Kernel` (pyccel/CUDA pair), `CudaKernel`, `CudaMarkerArguments`/`CudaDomainArguments`, one demo kernel pair and a test on both backends. Also adds this document and the `gpu` optional dependency. + `Kernel` (pyccel/CUDA pair), `CudaKernel`, `CudaMarkerArguments`/`CudaDomainArguments`, and one test file with a demo kernel pair run on both backends. Also adds this document and the `gpu` optional dependency. - [ ] **PR 2: CUDA source files** โ€” `CudaKernel` loads CUDA source from a `_cuda.cu` file next to the pyccel file; `.cu`/`.cuh` files are shipped as package data. - [ ] **PR 3: Kernel catalog** โ€” kernels are defined once in the `__init__.py` of the folder that contains them; a missing CUDA kernel raises an error on the GPU backend. - [ ] **PR 4: `Pusher` and propagators accept `Kernel`** โ€” replace `PyccelKernel(...)` in the propagators by catalog lookups (no CUDA kernels yet, so no behaviour change on CPU). @@ -47,9 +47,7 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke |---|---| | `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily, one thread per marker), `Kernel` (pyccel/CUDA pair, `get_kernel()` picks by backend) | | `src/struphy/utils/cuda_arguments.py` | `CudaMarkerArguments`, `CudaDomainArguments`: same constructor arguments as the pyccel classes, hold CuPy arrays, flatten them into the CUDA kernel arguments | -| `src/struphy/pic/pushing/demo_kernels.py` | pyccel `push_eta_linear` | -| `src/struphy/pic/pushing/demo_cuda.py` | CUDA `push_eta_linear` (source string), the `Kernel` pair and a demo `main()` comparing both backends | -| `src/struphy/pic/tests/test_kernel_backends.py` | tests on both backends | +| `src/struphy/pic/tests/test_kernel_backends.py` | the demo kernel pair `push_eta_linear` (pyccel function compiled with `epyccel` at test time, CUDA source string) and tests on both backends | Things we learned in the proof of concept: @@ -100,7 +98,7 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - `CudaKernel.from_file(path, name)` reads `_cuda.cu`. The kernel is compiled lazily on first call. CuPy caches compiled kernels on disk (`~/.cupy/kernel_cache`), so the compile cost is paid once per machine. - Headers are found through NVRTC include paths (`cupy.RawModule(code=..., options=("-I",))`), so a `.cu` file can `#include "struphy/bsplines/bsplines_kernels.cuh"`. - Add `"**/*.cu"` and `"**/*.cuh"` to `[tool.setuptools.package-data]` in `pyproject.toml`. -- Move the demo kernel's CUDA source from the string in `demo_cuda.py` into `demo_cuda.cu`. +- Move the test kernel's CUDA source from the string in `test_kernel_backends.py` into a `.cu` file. ### PR 3: Kernel catalog diff --git a/src/struphy/pic/pushing/demo_cuda.py b/src/struphy/pic/pushing/demo_cuda.py deleted file mode 100644 index ca292ef41..000000000 --- a/src/struphy/pic/pushing/demo_cuda.py +++ /dev/null @@ -1,88 +0,0 @@ -"""CUDA counterpart of :mod:`struphy.pic.pushing.demo_kernels`. - -Run ``python -m struphy.pic.pushing.demo_cuda`` to push markers with both backends and compare. -""" - -import cunumpy -import numpy as np -from cunumpy import PyccelKernel - -from struphy.geometry.domains import Cuboid -from struphy.kernel_arguments.pusher_args_kernels import MarkerArguments -from struphy.pic.pushing import demo_kernels -from struphy.utils.cuda_arguments import CudaDomainArguments, CudaMarkerArguments -from struphy.utils.kernel_backends import CudaKernel, Kernel, is_cuda_backend - -# Arguments: (dt, stage, CudaMarkerArguments, CudaDomainArguments), see struphy.utils.cuda_arguments. -PUSH_ETA_LINEAR_SRC = r""" -extern "C" __global__ -void push_eta_linear( - double dt, int stage, - double* markers, bool* valid_mks, int n_markers, int n_cols, - int Np, int vdim, int weight_idx, int first_diagnostics_idx, int first_init_idx, - int first_shift_idx, int residual_idx, int first_free_idx, int mu_idx, long long* bc_type, - int kind_map, double* params, long long* degree, - double* t1, double* t2, double* t3, - long long* ind1, long long* ind2, long long* ind3, - double* cx, double* cy, double* cz) -{ - int ip = blockDim.x * blockIdx.x + threadIdx.x; - - // only do something if particle is valid (i.e. not a hole or ghost) - if (ip >= n_markers || !valid_mks[ip]) return; - - double* mk = markers + (long long)ip * n_cols; - mk[0] += dt * mk[3]; - mk[1] += dt * mk[4]; - mk[2] += dt * mk[5]; -} -""" - -push_eta_linear = Kernel( - pyccel_kernel=PyccelKernel(demo_kernels.push_eta_linear), - cuda_kernel=CudaKernel(PUSH_ETA_LINEAR_SRC, "push_eta_linear"), -) - - -def make_demo_arguments(n_markers: int, seed: int = 0): - """Random markers and a Cuboid domain, as kernel arguments for the active cunumpy backend. - - The arrays are created on the active backend (on the device for CuPy); the arguments reference them without copies. - """ - rng = np.random.default_rng(seed) - markers = cunumpy.asarray(rng.random((n_markers, 25))) - valid_mks = cunumpy.asarray(rng.random(n_markers) > 0.1) - bc_type = cunumpy.zeros(3, dtype=int) - - domain = Cuboid() - if not is_cuda_backend(): - return MarkerArguments(markers, valid_mks, n_markers, 3, 6, 7, 8, 14, 17, 18, 4, bc_type), domain.args_domain - - args_markers = CudaMarkerArguments(markers, valid_mks, n_markers, 3, 6, 7, 8, 14, 17, 18, 4, bc_type) - args_domain = CudaDomainArguments( - domain.kind_map, - domain.params_numpy, - cunumpy.asarray(domain.degree), - *domain.T, - *domain.indN, - domain.cx, - domain.cy, - domain.cz, - ) - return args_markers, args_domain - - -def main(n_markers: int = 1_000_000, n_steps: int = 100, dt: float = 1e-3): - results = {} - for backend in ("numpy", "cupy"): - with cunumpy.use_backend(backend): - args_markers, args_domain = make_demo_arguments(n_markers) - for _ in range(n_steps): - push_eta_linear(dt, 0, args_markers, args_domain) - results[backend] = cunumpy.to_numpy(args_markers.markers) - - print(f"max |pyccel - cuda| = {np.max(np.abs(results['numpy'] - results['cupy'])):.2e}") - - -if __name__ == "__main__": - main() diff --git a/src/struphy/pic/pushing/demo_kernels.py b/src/struphy/pic/pushing/demo_kernels.py deleted file mode 100644 index 419a558e6..000000000 --- a/src/struphy/pic/pushing/demo_kernels.py +++ /dev/null @@ -1,35 +0,0 @@ -"""Minimal pusher kernel with a 1:1 CUDA counterpart in :mod:`struphy.pic.pushing.demo_cuda`, -used to test the pyccel/CUDA kernel dispatch in :mod:`struphy.utils.kernel_backends`.""" - -# do not remove; needed to identify dependencies -import struphy.kernel_arguments.pusher_args_kernels as pusher_args_kernels -from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments - - -def push_eta_linear( - dt: float, - stage: int, - args_markers: "MarkerArguments", - args_domain: "DomainArguments", -): - r"""Explicit Euler step of - - .. math:: - - \frac{\textnormal d \boldsymbol \eta_p(t)}{\textnormal d t} = \mathbf v_p - - for each valid marker :math:`p`, where :math:`\mathbf v_p` is constant (no mapping, no boundary conditions). - """ - - markers = args_markers.markers - n_markers = args_markers.n_markers - valid_mks = args_markers.valid_mks - - for ip in range(n_markers): - # only do something if particle is valid (i.e. not a hole or ghost) - if not valid_mks[ip]: - continue - - markers[ip, 0] += dt * markers[ip, 3] - markers[ip, 1] += dt * markers[ip, 4] - markers[ip, 2] += dt * markers[ip, 5] diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index 2f62fe7cd..c93b2bfda 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -1,36 +1,286 @@ -"""Proof of concept: 1:1 pyccel and CUDA kernels, dispatched by the cunumpy backend.""" +"""Proof of concept: 1:1 pyccel and CUDA kernels, dispatched by the cunumpy backend. + +The pyccel kernel :func:`push_eta_linear` is compiled with ``epyccel`` at test time (``struphy compile`` +skips test files); its CUDA counterpart is :data:`PUSH_ETA_LINEAR_SRC`. See ``CUDA_STRATEGY.md``. +""" + +import importlib +import inspect +import sys import cunumpy import numpy as np import pytest +from cunumpy import PyccelKernel -from struphy.pic.pushing.demo_cuda import make_demo_arguments, push_eta_linear +from struphy.geometry.domains import Cuboid +from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments +from struphy.utils.cuda_arguments import CudaDomainArguments, CudaMarkerArguments +from struphy.utils.kernel_backends import CudaKernel, Kernel, is_cuda_backend requires_cupy = pytest.mark.skipif(not cunumpy.cupy_available(), reason="CuPy/GPU not available") +N_COLS = 25 +MARKER_INDICES = (3, 6, 7, 8, 14, 17, 18, 4) # vdim, weight_idx, ..., mu_idx + + +# --------------------------------- +# the kernel pair: pyccel and CUDA +# --------------------------------- + + +def push_eta_linear( + dt: float, + stage: int, + args_markers: "MarkerArguments", + args_domain: "DomainArguments", +): + """Explicit Euler step eta <- eta + dt * v for each valid marker (pyccel kernel).""" + + markers = args_markers.markers + n_markers = args_markers.n_markers + valid_mks = args_markers.valid_mks + + for ip in range(n_markers): + # only do something if particle is valid (i.e. not a hole or ghost) + if not valid_mks[ip]: + continue + + markers[ip, 0] += dt * markers[ip, 3] + markers[ip, 1] += dt * markers[ip, 4] + markers[ip, 2] += dt * markers[ip, 5] + + +# Arguments: (dt, stage, CudaMarkerArguments, CudaDomainArguments), see struphy.utils.cuda_arguments. +CUDA_ARGS = r""" + double dt, int stage, + double* markers, bool* valid_mks, int n_markers, int n_cols, + int Np, int vdim, int weight_idx, int first_diagnostics_idx, int first_init_idx, + int first_shift_idx, int residual_idx, int first_free_idx, int mu_idx, long long* bc_type, + int kind_map, double* params, long long* degree, + double* t1, double* t2, double* t3, + long long* ind1, long long* ind2, long long* ind3, + double* cx, double* cy, double* cz +""" + +PUSH_ETA_LINEAR_SRC = f""" +extern "C" __global__ +void push_eta_linear({CUDA_ARGS}) +{{ + int ip = blockDim.x * blockIdx.x + threadIdx.x; + + // only do something if particle is valid (i.e. not a hole or ghost) + if (ip >= n_markers || !valid_mks[ip]) return; + + double* mk = markers + (long long)ip * n_cols; + mk[0] += dt * mk[3]; + mk[1] += dt * mk[4]; + mk[2] += dt * mk[5]; +}} +""" + +# writes the scalar arguments into the markers, to check that they arrive with the right types +WRITE_SCALARS_SRC = f""" +extern "C" __global__ +void write_scalars({CUDA_ARGS}) +{{ + int ip = blockDim.x * blockIdx.x + threadIdx.x; + if (ip >= n_markers) return; + + double* mk = markers + (long long)ip * n_cols; + mk[0] = dt; + mk[1] = stage; + mk[2] = n_cols; + mk[3] = first_init_idx; + mk[4] = mu_idx; + mk[5] = kind_map; +}} +""" + + +@pytest.fixture(scope="module") +def kernel(tmp_path_factory): + """The Kernel pair, with the pyccel kernel compiled by epyccel.""" + from pyccel import epyccel + + src_dir = tmp_path_factory.mktemp("pyccel_src") + src = "from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments\n\n\n" + src += inspect.getsource(push_eta_linear) + (src_dir / "poc_push_kernels.py").write_text(src) + + sys.path.insert(0, str(src_dir)) + try: + module = epyccel(importlib.import_module("poc_push_kernels"), language="fortran") + finally: + sys.path.remove(str(src_dir)) + + return Kernel( + pyccel_kernel=PyccelKernel(module.push_eta_linear), + cuda_kernel=CudaKernel(PUSH_ETA_LINEAR_SRC, "push_eta_linear"), + ) + + +def make_arguments(n_markers: int, seed: int = 0): + """Random markers (some holes) and a Cuboid domain, as kernel arguments for the active cunumpy backend. + + The arrays are created on the active backend (on the device for CuPy); the arguments reference them without copies. + """ + rng = np.random.default_rng(seed) + markers = cunumpy.asarray(rng.random((n_markers, N_COLS))) + valid_mks = cunumpy.asarray(rng.random(n_markers) > 0.1) + bc_type = cunumpy.zeros(3, dtype=int) + + domain = Cuboid() + if not is_cuda_backend(): + args_markers = MarkerArguments(markers, valid_mks, n_markers, *MARKER_INDICES, bc_type) + return args_markers, domain.args_domain + + args_markers = CudaMarkerArguments(markers, valid_mks, n_markers, *MARKER_INDICES, bc_type) + args_domain = CudaDomainArguments( + domain.kind_map, + domain.params_numpy, + cunumpy.asarray(domain.degree), + *domain.T, + *domain.indN, + domain.cx, + domain.cy, + domain.cz, + ) + return args_markers, args_domain + + +def expected_push(markers, valid_mks, dt, n_steps=1): + out = markers.copy() + for _ in range(n_steps): + out[valid_mks, 0:3] += dt * out[valid_mks, 3:6] + return out + + +BACKENDS = ["numpy", pytest.param("cupy", marks=requires_cupy)] -@pytest.mark.parametrize("backend", ["numpy", pytest.param("cupy", marks=requires_cupy)]) -def test_push_eta_linear(backend): - """The kernel of the active backend updates the marker array in place.""" + +# --------------------------------- +# tests +# --------------------------------- + + +@pytest.mark.parametrize("backend", BACKENDS) +def test_kernel_dispatch(kernel, backend): + with cunumpy.use_backend(backend): + assert is_cuda_backend() == (backend == "cupy") + expected = kernel.cuda_kernel if backend == "cupy" else kernel.pyccel_kernel + assert kernel.get_kernel() is expected + + +@pytest.mark.parametrize("backend", BACKENDS) +@pytest.mark.parametrize("n_markers", [1, 129, 1000]) +def test_push_eta_linear(kernel, backend, n_markers): + """One step, compared to the analytic result; holes are not touched. 129 is not a multiple of the block size.""" dt = 0.1 with cunumpy.use_backend(backend): - args_markers, args_domain = make_demo_arguments(1000) + args_markers, args_domain = make_arguments(n_markers) markers = args_markers.markers valid = cunumpy.to_numpy(args_markers.valid_mks) - expected = cunumpy.to_numpy(markers).copy() - expected[valid, 0:3] += dt * expected[valid, 3:6] + before = cunumpy.to_numpy(markers).copy() - push_eta_linear(dt, 0, args_markers, args_domain) + kernel(dt, 0, args_markers, args_domain) - assert np.allclose(cunumpy.to_numpy(markers), expected, rtol=1e-14, atol=0.0) + after = cunumpy.to_numpy(markers) + assert np.allclose(after, expected_push(before, valid, dt), rtol=1e-14, atol=0.0) + assert np.array_equal(after[~valid], before[~valid]) @requires_cupy -def test_cuda_arguments_reject_host_arrays(): - """Host arrays are never copied to the device.""" - with cunumpy.use_backend("numpy"): - host_markers, _ = make_demo_arguments(10) - with cunumpy.use_backend("cupy"), pytest.raises(TypeError, match="CuPy array"): - from struphy.utils.cuda_arguments import CudaMarkerArguments +def test_pyccel_cuda_agree(kernel): + """Same markers pushed for many steps on both backends (what used to be the demo).""" + dt, n_steps, n_markers = 1e-3, 100, 100_000 + results = {} + for backend in ("numpy", "cupy"): + with cunumpy.use_backend(backend): + args_markers, args_domain = make_arguments(n_markers, seed=1) + if backend == "numpy": + expected = expected_push(args_markers.markers, args_markers.valid_mks, dt, n_steps) + for _ in range(n_steps): + kernel(dt, 0, args_markers, args_domain) + results[backend] = cunumpy.to_numpy(args_markers.markers) + + # not bitwise equal: nvcc contracts x + dt * v into fused multiply-adds by default + assert np.allclose(results["numpy"], expected, rtol=1e-13, atol=0.0) + assert np.allclose(results["cupy"], results["numpy"], rtol=1e-12, atol=0.0) + + +@requires_cupy +def test_cuda_kernel_updates_device_array_in_place(kernel): + """The CUDA kernel works on the very array created on the device; nothing is replaced or copied.""" + with cunumpy.use_backend("cupy"): + args_markers, args_domain = make_arguments(1000) + markers = args_markers.markers + ptr = markers.data.ptr + + for _ in range(10): + kernel(0.1, 0, args_markers, args_domain) + + assert args_markers.markers is markers + assert markers.data.ptr == ptr + assert args_markers.values[0] is markers + - CudaMarkerArguments(host_markers.markers, host_markers.valid_mks, 10, 3, 6, 7, 8, 14, 17, 18, 4, np.zeros(3)) +@requires_cupy +def test_cuda_scalar_arguments(): + """Python scalars and the flattened argument classes arrive in the CUDA kernel with the right types and order.""" + write_scalars = CudaKernel(WRITE_SCALARS_SRC, "write_scalars") + with cunumpy.use_backend("cupy"): + args_markers, args_domain = make_arguments(10) + write_scalars(0.25, 3, args_markers, args_domain) + + row = cunumpy.to_numpy(args_markers.markers)[0, :6] + first_pusher_idx, mu_idx = MARKER_INDICES[3], MARKER_INDICES[7] + assert np.array_equal(row, [0.25, 3, N_COLS, first_pusher_idx, mu_idx, Cuboid().kind_map]) + + +@requires_cupy +def test_cuda_domain_arguments_reference_domain_arrays(): + with cunumpy.use_backend("cupy"): + domain = Cuboid() + args = CudaDomainArguments( + domain.kind_map, + domain.params_numpy, + cunumpy.asarray(domain.degree), + *domain.T, + *domain.indN, + domain.cx, + domain.cy, + domain.cz, + ) + assert len(args.values) == 12 + assert args.values[3] is domain.T[0] and args.values[8] is domain.indN[2] and args.values[9] is domain.cx + + +@requires_cupy +def test_cuda_arguments_reject_host_and_bad_arrays(): + """Host arrays are never copied to the device, and wrong dtypes or layouts are not converted.""" + import cupy as cp + + markers = cp.zeros((10, N_COLS)) + valid_mks = cp.ones(10, dtype=bool) + bc_type = cp.zeros(3, dtype=int) + + CudaMarkerArguments(markers, valid_mks, 10, *MARKER_INDICES, bc_type) # ok + for bad_markers in (markers.get(), markers.astype(np.float32), cp.zeros((N_COLS, 10)).T): + with pytest.raises(TypeError): + CudaMarkerArguments(bad_markers, valid_mks, 10, *MARKER_INDICES, bc_type) + with pytest.raises(TypeError): + CudaMarkerArguments(markers, valid_mks.get(), 10, *MARKER_INDICES, bc_type) + + +@requires_cupy +def test_cuda_kernel_rejects_pyccel_arguments(kernel): + """Passing the pyccel argument classes to the CUDA kernel fails instead of copying.""" + with cunumpy.use_backend("numpy"): + host_markers, host_domain = make_arguments(10) + with cunumpy.use_backend("cupy"): + cuda_markers, cuda_domain = make_arguments(10) + with pytest.raises(ValueError, match="CudaMarkerArguments"): + kernel(0.1, 0, host_markers, cuda_domain) + with pytest.raises(TypeError): + kernel(0.1, 0, cuda_markers, host_domain) From 8a30ba2aa7461b22ba60efaab303b9c43c4ec801 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 17:27:25 +0200 Subject: [PATCH 11/29] Pass CUDA kernel arguments as they are and n_threads explicitly CudaKernel no longer flattens argument objects or casts scalars at call time: the arguments must already be in the format of a cupy.RawKernel (flat tuple of CuPy arrays and NumPy scalars, built once at setup), and the number of threads is passed as n_threads. Add docstrings to the kernel and argument classes. Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 6 +- src/struphy/pic/tests/test_kernel_backends.py | 43 ++++++-- src/struphy/utils/cuda_arguments.py | 104 +++++++++++++++++- src/struphy/utils/kernel_backends.py | 101 ++++++++++++----- 4 files changed, 207 insertions(+), 47 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index 01e98612d..1112b154a 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -34,7 +34,7 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke ## Principles -- **1:1 correspondence.** Each CUDA kernel has the same name and the same arguments (in the same order) as its pyccel kernel. The call site does not know which one runs. +- **1:1 correspondence.** Each CUDA kernel has the same name and the same arguments (in the same order) as its pyccel kernel. For the CUDA kernel, the argument classes are flattened (`CudaMarkerArguments.values` etc.), and the caller also passes the number of threads (`n_threads`). - **The backend decides.** The cunumpy backend (`ARRAY_BACKEND=cupy` or `cunumpy.set_backend("cupy")`) selects the CUDA kernels; with NumPy the pyccel kernels run as today. - **No conversions at call time.** When a kernel is called, its arguments are already in the right format. There are no host/device copies per kernel call. - **Data already lives on the GPU.** On the CuPy backend, `xp` is `cupy`, so markers, spline coefficients etc. are CuPy arrays from the start. The CUDA argument objects only collect *references* to these arrays and raise if they get host arrays. @@ -45,7 +45,7 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke | File | Content | |---|---| -| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily, one thread per marker), `Kernel` (pyccel/CUDA pair, `get_kernel()` picks by backend) | +| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; arguments are passed as they are, plus `n_threads`), `Kernel` (pyccel/CUDA pair, `get_kernel()` picks by backend) | | `src/struphy/utils/cuda_arguments.py` | `CudaMarkerArguments`, `CudaDomainArguments`: same constructor arguments as the pyccel classes, hold CuPy arrays, flatten them into the CUDA kernel arguments | | `src/struphy/pic/tests/test_kernel_backends.py` | the demo kernel pair `push_eta_linear` (pyccel function compiled with `epyccel` at test time, CUDA source string) and tests on both backends | @@ -54,7 +54,7 @@ Things we learned in the proof of concept: - The pyccel-compiled argument classes (`MarkerArguments`, `DomainArguments`, `DerhamArguments`) hold references to their owner's arrays, but only accept **NumPy** arrays. Hence the CUDA counterparts in `cuda_arguments.py`. - Today, `Particles` builds `args_markers` from `_to_numpy_for_kernel(self.markers)`, i.e. from a **host copy** when the backend is CuPy. The same holds for `Domain` and `Derham`. - `Particles6D` and `Derham` cannot be created on the CuPy backend yet (PR 6, PR 7). `Domain` (e.g. `Cuboid`) can, and all its arrays are already CuPy arrays. -- `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Python `int`s are passed as 64-bit integers, so scalars are cast explicitly (`int` โ†’ `np.int32`, `float` โ†’ `np.float64`). +- `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Python `int`s are passed as 64-bit integers, so the caller passes NumPy scalars of the C type in the signature (`np.int32` for `int`, `np.float64` for `double`). `CudaKernel` does not convert anything; the flat argument tuple is built once, at setup. - `struphy compile` compiles every `.py` file whose name contains `kernels`. Non-pyccel modules must not contain `kernels` in their name; `.cu` files are ignored by it. - On an H100, the demo kernel pushes 10โถ markers in about 0.13 ms per step. diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index c93b2bfda..2f29ba224 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -149,6 +149,17 @@ def make_arguments(n_markers: int, seed: int = 0): return args_markers, args_domain +def kernel_args(dt: float, stage: int, args_markers, args_domain) -> tuple: + """Kernel arguments in the format of the kernel of the active backend (built once, before the calls). + + For CUDA, the flat tuple of a ``cupy.RawKernel``: NumPy scalars with the C types of the signature + followed by the flattened argument classes. + """ + if not is_cuda_backend(): + return dt, stage, args_markers, args_domain + return np.float64(dt), np.int32(stage), *args_markers.values, *args_domain.values + + def expected_push(markers, valid_mks, dt, n_steps=1): out = markers.copy() for _ in range(n_steps): @@ -182,8 +193,9 @@ def test_push_eta_linear(kernel, backend, n_markers): markers = args_markers.markers valid = cunumpy.to_numpy(args_markers.valid_mks) before = cunumpy.to_numpy(markers).copy() + args = kernel_args(dt, 0, args_markers, args_domain) - kernel(dt, 0, args_markers, args_domain) + kernel(*args, n_threads=n_markers) after = cunumpy.to_numpy(markers) assert np.allclose(after, expected_push(before, valid, dt), rtol=1e-14, atol=0.0) @@ -200,8 +212,9 @@ def test_pyccel_cuda_agree(kernel): args_markers, args_domain = make_arguments(n_markers, seed=1) if backend == "numpy": expected = expected_push(args_markers.markers, args_markers.valid_mks, dt, n_steps) + args = kernel_args(dt, 0, args_markers, args_domain) for _ in range(n_steps): - kernel(dt, 0, args_markers, args_domain) + kernel(*args, n_threads=n_markers) results[backend] = cunumpy.to_numpy(args_markers.markers) # not bitwise equal: nvcc contracts x + dt * v into fused multiply-adds by default @@ -216,9 +229,10 @@ def test_cuda_kernel_updates_device_array_in_place(kernel): args_markers, args_domain = make_arguments(1000) markers = args_markers.markers ptr = markers.data.ptr + args = kernel_args(0.1, 0, args_markers, args_domain) for _ in range(10): - kernel(0.1, 0, args_markers, args_domain) + kernel(*args, n_threads=1000) assert args_markers.markers is markers assert markers.data.ptr == ptr @@ -227,11 +241,11 @@ def test_cuda_kernel_updates_device_array_in_place(kernel): @requires_cupy def test_cuda_scalar_arguments(): - """Python scalars and the flattened argument classes arrive in the CUDA kernel with the right types and order.""" + """The scalars and the flattened argument classes arrive in the CUDA kernel with the right types and order.""" write_scalars = CudaKernel(WRITE_SCALARS_SRC, "write_scalars") with cunumpy.use_backend("cupy"): args_markers, args_domain = make_arguments(10) - write_scalars(0.25, 3, args_markers, args_domain) + write_scalars(*kernel_args(0.25, 3, args_markers, args_domain), n_threads=10) row = cunumpy.to_numpy(args_markers.markers)[0, :6] first_pusher_idx, mu_idx = MARKER_INDICES[3], MARKER_INDICES[7] @@ -274,13 +288,22 @@ def test_cuda_arguments_reject_host_and_bad_arrays(): @requires_cupy -def test_cuda_kernel_rejects_pyccel_arguments(kernel): - """Passing the pyccel argument classes to the CUDA kernel fails instead of copying.""" +def test_cuda_kernel_does_not_convert_arguments(kernel): + """Arguments are passed to the RawKernel as they are: host data or the argument objects themselves fail.""" with cunumpy.use_backend("numpy"): host_markers, host_domain = make_arguments(10) with cunumpy.use_backend("cupy"): cuda_markers, cuda_domain = make_arguments(10) - with pytest.raises(ValueError, match="CudaMarkerArguments"): - kernel(0.1, 0, host_markers, cuda_domain) with pytest.raises(TypeError): - kernel(0.1, 0, cuda_markers, host_domain) + kernel(np.float64(0.1), np.int32(0), cuda_markers, cuda_domain, n_threads=10) + with pytest.raises(TypeError): + kernel( + np.float64(0.1), + np.int32(0), + host_markers.markers, + *cuda_markers.values[1:], + *cuda_domain.values, + n_threads=10, + ) + with pytest.raises(ValueError, match="n_threads"): + kernel(*kernel_args(0.1, 0, cuda_markers, cuda_domain)) diff --git a/src/struphy/utils/cuda_arguments.py b/src/struphy/utils/cuda_arguments.py index b34bc19b3..a1e89dbf8 100644 --- a/src/struphy/utils/cuda_arguments.py +++ b/src/struphy/utils/cuda_arguments.py @@ -10,7 +10,24 @@ def _cupy_array(name: str, arr, dtype): - """Return ``arr`` itself after checking it is a C-contiguous CuPy array of ``dtype``.""" + """Check that ``arr`` is a C-contiguous CuPy array of ``dtype`` (never converts or copies). + + Parameters + ---------- + name : str + Name of the argument, for error messages. + + arr : cupy.ndarray + The array to check. + + dtype : type + Expected dtype. + + Returns + ------- + cupy.ndarray + ``arr`` itself. + """ if not hasattr(arr, "__cuda_array_interface__"): raise TypeError(f"{name} must be a CuPy array, got {type(arr)}.") if arr.dtype != dtype or not arr.flags.c_contiguous: @@ -21,9 +38,61 @@ def _cupy_array(name: str, arr, dtype): class CudaMarkerArguments: """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.MarkerArguments`. - CUDA signature: ``double* markers, bool* valid_mks, int n_markers, int n_cols, int Np, int vdim, - int weight_idx, int first_diagnostics_idx, int first_init_idx, int first_shift_idx, + CUDA signature of :attr:`values`: ``double* markers, bool* valid_mks, int n_markers, int n_cols, int Np, + int vdim, int weight_idx, int first_diagnostics_idx, int first_init_idx, int first_shift_idx, int residual_idx, int first_free_idx, int mu_idx, long long* bc_type`` + + Parameters + ---------- + markers : cupy.ndarray[float] + Markers array (C-contiguous, float64). + + valid_mks : cupy.ndarray[bool] + True for valid markers (not holes or ghosts). + + Np : int + Total number of particles. + + vdim : int + Dimension of velocity space. + + weight_idx : int + Column index of particle weight. + + first_diagnostics_idx : int + Starting index for diagnostics columns. + + first_pusher_idx : int + Starting buffer marker index number for pusher. + + first_shift_idx : int + First index for storing shifts due to boundary conditions in eta-space. + + residual_idx : int + Column for storing the residual in iterative pushers. + + first_free_idx : int + First index for storing auxiliary quantities for each particle. + + mu_idx : int + Column index of particle magnetic moment. + + bc_type : cupy.ndarray[int] + Kinetic boundary condition in each logical direction (int64). + + Attributes + ---------- + markers : cupy.ndarray[float] + The markers array passed in (no copy). + + valid_mks : cupy.ndarray[bool] + The array of valid markers passed in (no copy). + + n_markers : int + Number of rows of ``markers``, e.g. for the number of CUDA threads. + + values : tuple + Flat CUDA kernel arguments, see the signature above. """ def __init__( @@ -70,8 +139,33 @@ def __init__( class CudaDomainArguments: """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments`. - CUDA signature: ``int kind_map, double* params, long long* degree, double* t1, double* t2, double* t3, - long long* ind1, long long* ind2, long long* ind3, double* cx, double* cy, double* cz`` + CUDA signature of :attr:`values`: ``int kind_map, double* params, long long* degree, double* t1, + double* t2, double* t3, long long* ind1, long long* ind2, long long* ind3, double* cx, double* cy, double* cz`` + + Parameters + ---------- + kind_map : int + Mapping identifier of :class:`~struphy.geometry.base.Domain`. + + params : cupy.ndarray[float] + Mapping parameters. + + degree : cupy.ndarray[int] + Spline degrees of the mapping. + + t1, t2, t3 : cupy.ndarray[float] + Knot sequences of the mapping. + + ind1, ind2, ind3 : cupy.ndarray[int] + Indices of non-vanishing splines in format (number of mapping grid cells, degree + 1). + + cx, cy, cz : cupy.ndarray[float] + Spline coefficients (control points) of the mapping. + + Attributes + ---------- + values : tuple + Flat CUDA kernel arguments, see the signature above. """ def __init__(self, kind_map: int, params, degree, t1, t2, t3, ind1, ind2, ind3, cx, cy, cz): diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index cbf5b1b98..a19a1d7ed 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -1,22 +1,48 @@ -"""Pairs of pyccel and CUDA kernels, selected at runtime from the cunumpy backend.""" +"""Pairs of pyccel and CUDA kernels, selected at runtime from the cunumpy backend. + +A :class:`Kernel` holds a pyccel kernel and its 1:1 corresponding CUDA kernel (:class:`CudaKernel`) +and calls the one matching the active cunumpy backend, see :func:`is_cuda_backend`. + +The arguments must already be in the format of the kernel that is called; nothing is converted +at call time. For the CUDA kernel, this is the flat tuple of a ``cupy.RawKernel``: CuPy arrays and +NumPy scalars of the C types in the kernel signature (e.g. ``np.float64`` for ``double``, +``np.int32`` for ``int``). The argument classes are flattened with +:attr:`~struphy.utils.cuda_arguments.CudaMarkerArguments.values` etc. This tuple is meant to be +built once, at setup, not before every call. +""" import math -import numpy as np from cunumpy import PyccelKernel from cunumpy.xp import array_backend def is_cuda_backend() -> bool: - """Whether the active cunumpy backend is CuPy.""" + """Whether the active cunumpy backend is CuPy. + + Returns + ------- + bool + True if the backend is ``"cupy"``, False if it is ``"numpy"``. + """ return array_backend.backend == "cupy" class CudaKernel: - """A ``cupy.RawKernel`` called with the 1:1 corresponding arguments of its pyccel counterpart. + """A ``cupy.RawKernel``, the CUDA counterpart of a pyccel kernel. + + The kernel is compiled on the first call. Arguments are passed to the ``cupy.RawKernel`` as they are. + + Parameters + ---------- + source : str + CUDA C source code containing the ``extern "C" __global__`` function ``name``. + + name : str + Name of the kernel function in ``source``. - Argument classes must already be the CUDA versions (:mod:`struphy.utils.cuda_arguments`); - no arrays are converted or copied at call time. One thread is launched per marker. + block_size : int + Number of threads per block. """ def __init__(self, source: str, name: str, block_size: int = 128): @@ -25,43 +51,60 @@ def __init__(self, source: str, name: str, block_size: int = 128): self._block_size = block_size self._raw_kernel = None - def __call__(self, *args): + def __call__(self, *args, n_threads: int): + """Launch the kernel. + + Parameters + ---------- + *args + Kernel arguments in the format of a ``cupy.RawKernel``: CuPy arrays and NumPy scalars + with the C types of the kernel signature. + + n_threads : int + Number of threads to launch (e.g. number of markers); rounded up to a multiple of the block size. + """ if self._raw_kernel is None: import cupy as cp self._raw_kernel = cp.RawKernel(self._source, self.name) - values = [] - n_threads = None - for arg in args: - if hasattr(arg, "values"): - values += arg.values - n_threads = n_threads or getattr(arg, "n_markers", None) - elif isinstance(arg, bool): - values.append(np.bool_(arg)) - elif isinstance(arg, int): - values.append(np.int32(arg)) - elif isinstance(arg, float): - values.append(np.float64(arg)) - else: - values.append(arg) - - if n_threads is None: - raise ValueError(f"{self.name}: no CudaMarkerArguments passed, cannot set the number of threads.") - grid = (math.ceil(n_threads / self._block_size),) - self._raw_kernel(grid, (self._block_size,), tuple(values)) + self._raw_kernel(grid, (self._block_size,), args) class Kernel: - """A pyccel kernel and its CUDA counterpart; calls the one matching the cunumpy backend.""" + """A pyccel kernel and its CUDA counterpart; calls the one matching the cunumpy backend. + + Parameters + ---------- + pyccel_kernel : PyccelKernel + The pyccel kernel, called on the NumPy backend. + + cuda_kernel : CudaKernel + The CUDA kernel, called on the CuPy backend. + """ def __init__(self, pyccel_kernel: PyccelKernel, cuda_kernel: CudaKernel): self.pyccel_kernel = pyccel_kernel self.cuda_kernel = cuda_kernel def get_kernel(self) -> PyccelKernel | CudaKernel: + """The kernel for the active cunumpy backend.""" return self.cuda_kernel if is_cuda_backend() else self.pyccel_kernel - def __call__(self, *args): - return self.get_kernel()(*args) + def __call__(self, *args, n_threads: int | None = None): + """Call the kernel for the active cunumpy backend. + + Parameters + ---------- + *args + Kernel arguments, already in the format of the kernel for the active backend. + + n_threads : int | None + Number of CUDA threads; required on the CuPy backend, ignored on the NumPy backend. + """ + if is_cuda_backend(): + if n_threads is None: + raise ValueError(f"{self.cuda_kernel.name}: n_threads is required on the CuPy backend.") + return self.cuda_kernel(*args, n_threads=n_threads) + return self.pyccel_kernel(*args) From 7d00df69705c0880ece9b489ee393ef4e5149408 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 17:51:40 +0200 Subject: [PATCH 12/29] Cast Python scalars and flatten argument classes in CudaKernel again Converting the scalars and joining the .values of the CUDA argument classes costs about 1 us per call (the kernel launch alone about 70 us), prevents silently misaligned arguments when a Python int is passed as a 64-bit value, and keeps the calls the same on both backends. Arrays are still never converted or copied; n_threads stays an explicit argument. Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 6 +-- src/struphy/pic/tests/test_kernel_backends.py | 41 +++++-------------- src/struphy/utils/kernel_backends.py | 37 ++++++++++++----- 3 files changed, 39 insertions(+), 45 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index 1112b154a..506b56b88 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -34,7 +34,7 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke ## Principles -- **1:1 correspondence.** Each CUDA kernel has the same name and the same arguments (in the same order) as its pyccel kernel. For the CUDA kernel, the argument classes are flattened (`CudaMarkerArguments.values` etc.), and the caller also passes the number of threads (`n_threads`). +- **1:1 correspondence.** Each CUDA kernel has the same name and the same arguments (in the same order) as its pyccel kernel. The CUDA kernel takes the CUDA versions of the argument classes, plus the number of threads (`n_threads`). - **The backend decides.** The cunumpy backend (`ARRAY_BACKEND=cupy` or `cunumpy.set_backend("cupy")`) selects the CUDA kernels; with NumPy the pyccel kernels run as today. - **No conversions at call time.** When a kernel is called, its arguments are already in the right format. There are no host/device copies per kernel call. - **Data already lives on the GPU.** On the CuPy backend, `xp` is `cupy`, so markers, spline coefficients etc. are CuPy arrays from the start. The CUDA argument objects only collect *references* to these arrays and raise if they get host arrays. @@ -45,7 +45,7 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke | File | Content | |---|---| -| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; arguments are passed as they are, plus `n_threads`), `Kernel` (pyccel/CUDA pair, `get_kernel()` picks by backend) | +| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; flattens the argument classes and casts Python scalars at each call, takes `n_threads`), `Kernel` (pyccel/CUDA pair, `get_kernel()` picks by backend) | | `src/struphy/utils/cuda_arguments.py` | `CudaMarkerArguments`, `CudaDomainArguments`: same constructor arguments as the pyccel classes, hold CuPy arrays, flatten them into the CUDA kernel arguments | | `src/struphy/pic/tests/test_kernel_backends.py` | the demo kernel pair `push_eta_linear` (pyccel function compiled with `epyccel` at test time, CUDA source string) and tests on both backends | @@ -54,7 +54,7 @@ Things we learned in the proof of concept: - The pyccel-compiled argument classes (`MarkerArguments`, `DomainArguments`, `DerhamArguments`) hold references to their owner's arrays, but only accept **NumPy** arrays. Hence the CUDA counterparts in `cuda_arguments.py`. - Today, `Particles` builds `args_markers` from `_to_numpy_for_kernel(self.markers)`, i.e. from a **host copy** when the backend is CuPy. The same holds for `Domain` and `Derham`. - `Particles6D` and `Derham` cannot be created on the CuPy backend yet (PR 6, PR 7). `Domain` (e.g. `Cuboid`) can, and all its arrays are already CuPy arrays. -- `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Python `int`s are passed as 64-bit integers, so the caller passes NumPy scalars of the C type in the signature (`np.int32` for `int`, `np.float64` for `double`). `CudaKernel` does not convert anything; the flat argument tuple is built once, at setup. +- `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Python `int`s are passed as 64-bit integers, so `CudaKernel` casts Python scalars to the C types of the signature (`int` โ†’ `np.int32`, `float` โ†’ `np.float64`) and flattens the argument classes at each call. This costs about 1 ยตs per call, compared to about 70 ยตs for launching the kernel; arrays are never converted. - `struphy compile` compiles every `.py` file whose name contains `kernels`. Non-pyccel modules must not contain `kernels` in their name; `.cu` files are ignored by it. - On an H100, the demo kernel pushes 10โถ markers in about 0.13 ms per step. diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index 2f29ba224..ae8a82786 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -149,17 +149,6 @@ def make_arguments(n_markers: int, seed: int = 0): return args_markers, args_domain -def kernel_args(dt: float, stage: int, args_markers, args_domain) -> tuple: - """Kernel arguments in the format of the kernel of the active backend (built once, before the calls). - - For CUDA, the flat tuple of a ``cupy.RawKernel``: NumPy scalars with the C types of the signature - followed by the flattened argument classes. - """ - if not is_cuda_backend(): - return dt, stage, args_markers, args_domain - return np.float64(dt), np.int32(stage), *args_markers.values, *args_domain.values - - def expected_push(markers, valid_mks, dt, n_steps=1): out = markers.copy() for _ in range(n_steps): @@ -193,9 +182,8 @@ def test_push_eta_linear(kernel, backend, n_markers): markers = args_markers.markers valid = cunumpy.to_numpy(args_markers.valid_mks) before = cunumpy.to_numpy(markers).copy() - args = kernel_args(dt, 0, args_markers, args_domain) - kernel(*args, n_threads=n_markers) + kernel(dt, 0, args_markers, args_domain, n_threads=n_markers) after = cunumpy.to_numpy(markers) assert np.allclose(after, expected_push(before, valid, dt), rtol=1e-14, atol=0.0) @@ -212,9 +200,8 @@ def test_pyccel_cuda_agree(kernel): args_markers, args_domain = make_arguments(n_markers, seed=1) if backend == "numpy": expected = expected_push(args_markers.markers, args_markers.valid_mks, dt, n_steps) - args = kernel_args(dt, 0, args_markers, args_domain) for _ in range(n_steps): - kernel(*args, n_threads=n_markers) + kernel(dt, 0, args_markers, args_domain, n_threads=n_markers) results[backend] = cunumpy.to_numpy(args_markers.markers) # not bitwise equal: nvcc contracts x + dt * v into fused multiply-adds by default @@ -229,10 +216,9 @@ def test_cuda_kernel_updates_device_array_in_place(kernel): args_markers, args_domain = make_arguments(1000) markers = args_markers.markers ptr = markers.data.ptr - args = kernel_args(0.1, 0, args_markers, args_domain) for _ in range(10): - kernel(*args, n_threads=1000) + kernel(0.1, 0, args_markers, args_domain, n_threads=1000) assert args_markers.markers is markers assert markers.data.ptr == ptr @@ -241,11 +227,11 @@ def test_cuda_kernel_updates_device_array_in_place(kernel): @requires_cupy def test_cuda_scalar_arguments(): - """The scalars and the flattened argument classes arrive in the CUDA kernel with the right types and order.""" + """Python scalars and the flattened argument classes arrive in the CUDA kernel with the right types and order.""" write_scalars = CudaKernel(WRITE_SCALARS_SRC, "write_scalars") with cunumpy.use_backend("cupy"): args_markers, args_domain = make_arguments(10) - write_scalars(*kernel_args(0.25, 3, args_markers, args_domain), n_threads=10) + write_scalars(0.25, 3, args_markers, args_domain, n_threads=10) row = cunumpy.to_numpy(args_markers.markers)[0, :6] first_pusher_idx, mu_idx = MARKER_INDICES[3], MARKER_INDICES[7] @@ -288,22 +274,15 @@ def test_cuda_arguments_reject_host_and_bad_arrays(): @requires_cupy -def test_cuda_kernel_does_not_convert_arguments(kernel): - """Arguments are passed to the RawKernel as they are: host data or the argument objects themselves fail.""" +def test_cuda_kernel_rejects_host_arrays(kernel): + """Arrays are never converted: host arrays and the pyccel argument classes fail; n_threads is required.""" with cunumpy.use_backend("numpy"): host_markers, host_domain = make_arguments(10) with cunumpy.use_backend("cupy"): cuda_markers, cuda_domain = make_arguments(10) with pytest.raises(TypeError): - kernel(np.float64(0.1), np.int32(0), cuda_markers, cuda_domain, n_threads=10) + kernel(0.1, 0, host_markers, cuda_domain, n_threads=10) with pytest.raises(TypeError): - kernel( - np.float64(0.1), - np.int32(0), - host_markers.markers, - *cuda_markers.values[1:], - *cuda_domain.values, - n_threads=10, - ) + kernel(0.1, 0, cuda_markers, host_domain, n_threads=10) with pytest.raises(ValueError, match="n_threads"): - kernel(*kernel_args(0.1, 0, cuda_markers, cuda_domain)) + kernel(0.1, 0, cuda_markers, cuda_domain) diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index a19a1d7ed..d5f7bda7e 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -3,16 +3,14 @@ A :class:`Kernel` holds a pyccel kernel and its 1:1 corresponding CUDA kernel (:class:`CudaKernel`) and calls the one matching the active cunumpy backend, see :func:`is_cuda_backend`. -The arguments must already be in the format of the kernel that is called; nothing is converted -at call time. For the CUDA kernel, this is the flat tuple of a ``cupy.RawKernel``: CuPy arrays and -NumPy scalars of the C types in the kernel signature (e.g. ``np.float64`` for ``double``, -``np.int32`` for ``int``). The argument classes are flattened with -:attr:`~struphy.utils.cuda_arguments.CudaMarkerArguments.values` etc. This tuple is meant to be -built once, at setup, not before every call. +Both kernels are called with the same arguments; the CUDA kernel takes the CUDA versions of the +argument classes (:mod:`struphy.utils.cuda_arguments`), which reference arrays on the device, and +the number of threads ``n_threads``. No arrays are converted or copied at call time. """ import math +import numpy as np from cunumpy import PyccelKernel from cunumpy.xp import array_backend @@ -31,7 +29,11 @@ def is_cuda_backend() -> bool: class CudaKernel: """A ``cupy.RawKernel``, the CUDA counterpart of a pyccel kernel. - The kernel is compiled on the first call. Arguments are passed to the ``cupy.RawKernel`` as they are. + The kernel is compiled on the first call. The arguments are flattened into the arguments of the + ``cupy.RawKernel`` at each call: the CUDA argument classes are replaced by their ``values``, and + Python scalars are cast to the C types of the kernel signature (``bool`` -> ``bool``, ``int`` -> ``int``, + ``float`` -> ``double``). This costs about a microsecond per call. Arrays are passed as they are and + must be CuPy arrays (``cupy`` raises otherwise); they are never converted or copied. Parameters ---------- @@ -57,8 +59,8 @@ def __call__(self, *args, n_threads: int): Parameters ---------- *args - Kernel arguments in the format of a ``cupy.RawKernel``: CuPy arrays and NumPy scalars - with the C types of the kernel signature. + The arguments of the pyccel kernel, with the CUDA versions of the argument classes + (:mod:`struphy.utils.cuda_arguments`), CuPy arrays and Python or NumPy scalars. n_threads : int Number of threads to launch (e.g. number of markers); rounded up to a multiple of the block size. @@ -68,8 +70,21 @@ def __call__(self, *args, n_threads: int): self._raw_kernel = cp.RawKernel(self._source, self.name) + values = [] + for arg in args: + if hasattr(arg, "values"): + values += arg.values + elif isinstance(arg, bool): + values.append(np.bool_(arg)) + elif isinstance(arg, int): + values.append(np.int32(arg)) + elif isinstance(arg, float): + values.append(np.float64(arg)) + else: + values.append(arg) + grid = (math.ceil(n_threads / self._block_size),) - self._raw_kernel(grid, (self._block_size,), args) + self._raw_kernel(grid, (self._block_size,), tuple(values)) class Kernel: @@ -98,7 +113,7 @@ def __call__(self, *args, n_threads: int | None = None): Parameters ---------- *args - Kernel arguments, already in the format of the kernel for the active backend. + Kernel arguments; on the CuPy backend with the CUDA versions of the argument classes. n_threads : int | None Number of CUDA threads; required on the CuPy backend, ignored on the NumPy backend. From ae44cf08198a7871fb23575421a46d204ca555a8 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 18:01:33 +0200 Subject: [PATCH 13/29] Remove the scalar casts from CudaKernel, keep flattening argument classes CUDA reads each kernel argument with the size declared in the signature, so Python int/float already arrive correctly in int/double parameters; the casts did not change the result, and did not prevent the actual failure cases (e.g. an integer passed to a double parameter). Checking scalars against the kernel signature is noted as a follow-up in CUDA_STRATEGY.md. Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 7 +++++-- src/struphy/pic/tests/test_kernel_backends.py | 2 +- src/struphy/utils/kernel_backends.py | 16 ++++------------ 3 files changed, 10 insertions(+), 15 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index 506b56b88..c03f157a8 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -45,7 +45,7 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke | File | Content | |---|---| -| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; flattens the argument classes and casts Python scalars at each call, takes `n_threads`), `Kernel` (pyccel/CUDA pair, `get_kernel()` picks by backend) | +| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; replaces the argument classes by their `values` at each call, takes `n_threads`), `Kernel` (pyccel/CUDA pair, `get_kernel()` picks by backend) | | `src/struphy/utils/cuda_arguments.py` | `CudaMarkerArguments`, `CudaDomainArguments`: same constructor arguments as the pyccel classes, hold CuPy arrays, flatten them into the CUDA kernel arguments | | `src/struphy/pic/tests/test_kernel_backends.py` | the demo kernel pair `push_eta_linear` (pyccel function compiled with `epyccel` at test time, CUDA source string) and tests on both backends | @@ -54,7 +54,8 @@ Things we learned in the proof of concept: - The pyccel-compiled argument classes (`MarkerArguments`, `DomainArguments`, `DerhamArguments`) hold references to their owner's arrays, but only accept **NumPy** arrays. Hence the CUDA counterparts in `cuda_arguments.py`. - Today, `Particles` builds `args_markers` from `_to_numpy_for_kernel(self.markers)`, i.e. from a **host copy** when the backend is CuPy. The same holds for `Domain` and `Derham`. - `Particles6D` and `Derham` cannot be created on the CuPy backend yet (PR 6, PR 7). `Domain` (e.g. `Cuboid`) can, and all its arrays are already CuPy arrays. -- `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Python `int`s are passed as 64-bit integers, so `CudaKernel` casts Python scalars to the C types of the signature (`int` โ†’ `np.int32`, `float` โ†’ `np.float64`) and flattens the argument classes at each call. This costs about 1 ยตs per call, compared to about 70 ยตs for launching the kernel; arrays are never converted. +- `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Each argument is read with the size declared in the signature, so Python `int`/`float` arrive correctly in `int`/`double` parameters, but a wrongly typed scalar (e.g. an integer for a `double`, or a value that overflows an `int`) gives a wrong value **without an error**. Casting Python scalars in `CudaKernel` does not prevent this, so it is not done; see the follow-up in [Open questions](#open-questions). +- Flattening the argument classes at each call (joining their `values`) costs well under 1 ยตs, compared to about 70 ยตs for launching the kernel. - `struphy compile` compiles every `.py` file whose name contains `kernels`. Non-pyccel modules must not contain `kernels` in their name; `.cu` files are ignored by it. - On an H100, the demo kernel pushes 10โถ markers in about 0.13 ms per step. @@ -179,6 +180,8 @@ Port the kernels in the order the target models need them, so that complete mode ## Open questions +- **Scalar types.** Scalars are not checked against the kernel signature (see [Current state](#current-state-pr-1)). `CudaKernel` could read the parameter types from the `extern "C"` signature once, when it is created, and cast each scalar to its declared type or raise if it does not fit (e.g. a Python `float` for an `int`, or an overflowing integer). This could go together with PR 8. + - **Marker layout.** The markers array is row-major (`n_markers ร— n_cols`). With one thread per marker, the memory accesses are strided. This is fine for now (each thread reads a few neighbouring columns), but a column-major or struct-of-arrays layout may be faster later. This would affect the CPU code too, so it is out of scope here. - **MPI + GPUs.** One GPU per MPI rank (`cunumpy.set_device(rank % n_gpus)`), and GPU-aware MPI for the marker exchange, so markers do not go through the host. - **Single-source alternatives.** Before porting a large number of kernels by hand, it may be worth checking whether some of them can be generated from the Python source (e.g. `cupyx.jit` or numba-cuda) instead of written twice. Hand-written CUDA stays the default. diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index ae8a82786..cc5b19465 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -227,7 +227,7 @@ def test_cuda_kernel_updates_device_array_in_place(kernel): @requires_cupy def test_cuda_scalar_arguments(): - """Python scalars and the flattened argument classes arrive in the CUDA kernel with the right types and order.""" + """Python scalars (not cast) and the flattened argument classes arrive in the CUDA kernel correctly and in order.""" write_scalars = CudaKernel(WRITE_SCALARS_SRC, "write_scalars") with cunumpy.use_backend("cupy"): args_markers, args_domain = make_arguments(10) diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index d5f7bda7e..a3ed29c6b 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -10,7 +10,6 @@ import math -import numpy as np from cunumpy import PyccelKernel from cunumpy.xp import array_backend @@ -29,11 +28,10 @@ def is_cuda_backend() -> bool: class CudaKernel: """A ``cupy.RawKernel``, the CUDA counterpart of a pyccel kernel. - The kernel is compiled on the first call. The arguments are flattened into the arguments of the - ``cupy.RawKernel`` at each call: the CUDA argument classes are replaced by their ``values``, and - Python scalars are cast to the C types of the kernel signature (``bool`` -> ``bool``, ``int`` -> ``int``, - ``float`` -> ``double``). This costs about a microsecond per call. Arrays are passed as they are and - must be CuPy arrays (``cupy`` raises otherwise); they are never converted or copied. + The kernel is compiled on the first call. At each call, the CUDA argument classes are replaced by their + ``values``; all other arguments are passed to the ``cupy.RawKernel`` as they are. Arrays must be CuPy + arrays (``cupy`` raises otherwise); they are never converted or copied. Python ``int`` and ``float`` + arrive correctly in ``int`` and ``double`` parameters; scalars are not checked against the kernel signature. Parameters ---------- @@ -74,12 +72,6 @@ def __call__(self, *args, n_threads: int): for arg in args: if hasattr(arg, "values"): values += arg.values - elif isinstance(arg, bool): - values.append(np.bool_(arg)) - elif isinstance(arg, int): - values.append(np.int32(arg)) - elif isinstance(arg, float): - values.append(np.float64(arg)) else: values.append(arg) From 49343cb0786e83fb296f389bb5f82759401b2352 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 16:10:02 +0200 Subject: [PATCH 14/29] Load CUDA kernels from _cuda.cu files Add CudaKernel.from_file, which reads the CUDA source from a _cuda.cu file and takes the kernel name from the file name, and ship .cu/.cuh files as package data. Step 2 of CUDA_STRATEGY.md. Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 5 ++-- pyproject.toml | 2 ++ src/struphy/pic/tests/test_kernel_backends.py | 23 ++++++++++++++++ src/struphy/utils/kernel_backends.py | 27 +++++++++++++++++++ 4 files changed, 54 insertions(+), 3 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index c03f157a8..cc599e30d 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -96,10 +96,9 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba ### PR 2: CUDA source files -- `CudaKernel.from_file(path, name)` reads `_cuda.cu`. The kernel is compiled lazily on first call. CuPy caches compiled kernels on disk (`~/.cupy/kernel_cache`), so the compile cost is paid once per machine. -- Headers are found through NVRTC include paths (`cupy.RawModule(code=..., options=("-I",))`), so a `.cu` file can `#include "struphy/bsplines/bsplines_kernels.cuh"`. +- `CudaKernel.from_file(path)` reads `_cuda.cu`; the kernel name is taken from the file name. The kernel is compiled lazily on first call. CuPy caches compiled kernels on disk (`~/.cupy/kernel_cache`), so the compile cost is paid once per machine. - Add `"**/*.cu"` and `"**/*.cuh"` to `[tool.setuptools.package-data]` in `pyproject.toml`. -- Move the test kernel's CUDA source from the string in `test_kernel_backends.py` into a `.cu` file. +- Later (PR 10), when the first shared header is needed: headers are found through NVRTC include paths (`cupy.RawModule(code=..., options=("-I",))`), so a `.cu` file can `#include "struphy/bsplines/bsplines_kernels.cuh"`. ### PR 3: Kernel catalog diff --git a/pyproject.toml b/pyproject.toml index ffcc885d1..045ec76b6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -140,6 +140,8 @@ struphy = "struphy.console.main:struphy" ] struphy = [ "compile_struphy.mk", + "**/*.cu", + "**/*.cuh", ] [tool.autopep8] diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index cc5b19465..da5467d60 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -286,3 +286,26 @@ def test_cuda_kernel_rejects_host_arrays(kernel): kernel(0.1, 0, cuda_markers, host_domain, n_threads=10) with pytest.raises(ValueError, match="n_threads"): kernel(0.1, 0, cuda_markers, cuda_domain) + + +@requires_cupy +def test_cuda_kernel_from_file(kernel, tmp_path): + """CUDA kernels can be loaded from _cuda.cu files; the name is taken from the file name.""" + path = tmp_path / "push_eta_linear_cuda.cu" + path.write_text(PUSH_ETA_LINEAR_SRC) + cuda_kernel = CudaKernel.from_file(path) + assert cuda_kernel.name == "push_eta_linear" + + results = {} + for backend in ("numpy", "cupy"): + with cunumpy.use_backend(backend): + args_markers, args_domain = make_arguments(1000) + if backend == "cupy": + cuda_kernel(0.1, 0, args_markers, args_domain, n_threads=1000) + else: + kernel.pyccel_kernel(0.1, 0, args_markers, args_domain) + results[backend] = cunumpy.to_numpy(args_markers.markers) + assert np.allclose(results["cupy"], results["numpy"], rtol=1e-14, atol=0.0) + + with pytest.raises(AssertionError, match="naming convention"): + CudaKernel.from_file(tmp_path / "push_eta_linear.cu") diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index a3ed29c6b..ff1d22dcf 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -9,6 +9,7 @@ """ import math +from pathlib import Path from cunumpy import PyccelKernel from cunumpy.xp import array_backend @@ -51,6 +52,32 @@ def __init__(self, source: str, name: str, block_size: int = 128): self._block_size = block_size self._raw_kernel = None + @classmethod + def from_file(cls, path: str | Path, name: str | None = None, block_size: int = 128) -> "CudaKernel": + """Load the CUDA source from a ``_cuda.cu`` file. + + Parameters + ---------- + path : str | Path + Path of the ``.cu`` file. + + name : str | None + Name of the kernel function; defaults to the file name without ``_cuda.cu``. + + block_size : int + Number of threads per block. + + Returns + ------- + CudaKernel + The (not yet compiled) CUDA kernel. + """ + path = Path(path) + if name is None: + assert path.name.endswith("_cuda.cu"), f"{path.name} does not follow the naming convention _cuda.cu" + name = path.name[: -len("_cuda.cu")] + return cls(path.read_text(), name, block_size=block_size) + def __call__(self, *args, n_threads: int): """Launch the kernel. From 650d020017d1ded7589d794f20bc9be5b074b5e4 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 16:10:41 +0200 Subject: [PATCH 15/29] Add KernelCatalog for packages with one folder per kernel KernelCatalog.from_package pairs /_kernels.py (pyccel) with /_cuda.cu (CUDA, optional). The CUDA kernel of a Kernel is now optional: on the CuPy backend, a kernel without a CUDA version raises NotImplementedError naming the expected .cu file. Step 3 of CUDA_STRATEGY.md. Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 2 +- src/struphy/pic/tests/test_kernel_backends.py | 66 ++++++++++- src/struphy/utils/kernel_backends.py | 107 ++++++++++++++++-- 3 files changed, 163 insertions(+), 12 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index cc599e30d..03344a3a9 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -110,7 +110,7 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba ``` - The error should come as early as possible: propagators/pushers call `get_kernel()` when they are set up, not only at the first time step. Then a GPU run fails right away instead of after the initialization. -- A small overview, e.g. `struphy compile --status` also printing "CUDA kernels: 3 of 60", helps to see what is left to port. +- `catalog.missing_cuda` lists the kernels without a CUDA version. Later, a small overview, e.g. `struphy compile --status` also printing "CUDA kernels: 3 of 60", helps to see what is left to port. ### PR 4: `Pusher` and propagators use `Kernel` diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index da5467d60..b9d2a321a 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -16,7 +16,7 @@ from struphy.geometry.domains import Cuboid from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments from struphy.utils.cuda_arguments import CudaDomainArguments, CudaMarkerArguments -from struphy.utils.kernel_backends import CudaKernel, Kernel, is_cuda_backend +from struphy.utils.kernel_backends import CudaKernel, Kernel, KernelCatalog, is_cuda_backend requires_cupy = pytest.mark.skipif(not cunumpy.cupy_available(), reason="CuPy/GPU not available") @@ -309,3 +309,67 @@ def test_cuda_kernel_from_file(kernel, tmp_path): with pytest.raises(AssertionError, match="naming convention"): CudaKernel.from_file(tmp_path / "push_eta_linear.cu") + + +@pytest.fixture +def catalog_package(tmp_path, monkeypatch): + """A package with one folder per kernel: push_a has a CUDA version, push_b has not. + + The pyccel kernels are not compiled here (PyccelKernel also wraps plain Python functions). + """ + root = tmp_path / "poc_catalog_pkg" + header = "from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments\n\n\n" + src = inspect.getsource(push_eta_linear) + for name, cuda in (("push_a", True), ("push_b", False)): + (root / name).mkdir(parents=True) + (root / name / "__init__.py").write_text("") + (root / name / f"{name}_kernels.py").write_text(header + src.replace("push_eta_linear", name)) + if cuda: + (root / name / f"{name}_cuda.cu").write_text(PUSH_ETA_LINEAR_SRC.replace("push_eta_linear", name)) + (root / "not_a_kernel").mkdir() + (root / "__init__.py").write_text( + "from struphy.utils.kernel_backends import KernelCatalog\n\ncatalog = KernelCatalog.from_package(__name__)\n" + ) + monkeypatch.syspath_prepend(str(tmp_path)) + yield importlib.import_module("poc_catalog_pkg").catalog + for mod in [m for m in sys.modules if m.startswith("poc_catalog_pkg")]: + del sys.modules[mod] + + +def test_catalog_discovers_kernels(catalog_package): + catalog = catalog_package + assert catalog.names == ["push_a", "push_b"] + assert "push_a" in catalog and "not_a_kernel" not in catalog + assert catalog.missing_cuda == ["push_b"] + assert catalog["push_a"].name == "push_a" + assert catalog["push_a"].cuda_kernel.name == "push_a" + assert catalog["push_b"].cuda_path.name == "push_b_cuda.cu" + + +@pytest.mark.parametrize("backend", BACKENDS) +def test_catalog_kernels_run(catalog_package, backend): + """The kernels from the catalog push on both backends; a missing CUDA kernel raises on the GPU.""" + dt = 0.1 + with cunumpy.use_backend(backend): + args_markers, args_domain = make_arguments(10) + valid = cunumpy.to_numpy(args_markers.valid_mks) + expected = expected_push(cunumpy.to_numpy(args_markers.markers), valid, dt) + + catalog_package["push_a"](dt, 0, args_markers, args_domain, n_threads=10) + assert np.allclose(cunumpy.to_numpy(args_markers.markers), expected, rtol=1e-14, atol=0.0) + + if backend == "numpy": + catalog_package["push_b"](dt, 0, args_markers, args_domain, n_threads=10) + else: + with pytest.raises(NotImplementedError, match="No CUDA version of kernel 'push_b'.*push_b_cuda.cu"): + catalog_package["push_b"](dt, 0, args_markers, args_domain, n_threads=10) + + +def test_kernel_without_cuda_version(): + kernel = Kernel(PyccelKernel(push_eta_linear)) + assert kernel.name == "push_eta_linear" + with cunumpy.use_backend("numpy"): + assert kernel.get_kernel() is kernel.pyccel_kernel + if cunumpy.cupy_available(): + with cunumpy.use_backend("cupy"), pytest.raises(NotImplementedError, match="push_eta_linear"): + kernel.get_kernel() diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index ff1d22dcf..b48a54df7 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -8,6 +8,7 @@ the number of threads ``n_threads``. No arrays are converted or copied at call time. """ +import importlib import math from pathlib import Path @@ -114,17 +115,43 @@ class Kernel: pyccel_kernel : PyccelKernel The pyccel kernel, called on the NumPy backend. - cuda_kernel : CudaKernel - The CUDA kernel, called on the CuPy backend. + cuda_kernel : CudaKernel | None + The CUDA kernel, called on the CuPy backend; None if it has not been ported yet + (then the kernel cannot run on the CuPy backend). + + cuda_path : Path | None + Where the CUDA kernel is expected, used in the error message if it is missing. """ - def __init__(self, pyccel_kernel: PyccelKernel, cuda_kernel: CudaKernel): + def __init__( + self, + pyccel_kernel: PyccelKernel, + cuda_kernel: CudaKernel | None = None, + cuda_path: Path | None = None, + ): self.pyccel_kernel = pyccel_kernel self.cuda_kernel = cuda_kernel + self.cuda_path = cuda_path + + @property + def name(self) -> str: + """Name of the kernel (the name of the pyccel kernel).""" + return self.pyccel_kernel.name def get_kernel(self) -> PyccelKernel | CudaKernel: - """The kernel for the active cunumpy backend.""" - return self.cuda_kernel if is_cuda_backend() else self.pyccel_kernel + """The kernel for the active cunumpy backend. + + Raises + ------ + NotImplementedError + On the CuPy backend, if there is no CUDA version of the kernel. + """ + if not is_cuda_backend(): + return self.pyccel_kernel + if self.cuda_kernel is None: + expected = "" if self.cuda_path is None else f" (expected {self.cuda_path})" + raise NotImplementedError(f"No CUDA version of kernel {self.name!r}{expected}.") + return self.cuda_kernel def __call__(self, *args, n_threads: int | None = None): """Call the kernel for the active cunumpy backend. @@ -137,8 +164,68 @@ def __call__(self, *args, n_threads: int | None = None): n_threads : int | None Number of CUDA threads; required on the CuPy backend, ignored on the NumPy backend. """ - if is_cuda_backend(): - if n_threads is None: - raise ValueError(f"{self.cuda_kernel.name}: n_threads is required on the CuPy backend.") - return self.cuda_kernel(*args, n_threads=n_threads) - return self.pyccel_kernel(*args) + kernel = self.get_kernel() + if not is_cuda_backend(): + return kernel(*args) + if n_threads is None: + raise ValueError(f"{kernel.name}: n_threads is required on the CuPy backend.") + return kernel(*args, n_threads=n_threads) + + +class KernelCatalog: + """The kernels of a package with one folder per kernel. + + For each subfolder ```` containing ``_kernels.py`` (pyccel), the function ```` in that + module is the pyccel kernel, and ``_cuda.cu`` in the same folder, if present, is the CUDA kernel:: + + catalog = KernelCatalog.from_package(__name__) # in the __init__.py of the package + kernel = catalog["push_eta_stage"] + + Parameters + ---------- + kernels : dict[str, Kernel] + The kernels by name. + """ + + def __init__(self, kernels: dict[str, Kernel]): + self._kernels = kernels + + @classmethod + def from_package(cls, package: str) -> "KernelCatalog": + """Collect the kernels in the subfolders of a package. + + Parameters + ---------- + package : str + Full name of the package, e.g. ``__name__`` in its ``__init__.py``. + + Returns + ------- + KernelCatalog + One :class:`Kernel` per subfolder ```` with a file ``_kernels.py``. + """ + root = Path(importlib.import_module(package).__file__).parent + kernels = {} + for folder in sorted(p for p in root.iterdir() if (p / f"{p.name}_kernels.py").is_file()): + name = folder.name + module = importlib.import_module(f"{package}.{name}.{name}_kernels") + cuda_path = folder / f"{name}_cuda.cu" + cuda_kernel = CudaKernel.from_file(cuda_path) if cuda_path.is_file() else None + kernels[name] = Kernel(PyccelKernel(getattr(module, name)), cuda_kernel, cuda_path) + return cls(kernels) + + def __getitem__(self, name: str) -> Kernel: + return self._kernels[name] + + def __contains__(self, name: str) -> bool: + return name in self._kernels + + @property + def names(self) -> list[str]: + """Names of all kernels.""" + return list(self._kernels) + + @property + def missing_cuda(self) -> list[str]: + """Names of the kernels without a CUDA version.""" + return [name for name, kernel in self._kernels.items() if kernel.cuda_kernel is None] From d7977f4ac091fd41c34273c851769c4d746a2cbe Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 16:34:17 +0200 Subject: [PATCH 16/29] Let Pusher accept a Kernel and choose the kernel once at setup Pusher now takes a Kernel or, as before, a PyccelKernel (wrapped into a Kernel without CUDA version) and calls get_kernel() in its constructor. On the CuPy backend, a pusher whose kernel has no CUDA version thus fails when it is created instead of in the time loop. No changes to the propagators, no behaviour change on the CPU. Step 4 of CUDA_STRATEGY.md. Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 9 ++-- src/struphy/pic/pushing/pusher.py | 21 +++++--- src/struphy/pic/tests/test_kernel_backends.py | 49 +++++++++++++++++++ 3 files changed, 68 insertions(+), 11 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index 03344a3a9..e68f7ffc3 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -9,7 +9,7 @@ The work is split into small PRs that can be reviewed and merged one at a time. `Kernel` (pyccel/CUDA pair), `CudaKernel`, `CudaMarkerArguments`/`CudaDomainArguments`, and one test file with a demo kernel pair run on both backends. Also adds this document and the `gpu` optional dependency. - [ ] **PR 2: CUDA source files** โ€” `CudaKernel` loads CUDA source from a `_cuda.cu` file next to the pyccel file; `.cu`/`.cuh` files are shipped as package data. - [ ] **PR 3: Kernel catalog** โ€” kernels are defined once in the `__init__.py` of the folder that contains them; a missing CUDA kernel raises an error on the GPU backend. -- [ ] **PR 4: `Pusher` and propagators accept `Kernel`** โ€” replace `PyccelKernel(...)` in the propagators by catalog lookups (no CUDA kernels yet, so no behaviour change on CPU). +- [ ] **PR 4: `Pusher` accepts `Kernel`** โ€” the kernel for the active backend is chosen once, when the pusher is created; a plain `PyccelKernel` is wrapped, so the propagators do not change (no behaviour change on CPU). - [ ] **PR 5: `Domain` on the GPU** โ€” `Domain.cuda_args_domain`, plus the separate fix for `Domain` deepcopy on the CuPy backend (`_build_args_domain` passes `params_numpy` without `_to_numpy_for_kernel`). - [ ] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend (e.g. `xp.prod` on Python lists in `pic/base.py`), plus `Particles.cuda_args_markers`. - [ ] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. @@ -112,10 +112,11 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - The error should come as early as possible: propagators/pushers call `get_kernel()` when they are set up, not only at the first time step. Then a GPU run fails right away instead of after the initialization. - `catalog.missing_cuda` lists the kernels without a CUDA version. Later, a small overview, e.g. `struphy compile --status` also printing "CUDA kernels: 3 of 60", helps to see what is left to port. -### PR 4: `Pusher` and propagators use `Kernel` +### PR 4: `Pusher` accepts `Kernel` -- `Pusher` currently asserts `isinstance(kernel, PyccelKernel)` (`pic/pushing/pusher.py`). Allow `Kernel` there and use `kernel.name` for profiling as today. -- On the GPU, the pusher passes the CUDA argument objects (`particles.cuda_args_markers`, `domain.cuda_args_domain`, ...) instead of the pyccel ones. The choice is made once when the pusher is set up, together with the kernel. This avoids calling a CUDA kernel with pyccel arguments or vice versa. +- `Pusher` takes a `Kernel` or, as before, a `PyccelKernel` (wrapped into a `Kernel` without CUDA version). It calls `get_kernel()` once in its constructor, so on the CuPy backend a pusher whose kernel has no CUDA version fails when it is created, not in the time loop. Since no pusher kernel has a CUDA version yet, this is the case for all pushers. +- The propagators do not change. They switch to catalog lookups once the kernels are split into folders (PR 9). +- Still to do (with PR 5โ€“7 and PR 11): on the GPU, the pusher must pass the CUDA argument objects (`particles.cuda_args_markers`, `domain.cuda_args_domain`, device arrays in `args_kernel`) instead of the pyccel ones. This choice is made once, together with the kernel, so that a CUDA kernel is never called with pyccel arguments or vice versa. ### PR 5โ€“7: Owners build their CUDA arguments diff --git a/src/struphy/pic/pushing/pusher.py b/src/struphy/pic/pushing/pusher.py index 7b2522ff2..04e220f2d 100644 --- a/src/struphy/pic/pushing/pusher.py +++ b/src/struphy/pic/pushing/pusher.py @@ -11,6 +11,7 @@ from struphy.kernel_arguments.pusher_args_kernels import DerhamArguments, DomainArguments from struphy.pic.base import Particles from struphy.pic.pushing.kernel_setup import KernelSetup +from struphy.utils.kernel_backends import CudaKernel, Kernel logger = logging.getLogger("struphy") @@ -60,8 +61,9 @@ class Pusher: particles : Particles Particles object holding the markers to push. - kernel : pyccelized function - The pusher kernel. + kernel : PyccelKernel | Kernel + The pusher kernel. A :class:`~struphy.utils.kernel_backends.Kernel` also holds its CUDA version; + on the CuPy backend, a kernel without CUDA version raises NotImplementedError. args_kernel : tuple Optional arguments passed to the kernel. @@ -114,7 +116,7 @@ class Pusher: def __init__( self, particles: Particles, - kernel: PyccelKernel, + kernel: PyccelKernel | Kernel, args_kernel: tuple, args_domain: DomainArguments, pushes_eta: bool, @@ -128,9 +130,14 @@ def __init__( mpi_sort: str = None, local_eval_only: bool = False, ): + # choose the kernel for the active backend once; on the CuPy backend this raises + # if there is no CUDA version (yet), see CUDA_STRATEGY.md + if isinstance(kernel, PyccelKernel): + kernel = Kernel(kernel) + assert isinstance(kernel, Kernel), f"{kernel} is not of type Kernel or PyccelKernel" + self._kernel = kernel.get_kernel() + self._particles = particles - assert isinstance(kernel, PyccelKernel), f"{kernel} is not of type PyccelKernel" - self._kernel = kernel self._newton = "newton" in kernel.name self._args_kernel = args_kernel self._args_domain = args_domain @@ -355,8 +362,8 @@ def particles(self): return self._particles @property - def kernel(self): - """The pyccelized pusher kernel.""" + def kernel(self) -> PyccelKernel | CudaKernel: + """The pusher kernel for the active backend (pyccel or CUDA).""" return self._kernel @property diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index b9d2a321a..0a3879efe 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -373,3 +373,52 @@ def test_kernel_without_cuda_version(): if cunumpy.cupy_available(): with cunumpy.use_backend("cupy"), pytest.raises(NotImplementedError, match="push_eta_linear"): kernel.get_kernel() + + +def make_pusher(kernel): + """Pusher for push_eta_stage (forward Euler) on 100 particles in a Cuboid.""" + from feectools.ddm.mpi import mpi as MPI + + from struphy import LoadingParameters + from struphy.ode.utils import ButcherTableau + from struphy.pic.particles import Particles6D + from struphy.pic.pushing.pusher import Pusher + + domain = Cuboid() + loading_params = LoadingParameters(Np=100, seed=1234, moments=(0.0, 0.0, 0.0, 1.0, 1.0, 1.0), spatial="uniform") + particles = Particles6D(comm_world=MPI.COMM_WORLD, loading_params=loading_params, domain=domain) + particles.draw_markers() + butcher = ButcherTableau("forward_euler") + return lambda: Pusher( + particles, + kernel, + (butcher.a_stage, butcher.b, butcher.c), + domain.args_domain, + pushes_eta=True, + alpha_in_kernel=1.0, + n_stages=butcher.n_stages, + local_eval_only=True, + ) + + +@pytest.mark.parametrize("wrap", [False, True]) +def test_pusher_accepts_kernel(wrap): + """Pusher takes a PyccelKernel (wrapped into a Kernel) or a Kernel, and runs the pyccel kernel on NumPy.""" + from struphy.pic.pushing import pusher_kernels + + pyccel_kernel = PyccelKernel(pusher_kernels.push_eta_stage) + with cunumpy.use_backend("numpy"): + pusher = make_pusher(Kernel(pyccel_kernel) if wrap else pyccel_kernel)() + assert pusher.kernel is pyccel_kernel + pusher(0.1) + + +@requires_cupy +def test_pusher_without_cuda_kernel_fails_at_setup(): + """On the CuPy backend, a pusher whose kernel has no CUDA version fails when it is created, not in the time loop.""" + from struphy.pic.pushing import pusher_kernels + + with cunumpy.use_backend("numpy"): + create_pusher = make_pusher(PyccelKernel(pusher_kernels.push_eta_stage)) + with cunumpy.use_backend("cupy"), pytest.raises(NotImplementedError, match="push_eta_stage"): + create_pusher() From db9dced934b2951ac8e07eab114168fe5fa388f0 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 18:13:02 +0200 Subject: [PATCH 17/29] Upgrade cunumpy and use cunumpy.get_backend() --- CUDA_STRATEGY.md | 2 +- pyproject.toml | 2 +- src/struphy/utils/kernel_backends.py | 4 ++-- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index c03f157a8..f7f110492 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -35,7 +35,7 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke ## Principles - **1:1 correspondence.** Each CUDA kernel has the same name and the same arguments (in the same order) as its pyccel kernel. The CUDA kernel takes the CUDA versions of the argument classes, plus the number of threads (`n_threads`). -- **The backend decides.** The cunumpy backend (`ARRAY_BACKEND=cupy` or `cunumpy.set_backend("cupy")`) selects the CUDA kernels; with NumPy the pyccel kernels run as today. +- **The backend decides.** The cunumpy backend (`ARRAY_BACKEND=cupy` or `cunumpy.set_backend("cupy")`, queried with `cunumpy.get_backend()`, cunumpy โ‰ฅ 0.2.0) selects the CUDA kernels; with NumPy the pyccel kernels run as today. - **No conversions at call time.** When a kernel is called, its arguments are already in the right format. There are no host/device copies per kernel call. - **Data already lives on the GPU.** On the CuPy backend, `xp` is `cupy`, so markers, spline coefficients etc. are CuPy arrays from the start. The CUDA argument objects only collect *references* to these arrays and raise if they get host arrays. - **No silent CPU fallback on the GPU.** A kernel without a CUDA version raises an error on the GPU backend. Falling back would mean copying data to the host and back at every call. diff --git a/pyproject.toml b/pyproject.toml index ffcc885d1..0a92e39c9 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -24,7 +24,7 @@ classifiers = [ ] dependencies = [ "numpy<=2.5.0", - "cunumpy>=0.1.4, <=0.1.5", + "cunumpy>=0.2.0, <=0.2.0", "pyccel>=2.2.0, <=2.2.3", "feectools >= 0.1.11, <=0.1.11", "scipy<=1.18.0", diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index a3ed29c6b..2a693c9ad 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -10,8 +10,8 @@ import math +import cunumpy from cunumpy import PyccelKernel -from cunumpy.xp import array_backend def is_cuda_backend() -> bool: @@ -22,7 +22,7 @@ def is_cuda_backend() -> bool: bool True if the backend is ``"cupy"``, False if it is ``"numpy"``. """ - return array_backend.backend == "cupy" + return cunumpy.get_backend() == "cupy" class CudaKernel: From 65ce485666b86d6e8f05f69a742af2edd75e9fab Mon Sep 17 00:00:00 2001 From: Max Date: Wed, 30 Sep 2026 20:24:21 +0200 Subject: [PATCH 18/29] Add Domain.cuda_args_domain and fix Domain deepcopy on the CuPy backend Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 15 ++++- src/struphy/geometry/base.py | 60 ++++++++++++----- src/struphy/geometry/tests/test_domain.py | 67 +++++++++++++++++++ src/struphy/pic/tests/test_kernel_backends.py | 24 +------ 4 files changed, 126 insertions(+), 40 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index c4668791a..d2c521bf8 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -10,7 +10,7 @@ The work is split into small PRs that can be reviewed and merged one at a time. - [ ] **PR 2: CUDA source files** โ€” `CudaKernel` loads CUDA source from a `_cuda.cu` file next to the pyccel file; `.cu`/`.cuh` files are shipped as package data. - [ ] **PR 3: Kernel catalog** โ€” kernels are defined once in the `__init__.py` of the folder that contains them; a missing CUDA kernel raises an error on the GPU backend. - [ ] **PR 4: `Pusher` accepts `Kernel`** โ€” the kernel for the active backend is chosen once, when the pusher is created; a plain `PyccelKernel` is wrapped, so the propagators do not change (no behaviour change on CPU). -- [ ] **PR 5: `Domain` on the GPU** โ€” `Domain.cuda_args_domain`, plus the separate fix for `Domain` deepcopy on the CuPy backend (`_build_args_domain` passes `params_numpy` without `_to_numpy_for_kernel`). +- [ ] **PR 5: `Domain` on the GPU** โ€” `Domain.cuda_args_domain` (built lazily from the domain's device arrays), plus the fix for `Domain` deepcopy/unpickling on the CuPy backend (`_build_args_domain` passed `params_numpy` without `_to_numpy_for_kernel`). - [ ] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend (e.g. `xp.prod` on Python lists in `pic/base.py`), plus `Particles.cuda_args_markers`. - [ ] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. - [ ] **PR 8: Shared CUDA headers for the argument classes** โ€” one `.cuh` per argument class instead of long flat kernel signatures. @@ -52,7 +52,7 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke Things we learned in the proof of concept: - The pyccel-compiled argument classes (`MarkerArguments`, `DomainArguments`, `DerhamArguments`) hold references to their owner's arrays, but only accept **NumPy** arrays. Hence the CUDA counterparts in `cuda_arguments.py`. -- Today, `Particles` builds `args_markers` from `_to_numpy_for_kernel(self.markers)`, i.e. from a **host copy** when the backend is CuPy. The same holds for `Domain` and `Derham`. +- Today, `Particles` builds `args_markers` from `_to_numpy_for_kernel(self.markers)`, i.e. from a **host copy** when the backend is CuPy. The same holds for `Derham`, and for `Domain.args_domain` (the CUDA version `Domain.cuda_args_domain` references the device arrays, PR 5). - `Particles6D` and `Derham` cannot be created on the CuPy backend yet (PR 6, PR 7). `Domain` (e.g. `Cuboid`) can, and all its arrays are already CuPy arrays. - `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Each argument is read with the size declared in the signature, so Python `int`/`float` arrive correctly in `int`/`double` parameters, but a wrongly typed scalar (e.g. an integer for a `double`, or a value that overflows an `int`) gives a wrong value **without an error**. Casting Python scalars in `CudaKernel` does not prevent this, so it is not done; see the follow-up in [Open questions](#open-questions). - Flattening the argument classes at each call (joining their `values`) costs well under 1 ยตs, compared to about 70 ยตs for launching the kernel. @@ -125,7 +125,16 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - First these classes must be creatable on the CuPy backend at all: - `Particles`: `xp.prod`/`xp.sum` on Python lists and similar (`pic/base.py`), probably more. - `Derham`: NumPy arrays from feectools reach `cupy.ascontiguousarray`. - - `Domain`: deepcopy on CuPy fails (see PR 5 above). + - `Domain`: deepcopy and unpickling on CuPy failed (see PR 5 below). + +### PR 5: `Domain` on the GPU + +- `Domain.cuda_args_domain`: `CudaDomainArguments` built from the domain's own device arrays, next to `args_domain` in `geometry/base.py`. It is built on first access, so CPU runs never build it; on the NumPy backend it raises `TypeError` (host arrays are never copied to the device). +- Arrays that already have the dtype and layout the CUDA kernels expect (`float64`/`int64`, C-contiguous) are referenced, not copied, e.g. the knot vectors `T` and `indN`. Otherwise one device copy is made when the arguments are built (e.g. `degree`, which is a tuple, or broadcast control points), never at kernel call time. +- Like `args_domain`, the CUDA arguments are rebuilt after a deepcopy or unpickling (they are dropped in `__deepcopy__`/`__getstate__` and built again on first access). +- Fix: `__init__` and the rebuild after deepcopy/unpickling now use the same builder, `_build_args_domain`, which converts **all** arrays to NumPy for the pyccel `DomainArguments`. Before, the rebuild passed `params_numpy` without conversion, so a deepcopy of a domain on the CuPy backend failed. +- Spline mappings whose control points are passed in as NumPy arrays on the CuPy backend still hold host arrays, so `cuda_args_domain` raises for them. Converting user input to `xp` in the `Spline` classes is left for when the first spline mapping is needed on the GPU. +- The pushers still receive `domain.args_domain` from the propagators; switching them to `cuda_args_domain` on the GPU comes with PR 6/7 and PR 11. ### PR 8: Argument structs in shared headers diff --git a/src/struphy/geometry/base.py b/src/struphy/geometry/base.py index d162c2f11..2e41811ff 100644 --- a/src/struphy/geometry/base.py +++ b/src/struphy/geometry/base.py @@ -16,6 +16,7 @@ from struphy.geometry import evaluation_kernels, transform_kernels from struphy.kernel_arguments.pusher_args_kernels import DomainArguments from struphy.linear_algebra import linalg_kron +from struphy.utils.cuda_arguments import CudaDomainArguments from struphy.utils.docstring_converter import rst_to_html, rst_to_latex, rst_to_markdown from struphy.utils.ipython_compat import HTML, display from struphy.utils.utils import __class_with_params_repr_no_defaults__, all_class_params_are_default, all_subclasses @@ -219,7 +220,11 @@ def __init__( "tran": dict_tran, } - self._args_domain = DomainArguments( + self._args_domain = self._build_args_domain() + + def _build_args_domain(self): + """Build runtime mapping arguments used by compiled evaluation kernels (host copies on the CuPy backend).""" + return DomainArguments( self.kind_map, _to_numpy_for_kernel(self.params_numpy), _to_numpy_for_kernel(xp.array(self.degree)), @@ -234,21 +239,28 @@ def __init__( _to_numpy_for_kernel(self.cz.copy()), # make sure we don't have stride = 0 ) - def _build_args_domain(self): - """Build runtime mapping arguments used by compiled evaluation kernels.""" - return DomainArguments( + def _build_cuda_args_domain(self) -> CudaDomainArguments: + """Build the CUDA kernel arguments from the domain's own (device) arrays. + + Arrays that already have the dtype and layout the CUDA kernels expect are referenced, not copied; + otherwise a device copy with the right dtype and layout is made once, here. Host arrays raise. + """ + + def device(arr, dtype): + if not hasattr(arr, "__cuda_array_interface__"): + raise TypeError( + f"{self.__class__.__name__}: CUDA domain arguments need CuPy arrays, got {type(arr)}; " + "create the domain on the CuPy backend." + ) + return xp.ascontiguousarray(arr, dtype=dtype) + + return CudaDomainArguments( self.kind_map, - self.params_numpy, - _to_numpy_for_kernel(xp.array(self.degree)), - _to_numpy_for_kernel(self.T[0]), - _to_numpy_for_kernel(self.T[1]), - _to_numpy_for_kernel(self.T[2]), - _to_numpy_for_kernel(self.indN[0]), - _to_numpy_for_kernel(self.indN[1]), - _to_numpy_for_kernel(self.indN[2]), - _to_numpy_for_kernel(self.cx.copy()), # make sure we don't have stride = 0 - _to_numpy_for_kernel(self.cy.copy()), # make sure we don't have stride = 0 - _to_numpy_for_kernel(self.cz.copy()), # make sure we don't have stride = 0 + device(self.params_numpy, np.float64), + device(xp.array(self.degree), np.int64), + *(device(t, np.float64) for t in self.T), + *(device(ind, np.int64) for ind in self.indN), + *(device(c, np.float64) for c in (self.cx, self.cy, self.cz)), ) def _can_build_args_domain(self): @@ -264,6 +276,8 @@ def _can_build_args_domain(self): return all(hasattr(self, attr) for attr in required_attrs) def _rebuild_args_domain(self): + # built lazily on first access, see cuda_args_domain + self._cuda_args_domain = None if self._can_build_args_domain(): self._args_domain = self._build_args_domain() else: @@ -275,7 +289,7 @@ def __deepcopy__(self, memo): memo[id(self)] = result for key, value in self.__dict__.items(): - if key == "_args_domain": + if key in ("_args_domain", "_cuda_args_domain"): continue setattr(result, key, copy.deepcopy(value, memo)) @@ -285,6 +299,7 @@ def __deepcopy__(self, memo): def __getstate__(self): state = self.__dict__.copy() state.pop("_args_domain", None) + state.pop("_cuda_args_domain", None) return state def __setstate__(self, state): @@ -475,6 +490,19 @@ def args_domain(self): return self._args_domain + @property + def cuda_args_domain(self) -> CudaDomainArguments: + """CUDA version of :attr:`args_domain`, referencing the domain's arrays on the device. + + Built on first access (never on the NumPy backend, where it raises TypeError), and rebuilt + after a deepcopy or unpickling, like :attr:`args_domain`. + """ + if getattr(self, "_cuda_args_domain", None) is None: + if not self._can_build_args_domain(): + raise AttributeError("CudaDomainArguments are not available because the domain state is incomplete.") + self._cuda_args_domain = self._build_cuda_args_domain() + return self._cuda_args_domain + @property def dict_transformations(self): """Dictionary of str->int for pull, push and transformation functions.""" diff --git a/src/struphy/geometry/tests/test_domain.py b/src/struphy/geometry/tests/test_domain.py index be996e003..5abe7f0b5 100644 --- a/src/struphy/geometry/tests/test_domain.py +++ b/src/struphy/geometry/tests/test_domain.py @@ -2,6 +2,7 @@ import logging import pickle +import cunumpy import pytest logger = logging.getLogger("struphy") @@ -984,3 +985,69 @@ def fun(e1, e2, e3): # test_pullback() # test_pushforward() # test_transform() + + +requires_cupy = pytest.mark.skipif(not cunumpy.cupy_available(), reason="CuPy/GPU not available") + + +def test_cuda_args_domain_needs_device_arrays(): + """On the NumPy backend the domain arrays are host arrays, which are never copied to the device.""" + from struphy import domains + + with cunumpy.use_backend("numpy"): + domain = domains.Cuboid() + with pytest.raises(TypeError, match="CuPy arrays"): + domain.cuda_args_domain + + +@requires_cupy +@pytest.mark.parametrize("mapping", ["Cuboid", "HollowTorus", "IGAPolarCylinder"]) +def test_cuda_args_domain(mapping): + """The CUDA domain arguments reference the domain's device arrays and match the pyccel arguments.""" + from struphy import domains + from struphy.utils.cuda_arguments import CudaDomainArguments + + with cunumpy.use_backend("cupy"): + domain = getattr(domains, mapping)() + args = domain.cuda_args_domain + assert isinstance(args, CudaDomainArguments) + assert domain.cuda_args_domain is args # built once + + kind_map, params, degree, t1, t2, t3, ind1, ind2, ind3, cx, cy, cz = args.values + assert int(kind_map) == domain.kind_map + # no copies of arrays that already have the right dtype and layout + assert t1 is domain.T[0] and ind3 is domain.indN[2] + + host = domain.args_domain + for dev, ref in ( + (params, host.params), + (degree, host.degree), + (t1, host.t1), + (t2, host.t2), + (t3, host.t3), + (ind1, host.ind1), + (ind2, host.ind2), + (ind3, host.ind3), + (cx, host.cx), + (cy, host.cy), + (cz, host.cz), + ): + assert (cunumpy.to_numpy(dev) == ref).all() + + +@requires_cupy +@pytest.mark.parametrize("mapping", ["Cuboid", "IGAPolarCylinder"]) +def test_domain_deepcopy_and_pickle_on_cupy(mapping): + """Deepcopy and unpickling on the CuPy backend rebuild both the pyccel and the CUDA arguments.""" + from struphy import domains + + with cunumpy.use_backend("cupy"): + domain = getattr(domains, mapping)() + cuda_args = domain.cuda_args_domain + + for other in (copy.deepcopy(domain), pickle.loads(pickle.dumps(domain, protocol=pickle.HIGHEST_PROTOCOL))): + assert other.args_domain.kind_map == domain.args_domain.kind_map + assert (other.args_domain.params == domain.args_domain.params).all() + other_cuda = other.cuda_args_domain + assert other_cuda is not cuda_args + assert other_cuda.values[3] is other.T[0] diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index 0a3879efe..2921e09fd 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -136,17 +136,7 @@ def make_arguments(n_markers: int, seed: int = 0): return args_markers, domain.args_domain args_markers = CudaMarkerArguments(markers, valid_mks, n_markers, *MARKER_INDICES, bc_type) - args_domain = CudaDomainArguments( - domain.kind_map, - domain.params_numpy, - cunumpy.asarray(domain.degree), - *domain.T, - *domain.indN, - domain.cx, - domain.cy, - domain.cz, - ) - return args_markers, args_domain + return args_markers, domain.cuda_args_domain def expected_push(markers, valid_mks, dt, n_steps=1): @@ -242,16 +232,8 @@ def test_cuda_scalar_arguments(): def test_cuda_domain_arguments_reference_domain_arrays(): with cunumpy.use_backend("cupy"): domain = Cuboid() - args = CudaDomainArguments( - domain.kind_map, - domain.params_numpy, - cunumpy.asarray(domain.degree), - *domain.T, - *domain.indN, - domain.cx, - domain.cy, - domain.cz, - ) + args = domain.cuda_args_domain + assert isinstance(args, CudaDomainArguments) assert len(args.values) == 12 assert args.values[3] is domain.T[0] and args.values[8] is domain.indN[2] and args.values[9] is domain.cx From fb72c7e63d618ce60f637f97f39cd5be42a0ac00 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Wed, 30 Sep 2026 21:23:31 +0200 Subject: [PATCH 19/29] Fix Domain.cuda_args_domain tests and backend dependence - Test only analytic mappings on the CuPy backend: spline mappings such as IGAPolarCylinder cannot be created there yet (interp_mapping passes CuPy arrays to scipy.sparse.csc_matrix); correct CUDA_STRATEGY.md. - Build the CUDA domain arguments with cupy instead of xp, so they can be built whichever backend is active (the arrays are on the device); test this. - Move the new tests above the __main__ block of test_domain.py. Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 2 +- src/struphy/geometry/base.py | 7 +++-- src/struphy/geometry/tests/test_domain.py | 37 ++++++++++++++++------- 3 files changed, 32 insertions(+), 14 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index d2c521bf8..6e75244a2 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -133,7 +133,7 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - Arrays that already have the dtype and layout the CUDA kernels expect (`float64`/`int64`, C-contiguous) are referenced, not copied, e.g. the knot vectors `T` and `indN`. Otherwise one device copy is made when the arguments are built (e.g. `degree`, which is a tuple, or broadcast control points), never at kernel call time. - Like `args_domain`, the CUDA arguments are rebuilt after a deepcopy or unpickling (they are dropped in `__deepcopy__`/`__getstate__` and built again on first access). - Fix: `__init__` and the rebuild after deepcopy/unpickling now use the same builder, `_build_args_domain`, which converts **all** arrays to NumPy for the pyccel `DomainArguments`. Before, the rebuild passed `params_numpy` without conversion, so a deepcopy of a domain on the CuPy backend failed. -- Spline mappings whose control points are passed in as NumPy arrays on the CuPy backend still hold host arrays, so `cuda_args_domain` raises for them. Converting user input to `xp` in the `Spline` classes is left for when the first spline mapping is needed on the GPU. +- Spline mappings (e.g. `IGAPolarCylinder`) cannot be created on the CuPy backend yet: `interp_mapping` passes CuPy arrays to `scipy.sparse.csc_matrix`. So `cuda_args_domain` is only available for analytic mappings for now; making the spline mappings work on the GPU is left for when the first one is needed there. - The pushers still receive `domain.args_domain` from the propagators; switching them to `cuda_args_domain` on the GPU comes with PR 6/7 and PR 11. ### PR 8: Argument structs in shared headers diff --git a/src/struphy/geometry/base.py b/src/struphy/geometry/base.py index 2e41811ff..e8fd64dee 100644 --- a/src/struphy/geometry/base.py +++ b/src/struphy/geometry/base.py @@ -246,18 +246,21 @@ def _build_cuda_args_domain(self) -> CudaDomainArguments: otherwise a device copy with the right dtype and layout is made once, here. Host arrays raise. """ + # cupy (not xp): the arrays are on the device, whichever backend is active now + import cupy as cp + def device(arr, dtype): if not hasattr(arr, "__cuda_array_interface__"): raise TypeError( f"{self.__class__.__name__}: CUDA domain arguments need CuPy arrays, got {type(arr)}; " "create the domain on the CuPy backend." ) - return xp.ascontiguousarray(arr, dtype=dtype) + return cp.ascontiguousarray(arr, dtype=dtype) return CudaDomainArguments( self.kind_map, device(self.params_numpy, np.float64), - device(xp.array(self.degree), np.int64), + cp.asarray(self.degree, dtype=np.int64), # a tuple, not an array of the domain *(device(t, np.float64) for t in self.T), *(device(ind, np.int64) for ind in self.indN), *(device(c, np.float64) for c in (self.cx, self.cy, self.cz)), diff --git a/src/struphy/geometry/tests/test_domain.py b/src/struphy/geometry/tests/test_domain.py index 5abe7f0b5..3fb9d0bb7 100644 --- a/src/struphy/geometry/tests/test_domain.py +++ b/src/struphy/geometry/tests/test_domain.py @@ -979,14 +979,6 @@ def fun(e1, e2, e3): # assert a.shape == mat_x.shape -if __name__ == "__main__": - # test_prepare_arg() - test_evaluation_mappings("DESCunit") - # test_pullback() - # test_pushforward() - # test_transform() - - requires_cupy = pytest.mark.skipif(not cunumpy.cupy_available(), reason="CuPy/GPU not available") @@ -1001,9 +993,24 @@ def test_cuda_args_domain_needs_device_arrays(): @requires_cupy -@pytest.mark.parametrize("mapping", ["Cuboid", "HollowTorus", "IGAPolarCylinder"]) +def test_cuda_args_domain_independent_of_active_backend(): + """A domain created on the CuPy backend holds device arrays; its CUDA arguments can be built on either backend.""" + from struphy import domains + + with cunumpy.use_backend("cupy"): + domain = domains.Cuboid() + with cunumpy.use_backend("numpy"): + args = domain.cuda_args_domain + assert args.values[3] is domain.T[0] + + +@requires_cupy +@pytest.mark.parametrize("mapping", ["Cuboid", "HollowTorus", "Colella"]) def test_cuda_args_domain(mapping): - """The CUDA domain arguments reference the domain's device arrays and match the pyccel arguments.""" + """The CUDA domain arguments reference the domain's device arrays and match the pyccel arguments. + + Only analytic mappings: spline mappings (e.g. IGAPolarCylinder) cannot be created on the CuPy backend yet. + """ from struphy import domains from struphy.utils.cuda_arguments import CudaDomainArguments @@ -1036,7 +1043,7 @@ def test_cuda_args_domain(mapping): @requires_cupy -@pytest.mark.parametrize("mapping", ["Cuboid", "IGAPolarCylinder"]) +@pytest.mark.parametrize("mapping", ["Cuboid", "Colella"]) def test_domain_deepcopy_and_pickle_on_cupy(mapping): """Deepcopy and unpickling on the CuPy backend rebuild both the pyccel and the CUDA arguments.""" from struphy import domains @@ -1051,3 +1058,11 @@ def test_domain_deepcopy_and_pickle_on_cupy(mapping): other_cuda = other.cuda_args_domain assert other_cuda is not cuda_args assert other_cuda.values[3] is other.T[0] + + +if __name__ == "__main__": + # test_prepare_arg() + test_evaluation_mappings("DESCunit") + # test_pullback() + # test_pushforward() + # test_transform() From 791960f6daf060d6761a315e03d8a1adb1e65639 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Thu, 1 Oct 2026 11:21:11 +0200 Subject: [PATCH 20/29] Add Argument baseclass --- src/struphy/utils/cuda_arguments.py | 87 +++++++++++++++++++--------- src/struphy/utils/kernel_backends.py | 6 +- 2 files changed, 64 insertions(+), 29 deletions(-) diff --git a/src/struphy/utils/cuda_arguments.py b/src/struphy/utils/cuda_arguments.py index a1e89dbf8..104934f2b 100644 --- a/src/struphy/utils/cuda_arguments.py +++ b/src/struphy/utils/cuda_arguments.py @@ -6,9 +6,20 @@ The order of :attr:`values` is the corresponding part of the CUDA kernel signature. """ +from abc import ABC, abstractmethod + import numpy as np +class Argument(ABC): + """Base class for objects that provide arguments to a CUDA kernel.""" + + @abstractmethod + def get_cuda_args(self) -> tuple: + """Return this object's arguments in CUDA kernel signature order.""" + raise NotImplementedError + + def _cupy_array(name: str, arr, dtype): """Check that ``arr`` is a C-contiguous CuPy array of ``dtype`` (never converts or copies). @@ -35,7 +46,7 @@ def _cupy_array(name: str, arr, dtype): return arr -class CudaMarkerArguments: +class CudaMarkerArguments(Argument): """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.MarkerArguments`. CUDA signature of :attr:`values`: ``double* markers, bool* valid_mks, int n_markers, int n_cols, int Np, @@ -113,30 +124,38 @@ def __init__( self.markers = _cupy_array("markers", markers, np.float64) self.valid_mks = _cupy_array("valid_mks", valid_mks, np.bool_) self.n_markers = markers.shape[0] - self.values = ( + self.n_cols = np.int32(markers.shape[1]) + self.Np = np.int32(Np) + self.vdim = np.int32(vdim) + self.weight_idx = np.int32(weight_idx) + self.first_diagnostics_idx = np.int32(first_diagnostics_idx) + self.first_pusher_idx = np.int32(first_pusher_idx) + self.first_shift_idx = np.int32(first_shift_idx) + self.residual_idx = np.int32(residual_idx) + self.first_free_idx = np.int32(first_free_idx) + self.mu_idx = np.int32(mu_idx) + self.bc_type = _cupy_array("bc_type", bc_type, np.int64) + + def get_cuda_args(self) -> tuple: + return ( self.markers, self.valid_mks, - *( - np.int32(i) - for i in ( - markers.shape[0], - markers.shape[1], - Np, - vdim, - weight_idx, - first_diagnostics_idx, - first_pusher_idx, - first_shift_idx, - residual_idx, - first_free_idx, - mu_idx, - ) - ), - _cupy_array("bc_type", bc_type, np.int64), + np.int32(self.n_markers), + self.n_cols, + self.Np, + self.vdim, + self.weight_idx, + self.first_diagnostics_idx, + self.first_pusher_idx, + self.first_shift_idx, + self.residual_idx, + self.first_free_idx, + self.mu_idx, + self.bc_type, ) -class CudaDomainArguments: +class CudaDomainArguments(Argument): """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments`. CUDA signature of :attr:`values`: ``int kind_map, double* params, long long* degree, double* t1, @@ -169,11 +188,25 @@ class CudaDomainArguments: """ def __init__(self, kind_map: int, params, degree, t1, t2, t3, ind1, ind2, ind3, cx, cy, cz): - self.values = ( - np.int32(kind_map), - _cupy_array("params", params, np.float64), - _cupy_array("degree", degree, np.int64), - *(_cupy_array("t", t, np.float64) for t in (t1, t2, t3)), - *(_cupy_array("ind", ind, np.int64) for ind in (ind1, ind2, ind3)), - *(_cupy_array("c", c, np.float64) for c in (cx, cy, cz)), + self.kind_map = np.int32(kind_map) + self.params = _cupy_array("params", params, np.float64) + self.degree = _cupy_array("degree", degree, np.int64) + self.t1, self.t2, self.t3 = (_cupy_array("t", t, np.float64) for t in (t1, t2, t3)) + self.ind1, self.ind2, self.ind3 = (_cupy_array("ind", ind, np.int64) for ind in (ind1, ind2, ind3)) + self.cx, self.cy, self.cz = (_cupy_array("c", c, np.float64) for c in (cx, cy, cz)) + + def get_cuda_args(self) -> tuple: + return ( + self.kind_map, + self.params, + self.degree, + self.t1, + self.t2, + self.t3, + self.ind1, + self.ind2, + self.ind3, + self.cx, + self.cy, + self.cz, ) diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index 2a693c9ad..f942ae4f9 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -13,6 +13,8 @@ import cunumpy from cunumpy import PyccelKernel +from struphy.utils.cuda_arguments import Argument + def is_cuda_backend() -> bool: """Whether the active cunumpy backend is CuPy. @@ -70,8 +72,8 @@ def __call__(self, *args, n_threads: int): values = [] for arg in args: - if hasattr(arg, "values"): - values += arg.values + if isinstance(arg, Argument): + values.extend(arg.get_cuda_args()) else: values.append(arg) From 206ace55abe5e1ee788f1040dabfe01bc20055c8 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Thu, 1 Oct 2026 14:52:41 +0200 Subject: [PATCH 21/29] Only set _args_domain once, no logic in the property --- src/struphy/geometry/base.py | 55 ++++++++----------- src/struphy/geometry/tests/test_domain.py | 33 ++++++----- src/struphy/pic/tests/test_kernel_backends.py | 4 +- 3 files changed, 44 insertions(+), 48 deletions(-) diff --git a/src/struphy/geometry/base.py b/src/struphy/geometry/base.py index 9277749b0..be9281675 100644 --- a/src/struphy/geometry/base.py +++ b/src/struphy/geometry/base.py @@ -220,7 +220,8 @@ def __init__( "tran": dict_tran, } - self._args_domain = self._build_args_domain() + self._args_backend = xp.get_backend() + self._make_args_domain() def _build_args_domain(self): """Build runtime mapping arguments used by compiled evaluation kernels (host copies on the CuPy backend).""" @@ -278,13 +279,15 @@ def _can_build_args_domain(self): ) return all(hasattr(self, attr) for attr in required_attrs) - def _rebuild_args_domain(self): - # built lazily on first access, see cuda_args_domain - self._cuda_args_domain = None + def _make_args_domain(self): + self._args_domain = None + self._pyccel_args_domain = None if self._can_build_args_domain(): - self._args_domain = self._build_args_domain() - else: - self._args_domain = None + self._pyccel_args_domain = self._build_args_domain() + if self._args_backend == "cupy": + self._args_domain = self._build_cuda_args_domain() + else: + self._args_domain = self._pyccel_args_domain def __deepcopy__(self, memo): cls = self.__class__ @@ -292,23 +295,26 @@ def __deepcopy__(self, memo): memo[id(self)] = result for key, value in self.__dict__.items(): - if key in ("_args_domain", "_cuda_args_domain"): + if key in ("_args_domain", "_pyccel_args_domain"): continue setattr(result, key, copy.deepcopy(value, memo)) - result._rebuild_args_domain() + result._make_args_domain() return result def __getstate__(self): state = self.__dict__.copy() state.pop("_args_domain", None) - state.pop("_cuda_args_domain", None) + state.pop("_pyccel_args_domain", None) return state def __setstate__(self, state): self.__dict__.update(state) + if "_args_backend" not in self.__dict__: + arrays = (getattr(self, "_cx", None), getattr(self, "_cy", None), getattr(self, "_cz", None)) + self._args_backend = "cupy" if any(hasattr(arr, "__cuda_array_interface__") for arr in arrays) else "numpy" self._args_domain = None - self._rebuild_args_domain() + self._make_args_domain() def __repr__(self): out = f"{self.__class__.__name__}(\n" @@ -484,28 +490,11 @@ def indN(self): @property def args_domain(self): - """Object for all parameters needed for evaluation of metric coefficients.""" - if getattr(self, "_args_domain", None) is None: - self._rebuild_args_domain() - + """Arguments for the backend selected when this domain was created.""" if self._args_domain is None: raise AttributeError("DomainArguments are not available because the domain state is incomplete.") - return self._args_domain - @property - def cuda_args_domain(self) -> CudaDomainArguments: - """CUDA version of :attr:`args_domain`, referencing the domain's arrays on the device. - - Built on first access (never on the NumPy backend, where it raises TypeError), and rebuilt - after a deepcopy or unpickling, like :attr:`args_domain`. - """ - if getattr(self, "_cuda_args_domain", None) is None: - if not self._can_build_args_domain(): - raise AttributeError("CudaDomainArguments are not available because the domain state is incomplete.") - self._cuda_args_domain = self._build_cuda_args_domain() - return self._cuda_args_domain - @property def dict_transformations(self): """Dictionary of str->int for pull, push and transformation functions.""" @@ -1092,7 +1081,7 @@ def _evaluate_metric_coefficient(self, *etas, which=0, **kwargs): n_inside = kernel( markers, which, - self.args_domain, + self._pyccel_args_domain, out, remove_outside, avoid_round_off, @@ -1138,7 +1127,7 @@ def _evaluate_metric_coefficient(self, *etas, which=0, **kwargs): E2, E3, which, - self.args_domain, + self._pyccel_args_domain, out, is_sparse_meshgrid, avoid_round_off, @@ -1308,7 +1297,7 @@ def _pull_push_transform(self, which, a, kind_fun, *etas, flat_eval=False, **kwa _to_numpy_for_kernel(markers), _to_numpy_for_kernel(self._transformation_ids[which]), _to_numpy_for_kernel(kind_int), - _to_numpy_for_kernel(self.args_domain), + _to_numpy_for_kernel(self._pyccel_args_domain), out_np, _to_numpy_for_kernel(remove_outside), ) @@ -1370,7 +1359,7 @@ def _pull_push_transform(self, which, a, kind_fun, *etas, flat_eval=False, **kwa _to_numpy_for_kernel(E3), _to_numpy_for_kernel(self._transformation_ids[which]), _to_numpy_for_kernel(kind_int), - _to_numpy_for_kernel(self.args_domain), + _to_numpy_for_kernel(self._pyccel_args_domain), _to_numpy_for_kernel(is_sparse_meshgrid), out_np, ) diff --git a/src/struphy/geometry/tests/test_domain.py b/src/struphy/geometry/tests/test_domain.py index 4ceb9002e..ab4838638 100644 --- a/src/struphy/geometry/tests/test_domain.py +++ b/src/struphy/geometry/tests/test_domain.py @@ -1090,26 +1090,33 @@ def test_hollow_cyl_df_finite_difference(poc): requires_cupy = pytest.mark.skipif(not cunumpy.cupy_available(), reason="CuPy/GPU not available") -def test_cuda_args_domain_needs_device_arrays(): - """On the NumPy backend the domain arrays are host arrays, which are never copied to the device.""" +def test_args_domain_selects_backend(): + """The argument object follows the active backend.""" from struphy import domains + from struphy.kernel_arguments.pusher_args_kernels import DomainArguments with cunumpy.use_backend("numpy"): domain = domains.Cuboid() - with pytest.raises(TypeError, match="CuPy arrays"): - domain.cuda_args_domain + args = domain.args_domain + assert isinstance(args, DomainArguments) + if cunumpy.cupy_available(): + with cunumpy.use_backend("cupy"): + assert domain.args_domain is args @requires_cupy -def test_cuda_args_domain_independent_of_active_backend(): - """A domain created on the CuPy backend holds device arrays; its CUDA arguments can be built on either backend.""" +def test_args_domain_backend_is_fixed_at_creation(): + """Changing the active backend does not change a domain's argument object.""" from struphy import domains + from struphy.utils.cuda_arguments import CudaDomainArguments with cunumpy.use_backend("cupy"): domain = domains.Cuboid() + cuda_args = domain.args_domain + assert isinstance(cuda_args, CudaDomainArguments) with cunumpy.use_backend("numpy"): - args = domain.cuda_args_domain - assert args.values[3] is domain.T[0] + args = domain.args_domain + assert args is cuda_args @requires_cupy @@ -1124,16 +1131,16 @@ def test_cuda_args_domain(mapping): with cunumpy.use_backend("cupy"): domain = getattr(domains, mapping)() - args = domain.cuda_args_domain + args = domain.args_domain assert isinstance(args, CudaDomainArguments) - assert domain.cuda_args_domain is args # built once + assert domain.args_domain is args # built once kind_map, params, degree, t1, t2, t3, ind1, ind2, ind3, cx, cy, cz = args.values assert int(kind_map) == domain.kind_map # no copies of arrays that already have the right dtype and layout assert t1 is domain.T[0] and ind3 is domain.indN[2] - host = domain.args_domain + host = domain._pyccel_args_domain for dev, ref in ( (params, host.params), (degree, host.degree), @@ -1158,12 +1165,12 @@ def test_domain_deepcopy_and_pickle_on_cupy(mapping): with cunumpy.use_backend("cupy"): domain = getattr(domains, mapping)() - cuda_args = domain.cuda_args_domain + cuda_args = domain.args_domain for other in (copy.deepcopy(domain), pickle.loads(pickle.dumps(domain, protocol=pickle.HIGHEST_PROTOCOL))): assert other.args_domain.kind_map == domain.args_domain.kind_map assert (other.args_domain.params == domain.args_domain.params).all() - other_cuda = other.cuda_args_domain + other_cuda = other.args_domain assert other_cuda is not cuda_args assert other_cuda.values[3] is other.T[0] diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index 2921e09fd..276878275 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -136,7 +136,7 @@ def make_arguments(n_markers: int, seed: int = 0): return args_markers, domain.args_domain args_markers = CudaMarkerArguments(markers, valid_mks, n_markers, *MARKER_INDICES, bc_type) - return args_markers, domain.cuda_args_domain + return args_markers, domain.args_domain def expected_push(markers, valid_mks, dt, n_steps=1): @@ -232,7 +232,7 @@ def test_cuda_scalar_arguments(): def test_cuda_domain_arguments_reference_domain_arrays(): with cunumpy.use_backend("cupy"): domain = Cuboid() - args = domain.cuda_args_domain + args = domain.args_domain assert isinstance(args, CudaDomainArguments) assert len(args.values) == 12 assert args.values[3] is domain.T[0] and args.values[8] is domain.indN[2] and args.values[9] is domain.cx From 9529d599b205e267a1678eb7ae9b3ad8ee9ce9b6 Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Thu, 1 Oct 2026 14:56:43 +0200 Subject: [PATCH 22/29] Renamed the helper methods --- src/struphy/geometry/base.py | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/src/struphy/geometry/base.py b/src/struphy/geometry/base.py index be9281675..b9ab0388a 100644 --- a/src/struphy/geometry/base.py +++ b/src/struphy/geometry/base.py @@ -221,9 +221,9 @@ def __init__( } self._args_backend = xp.get_backend() - self._make_args_domain() + self._initialize_domain_args() - def _build_args_domain(self): + def _build_pyccel_domain_args(self): """Build runtime mapping arguments used by compiled evaluation kernels (host copies on the CuPy backend).""" return DomainArguments( self.kind_map, @@ -240,7 +240,7 @@ def _build_args_domain(self): _to_numpy_for_kernel(self.cz.copy()), # make sure we don't have stride = 0 ) - def _build_cuda_args_domain(self) -> CudaDomainArguments: + def _build_cuda_domain_args(self) -> CudaDomainArguments: """Build the CUDA kernel arguments from the domain's own (device) arrays. Arrays that already have the dtype and layout the CUDA kernels expect are referenced, not copied; @@ -279,13 +279,13 @@ def _can_build_args_domain(self): ) return all(hasattr(self, attr) for attr in required_attrs) - def _make_args_domain(self): + def _initialize_domain_args(self): self._args_domain = None self._pyccel_args_domain = None if self._can_build_args_domain(): - self._pyccel_args_domain = self._build_args_domain() + self._pyccel_args_domain = self._build_pyccel_domain_args() if self._args_backend == "cupy": - self._args_domain = self._build_cuda_args_domain() + self._args_domain = self._build_cuda_domain_args() else: self._args_domain = self._pyccel_args_domain @@ -299,7 +299,7 @@ def __deepcopy__(self, memo): continue setattr(result, key, copy.deepcopy(value, memo)) - result._make_args_domain() + result._initialize_domain_args() return result def __getstate__(self): @@ -314,7 +314,7 @@ def __setstate__(self, state): arrays = (getattr(self, "_cx", None), getattr(self, "_cy", None), getattr(self, "_cz", None)) self._args_backend = "cupy" if any(hasattr(arr, "__cuda_array_interface__") for arr in arrays) else "numpy" self._args_domain = None - self._make_args_domain() + self._initialize_domain_args() def __repr__(self): out = f"{self.__class__.__name__}(\n" From 470b43cb58b5024b07790f62301a16f8ea73f3dc Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Thu, 1 Oct 2026 15:25:41 +0200 Subject: [PATCH 23/29] Particles on GPU --- CUDA_STRATEGY.md | 48 ++++++++++++++++++++++++----------------- src/struphy/pic/base.py | 38 +++++++++++++++++++++++++++----- 2 files changed, 61 insertions(+), 25 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index 6e75244a2..0cb4bf421 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -5,13 +5,13 @@ The work is split into small PRs that can be reviewed and merged one at a time. ## PR checklist -- [ ] **PR 1: Proof of concept** (branch `cuda-kernel-proof-of-concept`) +- [x] **PR 1: Proof of concept** (branch `cuda-kernel-proof-of-concept`) `Kernel` (pyccel/CUDA pair), `CudaKernel`, `CudaMarkerArguments`/`CudaDomainArguments`, and one test file with a demo kernel pair run on both backends. Also adds this document and the `gpu` optional dependency. -- [ ] **PR 2: CUDA source files** โ€” `CudaKernel` loads CUDA source from a `_cuda.cu` file next to the pyccel file; `.cu`/`.cuh` files are shipped as package data. -- [ ] **PR 3: Kernel catalog** โ€” kernels are defined once in the `__init__.py` of the folder that contains them; a missing CUDA kernel raises an error on the GPU backend. -- [ ] **PR 4: `Pusher` accepts `Kernel`** โ€” the kernel for the active backend is chosen once, when the pusher is created; a plain `PyccelKernel` is wrapped, so the propagators do not change (no behaviour change on CPU). -- [ ] **PR 5: `Domain` on the GPU** โ€” `Domain.cuda_args_domain` (built lazily from the domain's device arrays), plus the fix for `Domain` deepcopy/unpickling on the CuPy backend (`_build_args_domain` passed `params_numpy` without `_to_numpy_for_kernel`). -- [ ] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend (e.g. `xp.prod` on Python lists in `pic/base.py`), plus `Particles.cuda_args_markers`. +- [x] **PR 2: CUDA source files** โ€” `CudaKernel` loads CUDA source from a `_cuda.cu` file next to the pyccel file; `.cu`/`.cuh` files are shipped as package data. +- [x] **PR 3: Kernel catalog** โ€” kernels are defined once in the `__init__.py` of the folder that contains them; a missing CUDA kernel raises an error on the GPU backend. +- [x] **PR 4: `Pusher` accepts `Kernel`** โ€” the kernel for the active backend is chosen once, when the pusher is created; a plain `PyccelKernel` is wrapped, so the propagators do not change (no behaviour change on CPU). +- [x] **PR 5: `Domain` on the GPU** โ€” domain arguments are selected and stored at domain construction; CUDA arguments reference device arrays, and deepcopy/unpickling rebuilds the arguments from the copied or restored arrays. +- [x] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend, and `Particles.cuda_args_markers` provides the CUDA argument bundle. - [ ] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. - [ ] **PR 8: Shared CUDA headers for the argument classes** โ€” one `.cuh` per argument class instead of long flat kernel signatures. - [ ] **PR 9: One folder per kernel, starting with `pic/pushing`** โ€” pure refactor, no behaviour change. @@ -41,19 +41,21 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke - **No silent CPU fallback on the GPU.** A kernel without a CUDA version raises an error on the GPU backend. Falling back would mean copying data to the host and back at every call. - **Small steps.** Every PR keeps the CPU code path working and tested. -## Current state (PR 1) +## Current state (PR 6) | File | Content | |---|---| -| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; replaces the argument classes by their `values` at each call, takes `n_threads`), `Kernel` (pyccel/CUDA pair, `get_kernel()` picks by backend) | -| `src/struphy/utils/cuda_arguments.py` | `CudaMarkerArguments`, `CudaDomainArguments`: same constructor arguments as the pyccel classes, hold CuPy arrays, flatten them into the CUDA kernel arguments | +| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; expands `Argument.get_cuda_args()` and takes `n_threads`), `Kernel` and `KernelCatalog` for backend selection and discovery | +| `src/struphy/utils/cuda_arguments.py` | `Argument` contract plus `CudaMarkerArguments` and `CudaDomainArguments`; CUDA arrays are stored individually and returned in signature order by `get_cuda_args()` | +| `src/struphy/geometry/base.py` | `Domain.args_domain` is selected once at construction; CUDA domains use device arrays, while direct Pyccel geometry calls retain a host argument bundle | +| `src/struphy/pic/base.py` | `Particles` arrays are allocated by the active `cunumpy` backend; `cuda_args_markers` lazily packages device arrays for CUDA kernels | | `src/struphy/pic/tests/test_kernel_backends.py` | the demo kernel pair `push_eta_linear` (pyccel function compiled with `epyccel` at test time, CUDA source string) and tests on both backends | Things we learned in the proof of concept: - The pyccel-compiled argument classes (`MarkerArguments`, `DomainArguments`, `DerhamArguments`) hold references to their owner's arrays, but only accept **NumPy** arrays. Hence the CUDA counterparts in `cuda_arguments.py`. -- Today, `Particles` builds `args_markers` from `_to_numpy_for_kernel(self.markers)`, i.e. from a **host copy** when the backend is CuPy. The same holds for `Derham`, and for `Domain.args_domain` (the CUDA version `Domain.cuda_args_domain` references the device arrays, PR 5). -- `Particles6D` and `Derham` cannot be created on the CuPy backend yet (PR 6, PR 7). `Domain` (e.g. `Cuboid`) can, and all its arrays are already CuPy arrays. +- `Particles.args_markers` remains the host bundle used by existing Pyccel calls. `Particles.cuda_args_markers` references the CuPy marker, validity, and boundary-condition arrays for CUDA calls. `Derham` still needs its CUDA creation work (PR 7). +- `Domain.args_domain` returns the argument type selected at domain construction; on CuPy it is a `CudaDomainArguments` object. Direct Pyccel methods on `Domain` use a separate internal host bundle. - `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Each argument is read with the size declared in the signature, so Python `int`/`float` arrive correctly in `int`/`double` parameters, but a wrongly typed scalar (e.g. an integer for a `double`, or a value that overflows an `int`) gives a wrong value **without an error**. Casting Python scalars in `CudaKernel` does not prevent this, so it is not done; see the follow-up in [Open questions](#open-questions). - Flattening the argument classes at each call (joining their `values`) costs well under 1 ยตs, compared to about 70 ยตs for launching the kernel. - `struphy compile` compiles every `.py` file whose name contains `kernels`. Non-pyccel modules must not contain `kernels` in their name; `.cu` files are ignored by it. @@ -116,25 +118,31 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - `Pusher` takes a `Kernel` or, as before, a `PyccelKernel` (wrapped into a `Kernel` without CUDA version). It calls `get_kernel()` once in its constructor, so on the CuPy backend a pusher whose kernel has no CUDA version fails when it is created, not in the time loop. Since no pusher kernel has a CUDA version yet, this is the case for all pushers. - The propagators do not change. They switch to catalog lookups once the kernels are split into folders (PR 9). -- Still to do (with PR 5โ€“7 and PR 11): on the GPU, the pusher must pass the CUDA argument objects (`particles.cuda_args_markers`, `domain.cuda_args_domain`, device arrays in `args_kernel`) instead of the pyccel ones. This choice is made once, together with the kernel, so that a CUDA kernel is never called with pyccel arguments or vice versa. +- Still to do (PR 11): on the GPU, the pusher must pass `particles.cuda_args_markers`, device arrays in `args_kernel`, and the domain's already backend-selected `domain.args_domain`. This choice is made once, together with the kernel, so that a CUDA kernel is never called with pyccel arguments or vice versa. ### PR 5โ€“7: Owners build their CUDA arguments -- `Particles`, `Domain` and `Derham` own the arrays, so they build the CUDA argument objects from their own `xp` arrays (`cuda_args_markers`, `cuda_args_domain`, `cuda_args_derham`), in the same place where the pyccel argument objects are built today. The pyccel `args_*` stay as they are: they are used by 30+ modules through pyccel kernels. +- `Particles`, `Domain` and `Derham` own the arrays, so they build CUDA argument objects from their own `xp` arrays. `Particles.cuda_args_markers` is separate from the host `args_markers`; `Domain.args_domain` is selected once at construction. The pyccel argument bundles remain available for existing direct Pyccel calls. - The CUDA argument objects hold references. If an owner reallocates an array (today the markers are allocated once), it must rebuild its CUDA arguments at the same place, exactly like for the pyccel arguments. - First these classes must be creatable on the CuPy backend at all: - - `Particles`: `xp.prod`/`xp.sum` on Python lists and similar (`pic/base.py`), probably more. + - `Particles`: wrap Python lists in `xp.array` before reductions, and use host buffers for scalar MPI gathers (`pic/base.py`). - `Derham`: NumPy arrays from feectools reach `cupy.ascontiguousarray`. - `Domain`: deepcopy and unpickling on CuPy failed (see PR 5 below). -### PR 5: `Domain` on the GPU +### PR 5: `Domain` on the GPU (complete) -- `Domain.cuda_args_domain`: `CudaDomainArguments` built from the domain's own device arrays, next to `args_domain` in `geometry/base.py`. It is built on first access, so CPU runs never build it; on the NumPy backend it raises `TypeError` (host arrays are never copied to the device). +- `Domain.args_domain` is selected when the domain is created: `DomainArguments` for NumPy or `CudaDomainArguments` for CuPy. CUDA arguments are built from the domain's device arrays; NumPy data is never copied to the device. - Arrays that already have the dtype and layout the CUDA kernels expect (`float64`/`int64`, C-contiguous) are referenced, not copied, e.g. the knot vectors `T` and `indN`. Otherwise one device copy is made when the arguments are built (e.g. `degree`, which is a tuple, or broadcast control points), never at kernel call time. -- Like `args_domain`, the CUDA arguments are rebuilt after a deepcopy or unpickling (they are dropped in `__deepcopy__`/`__getstate__` and built again on first access). -- Fix: `__init__` and the rebuild after deepcopy/unpickling now use the same builder, `_build_args_domain`, which converts **all** arrays to NumPy for the pyccel `DomainArguments`. Before, the rebuild passed `params_numpy` without conversion, so a deepcopy of a domain on the CuPy backend failed. -- Spline mappings (e.g. `IGAPolarCylinder`) cannot be created on the CuPy backend yet: `interp_mapping` passes CuPy arrays to `scipy.sparse.csc_matrix`. So `cuda_args_domain` is only available for analytic mappings for now; making the spline mappings work on the GPU is left for when the first one is needed there. -- The pushers still receive `domain.args_domain` from the propagators; switching them to `cuda_args_domain` on the GPU comes with PR 6/7 and PR 11. +- The selected arguments and the internal host bundle are recreated after deepcopy or unpickling so they refer to the new domain's arrays. +- Direct Pyccel geometry methods use the host bundle; the public `args_domain` remains the backend-specific bundle selected at construction. +- Spline mappings (e.g. `IGAPolarCylinder`) cannot be created on the CuPy backend yet: `interp_mapping` passes CuPy arrays to `scipy.sparse.csc_matrix`. Consequently, CUDA `args_domain` is available only for analytic mappings for now; making spline mappings work on the GPU is left for when the first one is needed there. + +### PR 6: `Particles` on the GPU (complete) + +- Particle arrays, validity masks, and boundary-condition codes are allocated through `cunumpy`, so they live on CuPy when the CuPy backend is active. +- `cuda_args_markers` lazily builds `CudaMarkerArguments` from those device arrays. The existing `args_markers` remains the host argument bundle for current Pyccel kernel calls. +- Domain decomposition now wraps the Python `nprocs` list with `xp.array` before calling `xp.prod`. Scalar MPI gathers use small NumPy buffers and copy the results back to the active array backend, avoiding unsupported CuPy buffers in MPI calls. +- Full GPU particle pushing still depends on CUDA versions of the required kernels and on PR 7's `Derham` support. ### PR 8: Argument structs in shared headers diff --git a/src/struphy/pic/base.py b/src/struphy/pic/base.py index eb60e7bf9..a968493f1 100644 --- a/src/struphy/pic/base.py +++ b/src/struphy/pic/base.py @@ -56,6 +56,7 @@ ) from struphy.utils import utils from struphy.utils.clone_config import CloneConfig +from struphy.utils.cuda_arguments import CudaMarkerArguments if TYPE_CHECKING: # importing mpi4py.MPI initializes MPI, which is slow; only needed for annotations from mpi4py.MPI import Intracomm @@ -940,6 +941,28 @@ def args_markers(self) -> MarkerArguments: """Collection of mandatory arguments for pusher kernels.""" return self._args_markers + @property + def cuda_args_markers(self) -> CudaMarkerArguments: + """CUDA arguments for particle kernels, referencing the marker owner's device arrays.""" + if not hasattr(self, "_markers"): + raise AttributeError("CudaMarkerArguments are not available before the marker array is allocated.") + if getattr(self, "_cuda_args_markers", None) is None: + self._cuda_args_markers = CudaMarkerArguments( + self.markers, + self.valid_mks, + self.Np, + self.vdim, + self.index["weights"], + self.first_diagnostics_idx, + self.first_pusher_idx, + self.first_shift_idx, + self.residual_idx, + self.first_free_idx, + self.mu_idx, + self._bc_type, + ) + return self._cuda_args_markers + # ------------------------------------------- # Initial condition and background -> weights # ------------------------------------------- @@ -2337,10 +2360,12 @@ def gather_scalar_in_subcomm_array(self, scalar: int, out: xp.ndarray = None): _tmp[self.mpi_rank] = scalar if self.mpi_comm is not None: + gathered = np.empty(self.mpi_size, dtype=int) self.mpi_comm.Allgather( - _tmp[self.mpi_rank], - _tmp, + np.array([scalar], dtype=int), + gathered, ) + _tmp[:] = xp.asarray(gathered) return _tmp @@ -2365,10 +2390,12 @@ def gather_scalar_in_intercomm_array(self, scalar: int, out: xp.ndarray = None): _tmp[self.clone_id] = scalar if self.clone_config is not None: + gathered = np.empty(self.num_clones, dtype=int) self.clone_config.inter_comm.Allgather( - _tmp[self.clone_id], - _tmp, + np.array([scalar], dtype=int), + gathered, ) + _tmp[:] = xp.asarray(gathered) return _tmp @@ -2431,7 +2458,7 @@ def _get_domain_decomp(self, mpi_dims_mask: tuple | list = None): mm = (mm + 1) % 3 nprocs[mm] *= fac - assert xp.prod(nprocs) == self.mpi_size + assert xp.prod(xp.array(nprocs)) == self.mpi_size # domain decomposition breaks = [xp.linspace(0.0, 1.0, nproc + 1) for nproc in nprocs] @@ -2555,6 +2582,7 @@ def _allocate_marker_array(self, dry_run: bool = False): _to_numpy_for_kernel(self.mu_idx), _to_numpy_for_kernel(self._bc_type), ) + self._cuda_args_markers = None def _initialize_sorting_boxes(self): """Initializes the sorting boxes. From b2da69048c45ca2b895c03444639a9dfce33a71f Mon Sep 17 00:00:00 2001 From: Max Date: Thu, 1 Oct 2026 16:02:10 +0200 Subject: [PATCH 24/29] Derham on the GPU: create Derham on CuPy, add Derham.cuda_args_derham Needs the feectools changes on branch cupy-derham. Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 25 ++- src/struphy/feec/psydac_derham.py | 227 +++++++++++----------- src/struphy/feec/tests/test_derham_gpu.py | 66 +++++++ src/struphy/utils/cuda_arguments.py | 29 +++ 4 files changed, 229 insertions(+), 118 deletions(-) create mode 100644 src/struphy/feec/tests/test_derham_gpu.py diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index 0cb4bf421..c6389d7e9 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -12,7 +12,7 @@ The work is split into small PRs that can be reviewed and merged one at a time. - [x] **PR 4: `Pusher` accepts `Kernel`** โ€” the kernel for the active backend is chosen once, when the pusher is created; a plain `PyccelKernel` is wrapped, so the propagators do not change (no behaviour change on CPU). - [x] **PR 5: `Domain` on the GPU** โ€” domain arguments are selected and stored at domain construction; CUDA arguments reference device arrays, and deepcopy/unpickling rebuilds the arguments from the copied or restored arrays. - [x] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend, and `Particles.cuda_args_markers` provides the CUDA argument bundle. -- [ ] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. +- [x] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. Needs the companion feectools changes (branch `cupy-derham`), which keep the spline-space construction data on the host. - [ ] **PR 8: Shared CUDA headers for the argument classes** โ€” one `.cuh` per argument class instead of long flat kernel signatures. - [ ] **PR 9: One folder per kernel, starting with `pic/pushing`** โ€” pure refactor, no behaviour change. - [ ] **PR 10: Device versions of helper kernels** โ€” B-spline evaluation, mapping evaluation (per domain), small linear algebra, as `__device__` functions in `.cuh` headers. @@ -41,20 +41,21 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke - **No silent CPU fallback on the GPU.** A kernel without a CUDA version raises an error on the GPU backend. Falling back would mean copying data to the host and back at every call. - **Small steps.** Every PR keeps the CPU code path working and tested. -## Current state (PR 6) +## Current state (PR 7) | File | Content | |---|---| | `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; expands `Argument.get_cuda_args()` and takes `n_threads`), `Kernel` and `KernelCatalog` for backend selection and discovery | -| `src/struphy/utils/cuda_arguments.py` | `Argument` contract plus `CudaMarkerArguments` and `CudaDomainArguments`; CUDA arrays are stored individually and returned in signature order by `get_cuda_args()` | +| `src/struphy/utils/cuda_arguments.py` | `Argument` contract plus `CudaMarkerArguments`, `CudaDerhamArguments` and `CudaDomainArguments`; CUDA arrays are stored individually and returned in signature order by `get_cuda_args()` | | `src/struphy/geometry/base.py` | `Domain.args_domain` is selected once at construction; CUDA domains use device arrays, while direct Pyccel geometry calls retain a host argument bundle | | `src/struphy/pic/base.py` | `Particles` arrays are allocated by the active `cunumpy` backend; `cuda_args_markers` lazily packages device arrays for CUDA kernels | +| `src/struphy/feec/psydac_derham.py` | `Derham` can be created on the CuPy backend; spline-space data stays on the host, `cuda_args_derham` lazily holds device copies of degrees, knots and starts | | `src/struphy/pic/tests/test_kernel_backends.py` | the demo kernel pair `push_eta_linear` (pyccel function compiled with `epyccel` at test time, CUDA source string) and tests on both backends | Things we learned in the proof of concept: - The pyccel-compiled argument classes (`MarkerArguments`, `DomainArguments`, `DerhamArguments`) hold references to their owner's arrays, but only accept **NumPy** arrays. Hence the CUDA counterparts in `cuda_arguments.py`. -- `Particles.args_markers` remains the host bundle used by existing Pyccel calls. `Particles.cuda_args_markers` references the CuPy marker, validity, and boundary-condition arrays for CUDA calls. `Derham` still needs its CUDA creation work (PR 7). +- `Particles.args_markers` remains the host bundle used by existing Pyccel calls. `Particles.cuda_args_markers` references the CuPy marker, validity, and boundary-condition arrays for CUDA calls. Likewise `Derham.args_derham` stays the host bundle and `Derham.cuda_args_derham` holds the CUDA one. - `Domain.args_domain` returns the argument type selected at domain construction; on CuPy it is a `CudaDomainArguments` object. Direct Pyccel methods on `Domain` use a separate internal host bundle. - `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Each argument is read with the size declared in the signature, so Python `int`/`float` arrive correctly in `int`/`double` parameters, but a wrongly typed scalar (e.g. an integer for a `double`, or a value that overflows an `int`) gives a wrong value **without an error**. Casting Python scalars in `CudaKernel` does not prevent this, so it is not done; see the follow-up in [Open questions](#open-questions). - Flattening the argument classes at each call (joining their `values`) costs well under 1 ยตs, compared to about 70 ยตs for launching the kernel. @@ -126,7 +127,7 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - The CUDA argument objects hold references. If an owner reallocates an array (today the markers are allocated once), it must rebuild its CUDA arguments at the same place, exactly like for the pyccel arguments. - First these classes must be creatable on the CuPy backend at all: - `Particles`: wrap Python lists in `xp.array` before reductions, and use host buffers for scalar MPI gathers (`pic/base.py`). - - `Derham`: NumPy arrays from feectools reach `cupy.ascontiguousarray`. + - `Derham`: feectools moved its host data (knots, grids, collocation matrices) to CuPy before calling pyccel kernels (see PR 7 below). - `Domain`: deepcopy and unpickling on CuPy failed (see PR 5 below). ### PR 5: `Domain` on the GPU (complete) @@ -142,7 +143,19 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - Particle arrays, validity masks, and boundary-condition codes are allocated through `cunumpy`, so they live on CuPy when the CuPy backend is active. - `cuda_args_markers` lazily builds `CudaMarkerArguments` from those device arrays. The existing `args_markers` remains the host argument bundle for current Pyccel kernel calls. - Domain decomposition now wraps the Python `nprocs` list with `xp.array` before calling `xp.prod`. Scalar MPI gathers use small NumPy buffers and copy the results back to the active array backend, avoiding unsupported CuPy buffers in MPI calls. -- Full GPU particle pushing still depends on CUDA versions of the required kernels and on PR 7's `Derham` support. +- Full GPU particle pushing still depends on CUDA versions of the required kernels. + +### PR 7: `Derham` on the GPU (complete) + +- feectools (and the struphy code that builds `Derham`) followed `xp` everywhere, so on CuPy the knots, quadrature grids, collocation matrices and decomposition metadata became device arrays and then reached pyccel kernels, SciPy or MPI, which only take host arrays. The rule is now: **data that describes the spline spaces is host data on every backend**. Only the coefficients (`StencilVector` data) and stencil matrices live on the device. + - feectools, branch `cupy-derham` (separate PR, must be merged and released first): `core/bsplines.py` takes and returns NumPy arrays (CuPy inputs are copied to the host once), and so do the 1D interpolation/histopolation matrices and their factorizations (`fem/splines.py`, `BandedSolver`). Decomposition metadata (`ddm/cart.py`, `ddm/partition.py`), MPI counts and shapes in `linalg/kron.py`, and `compute_diag_len` in `linalg/stencil.py` are NumPy too. `StencilMatrix._tocoo_no_pads` had a duplicated kernel call on the CuPy data (merge leftover) and now builds the SciPy matrix from one host copy. + - struphy: the projection and quadrature grids of `Derham` (`get_pts_and_wts`, ...) and `spline_types_pyccel` are NumPy. `domain_array`, `index_array(_N/_D)` and `neighbours` are gathered with NumPy MPI buffers and then converted with `xp.asarray`, so they are device arrays on CuPy like `Particles.domain_array`. +- `Derham.args_derham` is built from the host knots, degrees and starts on both backends. `Derham.cuda_args_derham` lazily builds `CudaDerhamArguments` with one device copy of these small arrays. On NumPy it raises (host arrays are never copied to the device). The pyccel scratch arrays (`bn1`, ..., `bd3`) are not part of it; they become per-thread local arrays in CUDA (PR 10). +- Not supported on CuPy yet: + - Local projectors (`DerhamOptions.local_projectors=True`) raise `NotImplementedError` when the `Derham` is created. `CommutingProjectorLocal` builds its data with `xp` and calls pyccel kernels on it, like `Derham` did. + - Polar splines need a spline mapping, which cannot be created on CuPy yet (see PR 5). + - Field evaluation (`SplineFunction.__call__`, ...) still calls pyccel kernels with the coefficients, which are device arrays on CuPy. It needs CUDA evaluation kernels (PR 10+). +- Tests: `feec/tests/test_derham_gpu.py`. Without a GPU, a strict host stand-in for CuPy (rejects host/device mixing, cannot run kernels; not part of the repository) was used to create `Derham` on the "CuPy" backend with 1, 2 and 4 MPI processes and compare it with NumPy. ### PR 8: Argument structs in shared headers diff --git a/src/struphy/feec/psydac_derham.py b/src/struphy/feec/psydac_derham.py index f3ca2784d..3b3ce2a3e 100644 --- a/src/struphy/feec/psydac_derham.py +++ b/src/struphy/feec/psydac_derham.py @@ -44,6 +44,8 @@ from struphy.polar.extraction_operators import PolarExtractionBlocksC1 from struphy.polar.linear_operators import PolarExtractionOperator, PolarLinearOperator from struphy.topology.grids import TensorProductGrid +from struphy.utils.cuda_arguments import CudaDerhamArguments +from struphy.utils.kernel_backends import is_cuda_backend NonTrivialBC = LiteralOptions.OptsNonTrivialBoundaryCondition space_to_form = { @@ -57,14 +59,6 @@ logger = logging.getLogger("struphy") -def _to_numpy_for_kernel(value): - """Convert CuPy arrays to NumPy for compiled kernel calls.""" - if hasattr(value, "get"): - # This is a CuPy array - return value.get() - return value - - class DiscreteDerham: """Discrete 3D de Rham sequence built from four FE spaces. @@ -423,7 +417,8 @@ def __init__( self._quad_grid_spans[-1] = tuple(self.quad_grid_spans[-1]) self._quad_grid_bases[-1] = tuple(self.quad_grid_bases[-1]) - self._spline_types_pyccel[-1] = xp.array( + # inputs of pyccel kernels, kept on the host + self._spline_types_pyccel[-1] = np.array( self._spline_types_pyccel[-1], ) @@ -485,7 +480,7 @@ def spline_types(self) -> tuple[tuple[str]]: return self._spline_types @property - def spline_types_pyccel(self) -> tuple[xp.ndarray]: + def spline_types_pyccel(self) -> tuple[np.ndarray]: """Tuple of spline types in each direction as integers (0 for 'B', 1 for 'M') for each component of the vector space.""" return self._spline_types_pyccel @@ -608,6 +603,10 @@ def __init__( polar_splines = options.polar_splines # local commuting projectors local_projectors = options.local_projectors + if local_projectors and is_cuda_backend(): + raise NotImplementedError( + "Local projectors (DerhamOptions.local_projectors=True) are not supported on the CuPy backend yet." + ) # number of elements and spline degrees in each direction assert len(num_elements) == 3 @@ -899,14 +898,13 @@ def __init__( self._neighbours = self._get_neighbours() - # collect arguments for kernels + # collect arguments for kernels (the knots of feectools are host arrays on every array backend) self._args_derham = DerhamArguments( - _to_numpy_for_kernel(xp.array(self.degree)), - _to_numpy_for_kernel(self.V0fem.knots[0]), - _to_numpy_for_kernel(self.V0fem.knots[1]), - _to_numpy_for_kernel(self.V0fem.knots[2]), - _to_numpy_for_kernel(xp.array(self.V0.starts)), + np.array(self.degree), + *self.V0fem.knots, + np.array(self.V0.starts), ) + self._cuda_args_derham = None logger.debug("\nDERHAM:") logger.debug(f"{'number of elements:'.ljust(25)} {num_elements}") @@ -1481,6 +1479,17 @@ def args_derham(self): """Collection of mandatory arguments for pusher kernels.""" return self._args_derham + @property + def cuda_args_derham(self) -> CudaDerhamArguments: + """CUDA arguments for particle kernels; holds device copies of the spline degrees, knots and start indices.""" + if self._cuda_args_derham is None: + self._cuda_args_derham = CudaDerhamArguments( + xp.asarray(self.args_derham.pn), + *(xp.asarray(t) for t in self.V0fem.knots), + xp.asarray(self.args_derham.starts), + ) + return self._cuda_args_derham + # -------------------------- # methods: # -------------------------- @@ -1778,7 +1787,7 @@ def _discretize_space( ) # Create uniform grid - grids = [xp.linspace(xmin, xmax, num=ne + 1) for xmin, xmax, ne in zip(min_coords, max_coords, ncells)] + grids = [np.linspace(xmin, xmax, num=ne + 1) for xmin, xmax, ne in zip(min_coords, max_coords, ncells)] # Create 1D finite element spaces and precompute quadrature data spaces_1d = [ @@ -1925,11 +1934,11 @@ def _get_domain_array(self): else: nproc = 1 - # send buffer - dom_arr_loc = xp.zeros(9, dtype=float) + # send buffer (MPI buffers are host arrays on every array backend) + dom_arr_loc = np.zeros(9, dtype=float) # main array (receive buffers) - dom_arr = xp.zeros(nproc * 9, dtype=float) + dom_arr = np.zeros(nproc * 9, dtype=float) # Get global starts and ends of domain decomposition gl_s = self.domain_decomposition.starts @@ -1947,7 +1956,7 @@ def _get_domain_array(self): else: dom_arr[:] = dom_arr_loc - return dom_arr.reshape(nproc, 9) + return xp.asarray(dom_arr.reshape(nproc, 9)) def _get_index_array(self, decomposition): """ @@ -1972,11 +1981,11 @@ def _get_index_array(self, decomposition): else: nproc = 1 - # send buffer - ind_arr_loc = xp.zeros(6, dtype=int) + # send buffer (MPI buffers are host arrays on every array backend) + ind_arr_loc = np.zeros(6, dtype=int) # main array (receive buffers) - ind_arr = xp.zeros(nproc * 6, dtype=int) + ind_arr = np.zeros(nproc * 6, dtype=int) # Get global starts and ends of cart OR domain decomposition gl_s = decomposition.starts @@ -1993,7 +2002,7 @@ def _get_index_array(self, decomposition): else: ind_arr[:] = ind_arr_loc - return ind_arr.reshape(nproc, 6) + return xp.asarray(ind_arr.reshape(nproc, 6)) def _get_neighbours(self): """ @@ -2025,7 +2034,7 @@ def _get_neighbours(self): neighbours along the edges only have one 1, neighbours along the edges have no 1 in the index. """ - neighs = xp.empty((3, 3, 3), dtype=int) + neighs = np.empty((3, 3, 3), dtype=int) for i in range(3): for j in range(3): @@ -2034,7 +2043,7 @@ def _get_neighbours(self): ind = tuple(comp) neighs[ind] = self._get_neighbour_one_component(comp) - return neighs + return xp.asarray(neighs) def _get_neighbour_one_component(self, comp): """ @@ -2070,8 +2079,9 @@ def _get_neighbour_one_component(self, comp): if comp == [1, 1, 1]: return neigh_id - comp = xp.array(comp) - kinds = xp.array(kinds) + # computed on the host: the start/end indices below are an object array (with None entries) + comp = np.array(comp) + kinds = np.array(kinds) # if only one process: check if comp is neighbour in non-peridic directions, if this is not the case then return the rank as neighbour id if size == 1: @@ -2084,12 +2094,13 @@ def _get_neighbour_one_component(self, comp): # elements with index 2n are the starts and 2n + 1 are the ends. neigh_inds = [None] * 6 + index_array = xp.to_numpy(self.index_array) # in each direction find start/end index for neighbour for k, co in enumerate(comp): if co == 1: - neigh_inds[2 * k + 0] = self.index_array[rank, 2 * k + 0] - neigh_inds[2 * k + 1] = self.index_array[rank, 2 * k + 1] + neigh_inds[2 * k + 0] = index_array[rank, 2 * k + 0] + neigh_inds[2 * k + 1] = index_array[rank, 2 * k + 1] elif co == 0: neigh_inds[2 * k + 1] = gl_s[k] - 1 @@ -2106,15 +2117,15 @@ def _get_neighbour_one_component(self, comp): "Wrong value for component; must be 0 or 1 or 2 !", ) - neigh_inds = xp.array(neigh_inds) + neigh_inds = np.array(neigh_inds) # only use indices where information is present to find the neighbours rank - inds = xp.where(xp.not_equal(neigh_inds, None)) + inds = np.where(np.not_equal(neigh_inds, None)) # find ranks (row index of domain_array) which agree in start/end indices - index_temp = xp.squeeze(self.index_array[:, inds]) - unique_ranks = xp.where( - xp.equal(index_temp, neigh_inds[inds]).all(1), + index_temp = np.squeeze(index_array[:, inds]) + unique_ranks = np.where( + np.equal(index_temp, neigh_inds[inds]).all(1), )[0] # if any row satisfies condition, return its index (=rank of neighbour) @@ -3490,18 +3501,16 @@ def get_pts_and_wts(space_1d, start, end, n_quad=None, polar_shift=False): histopol_loc = space_1d.histopolation_grid[start : end + 2].copy() # make sure that greville points used for interpolation are in [0, 1] - # Use numpy for comparison since greville points are NumPy arrays - greville_loc_np = greville_loc.get() if hasattr(greville_loc, "get") else greville_loc - assert np.all(np.logical_and(greville_loc_np >= 0.0, greville_loc_np <= 1.0)) + assert np.all(np.logical_and(greville_loc >= 0.0, greville_loc <= 1.0)) # interpolation if space_1d.basis == "B": x_grid = greville_loc pts = greville_loc[:, None] - wts = xp.ones(pts.shape, dtype=float) + wts = np.ones(pts.shape, dtype=float) # sub-interval index is always 0 for interpolation. - subs = xp.zeros(pts.shape[0], dtype=int) + subs = np.zeros(pts.shape[0], dtype=int) # !! shift away first interpolation point in eta_1 direction for polar domains !! if pts[0] == 0.0 and polar_shift: @@ -3515,32 +3524,32 @@ def get_pts_and_wts(space_1d, start, end, n_quad=None, polar_shift=False): union_breaks = space_1d.breaks[:-1] # Make union of Greville and break points - # tmp = set(xp.round(space_1d.histopolation_grid, decimals=14)).union( - # xp.round(union_breaks, decimals=14), + # tmp = set(np.round(space_1d.histopolation_grid, decimals=14)).union( + # np.round(union_breaks, decimals=14), # ) # tmp = list(tmp) # tmp.sort() - # tmp_a = xp.array(tmp) + # tmp_a = np.array(tmp) - tmp = set(xp.round(space_1d.histopolation_grid, decimals=14).tolist()).union( - xp.round(union_breaks, decimals=14).tolist() + tmp = set(np.round(space_1d.histopolation_grid, decimals=14).tolist()).union( + np.round(union_breaks, decimals=14).tolist() ) tmp = sorted(tmp) - tmp_a = xp.array(tmp) + tmp_a = np.array(tmp) x_grid = tmp_a[ - xp.logical_and( + np.logical_and( tmp_a - >= xp.min( + >= np.min( histopol_loc, ) - 1e-14, - tmp_a <= xp.max(histopol_loc) + 1e-14, + tmp_a <= np.max(histopol_loc) + 1e-14, ) ] # determine subinterval index (= 0 or 1): - subs = xp.zeros(x_grid[:-1].size, dtype=int) + subs = np.zeros(x_grid[:-1].size, dtype=int) for n, x_h in enumerate(x_grid[:-1]): add = 1 for x_g in histopol_loc: @@ -3555,12 +3564,6 @@ def get_pts_and_wts(space_1d, start, end, n_quad=None, polar_shift=False): pts_loc, wts_loc = np.polynomial.legendre.leggauss(n_quad) - if "cupy" in xp.__name__: - import cupy as cp - - pts_loc = cp.array(pts_loc) - wts_loc = cp.array(wts_loc) - x, wts = bsp.quadrature_grid(x_grid, pts_loc, wts_loc) pts = x % 1.0 @@ -3572,7 +3575,7 @@ def get_pts_and_wts_quasi( space_1d: SplineSpace, *, polar_shift: bool = False, -) -> tuple[xp.ndarray, xp.ndarray]: +) -> tuple[np.ndarray, np.ndarray]: r"""Obtain local projection point sets and weights in one grid direction for the quasi-interpolation method. The quasi-interpolation points are :math:`\nu - \mu +p` equidistant points :math:`\{ x^i_j \}_{0 \leq j < \nu - \mu +p}` in the sub-interval :math:`Q = [\eta_\mu , \eta_\nu]` given by: @@ -3620,12 +3623,12 @@ def get_pts_and_wts_quasi( # interpolation if space_1d.basis == "B": if degree == 1 and h != 1.0: - x_grid = xp.linspace(-(degree - 1) * h, 1.0 - h + (h / 2.0), (N + degree - 1) * 2) + x_grid = np.linspace(-(degree - 1) * h, 1.0 - h + (h / 2.0), (N + degree - 1) * 2) else: - x_grid = xp.linspace(-(degree - 1) * h, 1.0 - h, (N + degree - 1) * 2 - 1) + x_grid = np.linspace(-(degree - 1) * h, 1.0 - h, (N + degree - 1) * 2 - 1) pts = x_grid[:, None] % 1.0 - wts = xp.ones(pts.shape, dtype=float) + wts = np.ones(pts.shape, dtype=float) # !! shift away first interpolation point in eta_1 direction for polar domains !! if pts[0] == 0.0 and polar_shift: @@ -3636,16 +3639,16 @@ def get_pts_and_wts_quasi( # The computation of histopolation points breaks in case we have num_elements=1 and periodic boundary conditions since we end up with only one x_grid point. # We need to build the histopolation points by hand in this scenario. if degree == 0 and h == 1.0: - x_grid = xp.array([0.0, 0.5, 1.0]) + x_grid = np.array([0.0, 0.5, 1.0]) elif degree == 0 and h != 1.0: - x_grid = xp.linspace(-degree * h, 1.0 - h + (h / 2.0), (N + degree) * 2) + x_grid = np.linspace(-degree * h, 1.0 - h + (h / 2.0), (N + degree) * 2) else: - x_grid = xp.linspace(-degree * h, 1.0 - h, (N + degree) * 2 - 1) + x_grid = np.linspace(-degree * h, 1.0 - h, (N + degree) * 2 - 1) n_quad = degree + 1 # Gauss - Legendre quadrature points and weights # products of basis functions are integrated exactly - pts_loc, wts_loc = xp.polynomial.legendre.leggauss(n_quad) + pts_loc, wts_loc = np.polynomial.legendre.leggauss(n_quad) x, wts = bsp.quadrature_grid(x_grid, pts_loc, wts_loc) pts = x % 1.0 @@ -3659,26 +3662,26 @@ def get_pts_and_wts_quasi( N_b = N + degree # Filling the quasi-interpolation points for i=0 and i=1 (since they are equal) - x_grid = xp.linspace(0.0, knots[degree + 1], degree + 1) - x_aux = xp.linspace(0.0, knots[degree + 1], degree + 1) - x_grid = xp.append(x_grid, x_aux) + x_grid = np.linspace(0.0, knots[degree + 1], degree + 1) + x_aux = np.linspace(0.0, knots[degree + 1], degree + 1) + x_grid = np.append(x_grid, x_aux) # Now we append those for 1 tuple: ) +class CudaDerhamArguments(Argument): + """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DerhamArguments`. + + CUDA signature of :meth:`get_cuda_args`: ``long long* pn, double* tn1, double* tn2, double* tn3, long long* starts`` + + The scratch arrays of the pyccel class (``bn1``, ..., ``bd3``) are not part of it; CUDA kernels use + per-thread local arrays instead. + + Parameters + ---------- + pn : cupy.ndarray[int] + Spline degrees of :class:`~struphy.feec.psydac_derham.Derham` (int64). + + tn1, tn2, tn3 : cupy.ndarray[float] + Knot sequences of :class:`~struphy.feec.psydac_derham.Derham`. + + starts : cupy.ndarray[int] + Start indices (current MPI process) of :class:`~struphy.feec.psydac_derham.Derham` (int64). + """ + + def __init__(self, pn, tn1, tn2, tn3, starts): + self.pn = _cupy_array("pn", pn, np.int64) + self.tn1, self.tn2, self.tn3 = (_cupy_array("tn", t, np.float64) for t in (tn1, tn2, tn3)) + self.starts = _cupy_array("starts", starts, np.int64) + + def get_cuda_args(self) -> tuple: + return (self.pn, self.tn1, self.tn2, self.tn3, self.starts) + + class CudaDomainArguments(Argument): """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments`. From 1d50ae3ade4f2b0f336e2f09f9060f87ea30cd12 Mon Sep 17 00:00:00 2001 From: Max Date: Thu, 1 Oct 2026 16:06:16 +0200 Subject: [PATCH 25/29] Shared CUDA header for the kernel argument classes Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 28 ++- src/struphy/geometry/tests/test_domain.py | 29 +-- src/struphy/kernel_arguments/pusher_args.cuh | 57 +++++ src/struphy/pic/tests/test_kernel_backends.py | 176 ++++++++++++--- src/struphy/utils/cuda_arguments.py | 212 ++++++++++++------ src/struphy/utils/kernel_backends.py | 13 +- 6 files changed, 376 insertions(+), 139 deletions(-) create mode 100644 src/struphy/kernel_arguments/pusher_args.cuh diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index c6389d7e9..5681f99ad 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -13,7 +13,7 @@ The work is split into small PRs that can be reviewed and merged one at a time. - [x] **PR 5: `Domain` on the GPU** โ€” domain arguments are selected and stored at domain construction; CUDA arguments reference device arrays, and deepcopy/unpickling rebuilds the arguments from the copied or restored arrays. - [x] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend, and `Particles.cuda_args_markers` provides the CUDA argument bundle. - [x] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. Needs the companion feectools changes (branch `cupy-derham`), which keep the spline-space construction data on the host. -- [ ] **PR 8: Shared CUDA headers for the argument classes** โ€” one `.cuh` per argument class instead of long flat kernel signatures. +- [x] **PR 8: Shared CUDA header for the argument classes** โ€” one struct per argument class in `kernel_arguments/pusher_args.cuh`, passed by value, instead of long flat kernel signatures. - [ ] **PR 9: One folder per kernel, starting with `pic/pushing`** โ€” pure refactor, no behaviour change. - [ ] **PR 10: Device versions of helper kernels** โ€” B-spline evaluation, mapping evaluation (per domain), small linear algebra, as `__device__` functions in `.cuh` headers. - [ ] **PR 11: First real CUDA kernel** โ€” `push_eta_stage` with a pyccel/CUDA parity test and an end-to-end run on the GPU. @@ -41,12 +41,13 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke - **No silent CPU fallback on the GPU.** A kernel without a CUDA version raises an error on the GPU backend. Falling back would mean copying data to the host and back at every call. - **Small steps.** Every PR keeps the CPU code path working and tested. -## Current state (PR 7) +## Current state (PR 8) | File | Content | |---|---| -| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; expands `Argument.get_cuda_args()` and takes `n_threads`), `Kernel` and `KernelCatalog` for backend selection and discovery | -| `src/struphy/utils/cuda_arguments.py` | `Argument` contract plus `CudaMarkerArguments`, `CudaDerhamArguments` and `CudaDomainArguments`; CUDA arrays are stored individually and returned in signature order by `get_cuda_args()` | +| `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily with the struphy headers on the include path; expands `Argument.get_cuda_args()` and takes `n_threads`), `Kernel` and `KernelCatalog` for backend selection and discovery | +| `src/struphy/utils/cuda_arguments.py` | `Argument` base class plus `CudaMarkerArguments`, `CudaDerhamArguments` and `CudaDomainArguments`; each references its device arrays and packs them once into its C struct, which `get_cuda_args()` returns | +| `src/struphy/kernel_arguments/pusher_args.cuh` | the C structs `MarkerArgs`, `DerhamArgs` and `DomainArgs` that CUDA kernels take in place of the pyccel argument classes | | `src/struphy/geometry/base.py` | `Domain.args_domain` is selected once at construction; CUDA domains use device arrays, while direct Pyccel geometry calls retain a host argument bundle | | `src/struphy/pic/base.py` | `Particles` arrays are allocated by the active `cunumpy` backend; `cuda_args_markers` lazily packages device arrays for CUDA kernels | | `src/struphy/feec/psydac_derham.py` | `Derham` can be created on the CuPy backend; spline-space data stays on the host, `cuda_args_derham` lazily holds device copies of degrees, knots and starts | @@ -58,7 +59,7 @@ Things we learned in the proof of concept: - `Particles.args_markers` remains the host bundle used by existing Pyccel calls. `Particles.cuda_args_markers` references the CuPy marker, validity, and boundary-condition arrays for CUDA calls. Likewise `Derham.args_derham` stays the host bundle and `Derham.cuda_args_derham` holds the CUDA one. - `Domain.args_domain` returns the argument type selected at domain construction; on CuPy it is a `CudaDomainArguments` object. Direct Pyccel methods on `Domain` use a separate internal host bundle. - `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Each argument is read with the size declared in the signature, so Python `int`/`float` arrive correctly in `int`/`double` parameters, but a wrongly typed scalar (e.g. an integer for a `double`, or a value that overflows an `int`) gives a wrong value **without an error**. Casting Python scalars in `CudaKernel` does not prevent this, so it is not done; see the follow-up in [Open questions](#open-questions). -- Flattening the argument classes at each call (joining their `values`) costs well under 1 ยตs, compared to about 70 ยตs for launching the kernel. +- Flattening the argument classes at each call costs well under 1 ยตs, compared to about 70 ยตs for launching the kernel. Since PR 8 each class is one struct, packed once when the argument object is created. - `struphy compile` compiles every `.py` file whose name contains `kernels`. Non-pyccel modules must not contain `kernels` in their name; `.cu` files are ignored by it. - On an H100, the demo kernel pushes 10โถ markers in about 0.13 ms per step. @@ -101,7 +102,7 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - `CudaKernel.from_file(path)` reads `_cuda.cu`; the kernel name is taken from the file name. The kernel is compiled lazily on first call. CuPy caches compiled kernels on disk (`~/.cupy/kernel_cache`), so the compile cost is paid once per machine. - Add `"**/*.cu"` and `"**/*.cuh"` to `[tool.setuptools.package-data]` in `pyproject.toml`. -- Later (PR 10), when the first shared header is needed: headers are found through NVRTC include paths (`cupy.RawModule(code=..., options=("-I",))`), so a `.cu` file can `#include "struphy/bsplines/bsplines_kernels.cuh"`. +- Shared headers are found through the NVRTC include path (done in PR 8: `CudaKernel` compiles with `-I`), so a `.cu` file can `#include "struphy/kernel_arguments/pusher_args.cuh"` or, later, `"struphy/bsplines/bsplines_kernels.cuh"`. ### PR 3: Kernel catalog @@ -157,13 +158,16 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - Field evaluation (`SplineFunction.__call__`, ...) still calls pyccel kernels with the coefficients, which are device arrays on CuPy. It needs CUDA evaluation kernels (PR 10+). - Tests: `feec/tests/test_derham_gpu.py`. Without a GPU, a strict host stand-in for CuPy (rejects host/device mixing, cannot run kernels; not part of the repository) was used to create `Derham` on the "CuPy" backend with 1, 2 and 4 MPI processes and compare it with NumPy. -### PR 8: Argument structs in shared headers +### PR 8: Argument structs in a shared header (complete) -Today every CUDA kernel repeats the full flat signature (26 parameters for markers and domain alone). CuPy does not check it, so adding a field to `MarkerArguments` would shift all following arguments of all CUDA kernels **without an error**. +Before, every CUDA kernel repeated the full flat signature (26 parameters for markers and domain alone). CuPy does not check it, so adding a field to `MarkerArguments` would have shifted all following arguments of all CUDA kernels **without an error**. -- Define `struct MarkerArgs { double* markers; bool* valid_mks; int n_markers; ... };` etc. in `kernel_arguments/pusher_args.cuh`, and pass one struct per argument class. -- To check first: how to pass a struct to a `cupy.RawKernel` (e.g. as a NumPy structured scalar with pointer fields). If this does not work well, keep the flat signature, but generate it from the Python class so that it is defined in one place. -- A test compares the struct layout (field names, types, order) with the Python class. +- `kernel_arguments/pusher_args.cuh` defines `struct MarkerArgs`, `DerhamArgs` and `DomainArgs`. A kernel takes them by value in the place of the pyccel argument classes, e.g. `void push_eta_linear(double dt, int stage, MarkerArgs args_markers, DomainArgs args_domain)`, and reads `args_markers.markers`, `args_markers.n_markers`, ... The member names are the attribute names of the pyccel classes; the only CUDA-specific member is `MarkerArgs.n_cols` (pyccel takes `markers.shape[1]`). +- Passing a struct: CuPy passes a NumPy scalar by value, copying `itemsize` bytes. Each `Argument` subclass lists its members as `fields = (("double*", "markers"), ...)`, from which `struct_dtype()` builds a NumPy structured dtype with `align=True` (C alignment and padding). Pointer members are `uint64` holding `array.data.ptr`. The struct (a `numpy.void`) is packed once, in the constructor, so a kernel call does no extra work. +- The struct holds device addresses: it is repacked when an argument object is deepcopied or unpickled (`__getstate__`/`__setstate__`), and owners that reallocate an array must rebuild their argument object, as before. +- Scalar members are checked when packing: a `float` for an `int` member raises `TypeError` and a value that does not fit raises `OverflowError`, instead of arriving truncated or wrapped around (NumPy assignment alone truncates `1.7` to `1`). +- Tests: the header is parsed and compared with `fields` (names, C types, order) without a GPU; on a GPU, an NVRTC-compiled kernel reports `sizeof` and the member offsets, which are compared with the dtype. The demo kernels use the structs. On the host, `offsetof`/`sizeof` from a C++ compiler agree with the dtypes (x86-64/arm64 lay these structs out like CUDA). +- Fixed along the way: the GPU-only tests still used the `values` attribute that was replaced by `get_cuda_args()`. ### PR 9: One folder per kernel @@ -210,7 +214,7 @@ Port the kernels in the order the target models need them, so that complete mode ## Open questions -- **Scalar types.** Scalars are not checked against the kernel signature (see [Current state](#current-state-pr-1)). `CudaKernel` could read the parameter types from the `extern "C"` signature once, when it is created, and cast each scalar to its declared type or raise if it does not fit (e.g. a Python `float` for an `int`, or an overflowing integer). This could go together with PR 8. +- **Scalar types.** Since PR 8 the members of the argument structs are checked when they are packed. Scalars passed directly to a kernel (e.g. `dt`, `stage`) are still not checked against the kernel signature (see [Current state](#current-state-pr-8)). `CudaKernel` could read the parameter types from the `extern "C"` signature once, when it is created, and cast each scalar to its declared type or raise if it does not fit (e.g. a Python `float` for an `int`, or an overflowing integer). - **Marker layout.** The markers array is row-major (`n_markers ร— n_cols`). With one thread per marker, the memory accesses are strided. This is fine for now (each thread reads a few neighbouring columns), but a column-major or struct-of-arrays layout may be faster later. This would affect the CPU code too, so it is out of scope here. - **MPI + GPUs.** One GPU per MPI rank (`cunumpy.set_device(rank % n_gpus)`), and GPU-aware MPI for the marker exchange, so markers do not go through the host. diff --git a/src/struphy/geometry/tests/test_domain.py b/src/struphy/geometry/tests/test_domain.py index ab4838638..092ac8b57 100644 --- a/src/struphy/geometry/tests/test_domain.py +++ b/src/struphy/geometry/tests/test_domain.py @@ -1135,26 +1135,18 @@ def test_cuda_args_domain(mapping): assert isinstance(args, CudaDomainArguments) assert domain.args_domain is args # built once - kind_map, params, degree, t1, t2, t3, ind1, ind2, ind3, cx, cy, cz = args.values - assert int(kind_map) == domain.kind_map + assert args.kind_map == domain.kind_map # no copies of arrays that already have the right dtype and layout - assert t1 is domain.T[0] and ind3 is domain.indN[2] + assert args.t1 is domain.T[0] and args.ind3 is domain.indN[2] + # the struct holds the device addresses of these arrays + (struct,) = args.get_cuda_args() + assert struct["kind_map"] == domain.kind_map host = domain._pyccel_args_domain - for dev, ref in ( - (params, host.params), - (degree, host.degree), - (t1, host.t1), - (t2, host.t2), - (t3, host.t3), - (ind1, host.ind1), - (ind2, host.ind2), - (ind3, host.ind3), - (cx, host.cx), - (cy, host.cy), - (cz, host.cz), - ): - assert (cunumpy.to_numpy(dev) == ref).all() + for name in ("params", "degree", "t1", "t2", "t3", "ind1", "ind2", "ind3", "cx", "cy", "cz"): + dev = getattr(args, name) + assert struct[name] == dev.data.ptr, name + assert (cunumpy.to_numpy(dev) == getattr(host, name)).all(), name @requires_cupy @@ -1172,7 +1164,8 @@ def test_domain_deepcopy_and_pickle_on_cupy(mapping): assert (other.args_domain.params == domain.args_domain.params).all() other_cuda = other.args_domain assert other_cuda is not cuda_args - assert other_cuda.values[3] is other.T[0] + assert other_cuda.t1 is other.T[0] + assert other_cuda.get_cuda_args()[0]["t1"] == other.T[0].data.ptr if __name__ == "__main__": diff --git a/src/struphy/kernel_arguments/pusher_args.cuh b/src/struphy/kernel_arguments/pusher_args.cuh new file mode 100644 index 000000000..02a32c5c2 --- /dev/null +++ b/src/struphy/kernel_arguments/pusher_args.cuh @@ -0,0 +1,57 @@ +// CUDA versions of the argument classes in pusher_args_kernels.py, one struct per class. +// +// CUDA kernels take these structs by value, in the place of the pyccel argument classes, e.g. +// +// #include "struphy/kernel_arguments/pusher_args.cuh" +// +// extern "C" __global__ +// void push_eta_stage(double dt, int stage, MarkerArgs args_markers, DomainArgs args_domain, ...) +// +// The structs are filled on the host by the classes in struphy/utils/cuda_arguments.py, whose `fields` list the +// members below in the same order and with the same types; test_cuda_argument_structs checks that they agree. +// The member names are the attribute names of the pyccel classes. Pointers are device pointers. +#pragma once + +// CUDA version of MarkerArguments (struphy.utils.cuda_arguments.CudaMarkerArguments). +struct MarkerArgs { + double* markers; // (n_markers, n_cols), row-major + bool* valid_mks; // (n_markers,), true for markers that are neither holes nor ghosts + int n_markers; + int n_cols; + int Np; + int vdim; + int weight_idx; + int first_diagnostics_idx; + int first_init_idx; + int first_shift_idx; + int residual_idx; + int first_free_idx; + int mu_idx; + long long* bc_type; // (3,) +}; + +// CUDA version of DerhamArguments (struphy.utils.cuda_arguments.CudaDerhamArguments). +// The scratch arrays of the pyccel class (bn1, ..., bd3) are local arrays in the kernels. +struct DerhamArgs { + long long* pn; // (3,) + double* tn1; + double* tn2; + double* tn3; + long long* starts; // (3,) +}; + +// CUDA version of DomainArguments (struphy.utils.cuda_arguments.CudaDomainArguments). +struct DomainArgs { + int kind_map; + double* params; + long long* degree; // (3,) + double* t1; + double* t2; + double* t3; + long long* ind1; // (number of mapping grid cells, degree + 1) + long long* ind2; + long long* ind3; + double* cx; // control points + double* cy; + double* cz; +}; diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index 276878275..2dfb911c5 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -6,7 +6,9 @@ import importlib import inspect +import re import sys +from pathlib import Path import cunumpy import numpy as np @@ -15,7 +17,13 @@ from struphy.geometry.domains import Cuboid from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments -from struphy.utils.cuda_arguments import CudaDomainArguments, CudaMarkerArguments +import struphy +from struphy.utils.cuda_arguments import ( + C_TYPES, + CudaDerhamArguments, + CudaDomainArguments, + CudaMarkerArguments, +) from struphy.utils.kernel_backends import CudaKernel, Kernel, KernelCatalog, is_cuda_backend requires_cupy = pytest.mark.skipif(not cunumpy.cupy_available(), reason="CuPy/GPU not available") @@ -51,49 +59,79 @@ def push_eta_linear( markers[ip, 2] += dt * markers[ip, 5] -# Arguments: (dt, stage, CudaMarkerArguments, CudaDomainArguments), see struphy.utils.cuda_arguments. -CUDA_ARGS = r""" - double dt, int stage, - double* markers, bool* valid_mks, int n_markers, int n_cols, - int Np, int vdim, int weight_idx, int first_diagnostics_idx, int first_init_idx, - int first_shift_idx, int residual_idx, int first_free_idx, int mu_idx, long long* bc_type, - int kind_map, double* params, long long* degree, - double* t1, double* t2, double* t3, - long long* ind1, long long* ind2, long long* ind3, - double* cx, double* cy, double* cz -""" +# Same arguments as the pyccel kernel; the argument classes are the structs of pusher_args.cuh. +PUSH_ETA_LINEAR_SRC = r""" +#include "struphy/kernel_arguments/pusher_args.cuh" -PUSH_ETA_LINEAR_SRC = f""" extern "C" __global__ -void push_eta_linear({CUDA_ARGS}) -{{ +void push_eta_linear(double dt, int stage, MarkerArgs args_markers, DomainArgs args_domain) +{ int ip = blockDim.x * blockIdx.x + threadIdx.x; // only do something if particle is valid (i.e. not a hole or ghost) - if (ip >= n_markers || !valid_mks[ip]) return; + if (ip >= args_markers.n_markers || !args_markers.valid_mks[ip]) return; - double* mk = markers + (long long)ip * n_cols; + double* mk = args_markers.markers + (long long)ip * args_markers.n_cols; mk[0] += dt * mk[3]; mk[1] += dt * mk[4]; mk[2] += dt * mk[5]; -}} +} """ -# writes the scalar arguments into the markers, to check that they arrive with the right types -WRITE_SCALARS_SRC = f""" +# writes scalar arguments and struct members into the markers, to check that they arrive with the right types +WRITE_SCALARS_SRC = r""" +#include "struphy/kernel_arguments/pusher_args.cuh" + extern "C" __global__ -void write_scalars({CUDA_ARGS}) -{{ +void write_scalars(double dt, int stage, MarkerArgs args_markers, DomainArgs args_domain) +{ int ip = blockDim.x * blockIdx.x + threadIdx.x; - if (ip >= n_markers) return; + if (ip >= args_markers.n_markers) return; - double* mk = markers + (long long)ip * n_cols; + double* mk = args_markers.markers + (long long)ip * args_markers.n_cols; mk[0] = dt; mk[1] = stage; - mk[2] = n_cols; - mk[3] = first_init_idx; - mk[4] = mu_idx; - mk[5] = kind_map; + mk[2] = args_markers.n_cols; + mk[3] = args_markers.first_init_idx; + mk[4] = args_markers.mu_idx; + mk[5] = args_domain.kind_map; + mk[6] = args_markers.bc_type[2]; + mk[7] = args_domain.t3[1]; +} +""" + +STRUCT_CLASSES = [CudaMarkerArguments, CudaDerhamArguments, CudaDomainArguments] +HEADER = Path(struphy.__file__).parent / "kernel_arguments" / "pusher_args.cuh" + + +def header_structs() -> dict: + """The structs of pusher_args.cuh, as {name: ((C type, member), ...)} in declaration order.""" + text = re.sub(r"//[^\n]*", "", HEADER.read_text()) + structs = {} + for name, body in re.findall(r"struct\s+(\w+)\s*\{(.*?)\};", text, flags=re.S): + members = re.findall(r"([A-Za-z_][\w ]*?\s*\**)\s*(\w+)\s*;", body) + structs[name] = tuple( + (" ".join(ctype.replace("*", " *").split()).replace(" *", "*"), m) for ctype, m in members + ) + return structs + + +def layout_kernel_source(cls) -> str: + """CUDA kernel writing sizeof and (offsetof, sizeof) of each member of the struct of cls into an int64 array.""" + # NVRTC has no standard headers (no offsetof), so offsets are taken from a local struct + lines = [f"{cls.struct_name} s;", f"out[0] = sizeof({cls.struct_name});"] + for i, (_, name) in enumerate(cls.fields): + lines.append(f"out[{2 * i + 1}] = (char*)&s.{name} - (char*)&s;") + lines.append(f"out[{2 * i + 2}] = sizeof(s.{name});") + body = "\n ".join(lines) + return f""" +#include "struphy/kernel_arguments/pusher_args.cuh" + +extern "C" __global__ +void struct_layout(long long* out) +{{ + if (blockDim.x * blockIdx.x + threadIdx.x != 0) return; + {body} }} """ @@ -212,30 +250,100 @@ def test_cuda_kernel_updates_device_array_in_place(kernel): assert args_markers.markers is markers assert markers.data.ptr == ptr - assert args_markers.values[0] is markers + assert args_markers.get_cuda_args()[0]["markers"] == ptr @requires_cupy def test_cuda_scalar_arguments(): - """Python scalars (not cast) and the flattened argument classes arrive in the CUDA kernel correctly and in order.""" + """Python scalars (not cast) and the struct members arrive in the CUDA kernel correctly and in order.""" write_scalars = CudaKernel(WRITE_SCALARS_SRC, "write_scalars") with cunumpy.use_backend("cupy"): args_markers, args_domain = make_arguments(10) + args_markers.bc_type[2] = 7 # read through the pointer in the struct write_scalars(0.25, 3, args_markers, args_domain, n_threads=10) - row = cunumpy.to_numpy(args_markers.markers)[0, :6] + row = cunumpy.to_numpy(args_markers.markers)[0, :8] first_pusher_idx, mu_idx = MARKER_INDICES[3], MARKER_INDICES[7] - assert np.array_equal(row, [0.25, 3, N_COLS, first_pusher_idx, mu_idx, Cuboid().kind_map]) + t3 = cunumpy.to_numpy(args_domain.t3)[1] + assert np.array_equal(row, [0.25, 3, N_COLS, first_pusher_idx, mu_idx, Cuboid().kind_map, 7, t3]) @requires_cupy def test_cuda_domain_arguments_reference_domain_arrays(): + """The struct holds the device addresses of the domain's own arrays.""" with cunumpy.use_backend("cupy"): domain = Cuboid() args = domain.args_domain assert isinstance(args, CudaDomainArguments) - assert len(args.values) == 12 - assert args.values[3] is domain.T[0] and args.values[8] is domain.indN[2] and args.values[9] is domain.cx + (struct,) = args.get_cuda_args() + assert struct["kind_map"] == domain.kind_map + assert struct["t1"] == domain.T[0].data.ptr and struct["ind3"] == domain.indN[2].data.ptr + assert struct["cx"] == domain.cx.data.ptr + + +def test_cuda_argument_structs_match_header(): + """The fields of the CUDA argument classes are the members of the structs in pusher_args.cuh (names, C types, order).""" + structs = header_structs() + assert sorted(structs) == sorted(cls.struct_name for cls in STRUCT_CLASSES) + for cls in STRUCT_CLASSES: + assert structs[cls.struct_name] == cls.fields, cls.struct_name + assert all(ctype in C_TYPES for ctype, _ in cls.fields) + + +def test_cuda_struct_members_are_pyccel_attributes(): + """1:1 correspondence: the struct members are named like the attributes of the pyccel argument classes. + + Only ``n_cols`` is CUDA-specific: pyccel kernels take it from ``markers.shape[1]``. + """ + text = (Path(struphy.__file__).parent / "kernel_arguments" / "pusher_args_kernels.py").read_text() + for cls in STRUCT_CLASSES: + for _, name in cls.fields: + assert f"self.{name} =" in text or name == "n_cols", f"{cls.struct_name}.{name}" + + +@requires_cupy +@pytest.mark.parametrize("cls", STRUCT_CLASSES, ids=lambda cls: cls.struct_name) +def test_cuda_struct_layout(cls): + """The NumPy dtype of each struct has the memory layout NVRTC gives the C struct (size, offsets, member sizes).""" + with cunumpy.use_backend("cupy"): + out = cunumpy.zeros(2 * len(cls.fields) + 1, dtype=np.int64) + CudaKernel(layout_kernel_source(cls), "struct_layout")(out, n_threads=1) + layout = cunumpy.to_numpy(out) + dtype = cls.struct_dtype() + expected = [dtype.itemsize] + for _, name in cls.fields: + field_dtype, offset = dtype.fields[name][:2] + expected += [offset, field_dtype.itemsize] + assert layout.tolist() == expected + + +@requires_cupy +def test_cuda_struct_follows_copies(): + """Deepcopies and unpickled copies repack the struct with the addresses of their own arrays.""" + import copy + import pickle + + with cunumpy.use_backend("cupy"): + args_markers, _ = make_arguments(10) + for other in (copy.deepcopy(args_markers), pickle.loads(pickle.dumps(args_markers))): + assert other.markers is not args_markers.markers + (struct,) = other.get_cuda_args() + assert struct["markers"] == other.markers.data.ptr and struct["bc_type"] == other.bc_type.data.ptr + assert struct["mu_idx"] == args_markers.mu_idx + + +@requires_cupy +def test_cuda_struct_scalars_are_checked(): + """Scalars are checked when the struct is packed: no silent truncation or wrap-around in the kernel.""" + import cupy as cp + + markers, valid_mks, bc_type = cp.zeros((10, N_COLS)), cp.ones(10, dtype=bool), cp.zeros(3, dtype=int) + indices = list(MARKER_INDICES) + with pytest.raises(TypeError): + CudaMarkerArguments(markers, valid_mks, 10.0, *indices, bc_type) + with pytest.raises(OverflowError, match="Np"): + CudaMarkerArguments(markers, valid_mks, 2**31, *indices, bc_type) + CudaMarkerArguments(markers, valid_mks, np.int64(10), *indices, bc_type) # NumPy integers are fine @requires_cupy diff --git a/src/struphy/utils/cuda_arguments.py b/src/struphy/utils/cuda_arguments.py index 70c54edaa..15c9da450 100644 --- a/src/struphy/utils/cuda_arguments.py +++ b/src/struphy/utils/cuda_arguments.py @@ -2,22 +2,94 @@ The compiled classes in :mod:`struphy.kernel_arguments.pusher_args_kernels` only accept NumPy arrays. The classes here take the same constructor arguments, keep references to **CuPy** arrays -(no copies, other arrays raise) and flatten them into the arguments of a ``cupy.RawKernel``. -The order of :attr:`values` is the corresponding part of the CUDA kernel signature. +(no copies, other arrays raise) and pack them into one C struct each, which CUDA kernels take by value. +The structs are declared in ``struphy/kernel_arguments/pusher_args.cuh``; :attr:`Argument.fields` lists +their members in declaration order, with the C types. """ -from abc import ABC, abstractmethod +import operator +from abc import ABC import numpy as np +# NumPy types of the struct members, by C type; pointers are passed as device addresses +C_TYPES = { + "int": np.int32, + "double": np.float64, + "double*": np.uint64, + "bool*": np.uint64, + "long long*": np.uint64, +} + class Argument(ABC): - """Base class for objects that provide arguments to a CUDA kernel.""" + """Base class for objects that are passed to a CUDA kernel as one C struct. + + A subclass names the struct (:attr:`struct_name`), lists its members (:attr:`fields`) and stores each member + as an attribute of the same name, then calls :meth:`_pack` at the end of its constructor. The struct is packed + once; kernel calls pass it as it is. + """ + + struct_name: str + """Name of the C struct in ``pusher_args.cuh``.""" + + fields: tuple[tuple[str, str], ...] + """``(C type, name)`` of each struct member, in declaration order.""" + + @classmethod + def struct_dtype(cls) -> np.dtype: + """NumPy dtype with the memory layout of the C struct (C alignment and padding). + + Returns + ------- + numpy.dtype + Structured dtype, one field per struct member. + """ + return np.dtype( + {"names": [name for _, name in cls.fields], "formats": [C_TYPES[ctype] for ctype, _ in cls.fields]}, + align=True, + ) + + def _pack(self): + """Pack the members into the struct; pointer members hold the device address of the array attribute. + + Scalars are checked against the C type of their member: a non-integer for an ``int`` or a value that + does not fit raises, instead of arriving in the kernel truncated or wrapped around. + """ + struct = np.zeros((), dtype=self.struct_dtype()) + for ctype, name in self.fields: + value = getattr(self, name) + if ctype.endswith("*"): + struct[name] = value.data.ptr + elif ctype == "int": + value = operator.index(value) # raises TypeError for floats + info = np.iinfo(C_TYPES[ctype]) + if not info.min <= value <= info.max: + raise OverflowError(f"{name} = {value} does not fit into a C {ctype}.") + struct[name] = value + else: + struct[name] = float(value) + self._struct = struct[()] + + def __getstate__(self): + # the struct holds device addresses, which are not valid for the arrays of a copy + state = self.__dict__.copy() + state.pop("_struct", None) + return state + + def __setstate__(self, state): + self.__dict__.update(state) + self._pack() - @abstractmethod def get_cuda_args(self) -> tuple: - """Return this object's arguments in CUDA kernel signature order.""" - raise NotImplementedError + """Return this object's arguments in CUDA kernel signature order: the packed struct. + + Returns + ------- + tuple + One ``numpy.void`` with the bytes of the C struct. + """ + return (self._struct,) def _cupy_array(name: str, arr, dtype): @@ -49,9 +121,7 @@ def _cupy_array(name: str, arr, dtype): class CudaMarkerArguments(Argument): """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.MarkerArguments`. - CUDA signature of :attr:`values`: ``double* markers, bool* valid_mks, int n_markers, int n_cols, int Np, - int vdim, int weight_idx, int first_diagnostics_idx, int first_init_idx, int first_shift_idx, - int residual_idx, int first_free_idx, int mu_idx, long long* bc_type`` + Passed to CUDA kernels as ``struct MarkerArgs``, see :attr:`fields`. Parameters ---------- @@ -101,11 +171,26 @@ class CudaMarkerArguments(Argument): n_markers : int Number of rows of ``markers``, e.g. for the number of CUDA threads. - - values : tuple - Flat CUDA kernel arguments, see the signature above. """ + struct_name = "MarkerArgs" + fields = ( + ("double*", "markers"), + ("bool*", "valid_mks"), + ("int", "n_markers"), + ("int", "n_cols"), + ("int", "Np"), + ("int", "vdim"), + ("int", "weight_idx"), + ("int", "first_diagnostics_idx"), + ("int", "first_init_idx"), + ("int", "first_shift_idx"), + ("int", "residual_idx"), + ("int", "first_free_idx"), + ("int", "mu_idx"), + ("long long*", "bc_type"), + ) + def __init__( self, markers, @@ -124,44 +209,25 @@ def __init__( self.markers = _cupy_array("markers", markers, np.float64) self.valid_mks = _cupy_array("valid_mks", valid_mks, np.bool_) self.n_markers = markers.shape[0] - self.n_cols = np.int32(markers.shape[1]) - self.Np = np.int32(Np) - self.vdim = np.int32(vdim) - self.weight_idx = np.int32(weight_idx) - self.first_diagnostics_idx = np.int32(first_diagnostics_idx) - self.first_pusher_idx = np.int32(first_pusher_idx) - self.first_shift_idx = np.int32(first_shift_idx) - self.residual_idx = np.int32(residual_idx) - self.first_free_idx = np.int32(first_free_idx) - self.mu_idx = np.int32(mu_idx) + self.n_cols = markers.shape[1] + self.Np = Np + self.vdim = vdim + self.weight_idx = weight_idx + self.first_diagnostics_idx = first_diagnostics_idx + self.first_init_idx = first_pusher_idx + self.first_shift_idx = first_shift_idx + self.residual_idx = residual_idx + self.first_free_idx = first_free_idx + self.mu_idx = mu_idx self.bc_type = _cupy_array("bc_type", bc_type, np.int64) - - def get_cuda_args(self) -> tuple: - return ( - self.markers, - self.valid_mks, - np.int32(self.n_markers), - self.n_cols, - self.Np, - self.vdim, - self.weight_idx, - self.first_diagnostics_idx, - self.first_pusher_idx, - self.first_shift_idx, - self.residual_idx, - self.first_free_idx, - self.mu_idx, - self.bc_type, - ) + self._pack() class CudaDerhamArguments(Argument): """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DerhamArguments`. - CUDA signature of :meth:`get_cuda_args`: ``long long* pn, double* tn1, double* tn2, double* tn3, long long* starts`` - - The scratch arrays of the pyccel class (``bn1``, ..., ``bd3``) are not part of it; CUDA kernels use - per-thread local arrays instead. + Passed to CUDA kernels as ``struct DerhamArgs``, see :attr:`fields`. The scratch arrays of the pyccel class + (``bn1``, ..., ``bd3``) are not part of it; CUDA kernels use per-thread local arrays instead. Parameters ---------- @@ -175,20 +241,26 @@ class CudaDerhamArguments(Argument): Start indices (current MPI process) of :class:`~struphy.feec.psydac_derham.Derham` (int64). """ + struct_name = "DerhamArgs" + fields = ( + ("long long*", "pn"), + ("double*", "tn1"), + ("double*", "tn2"), + ("double*", "tn3"), + ("long long*", "starts"), + ) + def __init__(self, pn, tn1, tn2, tn3, starts): self.pn = _cupy_array("pn", pn, np.int64) self.tn1, self.tn2, self.tn3 = (_cupy_array("tn", t, np.float64) for t in (tn1, tn2, tn3)) self.starts = _cupy_array("starts", starts, np.int64) - - def get_cuda_args(self) -> tuple: - return (self.pn, self.tn1, self.tn2, self.tn3, self.starts) + self._pack() class CudaDomainArguments(Argument): """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments`. - CUDA signature of :attr:`values`: ``int kind_map, double* params, long long* degree, double* t1, - double* t2, double* t3, long long* ind1, long long* ind2, long long* ind3, double* cx, double* cy, double* cz`` + Passed to CUDA kernels as ``struct DomainArgs``, see :attr:`fields`. Parameters ---------- @@ -209,33 +281,29 @@ class CudaDomainArguments(Argument): cx, cy, cz : cupy.ndarray[float] Spline coefficients (control points) of the mapping. - - Attributes - ---------- - values : tuple - Flat CUDA kernel arguments, see the signature above. """ + struct_name = "DomainArgs" + fields = ( + ("int", "kind_map"), + ("double*", "params"), + ("long long*", "degree"), + ("double*", "t1"), + ("double*", "t2"), + ("double*", "t3"), + ("long long*", "ind1"), + ("long long*", "ind2"), + ("long long*", "ind3"), + ("double*", "cx"), + ("double*", "cy"), + ("double*", "cz"), + ) + def __init__(self, kind_map: int, params, degree, t1, t2, t3, ind1, ind2, ind3, cx, cy, cz): - self.kind_map = np.int32(kind_map) + self.kind_map = kind_map self.params = _cupy_array("params", params, np.float64) self.degree = _cupy_array("degree", degree, np.int64) self.t1, self.t2, self.t3 = (_cupy_array("t", t, np.float64) for t in (t1, t2, t3)) self.ind1, self.ind2, self.ind3 = (_cupy_array("ind", ind, np.int64) for ind in (ind1, ind2, ind3)) self.cx, self.cy, self.cz = (_cupy_array("c", c, np.float64) for c in (cx, cy, cz)) - - def get_cuda_args(self) -> tuple: - return ( - self.kind_map, - self.params, - self.degree, - self.t1, - self.t2, - self.t3, - self.ind1, - self.ind2, - self.ind3, - self.cx, - self.cy, - self.cz, - ) + self._pack() diff --git a/src/struphy/utils/kernel_backends.py b/src/struphy/utils/kernel_backends.py index c6a8816ef..6c22641b4 100644 --- a/src/struphy/utils/kernel_backends.py +++ b/src/struphy/utils/kernel_backends.py @@ -6,6 +6,9 @@ Both kernels are called with the same arguments; the CUDA kernel takes the CUDA versions of the argument classes (:mod:`struphy.utils.cuda_arguments`), which reference arrays on the device, and the number of threads ``n_threads``. No arrays are converted or copied at call time. + +CUDA sources can include struphy headers relative to the parent folder of the ``struphy`` package, +e.g. ``#include "struphy/kernel_arguments/pusher_args.cuh"`` for the argument structs. """ import importlib @@ -17,6 +20,9 @@ from struphy.utils.cuda_arguments import Argument +INCLUDE_DIR = Path(__file__).resolve().parents[2] +"""NVRTC include path of the CUDA kernels: the folder that contains the ``struphy`` package.""" + def is_cuda_backend() -> bool: """Whether the active cunumpy backend is CuPy. @@ -32,8 +38,9 @@ def is_cuda_backend() -> bool: class CudaKernel: """A ``cupy.RawKernel``, the CUDA counterpart of a pyccel kernel. - The kernel is compiled on the first call. At each call, the CUDA argument classes are replaced by their - ``values``; all other arguments are passed to the ``cupy.RawKernel`` as they are. Arrays must be CuPy + The kernel is compiled on the first call, with :data:`INCLUDE_DIR` on the include path. At each call, the + CUDA argument classes are replaced by their structs; all other arguments are passed to the ``cupy.RawKernel`` + as they are. Arrays must be CuPy arrays (``cupy`` raises otherwise); they are never converted or copied. Python ``int`` and ``float`` arrive correctly in ``int`` and ``double`` parameters; scalars are not checked against the kernel signature. @@ -96,7 +103,7 @@ def __call__(self, *args, n_threads: int): if self._raw_kernel is None: import cupy as cp - self._raw_kernel = cp.RawKernel(self._source, self.name) + self._raw_kernel = cp.RawKernel(self._source, self.name, options=(f"-I{INCLUDE_DIR}",)) values = [] for arg in args: From b785d68f1da0972ce4e676f98fbe8c662e4d88bd Mon Sep 17 00:00:00 2001 From: Max Date: Thu, 1 Oct 2026 17:05:18 +0200 Subject: [PATCH 26/29] CUDA_STRATEGY: PR 7 depends on feectools#85 Co-Authored-By: Claude Opus 5.5 --- CUDA_STRATEGY.md | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index c6389d7e9..806a053d9 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -12,7 +12,7 @@ The work is split into small PRs that can be reviewed and merged one at a time. - [x] **PR 4: `Pusher` accepts `Kernel`** โ€” the kernel for the active backend is chosen once, when the pusher is created; a plain `PyccelKernel` is wrapped, so the propagators do not change (no behaviour change on CPU). - [x] **PR 5: `Domain` on the GPU** โ€” domain arguments are selected and stored at domain construction; CUDA arguments reference device arrays, and deepcopy/unpickling rebuilds the arguments from the copied or restored arrays. - [x] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend, and `Particles.cuda_args_markers` provides the CUDA argument bundle. -- [x] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. Needs the companion feectools changes (branch `cupy-derham`), which keep the spline-space construction data on the host. +- [x] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. Needs feectools on the CuPy backend (struphy-hub/feectools#85). - [ ] **PR 8: Shared CUDA headers for the argument classes** โ€” one `.cuh` per argument class instead of long flat kernel signatures. - [ ] **PR 9: One folder per kernel, starting with `pic/pushing`** โ€” pure refactor, no behaviour change. - [ ] **PR 10: Device versions of helper kernels** โ€” B-spline evaluation, mapping evaluation (per domain), small linear algebra, as `__device__` functions in `.cuh` headers. @@ -127,7 +127,7 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - The CUDA argument objects hold references. If an owner reallocates an array (today the markers are allocated once), it must rebuild its CUDA arguments at the same place, exactly like for the pyccel arguments. - First these classes must be creatable on the CuPy backend at all: - `Particles`: wrap Python lists in `xp.array` before reductions, and use host buffers for scalar MPI gathers (`pic/base.py`). - - `Derham`: feectools moved its host data (knots, grids, collocation matrices) to CuPy before calling pyccel kernels (see PR 7 below). + - `Derham`: feectools and struphy moved host data (knots, grids, collocation matrices) to CuPy before calling pyccel kernels (see PR 7 below). - `Domain`: deepcopy and unpickling on CuPy failed (see PR 5 below). ### PR 5: `Domain` on the GPU (complete) @@ -147,15 +147,15 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba ### PR 7: `Derham` on the GPU (complete) -- feectools (and the struphy code that builds `Derham`) followed `xp` everywhere, so on CuPy the knots, quadrature grids, collocation matrices and decomposition metadata became device arrays and then reached pyccel kernels, SciPy or MPI, which only take host arrays. The rule is now: **data that describes the spline spaces is host data on every backend**. Only the coefficients (`StencilVector` data) and stencil matrices live on the device. - - feectools, branch `cupy-derham` (separate PR, must be merged and released first): `core/bsplines.py` takes and returns NumPy arrays (CuPy inputs are copied to the host once), and so do the 1D interpolation/histopolation matrices and their factorizations (`fem/splines.py`, `BandedSolver`). Decomposition metadata (`ddm/cart.py`, `ddm/partition.py`), MPI counts and shapes in `linalg/kron.py`, and `compute_diag_len` in `linalg/stencil.py` are NumPy too. `StencilMatrix._tocoo_no_pads` had a duplicated kernel call on the CuPy data (merge leftover) and now builds the SciPy matrix from one host copy. - - struphy: the projection and quadrature grids of `Derham` (`get_pts_and_wts`, ...) and `spline_types_pyccel` are NumPy. `domain_array`, `index_array(_N/_D)` and `neighbours` are gathered with NumPy MPI buffers and then converted with `xp.asarray`, so they are device arrays on CuPy like `Particles.domain_array`. +- feectools and the struphy code that builds `Derham` followed `xp` everywhere. So on CuPy, the knots, quadrature grids and decomposition metadata became device arrays and then reached pyccel kernels, SciPy or MPI, which only take host arrays. + - feectools: struphy-hub/feectools#85, the first of the feectools CUDA PRs, makes feectools run on the CuPy backend. It must be merged, and the submodule or the feectools version bumped, before `Derham` can be created on CuPy. + - struphy: data that describes the spline spaces is host data on every backend. Only the coefficients (`StencilVector` data) and stencil matrices live on the device. The projection and quadrature grids of `Derham` (`get_pts_and_wts`, ...) and `spline_types_pyccel` are NumPy. `domain_array`, `index_array(_N/_D)` and `neighbours` are gathered with NumPy MPI buffers and then converted with `xp.asarray`, so they are device arrays on CuPy like `Particles.domain_array`. - `Derham.args_derham` is built from the host knots, degrees and starts on both backends. `Derham.cuda_args_derham` lazily builds `CudaDerhamArguments` with one device copy of these small arrays. On NumPy it raises (host arrays are never copied to the device). The pyccel scratch arrays (`bn1`, ..., `bd3`) are not part of it; they become per-thread local arrays in CUDA (PR 10). - Not supported on CuPy yet: - Local projectors (`DerhamOptions.local_projectors=True`) raise `NotImplementedError` when the `Derham` is created. `CommutingProjectorLocal` builds its data with `xp` and calls pyccel kernels on it, like `Derham` did. - Polar splines need a spline mapping, which cannot be created on CuPy yet (see PR 5). - Field evaluation (`SplineFunction.__call__`, ...) still calls pyccel kernels with the coefficients, which are device arrays on CuPy. It needs CUDA evaluation kernels (PR 10+). -- Tests: `feec/tests/test_derham_gpu.py`. Without a GPU, a strict host stand-in for CuPy (rejects host/device mixing, cannot run kernels; not part of the repository) was used to create `Derham` on the "CuPy" backend with 1, 2 and 4 MPI processes and compare it with NumPy. +- Tests: `feec/tests/test_derham_gpu.py`. Without a GPU, a strict host stand-in for CuPy (rejects host/device mixing, cannot run kernels; not part of the repository) was used, on top of struphy-hub/feectools#85. With it, a `Derham` created on the "CuPy" backend matches the NumPy one on 1, 2 and 4 MPI processes. ### PR 8: Argument structs in shared headers From 0c4cddc5119f8839c18aba325eb313ce5f79758b Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Thu, 1 Oct 2026 17:22:34 +0200 Subject: [PATCH 27/29] Set self._args_markers in the __init__ --- CUDA_STRATEGY.md | 16 +++-- .../pic/accumulation/particles_to_grid.py | 4 +- src/struphy/pic/base.py | 66 +++++++++---------- 3 files changed, 41 insertions(+), 45 deletions(-) diff --git a/CUDA_STRATEGY.md b/CUDA_STRATEGY.md index 0cb4bf421..fbfe99639 100644 --- a/CUDA_STRATEGY.md +++ b/CUDA_STRATEGY.md @@ -11,7 +11,7 @@ The work is split into small PRs that can be reviewed and merged one at a time. - [x] **PR 3: Kernel catalog** โ€” kernels are defined once in the `__init__.py` of the folder that contains them; a missing CUDA kernel raises an error on the GPU backend. - [x] **PR 4: `Pusher` accepts `Kernel`** โ€” the kernel for the active backend is chosen once, when the pusher is created; a plain `PyccelKernel` is wrapped, so the propagators do not change (no behaviour change on CPU). - [x] **PR 5: `Domain` on the GPU** โ€” domain arguments are selected and stored at domain construction; CUDA arguments reference device arrays, and deepcopy/unpickling rebuilds the arguments from the copied or restored arrays. -- [x] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend, and `Particles.cuda_args_markers` provides the CUDA argument bundle. +- [x] **PR 6: `Particles` on the GPU** โ€” `Particles` can be created on the CuPy backend, and `Particles.args_markers` is selected as the CUDA or Pyccel argument bundle at construction. - [ ] **PR 7: `Derham` on the GPU** โ€” `Derham` can be created on the CuPy backend, plus `Derham.cuda_args_derham`. - [ ] **PR 8: Shared CUDA headers for the argument classes** โ€” one `.cuh` per argument class instead of long flat kernel signatures. - [ ] **PR 9: One folder per kernel, starting with `pic/pushing`** โ€” pure refactor, no behaviour change. @@ -48,13 +48,13 @@ CUDA kernels can be added one by one. If the code runs on the GPU and needs a ke | `src/struphy/utils/kernel_backends.py` | `is_cuda_backend()`, `CudaKernel` (wraps a `cupy.RawKernel`, compiled lazily; expands `Argument.get_cuda_args()` and takes `n_threads`), `Kernel` and `KernelCatalog` for backend selection and discovery | | `src/struphy/utils/cuda_arguments.py` | `Argument` contract plus `CudaMarkerArguments` and `CudaDomainArguments`; CUDA arrays are stored individually and returned in signature order by `get_cuda_args()` | | `src/struphy/geometry/base.py` | `Domain.args_domain` is selected once at construction; CUDA domains use device arrays, while direct Pyccel geometry calls retain a host argument bundle | -| `src/struphy/pic/base.py` | `Particles` arrays are allocated by the active `cunumpy` backend; `cuda_args_markers` lazily packages device arrays for CUDA kernels | +| `src/struphy/pic/base.py` | `Particles` arrays and `args_markers` use the backend selected at construction; direct Pyccel methods retain a private host bundle | | `src/struphy/pic/tests/test_kernel_backends.py` | the demo kernel pair `push_eta_linear` (pyccel function compiled with `epyccel` at test time, CUDA source string) and tests on both backends | Things we learned in the proof of concept: - The pyccel-compiled argument classes (`MarkerArguments`, `DomainArguments`, `DerhamArguments`) hold references to their owner's arrays, but only accept **NumPy** arrays. Hence the CUDA counterparts in `cuda_arguments.py`. -- `Particles.args_markers` remains the host bundle used by existing Pyccel calls. `Particles.cuda_args_markers` references the CuPy marker, validity, and boundary-condition arrays for CUDA calls. `Derham` still needs its CUDA creation work (PR 7). +- `Particles.args_markers` is the CUDA or Pyccel bundle selected at construction. Existing direct Pyccel methods use a private host bundle. `Derham` still needs its CUDA creation work (PR 7). - `Domain.args_domain` returns the argument type selected at domain construction; on CuPy it is a `CudaDomainArguments` object. Direct Pyccel methods on `Domain` use a separate internal host bundle. - `cupy.RawKernel` accepts only device arrays (host arrays raise) and does **not** check the kernel signature. Each argument is read with the size declared in the signature, so Python `int`/`float` arrive correctly in `int`/`double` parameters, but a wrongly typed scalar (e.g. an integer for a `double`, or a value that overflows an `int`) gives a wrong value **without an error**. Casting Python scalars in `CudaKernel` does not prevent this, so it is not done; see the follow-up in [Open questions](#open-questions). - Flattening the argument classes at each call (joining their `values`) costs well under 1 ยตs, compared to about 70 ยตs for launching the kernel. @@ -118,11 +118,11 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba - `Pusher` takes a `Kernel` or, as before, a `PyccelKernel` (wrapped into a `Kernel` without CUDA version). It calls `get_kernel()` once in its constructor, so on the CuPy backend a pusher whose kernel has no CUDA version fails when it is created, not in the time loop. Since no pusher kernel has a CUDA version yet, this is the case for all pushers. - The propagators do not change. They switch to catalog lookups once the kernels are split into folders (PR 9). -- Still to do (PR 11): on the GPU, the pusher must pass `particles.cuda_args_markers`, device arrays in `args_kernel`, and the domain's already backend-selected `domain.args_domain`. This choice is made once, together with the kernel, so that a CUDA kernel is never called with pyccel arguments or vice versa. +- The pusher already passes the backend-selected `particles.args_markers` and `domain.args_domain`. PR 11 still needs to provide device arrays in `args_kernel` and the CUDA implementation of the first real pusher kernel. ### PR 5โ€“7: Owners build their CUDA arguments -- `Particles`, `Domain` and `Derham` own the arrays, so they build CUDA argument objects from their own `xp` arrays. `Particles.cuda_args_markers` is separate from the host `args_markers`; `Domain.args_domain` is selected once at construction. The pyccel argument bundles remain available for existing direct Pyccel calls. +- `Particles`, `Domain` and `Derham` own the arrays, so they build CUDA argument objects from their own `xp` arrays. `Particles.args_markers` and `Domain.args_domain` are selected once at construction; private host bundles remain available for existing direct Pyccel calls. - The CUDA argument objects hold references. If an owner reallocates an array (today the markers are allocated once), it must rebuild its CUDA arguments at the same place, exactly like for the pyccel arguments. - First these classes must be creatable on the CuPy backend at all: - `Particles`: wrap Python lists in `xp.array` before reductions, and use host buffers for scalar MPI gathers (`pic/base.py`). @@ -140,7 +140,7 @@ kernel = catalog["push_eta_stage"] # Kernel: pyccel or CUDA depending on the ba ### PR 6: `Particles` on the GPU (complete) - Particle arrays, validity masks, and boundary-condition codes are allocated through `cunumpy`, so they live on CuPy when the CuPy backend is active. -- `cuda_args_markers` lazily builds `CudaMarkerArguments` from those device arrays. The existing `args_markers` remains the host argument bundle for current Pyccel kernel calls. +- `args_markers` is built as `CudaMarkerArguments` from device arrays on CuPy, or as `MarkerArguments` on NumPy. A private host bundle remains for direct Pyccel calls. - Domain decomposition now wraps the Python `nprocs` list with `xp.array` before calling `xp.prod`. Scalar MPI gathers use small NumPy buffers and copy the results back to the active array backend, avoiding unsupported CuPy buffers in MPI calls. - Full GPU particle pushing still depends on CUDA versions of the required kernels and on PR 7's `Derham` support. @@ -175,10 +175,12 @@ The pusher kernels call helpers from other pyccel modules: B-spline evaluation ( - Parity test: same markers, both backends, results agree to round-off (`rtol ~ 1e-13`). - End-to-end: run a propagator that only needs this kernel with `ARRAY_BACKEND=cupy`, and check that no host/device transfers happen inside the time loop (e.g. with `nsys` or by counting CuPy memory copies). -### PR 12+: Port kernels one by one +### PR 12+: Port kernels and particle boundary handling For each kernel: add `_cuda.cu`, a parity test is added automatically by the catalog (every kernel with a CUDA version is run on both backends with the same inputs), and the kernel is removed from the "missing" list. +- Port particle kinetic boundary handling to CUDA, including the `reflect` helper currently called from `Particles.apply_kinetic_bc`. The CUDA path must use `Particles.args_markers` and `Domain.args_domain`; remove the temporary direct-Pyccel use of `Domain._pyccel_args_domain` from this path once reflection runs in a CUDA kernel. + ## Porting order Port the kernels in the order the target models need them, so that complete models can run on the GPU as early as possible. Proposed: diff --git a/src/struphy/pic/accumulation/particles_to_grid.py b/src/struphy/pic/accumulation/particles_to_grid.py index ddd638c73..358fe9dea 100644 --- a/src/struphy/pic/accumulation/particles_to_grid.py +++ b/src/struphy/pic/accumulation/particles_to_grid.py @@ -227,7 +227,7 @@ def _accumulate(self, *optional_args): # accumulate into matrix (and vector) with markers with ProfileManager.profile_region("kernel: " + self.kernel.name): self.kernel( - self.particles.args_markers, + self.particles._pyccel_args_markers, self.derham.args_derham, self.args_domain, *self._args_data, @@ -555,7 +555,7 @@ def _accumulate(self, *optional_args): # accumulate into matrix (and vector) with markers with ProfileManager.profile_region("kernel: " + self.kernel.name): self.kernel( - self.particles.args_markers, + self.particles._pyccel_args_markers, self.derham.args_derham, self.args_domain, *self._args_data, diff --git a/src/struphy/pic/base.py b/src/struphy/pic/base.py index abbfdb7ed..ed70019d7 100644 --- a/src/struphy/pic/base.py +++ b/src/struphy/pic/base.py @@ -219,7 +219,7 @@ def __init__( equation_params: dict = None, dry_run: bool = False, ): - + self._args_backend = xp.get_backend() self._clone_config = clone_config if self.clone_config is None: self._mpi_comm = comm_world @@ -937,32 +937,10 @@ def f_jacobian_coords(self, new): self.markers[~self.holes, self.f_jacobian_coords_index] = new @property - def args_markers(self) -> MarkerArguments: - """Collection of mandatory arguments for pusher kernels.""" + def args_markers(self) -> MarkerArguments | CudaMarkerArguments: + """Arguments for marker kernels, selected when this particle object is created.""" return self._args_markers - @property - def cuda_args_markers(self) -> CudaMarkerArguments: - """CUDA arguments for particle kernels, referencing the marker owner's device arrays.""" - if not hasattr(self, "_markers"): - raise AttributeError("CudaMarkerArguments are not available before the marker array is allocated.") - if getattr(self, "_cuda_args_markers", None) is None: - self._cuda_args_markers = CudaMarkerArguments( - self.markers, - self.valid_mks, - self.Np, - self.vdim, - self.index["weights"], - self.first_diagnostics_idx, - self.first_pusher_idx, - self.first_shift_idx, - self.residual_idx, - self.first_free_idx, - self.mu_idx, - self._bc_type, - ) - return self._cuda_args_markers - # ------------------------------------------- # Initial condition and background -> weights # ------------------------------------------- @@ -1916,7 +1894,7 @@ def apply_kinetic_bc(self, newton=False): # flip velocity reflect( self.markers, - self.domain.args_domain, + self.domain._pyccel_args_domain, outside_inds_per_axis[axis], axis, ) @@ -2151,8 +2129,8 @@ def eval_velocity( func( alpha=xp.array((0.0, 0.0, 0.0)), output_indices=xp.array((first_free_idx, first_free_idx + 1, first_free_idx + 2), dtype=int), - args_markers=self.args_markers, - args_domain=self.domain.args_domain, + args_markers=self._pyccel_args_markers, + args_domain=self.domain._pyccel_args_domain, boxes=self.sorting_boxes.boxes, neighbours=self.sorting_boxes.neighbours, holes=self.holes, @@ -2262,8 +2240,8 @@ def eval_div_viscosity( func( alpha=xp.array((0.0, 0.0, 0.0)), output_indices=xp.array((first_free_idx, first_free_idx + 1, first_free_idx + 2), dtype=int), - args_markers=self.args_markers, - args_domain=self.domain.args_domain, + args_markers=self._pyccel_args_markers, + args_domain=self.domain._pyccel_args_domain, boxes=self.sorting_boxes.boxes, neighbours=self.sorting_boxes.neighbours, holes=self.holes, @@ -2281,8 +2259,8 @@ def eval_div_viscosity( func( alpha=xp.array((0.0, 0.0, 0.0)), output_indices=xp.arange(first_free_idx + 3, first_free_idx + 12, dtype=int), - args_markers=self.args_markers, - args_domain=self.domain.args_domain, + args_markers=self._pyccel_args_markers, + args_domain=self.domain._pyccel_args_domain, boxes=self.sorting_boxes.boxes, neighbours=self.sorting_boxes.neighbours, holes=self.holes, @@ -2573,7 +2551,7 @@ def _allocate_marker_array(self, dry_run: bool = False): self._lost_markers = xp.zeros((int(self.n_rows * 0.5), 10), dtype=float) # arguments for kernels - self._args_markers = MarkerArguments( + self._pyccel_args_markers = MarkerArguments( _to_numpy_for_kernel(self.markers), _to_numpy_for_kernel(self.valid_mks), _to_numpy_for_kernel(self.Np), @@ -2587,7 +2565,23 @@ def _allocate_marker_array(self, dry_run: bool = False): _to_numpy_for_kernel(self.mu_idx), _to_numpy_for_kernel(self._bc_type), ) - self._cuda_args_markers = None + if self._args_backend == "cupy": + self._args_markers = CudaMarkerArguments( + self.markers, + self.valid_mks, + self.Np, + self.vdim, + self.index["weights"], + self.first_diagnostics_idx, + self.first_pusher_idx, + self.first_shift_idx, + self.residual_idx, + self.first_free_idx, + self.mu_idx, + self._bc_type, + ) + else: + self._args_markers = self._pyccel_args_markers def _initialize_sorting_boxes(self): """Initializes the sorting boxes. @@ -4282,7 +4276,7 @@ def _eval_sph( func = PyccelKernel(box_based_evaluation_meshgrid) func( - self.args_markers, + self._pyccel_args_markers, eta1, eta2, eta3, @@ -4309,7 +4303,7 @@ def _eval_sph( elif len(_shp) == 3: func = PyccelKernel(naive_evaluation_meshgrid) func( - self.args_markers, + self._pyccel_args_markers, eta1, eta2, eta3, From ac7eb6bfb766549cfde121722ff7ac0b9e7e229f Mon Sep 17 00:00:00 2001 From: Max Lindqvist Date: Fri, 2 Oct 2026 14:45:07 +0200 Subject: [PATCH 28/29] Remove cuda_args_derham --- src/struphy/feec/psydac_derham.py | 26 ++++++++----------- src/struphy/feec/tests/test_derham_gpu.py | 25 ++++++++++++------ .../pic/accumulation/particles_to_grid.py | 4 +-- 3 files changed, 30 insertions(+), 25 deletions(-) diff --git a/src/struphy/feec/psydac_derham.py b/src/struphy/feec/psydac_derham.py index 10ba7c2e6..e16af24c5 100644 --- a/src/struphy/feec/psydac_derham.py +++ b/src/struphy/feec/psydac_derham.py @@ -899,12 +899,19 @@ def __init__( self._neighbours = self._get_neighbours() # collect arguments for kernels (the knots of feectools are host arrays on every array backend) - self._args_derham = DerhamArguments( + self._pyccel_args_derham = DerhamArguments( np.array(self.degree), *self.V0fem.knots, np.array(self.V0.starts), ) - self._cuda_args_derham = None + if is_cuda_backend(): + self._args_derham = CudaDerhamArguments( + xp.asarray(self._pyccel_args_derham.pn), + *(xp.asarray(t) for t in self.V0fem.knots), + xp.asarray(self._pyccel_args_derham.starts), + ) + else: + self._args_derham = self._pyccel_args_derham logger.debug("\nDERHAM:") logger.debug(f"{'number of elements:'.ljust(25)} {num_elements}") @@ -1475,21 +1482,10 @@ def div_bcfree(self): return self._div_bcfree @property - def args_derham(self): - """Collection of mandatory arguments for pusher kernels.""" + def args_derham(self) -> DerhamArguments | CudaDerhamArguments: + """Mandatory pusher kernel arguments for the backend used at initialization.""" return self._args_derham - @property - def cuda_args_derham(self) -> CudaDerhamArguments: - """CUDA arguments for particle kernels; holds device copies of the spline degrees, knots and start indices.""" - if self._cuda_args_derham is None: - self._cuda_args_derham = CudaDerhamArguments( - xp.asarray(self.args_derham.pn), - *(xp.asarray(t) for t in self.V0fem.knots), - xp.asarray(self.args_derham.starts), - ) - return self._cuda_args_derham - # -------------------------- # methods: # -------------------------- diff --git a/src/struphy/feec/tests/test_derham_gpu.py b/src/struphy/feec/tests/test_derham_gpu.py index b802a4a88..93a165b04 100644 --- a/src/struphy/feec/tests/test_derham_gpu.py +++ b/src/struphy/feec/tests/test_derham_gpu.py @@ -24,18 +24,24 @@ def make_derham(bcs=(None, None, None), local_projectors=False): return Derham(TensorProductGrid(num_elements=(8, 6, 4)), options, comm=MPI.COMM_WORLD) -def test_cuda_args_derham_needs_device_arrays(): - """On the NumPy backend the Derham arrays are host arrays, which are never copied to the device.""" +def test_args_derham_on_numpy(): + """On the NumPy backend the general arguments are the Pyccel host arguments.""" + from struphy.kernel_arguments.pusher_args_kernels import DerhamArguments + with cunumpy.use_backend("numpy"): derham = make_derham() - with pytest.raises(TypeError, match="CuPy array"): - derham.cuda_args_derham + assert isinstance(derham.args_derham, DerhamArguments) + assert derham.args_derham is derham._pyccel_args_derham + for name in ("pn", "tn1", "tn2", "tn3", "starts"): + assert isinstance(getattr(derham.args_derham, name), np.ndarray), name @requires_cupy @pytest.mark.parametrize("bcs", [(None, None, None), (("dirichlet", "free"), None, ("free", "dirichlet"))]) def test_derham_on_cupy(bcs): """Same decomposition and kernel arguments on both backends, and CUDA arguments holding device copies.""" + from struphy.utils.cuda_arguments import CudaDerhamArguments + derhams = {} for backend in ("numpy", "cupy"): with cunumpy.use_backend(backend): @@ -48,11 +54,14 @@ def test_derham_on_cupy(bcs): # the pyccel arguments are host arrays on both backends (feectools knots are host arrays) for name in ("pn", "tn1", "tn2", "tn3", "starts"): - assert np.array_equal(getattr(device.args_derham, name), getattr(host.args_derham, name)), name + assert isinstance(getattr(device._pyccel_args_derham, name), np.ndarray), name + assert np.array_equal(getattr(device._pyccel_args_derham, name), getattr(host._pyccel_args_derham, name)), name - with cunumpy.use_backend("cupy"): - args = device.cuda_args_derham - assert device.cuda_args_derham is args # built once + # Arguments keep the construction backend even when accessed from the NumPy backend. + with cunumpy.use_backend("numpy"): + args = device.args_derham + assert isinstance(args, CudaDerhamArguments) + assert device.args_derham is args expected = (host.args_derham.pn, *host.V0fem.knots, host.args_derham.starts) for name, value in zip(("pn", "tn1", "tn2", "tn3", "starts"), expected): assert cunumpy.is_gpu(getattr(args, name)), name diff --git a/src/struphy/pic/accumulation/particles_to_grid.py b/src/struphy/pic/accumulation/particles_to_grid.py index 358fe9dea..571d9be9a 100644 --- a/src/struphy/pic/accumulation/particles_to_grid.py +++ b/src/struphy/pic/accumulation/particles_to_grid.py @@ -228,7 +228,7 @@ def _accumulate(self, *optional_args): with ProfileManager.profile_region("kernel: " + self.kernel.name): self.kernel( self.particles._pyccel_args_markers, - self.derham.args_derham, + self.derham._pyccel_args_derham, self.args_domain, *self._args_data, *optional_args, @@ -556,7 +556,7 @@ def _accumulate(self, *optional_args): with ProfileManager.profile_region("kernel: " + self.kernel.name): self.kernel( self.particles._pyccel_args_markers, - self.derham.args_derham, + self.derham._pyccel_args_derham, self.args_domain, *self._args_data, *optional_args, From ea450e429327f7a9507e11a851dc83a138f33767 Mon Sep 17 00:00:00 2001 From: Max Date: Fri, 2 Oct 2026 16:35:00 +0200 Subject: [PATCH 29/29] Drop duplicate CudaDerhamArguments left by the devel merge; fix import order --- src/struphy/pic/tests/test_kernel_backends.py | 2 +- src/struphy/utils/cuda_arguments.py | 29 ------------------- 2 files changed, 1 insertion(+), 30 deletions(-) diff --git a/src/struphy/pic/tests/test_kernel_backends.py b/src/struphy/pic/tests/test_kernel_backends.py index 2dfb911c5..215af37d4 100644 --- a/src/struphy/pic/tests/test_kernel_backends.py +++ b/src/struphy/pic/tests/test_kernel_backends.py @@ -15,9 +15,9 @@ import pytest from cunumpy import PyccelKernel +import struphy from struphy.geometry.domains import Cuboid from struphy.kernel_arguments.pusher_args_kernels import DomainArguments, MarkerArguments -import struphy from struphy.utils.cuda_arguments import ( C_TYPES, CudaDerhamArguments, diff --git a/src/struphy/utils/cuda_arguments.py b/src/struphy/utils/cuda_arguments.py index aa49d7229..15c9da450 100644 --- a/src/struphy/utils/cuda_arguments.py +++ b/src/struphy/utils/cuda_arguments.py @@ -257,35 +257,6 @@ def __init__(self, pn, tn1, tn2, tn3, starts): self._pack() -class CudaDerhamArguments(Argument): - """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DerhamArguments`. - - CUDA signature of :meth:`get_cuda_args`: ``long long* pn, double* tn1, double* tn2, double* tn3, long long* starts`` - - The scratch arrays of the pyccel class (``bn1``, ..., ``bd3``) are not part of it; CUDA kernels use - per-thread local arrays instead. - - Parameters - ---------- - pn : cupy.ndarray[int] - Spline degrees of :class:`~struphy.feec.psydac_derham.Derham` (int64). - - tn1, tn2, tn3 : cupy.ndarray[float] - Knot sequences of :class:`~struphy.feec.psydac_derham.Derham`. - - starts : cupy.ndarray[int] - Start indices (current MPI process) of :class:`~struphy.feec.psydac_derham.Derham` (int64). - """ - - def __init__(self, pn, tn1, tn2, tn3, starts): - self.pn = _cupy_array("pn", pn, np.int64) - self.tn1, self.tn2, self.tn3 = (_cupy_array("tn", t, np.float64) for t in (tn1, tn2, tn3)) - self.starts = _cupy_array("starts", starts, np.int64) - - def get_cuda_args(self) -> tuple: - return (self.pn, self.tn1, self.tn2, self.tn3, self.starts) - - class CudaDomainArguments(Argument): """CUDA version of :class:`~struphy.kernel_arguments.pusher_args_kernels.DomainArguments`.