From 1b5645a3a64c3af551a41f4adb71da63bd6a7283 Mon Sep 17 00:00:00 2001 From: Max Date: Sat, 3 Oct 2026 15:09:47 +0200 Subject: [PATCH 1/2] Refactor: Moved helper methods to modules --- CHANGELOG.md | 5 + README.md | 64 +- docs/source/api.md | 313 +++++---- docs/source/best-practices.md | 4 +- docs/source/examples/particle-pusher.md | 14 +- docs/source/examples/particle_recipes.py | 10 +- docs/source/examples/portable-script.md | 8 +- docs/source/guides/backends.md | 4 +- docs/source/guides/data-movement.md | 10 +- docs/source/guides/gpu-devices.md | 16 +- docs/source/guides/mpi.md | 32 +- docs/source/guides/particle-codes.md | 16 +- docs/source/guides/portable-code.md | 8 +- docs/source/guides/profiling.md | 8 +- docs/source/guides/solvers.md | 20 +- docs/source/installation.md | 6 +- docs/source/kernels/accumulation.md | 10 +- docs/source/kernels/arguments.md | 30 +- docs/source/kernels/cuda-kernel.md | 28 +- docs/source/kernels/debugging.md | 10 +- docs/source/kernels/dispatch.md | 32 +- docs/source/kernels/overview.md | 2 +- docs/source/kernels/pyccel-kernel.md | 14 +- docs/source/kernels/testing.md | 26 +- docs/source/pyodide.md | 2 +- docs/source/quickstart.md | 4 +- docs/source/troubleshooting.md | 10 +- pyproject.toml | 2 +- src/cunumpy/LLM_GUIDE.md | 165 ++--- src/cunumpy/__init__.py | 211 ++---- src/cunumpy/__init__.pyi | 86 +-- src/cunumpy/_fake_cupy.py | 6 +- src/cunumpy/algorithms.py | 30 + src/cunumpy/cuda/__init__.py | 86 +++ src/cunumpy/cuda/include/cunumpy/morton.cuh | 2 +- src/cunumpy/cuda/include/cunumpy/random.cuh | 2 +- src/cunumpy/cuda_kernel.py | 24 +- src/cunumpy/dispatch.py | 56 +- src/cunumpy/emulation.py | 4 +- src/cunumpy/fusion.py | 6 +- src/cunumpy/kernel.py | 18 +- src/cunumpy/kernel_testing.py | 634 +++++++++++++++++ src/cunumpy/kernels.py | 54 ++ src/cunumpy/memory.py | 22 + src/cunumpy/morton.py | 6 +- src/cunumpy/mpi.py | 34 + src/cunumpy/petsc.py | 4 +- src/cunumpy/philox.py | 4 +- src/cunumpy/profiling.py | 32 + src/cunumpy/random_streams.py | 6 +- src/cunumpy/rng.py | 39 ++ src/cunumpy/staging.py | 2 +- src/cunumpy/testing.py | 641 +----------------- src/cunumpy/transfers.py | 16 +- src/cunumpy/xp.py | 20 +- tests/portable/test_numpy_runtime.py | 8 +- tests/unit/pyccel_kernels.py | 2 +- tests/unit/test_array_api_compat_backend.py | 2 +- tests/unit/test_cuda_kernel.py | 82 +-- tests/unit/test_cunumpy.py | 57 +- tests/unit/test_device_binding.py | 20 +- tests/unit/test_emulation.py | 6 +- tests/unit/test_fusion.py | 16 +- tests/unit/test_kernel_dispatch.py | 10 +- tests/unit/test_kernel_dispatch_arrays.py | 50 +- ...test_testing.py => test_kernel_testing.py} | 25 +- tests/unit/test_mirror.py | 9 +- tests/unit/test_morton.py | 64 +- tests/unit/test_mpi_cuda_aware.py | 26 +- tests/unit/test_namespaces.py | 89 +++ tests/unit/test_petsc.py | 23 +- tests/unit/test_philox.py | 44 +- tests/unit/test_porting_helpers.py | 68 +- tests/unit/test_profiling.py | 43 +- tests/unit/test_pyccel_kernel.py | 4 +- tests/unit/test_random_streams.py | 7 +- tests/unit/test_staging.py | 6 +- tests/unit/test_transfers.py | 55 +- 78 files changed, 1974 insertions(+), 1660 deletions(-) create mode 100644 src/cunumpy/algorithms.py create mode 100644 src/cunumpy/cuda/__init__.py create mode 100644 src/cunumpy/kernel_testing.py create mode 100644 src/cunumpy/kernels.py create mode 100644 src/cunumpy/memory.py create mode 100644 src/cunumpy/mpi.py create mode 100644 src/cunumpy/profiling.py create mode 100644 src/cunumpy/rng.py rename tests/unit/{test_testing.py => test_kernel_testing.py} (94%) create mode 100644 tests/unit/test_namespaces.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 0baab30..76a0c91 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Changed (breaking, with deprecation) +- The helpers moved from the top level of `cunumpy` to submodules, so that the top level is the NumPy/CuPy namespace plus backend selection and array conversion, and no helper hides a NumPy or CuPy name (`xp.fuse` hid `cupy.fuse`): `cunumpy.cuda` (CUDA only: `CudaKernel`, `CudaKernelVariants`, `CudaStruct*`, `CudaArguments`, `CudaParameter`, header tools, debug mode, device selection and memory, `stream`, `pin_memory`), `cunumpy.kernels` (`Kernel`, `KernelCatalog`, `PyccelKernel`, `KernelArguments`, `PyccelStructArguments`, host implementations, `as_kernel_array`, `kernel_output`, `fuse`), `cunumpy.rng` (`random_streams`, `RandomStreams`, `get_rng`, `philox_*`), `cunumpy.algorithms` (`morton_*`, `sort_by_key`, `segment_sum`), `cunumpy.mpi` (`mpi_buffer`, CUDA-aware MPI, `local_rank`, `synchronize_for_mpi`), `cunumpy.profiling` (`timed_region`, `Timing`, `nvtx_range`, transfer counting), `cunumpy.memory` (`HostStaging`, `StagedCopy`, `DeviceMirror`) and `cunumpy.petsc` (`petsc_vec`). All are imported by `import cunumpy`. The old top-level names still work and raise a `DeprecationWarning` naming the new place; they will be removed in 0.6. +- `cunumpy.testing` is now `cunumpy.kernel_testing`. Once imported, `cunumpy.testing` replaced NumPy's `xp.testing`, so `xp.testing.assert_allclose` failed in every test that ran after an `import cunumpy.testing`. `cunumpy.testing` still works, with a `DeprecationWarning`, until 0.6. +- Importing `cunumpy.cuda` makes `xp.cuda` the cunumpy submodule instead of CuPy's `cupy.cuda`; use `import cupy` for the latter. + ### Fixed - `CudaKernel`'s header hash (`-DCUNUMPY_INCLUDE_HASH`) now covers the headers shipped with cunumpy (`cunumpy/atomic.cuh`, `reduce.cuh`, ...), also when included in angle brackets. Before, an upgrade of cunumpy that changed one of them left CuPy's kernel cache serving the kernel compiled with the old header. `resolve_includes(..., angle_dirs=...)` tracks angle-bracket includes found in the given directories. - `xp.testing.assert_kernels_agree` reads `CudaStructArguments` objects and struct values through their struct fields, so their arrays get the same names as the attributes of the host argument object (before, arrays behind properties were named after the private attribute holding the owner, and the comparison failed with "do not have the same array arguments"). diff --git a/README.md b/README.md index d485f4c..59bb42f 100644 --- a/README.md +++ b/README.md @@ -18,6 +18,24 @@ move existing arrays just because the selected backend changes. This guide covers backend selection, array movement, mixed CPU/GPU workflows, and the helper APIs CuNumpy provides around NumPy and CuPy. +The top level of `cunumpy` is the NumPy (or CuPy) namespace plus backend +selection and array conversion. The helpers are in submodules, so that they +never hide a NumPy name: + +| Submodule | Contents | +|---|---| +| `xp.cuda` | CUDA only: `CudaKernel`, `CudaStruct`, CUDA headers, devices, streams | +| `xp.kernels` | `Kernel`, `KernelCatalog`, `PyccelKernel`, host implementations, `fuse` | +| `xp.rng` | `random_streams`, `get_rng`, `philox_*` | +| `xp.algorithms` | `morton_*`, `sort_by_key`, `segment_sum` | +| `xp.mpi` | `mpi_buffer`, CUDA-aware MPI | +| `xp.profiling` | `timed_region`, `nvtx_range`, `count_transfers` | +| `xp.memory` | `HostStaging`, `DeviceMirror` | +| `xp.petsc` | `petsc_vec` | +| `cunumpy.kernel_testing` | pytest helpers for host/CUDA kernel pairs | + +Everything except `xp.cuda` works on both backends. + ## Install ```bash @@ -132,7 +150,7 @@ anything was counted. Only transfers made through CuNumpy are seen; raw `cupy.ndarray.get()` or `cupy.asarray()` calls need a profiler such as `nsys`. ```python -with xp.count_transfers() as counter: +with xp.profiling.count_transfers() as counter: propagator(dt) assert counter.total == 0, counter.report() @@ -145,7 +163,7 @@ CuPy have similar generator APIs, though exact bit-for-bit sequences are not guaranteed to match between libraries: ```python -rng = xp.get_rng(seed=42) +rng = xp.rng.get_rng(seed=42) samples = rng.normal(size=1000) ``` @@ -162,9 +180,9 @@ These helpers are useful for multi-GPU programs and for understanding CuPy's memory behavior: ```python -print("visible GPUs:", xp.device_count()) -xp.set_device(0) # selects CUDA device 0 when CuPy is active -print("memory (free, total):", xp.memory_info()) +print("visible GPUs:", xp.cuda.device_count()) +xp.cuda.set_device(0) # selects CUDA device 0 when CuPy is active +print("memory (free, total):", xp.cuda.memory_info()) ``` `set_device()` is a no-op on NumPy. `device_count()` checks visible CUDA @@ -193,12 +211,12 @@ For MPI programs with one rank per GPU, the startup sequence is: ```python xp.set_backend("cupy") -xp.bind_local_device() # before MPI_Init +xp.cuda.bind_local_device() # before MPI_Init from mpi4py import MPI # MPI_Init -xp.require_cuda_aware_mpi() # once, on all ranks +xp.mpi.require_cuda_aware_mpi() # once, on all ranks -xp.synchronize_for_mpi(send, recv) +xp.mpi.synchronize_for_mpi(send, recv) MPI.COMM_WORLD.Sendrecv(send, dest, recvbuf=recv, source=source) ``` @@ -214,7 +232,7 @@ on NumPy. GPU work is asynchronous, so synchronize before reading results on the host: ```python -with xp.stream(): +with xp.cuda.stream(): device = xp.to_cupy(host) transformed = xp.fft.fft(device) @@ -229,12 +247,12 @@ device before reading the clock (on NumPy it is a plain timer), and no-ops or plain timers on NumPy, and `nvtx_range` also works as a decorator: ```python -with xp.timed_region("fft") as timing: +with xp.profiling.timed_region("fft") as timing: transformed = xp.fft.fft(device) print(timing.elapsed, timing.synced) -@xp.nvtx_range("step") +@xp.profiling.nvtx_range("step") def step(dt): ... ``` @@ -256,7 +274,7 @@ def scale_in_place(values, factor): return values -scale = xp.PyccelKernel(scale_in_place, outputs=(0,)) +scale = xp.kernels.PyccelKernel(scale_in_place, outputs=(0,)) with xp.use_backend("cupy"): values = xp.arange(5, dtype=xp.float64) @@ -304,7 +322,7 @@ def axpy(a, x, y, n): # host version, e.g. compiled with Pyccel y[i] += a * x[i] -kernel = xp.Kernel(axpy, xp.CudaKernel(AXPY, "axpy")) +kernel = xp.kernels.Kernel(axpy, xp.cuda.CudaKernel(AXPY, "axpy")) with xp.use_backend("cupy"): x = xp.arange(1000, dtype=xp.float64) @@ -327,8 +345,8 @@ the matching memory layout) and packs values into it, which the kernel takes as one parameter: ```python -Vec = xp.CudaStruct("Vec", [("data", "double*"), ("n", "int")]) -scale = xp.CudaKernel( +Vec = xp.cuda.CudaStruct("Vec", [("data", "double*"), ("n", "int")]) +scale = xp.cuda.CudaKernel( Vec.declaration + r""" extern "C" __global__ void scale(Vec v, double a) { @@ -357,7 +375,7 @@ on the CUDA path, so the call site is the same on both backends and each form can be built lazily on first access (a CPU run never builds device arguments): ```python -class ParticleArguments(xp.KernelArguments): +class ParticleArguments(xp.kernels.KernelArguments): def __init__(self, markers): self.markers = markers self._host = None @@ -387,9 +405,9 @@ class MarkerArguments: def __init__(self, markers: "float[:, :]", n_markers: int, valid: "bool[:]"): ... -MarkerArgs = xp.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs") +MarkerArgs = xp.cuda.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs") MarkerArgs.to_header("marker_args.cuh") # Array2D markers; long long n_markers; ... -push = xp.CudaKernel( +push = xp.cuda.CudaKernel( r""" #include "marker_args.cuh" #include @@ -414,8 +432,8 @@ details. Kernels run asynchronously, so a CUDA error (an illegal memory access, say) normally surfaces at a later `.get()` or MPI call, far from the kernel that -caused it. In debug mode, enabled with `xp.set_cuda_debug(True)`, the -context manager `xp.cuda_debug()`, `CudaKernel(..., debug=True)` or the +caused it. In debug mode, enabled with `xp.cuda.set_cuda_debug(True)`, the +context manager `xp.cuda.cuda_debug()`, `CudaKernel(..., debug=True)` or the environment variable `CUNUMPY_CUDA_DEBUG=1`, kernels are compiled with `-lineinfo` and `-DCUNUMPY_BOUNDS_CHECK` and every launch is synchronized, so the error is raised as a `RuntimeError` naming the kernel and its launch shape. @@ -425,7 +443,7 @@ next step is NVIDIA's memory checker: ## Test kernel pairs -`cunumpy.testing` helps to test the ports with pytest. `assert_kernels_agree` +`cunumpy.kernel_testing` helps to test the ports with pytest. `assert_kernels_agree` builds the arguments on both backends, runs the host and the CUDA kernel and compares the arrays they wrote; with `catalog.parity_cases()`, one parametrised test covers every ported kernel of a catalog. `BACKENDS` and @@ -436,7 +454,7 @@ kernel: ```python import pytest -from cunumpy.testing import assert_kernels_agree +from cunumpy.kernel_testing import assert_kernels_agree def make_args(backend, seed): @@ -460,7 +478,7 @@ the one transfer per accumulation is explicit. The shipped header writes: ```python -mirror = xp.DeviceMirror(vector._data) +mirror = xp.memory.DeviceMirror(vector._data) mirror.zero() accumulate(markers, mirror.device, n_threads=n_markers) mirror.to_host() # vector._data holds the result on both backends diff --git a/docs/source/api.md b/docs/source/api.md index 9fdffdf..1ef95eb 100644 --- a/docs/source/api.md +++ b/docs/source/api.md @@ -27,6 +27,34 @@ NumPy and CuPy are not interchangeable for every function or object. A function that needs to follow an input array's location should use `get_array_module(array)` instead of assuming the global backend matches it. +## Module layout + +The top level of `cunumpy` is the NumPy (or CuPy) namespace plus the functions +that select the backend and convert arrays. Everything else is in a submodule, +imported with `cunumpy` (`xp.cuda.CudaKernel`, `xp.rng.random_streams`, ...). +The submodules are named so that they do not hide a NumPy name (`rng`, not +`random`): + +| Submodule | Backends | Contents | +|---|---|---| +| `cunumpy` | both | NumPy/CuPy namespace, backend selection, array inspection and conversion, `synchronize`, `scipy`, `require_version` | +| `cunumpy.cuda` | CUDA only | `CudaKernel`, `CudaKernelVariants`, `CudaStruct`, `CudaStructArguments`, `CudaArguments`, CUDA headers, debug mode, device selection and memory, `stream`, `pin_memory` | +| `cunumpy.kernels` | both | `Kernel`, `KernelCatalog`, `PyccelKernel`, `KernelArguments`, `PyccelStructArguments`, host implementations, `as_kernel_array`, `kernel_output`, `fuse` | +| `cunumpy.rng` | both | `random_streams`, `get_rng`, `philox_*` | +| `cunumpy.algorithms` | both | `morton_*`, `sort_by_key`, `segment_sum` | +| `cunumpy.mpi` | both | `mpi_buffer`, CUDA-aware MPI detection, `local_rank`, `synchronize_for_mpi` | +| `cunumpy.profiling` | both | `timed_region`, `nvtx_range`, `count_transfers`, `assert_no_transfers` | +| `cunumpy.memory` | both | `HostStaging`, `DeviceMirror` | +| `cunumpy.petsc` | both | `petsc_vec` | +| `cunumpy.kernel_testing` | both | pytest helpers for host/CUDA kernel pairs (not imported by `import cunumpy`) | + +"Both" means the functions work on NumPy and CuPy arrays; the functions of +`cunumpy.cuda` do nothing (or return `None`/`0`) on the NumPy backend. + +Before cunumpy 0.5 these names were at the top level (`xp.CudaKernel`). The +old names still work until cunumpy 0.6 and raise a `DeprecationWarning` that +names the new place. + ## Version ### `require_version(minimum)` @@ -180,7 +208,7 @@ assert xp.get_array_backend(normalized) == xp.get_backend() Each conversion returns a suitable array; it does not change the active backend or mutate the source. -### `segment_sum(values, keys, n_segments)` +### `algorithms.segment_sum(values, keys, n_segments)` `out[k] = sum(values[i] for keys[i] == k)` on the backend of `keys`, with `bincount` under the hood: the reduction step of a sort-then-reduce @@ -189,13 +217,13 @@ separately); a negative key drops the value; keys must be smaller than `n_segments`. The result keeps a floating-point or complex dtype and is `float64` otherwise. -### `sort_by_key(keys, *arrays)` +### `algorithms.sort_by_key(keys, *arrays)` Stable argsort of the 1D `keys` (CuPy's radix sort on the device), applied to every array along axis 0, in one call: ```python -keys, order, positions, charges = xp.sort_by_key(keys, positions, charges) +keys, order, positions, charges = xp.algorithms.sort_by_key(keys, positions, charges) ``` Returns `(keys[order], order, *(a[order] for a in arrays))`, `order` as @@ -207,14 +235,14 @@ A transfer inside a time loop is the classic performance bug of a GPU port: every step then waits for the device and copies an array. These helpers let a test verify that a block of code does not transfer at all. -### `count_transfers()` +### `profiling.count_transfers()` Context manager yielding a `TransferCounter` that records every host/device transfer made through CuNumpy while the block runs, with the call site of each: ```python -with xp.count_transfers() as counter: +with xp.profiling.count_transfers() as counter: propagator(dt) assert counter.total == 0, counter.report() @@ -260,7 +288,7 @@ not thread-safe. `float(device_array)`, and implicit conversions inside other libraries are not counted. Use `nsys` (or CuPy's profiling hooks) to find those. -### `assert_no_transfers()` +### `profiling.assert_no_transfers()` Context manager that raises `AssertionError` with the counter's `report()` if the block makes a transfer through CuNumpy. It yields the `TransferCounter` @@ -268,7 +296,7 @@ too. An exception raised inside the block propagates as it is: ```python def test_time_step_stays_on_the_device(): - with xp.assert_no_transfers(): + with xp.profiling.assert_no_transfers(): propagator(dt) ``` @@ -293,14 +321,14 @@ running on CuPy, and host data is never copied to the device implicitly. If `ValueError`; `name` is the argument name used in error messages. ```python -class DeviceParticles(xp.CudaArguments): +class DeviceParticles(xp.cuda.CudaArguments): def __init__(self, markers, degree): self.markers = xp.as_device_array(markers, np.float64, ndim=2, name="markers") self.degree = xp.as_device_array(degree, np.int32, ndim=1, name="degree") super().__init__(self.markers, self.degree, self.markers.shape[0]) ``` -### `as_kernel_array(value, like, dtype=None)`, `kernel_output(out, like, dtype=None)` +### `kernels.as_kernel_array(value, like, dtype=None)`, `kernels.kernel_output(out, like, dtype=None)` For the arguments of a `Kernel` with `dispatch="arrays"`, whose choice follows the arrays. `as_kernel_array` returns `value` on the side of `like` (a CuPy @@ -313,31 +341,31 @@ contents are written into `out` (on its own side) when the block ends without an error. ```python -with xp.kernel_output(result, like=field, dtype=float) as buffer: - gather(xp.as_kernel_array(positions, like=field, dtype=float), field, buffer) +with xp.kernels.kernel_output(result, like=field, dtype=float) as buffer: + gather(xp.kernels.as_kernel_array(positions, like=field, dtype=float), field, buffer) ``` ## Random numbers and dtype -### `get_rng(seed=None)` +### `rng.get_rng(seed=None)` Returns a NumPy or CuPy `Generator` matching the active backend: ```python -rng = xp.get_rng(seed=7) +rng = xp.rng.get_rng(seed=7) samples = rng.uniform(size=100) ``` The generator APIs are similar, but seeds do not guarantee identical random sequences across NumPy and CuPy. -### `random_streams` +### `rng.random_streams` ```python -xp.random_streams.seed(42, rank=comm.Get_rank(), bit_generator="PCG64") -v = xp.random_streams.normal(0.0, v_th, (n, 3)) -rng = xp.random_streams.generator() # numpy or cupy Generator -own = xp.random_streams.make_generator(seed) # a component's own generator +xp.rng.random_streams.seed(42, rank=comm.Get_rank(), bit_generator="PCG64") +v = xp.rng.random_streams.normal(0.0, v_th, (n, 3)) +rng = xp.rng.random_streams.generator() # numpy or cupy Generator +own = xp.rng.random_streams.make_generator(seed) # a component's own generator ``` One seeded random generator per process and backend, for reproducible MPI @@ -356,7 +384,7 @@ backend=None)` returns a separate generator for a component with a seed of its own, and the process generator otherwise. `random`, `standard_normal`, `normal` and `uniform` draw from the process generator (or `rng=`); `normal` and `uniform` fall back to `standard_normal` and `random` for CuPy generators -without those methods. `xp.RandomStreams()` makes an independent instance. +without those methods. `xp.rng.RandomStreams()` makes an independent instance. ### `default_float_dtype()` @@ -375,18 +403,18 @@ Returns whether CuPy can be imported and reports itself functional. The result is cached for the process. This checks availability, not whether every GPU-specific operation will succeed later. -### `device_count()` +### `cuda.device_count()` Returns the number of visible CUDA devices. Returns zero when CuPy/CUDA is unavailable or querying the runtime fails. This is independent of the active backend, so it may return a positive number while NumPy is selected. -### `set_device(device_id)` +### `cuda.set_device(device_id)` Selects a CUDA device when CuPy is active; it is a no-op on NumPy. The device must be valid for the current CUDA process. -### `set_device_for_rank(rank, devices_per_node=None)` +### `cuda.set_device_for_rank(rank, devices_per_node=None)` Selects a device using `rank % devices_per_node` and returns its ID. If `devices_per_node` is omitted, it uses `device_count()`. When no devices are @@ -396,10 +424,10 @@ one-rank-per-GPU MPI layout. Use `set_device()` directly when the scheduler's mapping differs: ```python -device_id = xp.set_device_for_rank(mpi_rank) +device_id = xp.cuda.set_device_for_rank(mpi_rank) ``` -### `local_rank()` +### `mpi.local_rank()` The rank of the process within its node, read from the environment variables that MPI launchers export (Open MPI, MVAPICH2, Intel MPI/MPICH, PMI, Cray @@ -407,7 +435,7 @@ PALS, Slurm, `LOCAL_RANK`), or `0` if none is set. The launcher sets them before `MPI_Init`, so this works before MPI is initialized and without importing `mpi4py`. -### `bind_local_device()` +### `cuda.bind_local_device()` Selects device `local_rank() % device_count()` for this process and creates its CUDA context. Returns the device id, or `None` on the NumPy backend or without @@ -420,14 +448,14 @@ device 0. If the launcher gives each rank its own device through import cunumpy as xp xp.set_backend("cupy") -xp.bind_local_device() +xp.cuda.bind_local_device() from mpi4py import MPI # initializes MPI after the device is bound ``` Unlike `set_device_for_rank()`, it needs no MPI rank, and it uses the rank within the node rather than assuming contiguous ranks per node. -### `mpi_is_cuda_aware(comm=None, *, method="probe")` +### `mpi.mpi_is_cuda_aware(comm=None, *, method="probe")` Checks whether the MPI library can send and receive device (CuPy) buffers, which needs a CUDA-aware MPI build; with a plain build, passing a CuPy array @@ -452,7 +480,7 @@ means the same thing as `False`. Call it once at startup, after `bind_local_device()` and `MPI_Init`, before any communication of device buffers. -### `require_cuda_aware_mpi(comm=None)` +### `mpi.require_cuda_aware_mpi(comm=None)` Raises `RuntimeError`, explaining how to get a CUDA-aware build (Open MPI `--with-cuda`, MPICH with a CUDA-enabled UCX, the site's CUDA-aware MPI @@ -464,16 +492,16 @@ GPU: import cunumpy as xp xp.set_backend("cupy") -xp.bind_local_device() # 1. select the GPU, before MPI_Init +xp.cuda.bind_local_device() # 1. select the GPU, before MPI_Init from mpi4py import MPI # 2. MPI_Init, on the bound device -xp.require_cuda_aware_mpi() # 3. clear error instead of a segfault later +xp.mpi.require_cuda_aware_mpi() # 3. clear error instead of a segfault later -xp.synchronize_for_mpi(send, recv) # 4. before every MPI call with device buffers +xp.mpi.synchronize_for_mpi(send, recv) # 4. before every MPI call with device buffers MPI.COMM_WORLD.Sendrecv(send, dest, recvbuf=recv, source=source) ``` -### `synchronize_for_mpi(*arrays)` +### `mpi.synchronize_for_mpi(*arrays)` Waits for the work pending on the current stream if at least one of `arrays` is a CuPy array; `None` entries and host arrays are ignored, so it costs @@ -484,14 +512,14 @@ kernel is still writing would be sent as it is at that moment, without an error: ```python -xp.synchronize_for_mpi(send_buffer, recv_buffer) +xp.mpi.synchronize_for_mpi(send_buffer, recv_buffer) comm.Sendrecv(send_buffer, dest, recvbuf=recv_buffer, source=source) ``` No synchronization is needed after MPI returns: kernels launched afterwards see the received data. -### `mpi_buffer(array, *, send=True, recv=False, cuda_aware=None)` +### `mpi.mpi_buffer(array, *, send=True, recv=False, cuda_aware=None)` Context manager yielding the buffer to pass to MPI for `array`: a host array unchanged; a device array unchanged (after `synchronize_for_mpi`) when MPI is @@ -501,26 +529,26 @@ before the block (`send`) and copied back after it (`recv`), both counted by `mpi_is_cuda_aware()` or `set_mpi_cuda_aware()`; without one, a device array raises `RuntimeError`. -### `set_mpi_cuda_aware(value)`, `get_mpi_cuda_aware()` +### `mpi.set_mpi_cuda_aware(value)`, `mpi.get_mpi_cuda_aware()` Record (or read) whether MPI can take device buffers, for `mpi_buffer()`. `mpi_is_cuda_aware()` records its own result; set it by hand when the answer is known otherwise, or `None` to forget it. -### `memory_info()` +### `cuda.memory_info()` Returns `(free_bytes, total_bytes)` reported by the CUDA runtime for the active device, or `None` on NumPy. The values cover the device, not only allocations owned by CuPy. -### `free_memory()` +### `cuda.free_memory()` Releases currently free blocks in CuPy's device and pinned-host memory pools. It is a no-op on NumPy. It does not release blocks still referenced by live arrays. CuPy normally caches freed allocations for reuse, so cached memory does not necessarily indicate a leak. -### `max_shared_memory_per_block(device=None, *, opt_in=False)` +### `cuda.max_shared_memory_per_block(device=None, *, opt_in=False)` The bytes of shared memory a block may use on the current (or given) device, from the device attributes, e.g. to decide whether a per-block copy of a grid @@ -529,7 +557,7 @@ only after setting `max_dynamic_shared_size_bytes` on its compiled `cupy.RawKernel`. Without CuPy it returns `DEFAULT_SHARED_MEMORY_PER_BLOCK` (48 KiB, which every CUDA device provides). -### `cuda_include_dir()` +### `cuda.cuda_include_dir()` Returns the directory (as `str`) of the CUDA headers shipped with CuNumpy, `cunumpy/array_view.cuh`, `cunumpy/atomic.cuh`, `cunumpy/index.cuh` and `cunumpy/reduce.cuh`. `CudaKernel` adds it to its NVRTC options as @@ -537,7 +565,7 @@ Returns the directory (as `str`) of the CUDA headers shipped with CuNumpy, `#include ` without configuration. Use it to pass the same headers to other compilers. -### `pin_memory(array)` +### `cuda.pin_memory(array)` Copies a host array to page-locked (pinned) host memory. Pinned memory can improve host/device transfer throughput in suitable asynchronous workloads. @@ -550,14 +578,14 @@ Waits for queued work on the current CUDA device to finish. This is useful before reading asynchronously computed results from host code. It is a no-op on NumPy. -### `stream()` +### `cuda.stream()` Context manager that creates a non-blocking CuPy stream and yields it. Work issued in the block is enqueued on that stream. On NumPy, yields `None` and does nothing: ```python -with xp.stream() as work_stream: +with xp.cuda.stream() as work_stream: result = xp.to_cupy(host_values) * 2 work_stream.synchronize() # on CuPy; the yielded value is None on NumPy @@ -573,7 +601,7 @@ the launch, not the kernel, and regions of an application profiler are not visible to `nsys`. These helpers address both; they are no-ops (or plain timers) on NumPy, so instrumented code runs unchanged on both backends. -### `nvtx_range(name, color=None)` +### `profiling.nvtx_range(name, color=None)` Context manager and decorator marking a code region as an NVTX range. On CuPy it calls `cupy.cuda.nvtx.RangePush(name)` on entry and `RangePop()` on exit @@ -585,16 +613,16 @@ instance may be nested or re-entered, e.g. as the decorator of a recursive function. ```python -with xp.nvtx_range("push markers"): +with xp.profiling.nvtx_range("push markers"): kernel(markers, dt, n_threads=n) -@xp.nvtx_range("accumulate") +@xp.profiling.nvtx_range("accumulate") def accumulate(particles, grid): ... ``` -### `timed_region(name, *, sync=True)` +### `profiling.timed_region(name, *, sync=True)` Context manager timing a code region, including the device work it queues. It yields a `Timing` object whose `elapsed` (seconds, from @@ -606,23 +634,23 @@ the clock; `synced` records whether that happened. It also pushes an is `False`. With `sync=False` only the host time is measured. ```python -with xp.timed_region("push markers") as timing: +with xp.profiling.timed_region("push markers") as timing: kernel(markers, dt, n_threads=n) print(f"{timing.name}: {timing.elapsed:.4f} s, synced={timing.synced}") ``` -### `Timing` +### `profiling.Timing` Dataclass returned by `timed_region()`, with the fields `name` (`str`), `elapsed` (`float`, `None` until the block exits) and `synced` (`bool`). -## `PyccelKernel` +## `kernels.PyccelKernel` ### Constructor ```python -xp.PyccelKernel( +xp.kernels.PyccelKernel( kernel, use_cupy=None, object_modules=(), @@ -663,7 +691,7 @@ def scale_and_shift(scale, values, out): out[:] = scale * values + 1 return out -kernel = xp.PyccelKernel(scale_and_shift, outputs=(2,)) +kernel = xp.kernels.PyccelKernel(scale_and_shift, outputs=(2,)) with xp.use_backend("cupy"): values = xp.arange(8, dtype=xp.float64) @@ -701,12 +729,12 @@ lists) are converted back using `is_array`; dictionaries in return values are not recursively converted. On the NumPy path, the original return value and normal Python mutation and exception behavior are preserved. -## `CudaKernel` +## `cuda.CudaKernel` ### Constructor ```python -xp.CudaKernel( +xp.cuda.CudaKernel( source, name, *, @@ -719,8 +747,8 @@ xp.CudaKernel( check_signature=True, debug=None, ) -xp.CudaKernel.from_file(path, name=None, *, suffix="_cuda.cu", **kwargs) -xp.CudaKernel.all_from_file(path, **kwargs) +xp.cuda.CudaKernel.from_file(path, name=None, *, suffix="_cuda.cu", **kwargs) +xp.cuda.CudaKernel.all_from_file(path, **kwargs) ``` Wraps the `__global__` function `name` in the CUDA C `source` (declared @@ -737,11 +765,11 @@ is added to the include directories and is the `source_dir`. `all_from_file` loads every `__global__` function of a file, for files that group several small kernels, and returns a `dict` of kernels by name in the order of the source. The kernels share the source and options, so CuPy -compiles the file once. `xp.cuda_kernel_names(source)` lists the `__global__` +compiles the file once. `xp.cuda.cuda_kernel_names(source)` lists the `__global__` functions of a source string (ignoring comments). ```python -kernels = xp.CudaKernel.all_from_file("small_kernels.cu", block_size=64) +kernels = xp.cuda.CudaKernel.all_from_file("small_kernels.cu", block_size=64) kernels["scale"](x, 2.0, x.size, n_threads=x.size) kernels["shift"](x, 1.0, x.size, n_threads=x.size) ``` @@ -782,7 +810,7 @@ resolves the quoted includes of its source when it compiles and adds a define with a hash of their contents to the options: ```python -kernel = xp.CudaKernel.from_file("push/push_cuda.cu", include_dirs=[src_root]) +kernel = xp.cuda.CudaKernel.from_file("push/push_cuda.cu", include_dirs=[src_root]) kernel.included_headers # (Path('push/helpers.cuh'), Path('.../common.cuh')) kernel.options # ('-Ipush', '-I') kernel.compile_options() # options + ('-DCUNUMPY_INCLUDE_HASH=0x3f9a...',) @@ -807,10 +835,10 @@ kernel.compile_options() # options + ('-DCUNUMPY_INCLUDE_HASH=0x3f9a...',) The two building blocks are available on their own: -* `xp.resolve_includes(source, include_dirs=(), *, base_dir=None, +* `xp.cuda.resolve_includes(source, include_dirs=(), *, base_dir=None, angle_dirs=())`: the resolved header paths of a source, as a list; `angle_dirs` are searched last and also for `#include `. -* `xp.include_hash(paths)`: the first 16 hex digits of the SHA-256 digest of +* `xp.cuda.include_hash(paths)`: the first 16 hex digits of the SHA-256 digest of the contents of the files, in order. ### Calling @@ -831,7 +859,7 @@ shape is given either by `n_threads` or by `grid`: * `shared_mem`: dynamic shared memory per block in bytes, for `extern __shared__` arrays. Above 48 KiB (the limit every device has) the compiled kernel's `max_dynamic_shared_size_bytes` is raised to `shared_mem` - once, up to the device's opt-in limit (`xp.max_shared_memory_per_block( + once, up to the device's opt-in limit (`xp.cuda.max_shared_memory_per_block( opt_in=True)`); a larger request raises `ValueError` before the launch. Without `n_threads` and `grid`, a launch uses `n_threads_from(args)` if the @@ -840,7 +868,7 @@ argument tuple, or `"first_array"` for the length of the first array argument (its first axis), i.e. one thread per marker: ```python -push = xp.CudaKernel.from_file("push_cuda.cu", n_threads_from="first_array") +push = xp.cuda.CudaKernel.from_file("push_cuda.cu", n_threads_from="first_array") push(positions, velocities, e_field, dt) # n_threads = positions.shape[0] ``` @@ -862,7 +890,7 @@ extern "C" __global__ void block_sum(const double* x, double* out, int n) { if (threadIdx.x == 0) out[blockIdx.x] = buffer[0]; } """ -block_sum = xp.CudaKernel(BLOCK_SUM, "block_sum", block_size=128) +block_sum = xp.cuda.CudaKernel(BLOCK_SUM, "block_sum", block_size=128) (n_blocks,), _ = block_sum.launch_shape(x.size) partial = xp.zeros(n_blocks) block_sum(x, partial, x.size, n_threads=x.size, shared_mem=128 * 8) @@ -910,9 +938,9 @@ itself costs about as much), which is negligible for kernels that run for C types are mapped to NumPy dtypes as on Linux (LP64): `int` is `int32`, `long` and `long long` are `int64`, `float` is `float32`, `double` is `float64`, `complex` is `complex128`; fixed-width types such as -`int64_t` and `size_t` are supported too. `xp.ctype_of(dtype)` gives the C -type of a dtype (`xp.ctype_of(np.float64) == "double"`), e.g. to generate -source. `xp.parse_cuda_signature(source, name, *, structs=(), +`int64_t` and `size_t` are supported too. `xp.cuda.ctype_of(dtype)` gives the C +type of a dtype (`xp.cuda.ctype_of(np.float64) == "double"`), e.g. to generate +source. `xp.cuda.parse_cuda_signature(source, name, *, structs=(), template_args=None)` returns the parsed parameters (`CudaParameter` tuples of `name`, `ctype`, `dtype`, `pointer`, `struct`, `view_ndim`). @@ -928,13 +956,13 @@ void scale_column(Array2D a, long long column, double factor) { a(i, column) *= factor; } """ -scale_column = xp.CudaKernel(SCALE_COLUMN, "scale_column") +scale_column = xp.cuda.CudaKernel(SCALE_COLUMN, "scale_column") view = markers[::2, 1:5] # non-contiguous is fine scale_column(view, 1, 10.0, n_threads=view.shape[0]) ``` cunumpy ships CUDA headers that every `CudaKernel` finds automatically; -`xp.cuda_include_dir()` returns their directory (a `str`) for other compilers +`xp.cuda.cuda_include_dir()` returns their directory (a `str`) for other compilers (`-I`). `cunumpy/array_view.cuh` defines the strided views `Array1D` to @@ -974,7 +1002,7 @@ __global__ void scale(T* x, T factor, int n) { if (i < n) x[i] = factor * x[i] * (T)N; } """ -scale_f64 = xp.CudaKernel(SCALE, "scale", template_args=(np.float64, 3)) +scale_f64 = xp.cuda.CudaKernel(SCALE, "scale", template_args=(np.float64, 3)) scale_f64(x, 2.0, x.size, n_threads=x.size) # instantiation scale ``` @@ -983,8 +1011,8 @@ dimensions and dtype), `CudaKernelVariants` creates and caches one kernel per key: ```python -matvec = xp.CudaKernelVariants( - lambda ndim, dtype: xp.CudaKernel(make_source(ndim, xp.ctype_of(dtype)), "matvec") +matvec = xp.cuda.CudaKernelVariants( + lambda ndim, dtype: xp.cuda.CudaKernel(make_source(ndim, xp.cuda.ctype_of(dtype)), "matvec") ) matvec.get(3, np.float64)(mat, x, out, n_threads=out.size) # created once matvec.compile_all([(3, np.float64), (3, np.complex128)]) # at setup @@ -998,13 +1026,13 @@ threads (see `KernelCatalog.compile_all`). ### Debugging ```python -xp.set_cuda_debug(enabled) -xp.get_cuda_debug() -xp.cuda_debug(enabled=True) # context manager -xp.CudaKernel(..., debug=None) +xp.cuda.set_cuda_debug(enabled) +xp.cuda.get_cuda_debug() +xp.cuda.cuda_debug(enabled=True) # context manager +xp.cuda.CudaKernel(..., debug=None) kernel.debug_active() kernel.compile_options() -xp.DEBUG_OPTIONS # ("-lineinfo", "-DCUNUMPY_BOUNDS_CHECK") +xp.cuda.DEBUG_OPTIONS # ("-lineinfo", "-DCUNUMPY_BOUNDS_CHECK") ``` Kernel launches are asynchronous: a CUDA error such as an illegal memory @@ -1022,10 +1050,10 @@ that caused it. In debug mode, a `CudaKernel` `RuntimeError` that names the kernel and its grid and block, with the CuPy error chained. -Debug mode is enabled globally with `xp.set_cuda_debug(True)`, temporarily -with the context manager `xp.cuda_debug()`, or before starting Python with +Debug mode is enabled globally with `xp.cuda.set_cuda_debug(True)`, temporarily +with the context manager `xp.cuda.cuda_debug()`, or before starting Python with the environment variable `CUNUMPY_CUDA_DEBUG=1` (`true`, `yes` and `on` work -too); `xp.get_cuda_debug()` returns the current setting. A kernel created with +too); `xp.cuda.get_cuda_debug()` returns the current setting. A kernel created with `debug=None` (the default) reads the global setting at every launch, so enabling it also affects kernels created earlier; `debug=True` or `debug=False` fix the mode for one kernel. Only the compile options are fixed @@ -1036,8 +1064,8 @@ kernel now, and `kernel.compile_options()` returns the options a compilation now would use. ```python -with xp.cuda_debug(): - kernel = xp.CudaKernel(SOURCE, "kernel") +with xp.cuda.cuda_debug(): + kernel = xp.cuda.CudaKernel(SOURCE, "kernel") kernel(x, y, n, n_threads=n) # RuntimeError: CUDA error after launching kernel 'kernel' ... ``` @@ -1052,10 +1080,10 @@ CUNUMPY_CUDA_DEBUG=1 compute-sanitizer python -m pytest tests/unit/test_my_kerne Note that after an illegal memory access the CUDA context is unusable; the process (or the pytest run) has to be restarted. -## `CudaStruct` +## `cuda.CudaStruct` ```python -Particles = xp.CudaStruct( +Particles = xp.cuda.CudaStruct( "Particles", [("x", "double*"), ("v", "double*"), ("n", "int"), ("charge", "double")], ) @@ -1065,7 +1093,7 @@ extern "C" __global__ void push(Particles p, double dt) { if (i < p.n) p.x[i] += dt * p.charge * p.v[i]; } """ -push = xp.CudaKernel(source, "push", structs=[Particles]) +push = xp.cuda.CudaKernel(source, "push", structs=[Particles]) push(Particles(x=x, v=v, n=x.size, charge=-1.0), 0.1, n_threads=x.size) ``` @@ -1114,7 +1142,7 @@ class MarkerArguments: # the pyccel argument class, e.g. in struphy def __init__(self, markers: "float[:, :]", n_markers: int, valid: "bool[:]"): ... -MarkerArgs = xp.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs") +MarkerArgs = xp.cuda.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs") print(MarkerArgs.declaration) # struct MarkerArgs { # Array2D markers; @@ -1151,14 +1179,14 @@ or an annotation cannot be mapped. ### Generating headers ```python -xp.write_cuda_header("pusher_args.cuh", [MarkerArgs, DomainArgs]) +xp.cuda.write_cuda_header("pusher_args.cuh", [MarkerArgs, DomainArgs]) ``` `struct.to_header(path=None, *, guard=None, includes=())` returns the struct definition as a header: an include guard (`_CUH` by default), `#include "cunumpy/array_view.cuh"` if the struct has array view fields, the `includes` (file names or `#include` lines), and the definition. With `path` -the header is also written. `xp.write_cuda_header(path, structs, guard=None, +the header is also written. `xp.cuda.write_cuda_header(path, structs, guard=None, *, includes=())` writes several structs to one header (the guard defaults to the file name, `pusher_args.cuh` -> `PUSHER_ARGS_CUH`) and returns the source. @@ -1167,17 +1195,17 @@ to the kernels that `#include` it, and keep it in sync with a test: ```python def test_pusher_args_header_is_up_to_date(): - generated = xp.write_cuda_header(tmp_path / "pusher_args.cuh", [MarkerArgs, DomainArgs]) + generated = xp.cuda.write_cuda_header(tmp_path / "pusher_args.cuh", [MarkerArgs, DomainArgs]) assert Path("kernels/pusher_args.cuh").read_text() == generated ``` Kernels created with `structs=[MarkerArgs, ...]` also check a definition in their own source against the Python definition (`check_source`). -## `CudaStructArguments` +## `cuda.CudaStructArguments` ```python -class MarkerArguments(xp.CudaStructArguments): +class MarkerArguments(xp.cuda.CudaStructArguments): struct_name = "MarkerArgs" fields = (("markers", "Array2D"), ("valid", "bool*"), ("n_markers", "int")) @@ -1187,7 +1215,7 @@ class MarkerArguments(xp.CudaStructArguments): self.n_markers = markers.shape[0] self.pack() -push = xp.CudaKernel(source, "push", structs=[MarkerArguments.struct]) +push = xp.cuda.CudaKernel(source, "push", structs=[MarkerArguments.struct]) push(MarkerArguments(markers, valid), dt, n_threads=markers.shape[0]) ``` @@ -1217,7 +1245,7 @@ built once per subclass when the class is defined and is the class attribute base class (its instances cannot be packed); setting only one raises `TypeError`. Subclasses of a complete class inherit its struct. -## `PyccelStructArguments` +## `kernels.PyccelStructArguments` `CudaStructArguments` with a host form (the `KernelArguments` protocol). Class attributes, besides `struct_name` and `fields`: @@ -1236,10 +1264,10 @@ the attributes was replaced (array identity or address, scalar value). array fields are device arrays; objects holding host arrays are copied and pickled without packing, and the host object is never pickled. -## `CudaArguments` +## `cuda.CudaArguments` ```python -class Particles(xp.CudaArguments): +class Particles(xp.cuda.CudaArguments): def __init__(self, positions, velocities): self.positions = positions super().__init__(positions, velocities, positions.shape[0]) @@ -1256,10 +1284,10 @@ arrays) and matching device argument objects that reference the same data on the device, and pass either to the same call. A `CudaArguments` object may also return struct values (`CudaStructValue.packed`) among its values. -## `KernelArguments` +## `kernels.KernelArguments` ```python -class ParticleArguments(xp.KernelArguments): +class ParticleArguments(xp.kernels.KernelArguments): def __init__(self, particles): self._particles = particles self._host = None @@ -1316,7 +1344,7 @@ object. The owner is responsible for invalidating the cache (setting the stored forms to `None`, or replacing the `ParticleArguments` object) when its arrays are replaced, e.g. after resizing, `deepcopy` or unpickling. -### `resolve_host_args(args, kwargs=None)` +### `kernels.resolve_host_args(args, kwargs=None)` Returns `(args, kwargs)` with every top-level argument whose type defines a callable `__host_args__()` replaced by its result; everything else is passed @@ -1324,14 +1352,14 @@ through untouched. `Kernel` and `PyccelKernel` call it before the host kernel; it is exported for code that calls host kernels by other means: ```python -args, kwargs = xp.resolve_host_args((particles.kernel_args, dt), {"out": out}) +args, kwargs = xp.kernels.resolve_host_args((particles.kernel_args, dt), {"out": out}) host_push(*args, **kwargs) ``` -## `Kernel` +## `kernels.Kernel` ```python -xp.Kernel( +xp.kernels.Kernel( host_kernel, cuda_kernel=None, *, @@ -1406,10 +1434,10 @@ Properties: `name`, `host_kernel`, `cuda_kernel`, `has_cuda`, `missing_cuda`, * `Kernel.__call__` needs no `n_threads` when the CUDA kernel has `n_threads_from`. -## `KernelCatalog` +## `kernels.KernelCatalog` ```python -catalog = xp.KernelCatalog.from_package( +catalog = xp.kernels.KernelCatalog.from_package( package, *, host_suffix="_kernels", @@ -1435,7 +1463,7 @@ file are ignored by the catalog; they can be loaded with ```text my_kernels/ -├── __init__.py # catalog = xp.KernelCatalog.from_package(__name__) +├── __init__.py # catalog = xp.kernels.KernelCatalog.from_package(__name__) ├── push/ │ ├── push_kernels.py # def push(...): ... │ └── push_cuda.cu # __global__ void push(...) @@ -1469,7 +1497,7 @@ A Pyccel package with the layout `/_pyccel.py` and `/_cuda.cu`, compiled host kernels and dispatch by argument: ```python -catalog = xp.KernelCatalog.from_package( +catalog = xp.kernels.KernelCatalog.from_package( __name__, host_suffix="_pyccel", dispatch="arrays", @@ -1501,11 +1529,11 @@ that have a CUDA kernel, for a parametrised parity test (see "Testing utilities"). `KernelCatalog(kernels)` and `catalog.register(kernel, name=None)` build a catalog by hand. -## `Kernel.from_folder` +## `kernels.Kernel.from_folder` ```python # my_sim/kernels/push/__init__.py -kernel = xp.Kernel.from_folder(__name__, host_suffix="_pyccel", dispatch="arrays", +kernel = xp.kernels.Kernel.from_folder(__name__, host_suffix="_pyccel", dispatch="arrays", compile_host=compile_kernels) kernel.implementations # ("pyccel", "numpy", "python", "cuda") kernel.selected() # "pyccel": what a call with host arrays runs now @@ -1525,18 +1553,18 @@ files, as loaders by name (`{"numpy": lambda: push_numpy}`). Takes the options o no host kernel module and `ModuleNotFoundError` if `package` is not a package. `kernel.selected(device=True)` names the implementation for device arguments. -## `HostImplementations`, `set_kernel_implementation` +## `kernels.HostImplementations`, `kernels.set_kernel_implementation` ```python -host = xp.HostImplementations("push", {"pyccel": load_compiled, "numpy": lambda: push_numpy, +host = xp.kernels.HostImplementations("push", {"pyccel": load_compiled, "numpy": lambda: push_numpy, "python": lambda: push}) host(*args) # the default implementation -xp.set_kernel_implementation("numpy") # every kernel: like xp.set_backend -with xp.use_kernel_implementation("python"): # like xp.use_backend +xp.kernels.set_kernel_implementation("numpy") # every kernel: like xp.set_backend +with xp.kernels.use_kernel_implementation("python"): # like xp.use_backend host(*args) ``` -The host implementations of one kernel (names from `xp.HOST_IMPLEMENTATIONS`: +The host implementations of one kernel (names from `xp.kernels.HOST_IMPLEMENTATIONS`: `"pyccel"`, `"numba"`, `"numpy"`, `"python"`; `"python"` is required), each given as a loader that returns the function or raises if it is unavailable. Loaded on first use; `available(name)` loads and reports, `get(name)` returns it @@ -1551,10 +1579,10 @@ numba and NumPy, and else `"python"` with a `RuntimeWarning` (once). `get_kernel_implementation()` reads the setting; `None` is the default. The setting is global, not per thread, and applies to host calls only. -## `CompiledHostKernel` +## `kernels.CompiledHostKernel` ```python -kernel = xp.CompiledHostKernel(my_kernels_module, "push", compiler, fallback=push_numpy) +kernel = xp.kernels.CompiledHostKernel(my_kernels_module, "push", compiler, fallback=push_numpy) kernel(*args) ``` @@ -1573,7 +1601,7 @@ compiles now. ## Testing utilities ```python -from cunumpy.testing import ( +from cunumpy.kernel_testing import ( BACKENDS, assert_kernels_agree, device_function_kernel, @@ -1582,11 +1610,12 @@ from cunumpy.testing import ( ) ``` -`cunumpy.testing` holds helpers for testing kernels with pytest. It is not -imported by `import cunumpy` (so `xp.testing` remains NumPy's or CuPy's -`testing` module until `cunumpy.testing` is imported), and it imports pytest -only when one of its pytest objects is used, so `device_function_kernel` works -without pytest. +`cunumpy.kernel_testing` holds helpers for testing kernels with pytest. It is +not imported by `import cunumpy`, and it imports pytest only when one of its +pytest objects is used, so `device_function_kernel` works without pytest. +Before cunumpy 0.5 it was called `cunumpy.testing`, which replaced NumPy's +`xp.testing` once imported; that name still works, with a +`DeprecationWarning`, until cunumpy 0.6. ### `requires_cupy`, `BACKENDS`, `backend` @@ -1603,7 +1632,7 @@ def test_norm(backend): ``` The `backend` fixture does the same and activates the backend for the test; -import it into a `conftest.py` (`from cunumpy.testing import backend`) or the +import it into a `conftest.py` (`from cunumpy.kernel_testing import backend`) or the test module, then take `backend` as a test argument. ### `assert_kernels_agree(kernel, make_args, ...)` @@ -1788,10 +1817,10 @@ does not compile, or crashes (an out-of-bounds index with `-DCUNUMPY_BOUNDS_CHECK`, `__trap()`), raises `RuntimeError` with the compiler or program output. -## `HostStaging` +## `memory.HostStaging` ```python -staging = xp.HostStaging(rho.shape, rho.dtype, buffers=2) +staging = xp.memory.HostStaging(rho.shape, rho.dtype, buffers=2) copy = staging.copy(rho) # returns at once; rho may be overwritten ... if copy.ready(): @@ -1810,10 +1839,10 @@ buffer is reused `buffers` copies later; a stale result raises the NumPy backend copy at once. Device copies are counted by `count_transfers()`. The arrays must have the staging shape and dtype. -## `DeviceMirror` +## `memory.DeviceMirror` ```python -mirror = xp.DeviceMirror(host_array) +mirror = xp.memory.DeviceMirror(host_array) ``` Pairs a host NumPy array that another library owns and keeps using on the host @@ -1843,7 +1872,7 @@ The transfers are explicit so that one per accumulation is visible and bounded: ```python -mirror = xp.DeviceMirror(vector._data) +mirror = xp.memory.DeviceMirror(vector._data) mirror.zero() accumulate(markers, mirror.device, n_threads=n_markers) # a Kernel mirror.to_host() # vector._data now holds the result, same object @@ -1871,7 +1900,7 @@ The indexed helpers (also for `float`) address C-contiguous arrays of shape for `double` from compute capability 6.0 (sm_60) on; older devices use a compare-and-swap loop. -### `cunumpy/random.cuh` and `philox_uniform` +### `cunumpy/random.cuh` and `rng.philox_uniform` Counter-based random numbers (Philox4x32-10, as in Random123 and cuRAND): a pure function of a key and a counter, with no generator state, so each thread @@ -1892,19 +1921,19 @@ void cunumpy_normal2(seed, stream, counter, &z0, &z1); ```python ids = xp.arange(n, dtype=xp.uint64) -u0, u1 = xp.philox_uniform2(seed, ids, step) # == cunumpy_uniform2 in thread i -z0, z1 = xp.philox_normal2(seed, ids, step) -words = xp.philox4x32_10(counter_words, key0, key1) # the raw generator +u0, u1 = xp.rng.philox_uniform2(seed, ids, step) # == cunumpy_uniform2 in thread i +z0, z1 = xp.rng.philox_normal2(seed, ids, step) +words = xp.rng.philox4x32_10(counter_words, key0, key1) # the raw generator ``` -`xp.philox_uniform`, `philox_uniform2`, `philox_normal`, `philox_normal2` and +`xp.rng.philox_uniform`, `philox_uniform2`, `philox_normal`, `philox_normal2` and `philox4x32_10` broadcast their arguments and return NumPy or CuPy arrays, matching the inputs. The uniform numbers equal the kernel's bit for bit; the normal numbers can differ in the last bits (`log`, `sqrt`, `sin`, `cos` on the GPU are not the host's). The generator passes the Random123 known-answer tests. Use a different `counter` for every random decision of a step. -### `cunumpy/morton.cuh` and `morton_keys` +### `cunumpy/morton.cuh` and `algorithms.morton_keys` Morton (Z-order) keys: the bits of a point's integer cell coordinates, interleaved into one `uint64`. Sorted by key, nearby points are nearby in @@ -1913,12 +1942,12 @@ same box form a contiguous range, the starting point of tree builds on the GPU. ```python -keys = xp.morton_keys(positions, lower, upper, levels) # (n, 2|3) -> (n,) uint64 -keys, order, positions = xp.sort_by_key(keys, positions) +keys = xp.algorithms.morton_keys(positions, lower, upper, levels) # (n, 2|3) -> (n,) uint64 +keys, order, positions = xp.algorithms.sort_by_key(keys, positions) node = keys >> np.uint64(ndim * (levels - level)) # node index at `level` -cells = xp.morton_decode(node, ndim) # its integer coordinates -key = xp.morton_encode(ix, iy) # from integer cells -scales = xp.morton_scales(lower, upper, levels) # 2**levels / (upper - lower) +cells = xp.algorithms.morton_decode(node, ndim) # its integer coordinates +key = xp.algorithms.morton_encode(ix, iy) # from integer cells +scales = xp.algorithms.morton_scales(lower, upper, levels) # 2**levels / (upper - lower) ``` ```c @@ -1932,7 +1961,7 @@ unsigned long long cunumpy_morton_spread2(v); // and _compact2, _spread3, ``` `levels` is the number of bits per axis, at most 32 in 2D and 21 in 3D -(`xp.MAX_MORTON_LEVELS`). Axis 0 is the lowest bit of every group of `ndim` +(`xp.algorithms.MAX_MORTON_LEVELS`). Axis 0 is the lowest bit of every group of `ndim` bits; the top group is the child of the root. The cell along an axis is `floor((x - lower) * scale)` clipped to `[0, 2**levels - 1]`: points on a cell boundary go to the upper cell, points outside the box to the nearest face, and @@ -2001,10 +2030,10 @@ Sparse matrices assembled on the host move to the device once, with the constructor of the device type: `xp.scipy.sparse.csr_matrix(host_matrix)` on the CuPy backend copies a SciPy matrix; `matrix.get()` copies back. -## `fuse(function=None, *, kernel_name=None)` +## `kernels.fuse(function=None, *, kernel_name=None)` ```python -@xp.fuse +@xp.kernels.fuse def pressure(rho, T, gamma): return (gamma - 1.0) * rho * T ``` @@ -2019,11 +2048,11 @@ ufuncs, `xp.where`, and supported reductions as the last operation; no Python control flow on array values or indexing. Test the CuPy path: a function `cupy.fuse` cannot trace raises at its first call with CuPy arrays. -## `petsc_vec(array, comm=None)` +## `petsc.petsc_vec(array, comm=None)` ```python -b_vec = xp.petsc_vec(b) # b: NumPy or CuPy array, shared, never copied -x_vec = xp.petsc_vec(x) +b_vec = xp.petsc.petsc_vec(b) # b: NumPy or CuPy array, shared, never copied +x_vec = xp.petsc.petsc_vec(x) xp.synchronize() ksp.solve(b_vec, x_vec) # PETSc writes into x xp.synchronize() diff --git a/docs/source/best-practices.md b/docs/source/best-practices.md index 9ba9873..ede9dca 100644 --- a/docs/source/best-practices.md +++ b/docs/source/best-practices.md @@ -11,7 +11,7 @@ A condensed checklist. Each item links to the guide with the reasoning. `set_backend()`. Library code never calls `set_backend()`. ([Choosing a backend](guides/backends.md)) * Read back `xp.get_backend()` after requesting CuPy; log it with - `xp.device_count()` and `xp.__version__`. + `xp.cuda.device_count()` and `xp.__version__`. * Use `use_backend()` for scoped switches (tests, CPU reference computations), never from several threads at once. * Functions that receive arrays follow them with `get_array_module()`; functions @@ -64,7 +64,7 @@ A condensed checklist. Each item links to the guide with the reasoning. kernels](kernels/testing.md)) * One `assert_kernels_agree` test over `catalog.parity_cases()`. * `emulate_cuda_kernel` tests, so CPU-only CI checks the CUDA arithmetic. -* Seed `xp.random_streams` with `(seed, rank)` and draw only from it. +* Seed `xp.rng.random_streams` with `(seed, rank)` and draw only from it. * Debug crashes with `CUNUMPY_CUDA_DEBUG=1`, then `compute-sanitizer`. ([Debugging](kernels/debugging.md)) * Time with `timed_region()`, profile with `nvtx_range()` and `nsys`. diff --git a/docs/source/examples/particle-pusher.md b/docs/source/examples/particle-pusher.md index b1f3f9d..664b6c3 100644 --- a/docs/source/examples/particle-pusher.md +++ b/docs/source/examples/particle-pusher.md @@ -80,7 +80,7 @@ import cunumpy as xp OUTPUTS = {"push": (0,), "deposit": (1,), "accelerate": (1,)} -catalog = xp.KernelCatalog.from_package( +catalog = xp.kernels.KernelCatalog.from_package( __name__, missing_cuda="fallback", host_options=lambda name: {"outputs": OUTPUTS[name]}, @@ -145,7 +145,7 @@ xp.set_backend("cupy") print(catalog.summary()) # CUDA kernels: 0 of 3 (missing: accelerate, deposit, push) sim = Simulation() -with xp.count_transfers() as counter: +with xp.profiling.count_transfers() as counter: sim.step(0.1) print(counter.report()) ``` @@ -197,7 +197,7 @@ import numpy as np import pytest import cunumpy as xp -from cunumpy.testing import assert_kernels_agree +from cunumpy.kernel_testing import assert_kernels_agree from pic.kernels import catalog N, N_CELLS, LENGTH = 10_000, 64, 2 * np.pi @@ -274,7 +274,7 @@ Now the parity test covers all three, and the step should not transfer anything. Turn that into a test: ```python -from cunumpy.testing import requires_cupy +from cunumpy.kernel_testing import requires_cupy from pic.simulation import Simulation @@ -283,7 +283,7 @@ def test_step_stays_on_device(): with xp.use_backend("cupy"): sim = Simulation(n_particles=N) sim.step(0.1) # warm-up: compiles the kernels - with xp.assert_no_transfers(): + with xp.profiling.assert_no_transfers(): sim.step(0.1) ``` @@ -303,9 +303,9 @@ if xp.cupy_backend: catalog.compile_all(jobs=4) sim = Simulation(n_particles=1_000_000) -with xp.timed_region("100 steps") as timing: +with xp.profiling.timed_region("100 steps") as timing: for _ in range(100): - with xp.nvtx_range("step"): + with xp.profiling.nvtx_range("step"): sim.step(0.05) print(f"{timing.elapsed / 100 * 1e3:.2f} ms/step, field energy {sim.field_energy():.4e}") ``` diff --git a/docs/source/examples/particle_recipes.py b/docs/source/examples/particle_recipes.py index 52847eb..a8bd8df 100644 --- a/docs/source/examples/particle_recipes.py +++ b/docs/source/examples/particle_recipes.py @@ -46,7 +46,7 @@ def sort_by_cell(positions, lower, cell_size, n_cells): def deposit_nearest_cell(cell, weights, n_cells): """Sum the weights per cell without atomics: the sort-then-reduce deposit.""" - return xp.segment_sum(weights, cell, n_cells) + return xp.algorithms.segment_sum(weights, cell, n_cells) def pack_for_ranks(markers, destination, n_ranks): @@ -65,7 +65,7 @@ def exchange(comm, markers, destination): """Send every marker to its destination rank (``MPI_Alltoallv``); return the received ones. Works for host and device arrays, with or without CUDA-aware MPI - (``xp.mpi_buffer`` stages device buffers through the host when needed). + (``xp.mpi.mpi_buffer`` stages device buffers through the host when needed). """ n_ranks = comm.Get_size() width = markers.shape[1] @@ -79,8 +79,8 @@ def displacements(counts): return np.concatenate([[0], np.cumsum(counts)[:-1]]) with ( - xp.mpi_buffer(sendbuf) as send, - xp.mpi_buffer(received, send=False, recv=True) as recv, + xp.mpi.mpi_buffer(sendbuf) as send, + xp.mpi.mpi_buffer(received, send=False, recv=True) as recv, ): comm.Alltoallv( [send, send_counts * width, displacements(send_counts) * width, None], @@ -96,5 +96,5 @@ def thermal_velocities(seed, particle_ids, step, v_th): (up to the last bits of the math functions), whatever the order of the particles or the number of ranks. """ - z0, z1 = xp.philox_normal2(seed, particle_ids, step) + z0, z1 = xp.rng.philox_normal2(seed, particle_ids, step) return v_th * z0, v_th * z1 diff --git a/docs/source/examples/portable-script.md b/docs/source/examples/portable-script.md index 492947f..1e0c26b 100644 --- a/docs/source/examples/portable-script.md +++ b/docs/source/examples/portable-script.md @@ -51,13 +51,13 @@ def main(): args = parser.parse_args() xp.set_backend("cupy" if args.gpu else "numpy") - print(f"backend={xp.get_backend()} devices={xp.device_count()} cunumpy={xp.__version__}") + print(f"backend={xp.get_backend()} devices={xp.cuda.device_count()} cunumpy={xp.__version__}") dx = 1.0 / args.n dt = 0.2 * dx**2 # stable for the explicit scheme u = xp.to_cunumpy(initial_condition(args.n, seed=0)) # one host-to-device copy - with xp.timed_region("time loop") as timing: + with xp.profiling.timed_region("time loop") as timing: for step in range(1, args.steps + 1): u = u + dt * laplacian(u, dx) if step % args.every == 0: @@ -85,9 +85,9 @@ is also a quick check that both backends compute the same thing. ## Variations * **Check for transfers in a test.** Wrap a few steps in - `xp.assert_no_transfers()` to make sure nobody adds a `to_numpy()` to the loop + `xp.profiling.assert_no_transfers()` to make sure nobody adds a `to_numpy()` to the loop later. -* **Profile.** Mark the update with `xp.nvtx_range("update")` and run +* **Profile.** Mark the update with `xp.profiling.nvtx_range("update")` and run `nsys profile -t cuda,nvtx python heat.py --gpu`. * **Use it as a library.** `laplacian()` follows its input via `get_array_module()`, so other code can call it with NumPy or CuPy arrays no diff --git a/docs/source/guides/backends.md b/docs/source/guides/backends.md index a39a432..05ba108 100644 --- a/docs/source/guides/backends.md +++ b/docs/source/guides/backends.md @@ -52,7 +52,7 @@ handy in conditionals: ```python if xp.cupy_backend: - xp.bind_local_device() + xp.cuda.bind_local_device() ``` ## Switch temporarily @@ -109,7 +109,7 @@ uses the second (see [Writing backend-agnostic code](portable-code.md)). * **Use `use_backend()` for scoped work.** It is exception-safe and makes the scope obvious; a bare `set_backend()` in the middle of a function changes the state for everything that runs afterwards. -* **Log the effective backend.** Print `xp.get_backend()`, `xp.device_count()` +* **Log the effective backend.** Print `xp.get_backend()`, `xp.cuda.device_count()` and `xp.__version__` in the run's output so results can be traced to the hardware they ran on. diff --git a/docs/source/guides/data-movement.md b/docs/source/guides/data-movement.md index 09a65b5..9a87d21 100644 --- a/docs/source/guides/data-movement.md +++ b/docs/source/guides/data-movement.md @@ -63,7 +63,7 @@ time loop. `count_transfers()` records every copy made through CuNumpy in a block, with the file and line that caused it: ```python -with xp.count_transfers() as counter: +with xp.profiling.count_transfers() as counter: for _ in range(10): step(state, dt) @@ -93,7 +93,7 @@ report: ```python def test_step_stays_on_device(): state = make_state() - with xp.assert_no_transfers(): + with xp.profiling.assert_no_transfers(): step(state, dt) ``` @@ -139,12 +139,12 @@ pattern. ## Pinned memory Host-to-device copies from page-locked ("pinned") host memory are faster and -can overlap with computation on a stream. `xp.pin_memory(host_array)` returns a +can overlap with computation on a stream. `xp.cuda.pin_memory(host_array)` returns a pinned copy: ```python -pinned = xp.pin_memory(np.load("snapshot.npy")) -with xp.stream(): +pinned = xp.cuda.pin_memory(np.load("snapshot.npy")) +with xp.cuda.stream(): device = xp.to_cupy(pinned) ``` diff --git a/docs/source/guides/gpu-devices.md b/docs/source/guides/gpu-devices.md index 4a5f2f6..7cd0db8 100644 --- a/docs/source/guides/gpu-devices.md +++ b/docs/source/guides/gpu-devices.md @@ -21,8 +21,8 @@ On a multi-GPU workstation, pick the device before creating arrays: ```python xp.set_backend("cupy") -print("GPUs:", xp.device_count()) -xp.set_device(1) # arrays created from now on live on GPU 1 +print("GPUs:", xp.cuda.device_count()) +xp.cuda.set_device(1) # arrays created from now on live on GPU 1 values = xp.zeros(10**6) ``` @@ -39,7 +39,7 @@ For MPI programs with one rank per GPU, use `bind_local_device()` instead ## Memory ```python -free, total = xp.memory_info() +free, total = xp.cuda.memory_info() print(f"{free / 2**30:.1f} of {total / 2**30:.1f} GiB free") ``` @@ -55,7 +55,7 @@ another library or between phases with very different memory needs: ```python del large_temporary -xp.free_memory() +xp.cuda.free_memory() ``` It cannot free memory still referenced by live arrays. If memory keeps @@ -69,18 +69,18 @@ This is what makes GPUs fast, and it has three practical consequences: 1. Reading a value on the host (`to_numpy()`, `float()`, `print`) waits for all queued work that produces it. This happens automatically. 2. Timing with `time.perf_counter()` around a launch measures the launch, not - the work. Use `xp.timed_region()` (see [Timing and profiling](profiling.md)). + the work. Use `xp.profiling.timed_region()` (see [Timing and profiling](profiling.md)). 3. Libraries that read device memory without CuPy's knowledge, most importantly MPI, need an explicit `xp.synchronize()` (or - `xp.synchronize_for_mpi()`) before they access a buffer. + `xp.mpi.synchronize_for_mpi()`) before they access a buffer. ## Streams Work on one stream runs in order; work on different streams may overlap. -`xp.stream()` creates a non-blocking stream and makes it current for the block: +`xp.cuda.stream()` creates a non-blocking stream and makes it current for the block: ```python -with xp.stream() as s: +with xp.cuda.stream() as s: device = xp.to_cupy(pinned_host) # copy and compute queued on s result = xp.fft.fft(device) diff --git a/docs/source/guides/mpi.md b/docs/source/guides/mpi.md index c6211ea..f64d54a 100644 --- a/docs/source/guides/mpi.md +++ b/docs/source/guides/mpi.md @@ -10,17 +10,17 @@ CuNumpy provides one helper per step. import cunumpy as xp xp.set_backend("cupy") -xp.bind_local_device() # 1. pick this rank's GPU, before MPI_Init +xp.cuda.bind_local_device() # 1. pick this rank's GPU, before MPI_Init from mpi4py import MPI # 2. MPI_Init happens here -xp.require_cuda_aware_mpi() # 3. fail clearly if MPI cannot take GPU buffers +xp.mpi.require_cuda_aware_mpi() # 3. fail clearly if MPI cannot take GPU buffers comm = MPI.COMM_WORLD send = xp.full(1000, comm.rank, dtype=xp.float64) recv = xp.empty_like(send) -xp.synchronize_for_mpi(send, recv) # 4. before every MPI call on device buffers +xp.mpi.synchronize_for_mpi(send, recv) # 4. before every MPI call on device buffers comm.Sendrecv(send, dest=(comm.rank + 1) % comm.size, recvbuf=recv, source=(comm.rank - 1) % comm.size) ``` @@ -35,10 +35,10 @@ Without device selection every rank on a node uses GPU 0. A CUDA-aware MPI also binds to whatever device is current when `MPI_Init` runs, so the device must be chosen *before* `from mpi4py import MPI`. At that point MPI cannot be asked for the rank yet, but launchers export the node-local rank in environment -variables, which `xp.local_rank()` reads (Open MPI, MVAPICH2, Intel MPI/MPICH, +variables, which `xp.mpi.local_rank()` reads (Open MPI, MVAPICH2, Intel MPI/MPICH, PMI, Cray PALS, Slurm and `LOCAL_RANK`). -`xp.bind_local_device()` selects device `local_rank() % device_count()`, +`xp.cuda.bind_local_device()` selects device `local_rank() % device_count()`, creates its CUDA context, and returns the device id. If the launcher already restricts each rank to one GPU (`CUDA_VISIBLE_DEVICES` per rank, or Slurm's `--gpus-per-task=1`), every process sees one device and selects it. @@ -65,7 +65,7 @@ variant `mpi_is_cuda_aware(comm)` lets a program choose a fallback instead, such as staging buffers through the host: ```python -if xp.cupy_backend and not xp.mpi_is_cuda_aware(comm): +if xp.cupy_backend and not xp.mpi.mpi_is_cuda_aware(comm): stage_through_host = True ``` @@ -84,7 +84,7 @@ def exchange_halo(field, comm, left, right): # field has shape (nx + 2, ny): one ghost row on each side send_l, send_r = field[1], field[-2] recv_l, recv_r = xp.empty_like(send_l), xp.empty_like(send_r) - xp.synchronize_for_mpi(send_l, send_r) + xp.mpi.synchronize_for_mpi(send_l, send_r) comm.Sendrecv(send_l, dest=left, recvbuf=recv_r, source=right) comm.Sendrecv(send_r, dest=right, recvbuf=recv_l, source=left) field[0], field[-1] = recv_l, recv_r @@ -102,9 +102,9 @@ the NumPy backend) stages device buffers through the host. `mpi_buffer()` does the right thing for each case, so the MPI call is written once: ```python -xp.mpi_is_cuda_aware(comm) # once at startup; the answer is remembered +xp.mpi.mpi_is_cuda_aware(comm) # once at startup; the answer is remembered -with xp.mpi_buffer(send_r) as sendbuf, xp.mpi_buffer(recv_l, send=False, recv=True) as recvbuf: +with xp.mpi.mpi_buffer(send_r) as sendbuf, xp.mpi.mpi_buffer(recv_l, send=False, recv=True) as recvbuf: comm.Sendrecv(sendbuf, dest=right, recvbuf=recvbuf, source=left) ``` @@ -120,15 +120,15 @@ raises instead of guessing. ## Reproducible random numbers Each rank needs its own random stream, and a run is reproducible only if every -draw comes from a seeded generator. Seed `xp.random_streams` once, after MPI is +draw comes from a seeded generator. Seed `xp.rng.random_streams` once, after MPI is initialized, and draw from it everywhere: ```python -xp.random_streams.seed(config.seed, rank=comm.Get_rank()) +xp.rng.random_streams.seed(config.seed, rank=comm.Get_rank()) -positions = xp.random_streams.random((n, 3)) -velocities = xp.random_streams.normal(0.0, v_th, (n, 3)) -rng = xp.random_streams.generator() # for other distributions +positions = xp.rng.random_streams.random((n, 3)) +velocities = xp.rng.random_streams.normal(0.0, v_th, (n, 3)) +rng = xp.rng.random_streams.generator() # for other distributions ``` Rank `r` draws the stream `(seed, r)`; the same seed and number of ranks give @@ -149,9 +149,9 @@ srun --nodes=2 --ntasks-per-node=4 --gpus-per-task=1 python simulate.py --gpu Print the binding once at start-up to catch mapping errors early: ```python -device = xp.bind_local_device() +device = xp.cuda.bind_local_device() from mpi4py import MPI -print(f"rank {MPI.COMM_WORLD.rank}: local rank {xp.local_rank()}, device {device}") +print(f"rank {MPI.COMM_WORLD.rank}: local rank {xp.mpi.local_rank()}, device {device}") ``` ## Common failures diff --git a/docs/source/guides/particle-codes.md b/docs/source/guides/particle-codes.md index 660a25a..e3c468b 100644 --- a/docs/source/guides/particle-codes.md +++ b/docs/source/guides/particle-codes.md @@ -41,11 +41,11 @@ need: Apply `order` to every per-marker array (positions, velocities, weights, ids), e.g. by keeping them as columns of one `(n, k)` array. A stable sort keeps the result independent of how the markers were ordered before. -`xp.sort_by_key(keys, positions, velocities, weights)` does the argsort and +`xp.algorithms.sort_by_key(keys, positions, velocities, weights)` does the argsort and the reordering of several arrays in one call. For a tree code, or for better locality in 2D and 3D, sort by Morton key -(`xp.morton_keys`, see the API page) instead of by cell: the markers of every +(`xp.algorithms.morton_keys`, see the API page) instead of by cell: the markers of every quadtree or octree node are then a contiguous range of the sorted arrays. ## Deposit without atomics: sort, then reduce @@ -67,7 +67,7 @@ The two other GPU strategies, as kernels: varies between runs (results differ in the last bits). * **Per-block shared memory**: each block deposits into a copy of the grid in shared memory, then adds it to the global grid once per cell. Fast for small - grids. Check the size with `xp.max_shared_memory_per_block()` and pass + grids. Check the size with `xp.cuda.max_shared_memory_per_block()` and pass `shared_mem=` at the launch; above 48 KiB the kernel is set up for the larger limit automatically. @@ -86,7 +86,7 @@ void deposit(Array1D x, Array1D w, Array1D rho, } ``` -`cunumpy.testing.emulate_cuda_kernel(..., shared_mem=8 * nx)` runs such a +`cunumpy.kernel_testing.emulate_cuda_kernel(..., shared_mem=8 * nx)` runs such a kernel on the CPU, barriers included, so it can be checked against the host version without a GPU. @@ -104,7 +104,7 @@ group the markers by destination, exchange the counts, then the markers: :pyobject: exchange ``` -`xp.mpi_buffer` hands device arrays to MPI directly when it is CUDA-aware and +`xp.mpi.mpi_buffer` hands device arrays to MPI directly when it is CUDA-aware and stages them through host memory otherwise (see [MPI](mpi.md)). Only the counts are host arrays. Remove the markers that left with the compaction above, and append the received ones. @@ -133,13 +133,13 @@ v[i] = v_th * z0; Use a different counter for every random decision of a step (e.g. `4 * step + 0` for injection, `4 * step + 1` for collisions) so that they are independent. For draws that need not be per particle (e.g. a collision -operator's own sampling), `xp.random_streams` gives one seeded generator per +operator's own sampling), `xp.rng.random_streams` gives one seeded generator per rank. ## Write output without stalling the GPU ```python -staging = xp.HostStaging(rho.shape, rho.dtype) # once +staging = xp.memory.HostStaging(rho.shape, rho.dtype) # once pending = [] for step in range(n_steps): advance() @@ -189,4 +189,4 @@ raw.num_regs, raw.max_threads_per_block, raw.shared_size_bytes A kernel that uses many registers per thread cannot run 1024 threads per block; `max_threads_per_block` is the limit for this kernel on this device. Start with 128 or 256 threads per block, then time a few sizes on the target GPU -(`xp.timed_region`) for the kernels that dominate a step. +(`xp.profiling.timed_region`) for the kernels that dominate a step. diff --git a/docs/source/guides/portable-code.md b/docs/source/guides/portable-code.md index 02e04e2..7a942c4 100644 --- a/docs/source/guides/portable-code.md +++ b/docs/source/guides/portable-code.md @@ -82,11 +82,11 @@ CuPy, so using NumPy dtypes for declarations is fine. ## Random numbers -`xp.get_rng(seed)` returns a `Generator` of the active backend, so random data is +`xp.rng.get_rng(seed)` returns a `Generator` of the active backend, so random data is generated where it is used: ```python -rng = xp.get_rng(seed=42) +rng = xp.rng.get_rng(seed=42) velocities = rng.normal(0.0, 1.0, size=(n_particles, 3)) ``` @@ -149,13 +149,13 @@ def solve_banded(ab, b): ## Test on both backends -`cunumpy.testing.BACKENDS` parametrizes a test over NumPy and CuPy, skipping the +`cunumpy.kernel_testing.BACKENDS` parametrizes a test over NumPy and CuPy, skipping the CuPy case where there is no GPU, so the same test file runs on a laptop and in GPU CI: ```python import pytest -from cunumpy.testing import BACKENDS +from cunumpy.kernel_testing import BACKENDS import cunumpy as xp diff --git a/docs/source/guides/profiling.md b/docs/source/guides/profiling.md index ae594e9..7b9e4b3 100644 --- a/docs/source/guides/profiling.md +++ b/docs/source/guides/profiling.md @@ -8,7 +8,7 @@ them. CuNumpy's helpers handle both and run unchanged on the NumPy backend. ## Time a region: `timed_region` ```python -with xp.timed_region("field solve") as timing: +with xp.profiling.timed_region("field solve") as timing: solve(field) print(f"{timing.name}: {timing.elapsed * 1e3:.2f} ms (synced={timing.synced})") @@ -43,11 +43,11 @@ Timing tips: ## Mark regions for Nsight: `nvtx_range` ```python -with xp.nvtx_range("push markers"): +with xp.profiling.nvtx_range("push markers"): push(markers, dt, n_threads=n_markers) -@xp.nvtx_range("time step") +@xp.profiling.nvtx_range("time step") def step(state, dt): ... ``` @@ -73,7 +73,7 @@ slower than expected. `count_transfers()` lists every copy made through CuNumpy with its call site: ```python -with xp.count_transfers() as counter: +with xp.profiling.count_transfers() as counter: for _ in range(10): step(state, dt) print(counter.report()) diff --git a/docs/source/guides/solvers.md b/docs/source/guides/solvers.md index 058e7e3..f8f8290 100644 --- a/docs/source/guides/solvers.md +++ b/docs/source/guides/solvers.md @@ -3,8 +3,8 @@ Particle pushes and deposits are only part of a plasma code. Field solves need sparse matrices, iterative solvers and FFTs; fluid (MHD) updates are long chains of pointwise operations; diagnostics reduce over all cells or -particles. This page covers the three tools for that: `xp.scipy`, `xp.fuse` -and `xp.petsc_vec`, plus in-kernel reductions with `cunumpy/reduce.cuh`. +particles. This page covers the three tools for that: `xp.scipy`, `xp.kernels.fuse` +and `xp.petsc.petsc_vec`, plus in-kernel reductions with `cunumpy/reduce.cuh`. ## SciPy on both backends: `xp.scipy` @@ -53,21 +53,21 @@ def periodic_poisson(rho, length): (`erf`, `erfc`, Bessel functions `i0`, `i1`, `k0`, ...) are in `xp.scipy.special` on both backends. -## Fused elementwise updates: `xp.fuse` +## Fused elementwise updates: `xp.kernels.fuse` On the GPU, `(gamma - 1) * (E - 0.5 * rho * u**2)` runs as five kernels, each -reading and writing a full temporary array. `xp.fuse` turns the whole +reading and writing a full temporary array. `xp.kernels.fuse` turns the whole function into one kernel with `cupy.fuse` when it is called with CuPy arrays, and calls it unchanged with NumPy arrays: ```python -@xp.fuse +@xp.kernels.fuse def pressure(rho, mom, energy, gamma): u = mom / rho return (gamma - 1.0) * (energy - 0.5 * rho * u * u) -@xp.fuse +@xp.kernels.fuse def maxwellian(v, n, u, v_th): return n / (xp.sqrt(2.0 * np.pi) * v_th) * xp.exp(-0.5 * ((v - u) / v_th) ** 2) @@ -78,12 +78,12 @@ p = pressure(rho, mom, energy, 5.0 / 3.0) Memory-bound chains like these typically get several times faster. The function must be elementwise: arithmetic, comparisons, ufuncs (`xp.exp`, `xp.sqrt`, ...), `xp.where`; no `if` on array values, no indexing. Test the -CuPy path (`cunumpy.testing.BACKENDS`): `cupy.fuse` reports a function it +CuPy path (`cunumpy.kernel_testing.BACKENDS`): `cupy.fuse` reports a function it cannot trace at the first call with CuPy arrays. -## PETSc without copies: `xp.petsc_vec` +## PETSc without copies: `xp.petsc.petsc_vec` -When the field solve goes through PETSc, `xp.petsc_vec(array)` wraps an +When the field solve goes through PETSc, `xp.petsc.petsc_vec(array)` wraps an array as a PETSc vector that uses the array's memory, a CUDA vector for a CuPy array. The deposit writes into `rho`, PETSc reads it and writes `phi`, the gather reads `phi`, and no data leaves the GPU: @@ -93,7 +93,7 @@ from petsc4py import PETSc rho = xp.zeros(n_local) phi = xp.zeros(n_local) -rho_vec, phi_vec = xp.petsc_vec(rho), xp.petsc_vec(phi) +rho_vec, phi_vec = xp.petsc.petsc_vec(rho), xp.petsc.petsc_vec(phi) A = assemble_laplacian() # a PETSc Mat A.setType("aijcusparse") # keep the matrix on the GPU too diff --git a/docs/source/installation.md b/docs/source/installation.md index 3d2a3fe..9bd2b42 100644 --- a/docs/source/installation.md +++ b/docs/source/installation.md @@ -30,7 +30,7 @@ use it: import cunumpy as xp print("CuPy usable:", xp.cupy_available()) -print("visible GPUs:", xp.device_count()) +print("visible GPUs:", xp.cuda.device_count()) xp.set_backend("cupy") print("active backend:", xp.get_backend()) # 'cupy' if the GPU works @@ -43,7 +43,7 @@ is requested. See [Troubleshooting](troubleshooting.md) for the usual causes. | Extra | Installs | Use it for | | --- | --- | --- | -| `cunumpy[test]` | `pytest`, `coverage` | running the test suite, using `cunumpy.testing` | +| `cunumpy[test]` | `pytest`, `coverage` | running the test suite, using `cunumpy.kernel_testing` | | `cunumpy[test-compiled]` | the above plus `pyccel` | tests that compile host kernels with Pyccel | | `cunumpy[docs]` | Sphinx, MyST, the book theme | building this documentation | | `cunumpy[dev]` | all of the above plus formatters | developing CuNumpy itself | @@ -71,7 +71,7 @@ Tests that need a GPU are skipped automatically where CuPy is not functional. | `CUNUMPY_CUDA_DEBUG=1` | enable [CUDA debug mode](kernels/debugging.md) for all kernels | MPI launchers also export node-local rank variables (`OMPI_COMM_WORLD_LOCAL_RANK`, -`SLURM_LOCALID`, ...), which `xp.local_rank()` reads to pick a GPU per process. +`SLURM_LOCALID`, ...), which `xp.mpi.local_rank()` reads to pick a GPU per process. ## Build the documentation diff --git a/docs/source/kernels/accumulation.md b/docs/source/kernels/accumulation.md index f4135b9..8ebcc0b 100644 --- a/docs/source/kernels/accumulation.md +++ b/docs/source/kernels/accumulation.md @@ -35,7 +35,7 @@ compare-and-swap loop. ## `DeviceMirror` ```python -mirror = xp.DeviceMirror(host_array) +mirror = xp.memory.DeviceMirror(host_array) ``` | Member | CuPy backend | NumPy backend | @@ -79,11 +79,11 @@ def deposit_host(x, w, rho, n_particles, n_cells, dx): np.add.at(rho, cells[inside], w[:n_particles][inside] / dx) -deposit = xp.Kernel(deposit_host, xp.CudaKernel(DEPOSIT, "deposit")) +deposit = xp.kernels.Kernel(deposit_host, xp.cuda.CudaKernel(DEPOSIT, "deposit")) # owned by a host library, e.g. a distributed vector rho_host = np.zeros(64) -rho = xp.DeviceMirror(rho_host) +rho = xp.memory.DeviceMirror(rho_host) rng = np.random.default_rng(0) x = xp.to_cunumpy(rng.uniform(0.0, 1.0, 10_000)) @@ -103,13 +103,13 @@ host kernel writes into it directly, and `to_host()` does nothing. The alternative to atomics: bin the particles (sort or compute a cell key per particle), compute each particle's contribution into an array, and sum per -cell. The last step is `xp.segment_sum(values, keys, n_segments)`, on either +cell. The last step is `xp.algorithms.segment_sum(values, keys, n_segments)`, on either backend: ```python cell = ix + nx * (iy + ny * iz) # (n_particles,), -1 for outside weights = compute_weights(markers) # (n_particles, 8), one per corner -rho_cells = xp.segment_sum(weights, cell, nx * ny * nz) # (n_cells, 8) +rho_cells = xp.algorithms.segment_sum(weights, cell, nx * ny * nz) # (n_cells, 8) ``` Negative keys drop the value; a 2D `values` is summed column by column. Measure diff --git a/docs/source/kernels/arguments.md b/docs/source/kernels/arguments.md index ce3e163..2463f4b 100644 --- a/docs/source/kernels/arguments.md +++ b/docs/source/kernels/arguments.md @@ -28,7 +28,7 @@ import numpy as np import cunumpy as xp -class DeviceParticles(xp.CudaArguments): +class DeviceParticles(xp.cuda.CudaArguments): def __init__(self, positions, velocities): self.positions = xp.as_device_array(positions, np.float64, ndim=2, name="positions") self.velocities = xp.as_device_array(velocities, np.float64, ndim=2, name="velocities") @@ -39,7 +39,7 @@ PUSH = r""" extern "C" __global__ void push(double dt, double* x, const double* v, int n) { ... } """ -push = xp.CudaKernel(PUSH, "push") +push = xp.cuda.CudaKernel(PUSH, "push") particles = DeviceParticles(x, v) push(0.1, particles, n_threads=particles.positions.shape[0]) # -> push(0.1, x, v, n) ``` @@ -62,7 +62,7 @@ class MarkerArguments: # the host argument class, e.g. compiled with Pyccel self.n_markers = n_markers -class ParticleArguments(xp.KernelArguments): +class ParticleArguments(xp.kernels.KernelArguments): def __init__(self, particles): self._particles = particles self._host = None @@ -93,7 +93,7 @@ push(args, dt, n_threads=n_markers) # a Kernel: same call on both backends * **Invalidate the cache** when the underlying arrays are replaced (resizing, `deepcopy`, unpickling): reset the stored forms or create a new arguments object. Stale cached arguments point at the old arrays. -* `xp.resolve_host_args(args, kwargs)` applies the host replacement, for code +* `xp.kernels.resolve_host_args(args, kwargs)` applies the host replacement, for code that calls host kernels without `Kernel`. ## `CudaStruct`: one C struct, defined once @@ -103,7 +103,7 @@ source, its exact memory layout, and a packer for values. The kernel takes the struct by value as one parameter: ```python -Particles = xp.CudaStruct( +Particles = xp.cuda.CudaStruct( "Particles", [("x", "double*"), ("v", "double*"), ("n", "long long"), ("charge", "double")], ) @@ -114,7 +114,7 @@ extern "C" __global__ void push(Particles p, double dt) { if (i < p.n) p.x[i] += dt * p.charge * p.v[i]; } """ -push = xp.CudaKernel(PUSH, "push", structs=[Particles]) +push = xp.cuda.CudaKernel(PUSH, "push", structs=[Particles]) value = Particles(x=x, v=v, n=x.size, charge=-1.0) push(value, 0.1, n_threads=x.size) @@ -143,7 +143,7 @@ species, domain or grid, and kept next to the host argument object), subclass field values as attributes and is passed to kernels as it is: ```python -class CudaMarkerArguments(xp.CudaStructArguments): +class CudaMarkerArguments(xp.cuda.CudaStructArguments): struct_name = "MarkerArgs" fields = ( ("markers", "double*"), @@ -161,8 +161,8 @@ class CudaMarkerArguments(xp.CudaStructArguments): self.pack() -xp.write_cuda_header("kernels/marker_args.cuh", [CudaMarkerArguments.struct]) -push = xp.CudaKernel.from_file("kernels/push_cuda.cu", structs=[CudaMarkerArguments.struct]) +xp.cuda.write_cuda_header("kernels/marker_args.cuh", [CudaMarkerArguments.struct]) +push = xp.cuda.CudaKernel.from_file("kernels/push_cuda.cu", structs=[CudaMarkerArguments.struct]) args = CudaMarkerArguments(markers, valid_mks, weight_idx=6) push(args, dt, n_threads=args.n_markers) @@ -178,7 +178,7 @@ push(args, dt, n_threads=args.n_markers) array: ```python - class CudaMarkerArguments(xp.CudaStructArguments): + class CudaMarkerArguments(xp.cuda.CudaStructArguments): struct_name = "MarkerArgs" fields = (("markers", "Array2D"), ("n_markers", "int")) @@ -215,7 +215,7 @@ path: from my_sim.kernel_arguments import pusher_args_kernels # compiled by pyccel -class MarkerArguments(xp.PyccelStructArguments): +class MarkerArguments(xp.kernels.PyccelStructArguments): struct_name = "MarkerArgs" fields = (("markers", "Array2D"), ("Np", "long long"), ("n_markers", "int")) host_class = pusher_args_kernels.MarkerArguments @@ -271,7 +271,7 @@ class MarkerArguments: ... -MarkerArgs = xp.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs") +MarkerArgs = xp.cuda.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs") print(MarkerArgs.declaration) ``` @@ -297,7 +297,7 @@ the parameter is stored in (`self.first_init_idx = first_pusher_idx` gives a field `first_init_idx`), and skips the parameters in `exclude=`: ```python -MarkerArgs = xp.CudaStruct.from_pyccel_class( +MarkerArgs = xp.cuda.CudaStruct.from_pyccel_class( "my_sim/kernel_arguments/pusher_args_kernels.py", "MarkerArguments", "MarkerArgs" ) ``` @@ -321,7 +321,7 @@ Kernels in `.cu` files include the struct from a header. Generate it from the Python definition and commit it: ```python -xp.write_cuda_header("kernels/marker_args.cuh", [MarkerArgs, DomainArgs]) +xp.cuda.write_cuda_header("kernels/marker_args.cuh", [MarkerArgs, DomainArgs]) ``` The header gets an include guard (`MARKER_ARGS_CUH`), the `array_view.cuh` @@ -333,7 +333,7 @@ from pathlib import Path def test_marker_args_header_is_up_to_date(tmp_path): - generated = xp.write_cuda_header(tmp_path / "marker_args.cuh", [MarkerArgs, DomainArgs]) + generated = xp.cuda.write_cuda_header(tmp_path / "marker_args.cuh", [MarkerArgs, DomainArgs]) assert Path("kernels/marker_args.cuh").read_text() == generated ``` diff --git a/docs/source/kernels/cuda-kernel.md b/docs/source/kernels/cuda-kernel.md index 7687f97..fa444f8 100644 --- a/docs/source/kernels/cuda-kernel.md +++ b/docs/source/kernels/cuda-kernel.md @@ -19,7 +19,7 @@ void axpy(double a, const double* x, double* y, int n) { } """ -axpy = xp.CudaKernel(AXPY, "axpy") +axpy = xp.cuda.CudaKernel(AXPY, "axpy") xp.set_backend("cupy") x = xp.arange(10_000, dtype=xp.float64) @@ -59,7 +59,7 @@ every call: C types map to NumPy dtypes as on 64-bit Linux: `int` is `int32`, `long` and `long long` are `int64`, `float` is `float32`, `double` is `float64`. -`xp.ctype_of(np.float64)` returns `"double"`, useful when generating source. +`xp.cuda.ctype_of(np.float64)` returns `"double"`, useful when generating source. A wrong argument count, dtype or layout raises before anything is launched, with the parameter name in the message: @@ -115,7 +115,7 @@ extern "C" __global__ void block_sum(const double* x, double* out, int n) { if (threadIdx.x == 0) out[blockIdx.x] = buffer[0]; } """ -block_sum = xp.CudaKernel(BLOCK_SUM, "block_sum", block_size=128) +block_sum = xp.cuda.CudaKernel(BLOCK_SUM, "block_sum", block_size=128) (n_blocks,), (threads,) = block_sum.launch_shape(x.size) partial = xp.zeros(n_blocks) block_sum(x, partial, x.size, n_threads=x.size, shared_mem=threads * 8) @@ -128,7 +128,7 @@ Keeping CUDA source in `.cu` files gives editor support and lets kernels share headers: ```python -push = xp.CudaKernel.from_file("kernels/push/push_cuda.cu") # kernel name "push" +push = xp.cuda.CudaKernel.from_file("kernels/push/push_cuda.cu") # kernel name "push" ``` `from_file` derives the kernel name from the file name minus the `_cuda.cu` @@ -139,12 +139,12 @@ A file with several small kernels is loaded at once with `all_from_file`, which returns a dict by name; the kernels share one compilation: ```python -ops = xp.CudaKernel.all_from_file("kernels/vector_ops.cu", block_size=256) +ops = xp.cuda.CudaKernel.all_from_file("kernels/vector_ops.cu", block_size=256) ops["scale"](x, 2.0, x.size, n_threads=x.size) ops["shift"](x, 1.0, x.size, n_threads=x.size) ``` -`xp.cuda_kernel_names(source)` lists the `__global__` functions of a source. +`xp.cuda.cuda_kernel_names(source)` lists the `__global__` functions of a source. ## Headers and the compile cache @@ -156,10 +156,10 @@ Pass extra include directories with `include_dirs=[...]` and NVRTC flags with | `` | `CUNUMPY_THREAD_1D(i, n)`, `_2D`, `_3D`, `CUNUMPY_GRID_STRIDE_1D(i, n)` | | `` | strided views `Array1D` to `Array4D` | | `` | `cunumpy_atomic_add` and indexed 2D/3D variants, see [Accumulation kernels](accumulation.md) | -| `` | Morton (Z-order) keys `cunumpy_morton_key2(x, y, ...)`, `_key3`, equal to `xp.morton_keys` on the host | -| `` | counter-based random numbers `cunumpy_uniform(seed, stream, counter)`, `cunumpy_normal2(...)`, equal to `xp.philox_uniform` on the host | +| `` | Morton (Z-order) keys `cunumpy_morton_key2(x, y, ...)`, `_key3`, equal to `xp.algorithms.morton_keys` on the host | +| `` | counter-based random numbers `cunumpy_uniform(seed, stream, counter)`, `cunumpy_normal2(...)`, equal to `xp.rng.philox_uniform` on the host | -`xp.cuda_include_dir()` returns their directory for use with other compilers. +`xp.cuda.cuda_include_dir()` returns their directory for use with other compilers. CuPy's disk cache is keyed on the source string and the options only, so editing an included header would normally *not* trigger a recompile. @@ -187,7 +187,7 @@ void scale_column(Array2D a, long long column, double factor) { a(i, column) *= factor; } """ -scale_column = xp.CudaKernel(SCALE_COLUMN, "scale_column") +scale_column = xp.cuda.CudaKernel(SCALE_COLUMN, "scale_column") markers = xp.zeros((1000, 7)) view = markers[::2, 1:5] # non-contiguous view is fine @@ -224,8 +224,8 @@ __global__ void scale(T* x, T factor, int n) { if (i < n) x[i] *= factor; } """ -scale_f64 = xp.CudaKernel(SCALE, "scale", template_args=(np.float64,)) -scale_f32 = xp.CudaKernel(SCALE, "scale", template_args=(np.float32,)) +scale_f64 = xp.cuda.CudaKernel(SCALE, "scale", template_args=(np.float64,)) +scale_f32 = xp.cuda.CudaKernel(SCALE, "scale", template_args=(np.float32,)) ``` When the source itself is generated per variant (unrolled loops per dimension, @@ -234,10 +234,10 @@ on first use and caches it: ```python def make_matvec(ndim, dtype): - return xp.CudaKernel(generate_source(ndim, xp.ctype_of(dtype)), "matvec") + return xp.cuda.CudaKernel(generate_source(ndim, xp.cuda.ctype_of(dtype)), "matvec") -matvec = xp.CudaKernelVariants(make_matvec) +matvec = xp.cuda.CudaKernelVariants(make_matvec) matvec.get(3, np.float64)(mat, x, out, n_threads=out.size) matvec.compile_all([(3, np.float64), (3, np.complex128)], jobs=4) # at setup ``` diff --git a/docs/source/kernels/debugging.md b/docs/source/kernels/debugging.md index 1b702ff..16e363b 100644 --- a/docs/source/kernels/debugging.md +++ b/docs/source/kernels/debugging.md @@ -16,10 +16,10 @@ CUNUMPY_CUDA_DEBUG=1 python simulate.py # whole process ``` ```python -xp.set_cuda_debug(True) # globally, from now on -with xp.cuda_debug(): # for a block +xp.cuda.set_cuda_debug(True) # globally, from now on +with xp.cuda.cuda_debug(): # for a block ... -xp.CudaKernel(src, "push", debug=True) # one kernel, regardless of the global setting +xp.cuda.CudaKernel(src, "push", debug=True) # one kernel, regardless of the global setting ``` In debug mode a `CudaKernel`: @@ -36,8 +36,8 @@ In debug mode a `CudaKernel`: launched, so synchronize after `graph.launch()` to see them. ```python -with xp.cuda_debug(): - push = xp.CudaKernel(SOURCE, "push") +with xp.cuda.cuda_debug(): + push = xp.cuda.CudaKernel(SOURCE, "push") push(markers, dt, n, n_threads=n) # RuntimeError: CUDA error after launching kernel 'push' with grid (79,) and block (128,): ... ``` diff --git a/docs/source/kernels/dispatch.md b/docs/source/kernels/dispatch.md index b1c3463..1bbec2a 100644 --- a/docs/source/kernels/dispatch.md +++ b/docs/source/kernels/dispatch.md @@ -24,7 +24,7 @@ def axpy_host(a, x, y, n): y[i] += a * x[i] -axpy = xp.Kernel(axpy_host, xp.CudaKernel(AXPY, "axpy"), name="axpy") +axpy = xp.kernels.Kernel(axpy_host, xp.cuda.CudaKernel(AXPY, "axpy"), name="axpy") axpy(2.0, x, y, x.size, n_threads=x.size) ``` @@ -66,7 +66,7 @@ middle of a run. ```text my_sim/kernels/ -├── __init__.py # catalog = xp.KernelCatalog.from_package(__name__) +├── __init__.py # catalog = xp.kernels.KernelCatalog.from_package(__name__) ├── push/ │ ├── push_kernels.py # def push(...): ... host kernel │ └── push_cuda.cu # __global__ void push(...) CUDA kernel @@ -81,7 +81,7 @@ my_sim/kernels/ # my_sim/kernels/__init__.py import cunumpy as xp -catalog = xp.KernelCatalog.from_package(__name__, missing_cuda="fallback") +catalog = xp.kernels.KernelCatalog.from_package(__name__, missing_cuda="fallback") ``` ```python @@ -106,7 +106,7 @@ Options: ```python OUTPUTS = {"push": (0,), "deposit": (2,)} - catalog = xp.KernelCatalog.from_package( + catalog = xp.kernels.KernelCatalog.from_package( __name__, host_options=lambda name: {"outputs": OUTPUTS.get(name)}, ) @@ -132,7 +132,7 @@ where it is written: ```text my_sim/kernels/push/ -├── __init__.py # kernel = xp.Kernel.from_folder(__name__, ...) +├── __init__.py # kernel = xp.kernels.Kernel.from_folder(__name__, ...) ├── push_pyccel.py # "pyccel" (compiled with compile_host) and "python" (as is) ├── push_numba.py # "numba": def push(...) decorated with numba.njit ├── push_numpy.py # "numpy": vectorized @@ -143,7 +143,7 @@ my_sim/kernels/push/ # my_sim/kernels/push/__init__.py import cunumpy as xp -kernel = xp.Kernel.from_folder( +kernel = xp.kernels.Kernel.from_folder( __name__, host_suffix="_pyccel", compile_host=compile_kernels, # see below @@ -172,10 +172,10 @@ none is. Device arrays run the CUDA kernel. To choose, use the same pattern as for the array backend: ```python -xp.set_kernel_implementation("numpy") # like xp.set_backend -with xp.use_kernel_implementation("numba"): # like xp.use_backend +xp.kernels.set_kernel_implementation("numpy") # like xp.set_backend +with xp.kernels.use_kernel_implementation("numba"): # like xp.use_backend push(positions, velocities, dt) -xp.set_kernel_implementation(None) # back to the default +xp.kernels.set_kernel_implementation(None) # back to the default ``` or `CUNUMPY_KERNEL_IMPLEMENTATION=numpy` for a whole run (read at import, like @@ -205,7 +205,7 @@ def compile_kernels(module): return pyccel.epyccel(module, language="c") # add a cache in real code -catalog = xp.KernelCatalog.from_package( +catalog = xp.kernels.KernelCatalog.from_package( __name__, host_suffix="_pyccel", # push/push_pyccel.py next to push/push_cuda.cu compile_host=compile_kernels, @@ -217,7 +217,7 @@ catalog = xp.KernelCatalog.from_package( host implementations" above); `catalog["push"].host_kernel.kernel.available("pyccel")` reports whether the compiled version builds, and `catalog["push"].selected()` which version runs. To test the path of a machine without Pyccel, run the code -inside `with xp.use_kernel_implementation("numpy"):` (or set +inside `with xp.kernels.use_kernel_implementation("numpy"):` (or set `CUNUMPY_KERNEL_IMPLEMENTATION=numpy` for a whole run). Note that `epyccel` compiles again on every call; a code that compiles at run time usually keeps the builds in an on-disk cache keyed on the module source, so that only the first run after an edit compiles. @@ -234,7 +234,7 @@ for MPI, a path that has no GPU version yet. With `dispatch="arrays"` such calls run the host kernel: ```python -catalog = xp.KernelCatalog.from_package(__name__, dispatch="arrays") +catalog = xp.kernels.KernelCatalog.from_package(__name__, dispatch="arrays") catalog["gather"](positions, field, result, n_threads=n) # CUDA if positions are CuPy catalog["gather"](host_positions, host_field, host_result) # host kernel, also on CuPy @@ -246,16 +246,16 @@ a device-only argument object (`CudaArguments`, a struct value). A arguments go to the host function directly, without conversion. The arguments of one call must then all be on one side, with the dtype and -layout the kernels take. `xp.as_kernel_array(value, like, dtype)` brings an +layout the kernels take. `xp.kernels.as_kernel_array(value, like, dtype)` brings an input to the side of the main array `like` (no copy if it already fits), and -`xp.kernel_output(out, like, dtype)` gives the buffer for an output, copied +`xp.kernels.kernel_output(out, like, dtype)` gives the buffer for an output, copied back into `out` after the block if it had to be converted: ```python from functools import partial -convert = partial(xp.as_kernel_array, like=field, dtype=float) -with xp.kernel_output(result, like=field, dtype=float) as buffer: +convert = partial(xp.kernels.as_kernel_array, like=field, dtype=float) +with xp.kernels.kernel_output(result, like=field, dtype=float) as buffer: catalog["gather"](convert(positions), convert(field), buffer, n_threads=n) ``` diff --git a/docs/source/kernels/overview.md b/docs/source/kernels/overview.md index 3190b9b..775bdac 100644 --- a/docs/source/kernels/overview.md +++ b/docs/source/kernels/overview.md @@ -20,7 +20,7 @@ unchanged. | `KernelCatalog` | all `Kernel`s of a package, found by folder convention | [Pairing host and CUDA kernels](dispatch.md) | | `CudaArguments`, `KernelArguments`, `CudaStruct` | pass a group of arrays and scalars as one argument | [Kernel arguments and structs](arguments.md) | | `DeviceMirror`, `cunumpy/atomic.cuh` | scatter-add into a buffer owned by a host library | [Accumulation kernels](accumulation.md) | -| `cunumpy.testing` | check that host and CUDA kernels compute the same | [Testing kernels](testing.md) | +| `cunumpy.kernel_testing` | check that host and CUDA kernels compute the same | [Testing kernels](testing.md) | ## Which one do I need? diff --git a/docs/source/kernels/pyccel-kernel.md b/docs/source/kernels/pyccel-kernel.md index 8a9f319..8ce73dc 100644 --- a/docs/source/kernels/pyccel-kernel.md +++ b/docs/source/kernels/pyccel-kernel.md @@ -16,7 +16,7 @@ def smooth(field, out): out[0], out[-1] = field[0], field[-1] -smooth_kernel = xp.PyccelKernel(smooth, outputs=(1,)) +smooth_kernel = xp.kernels.PyccelKernel(smooth, outputs=(1,)) ``` On every call it decides whether conversion is needed (the active backend is @@ -44,10 +44,10 @@ copies back *every* converted array, which is correct but doubles the transfers. `outputs` lists the arguments that may be written: ```python -xp.PyccelKernel(smooth, outputs=(1,)) # positional argument 1 -xp.PyccelKernel(update, outputs=(0, -1)) # first and last argument -xp.PyccelKernel(solve, outputs=("out",)) # solve(a, b, out=out) -xp.PyccelKernel(norm, outputs=()) # writes nothing +xp.kernels.PyccelKernel(smooth, outputs=(1,)) # positional argument 1 +xp.kernels.PyccelKernel(update, outputs=(0, -1)) # first and last argument +xp.kernels.PyccelKernel(solve, outputs=("out",)) # solve(a, b, out=out) +xp.kernels.PyccelKernel(norm, outputs=()) # writes nothing ``` * Positional arguments are declared by index (negative indices count from the @@ -66,7 +66,7 @@ works. Instances of your own classes are traversed only if their class's module starts with one of the `object_modules` prefixes: ```python -kernel = xp.PyccelKernel(push, object_modules=("my_simulation.",), outputs=(0,)) +kernel = xp.kernels.PyccelKernel(push, object_modules=("my_simulation.",), outputs=(0,)) kernel(particles, dt) # particles.positions etc. are converted ``` @@ -94,7 +94,7 @@ time. `count_transfers()` records one `kernel_conversion` event per converted call, naming the kernel and the number of arrays: ```python -with xp.count_transfers() as counter: +with xp.profiling.count_transfers() as counter: step(state, dt) for event in counter.kernel_conversion_calls: print(event.where, event.description) diff --git a/docs/source/kernels/testing.md b/docs/source/kernels/testing.md index d7ff0e3..25b2426 100644 --- a/docs/source/kernels/testing.md +++ b/docs/source/kernels/testing.md @@ -1,12 +1,12 @@ # Testing kernels A GPU port is only as trustworthy as its comparison with the CPU version. -`cunumpy.testing` provides pytest helpers for exactly that, designed so that the +`cunumpy.kernel_testing` provides pytest helpers for exactly that, designed so that the same test suite runs on a laptop without a GPU (GPU cases are skipped) and on a GPU runner (everything runs). ```python -from cunumpy.testing import ( +from cunumpy.kernel_testing import ( BACKENDS, assert_kernels_agree, backend, @@ -15,7 +15,7 @@ from cunumpy.testing import ( ) ``` -`cunumpy.testing` is not imported by `import cunumpy`, and it imports pytest only +`cunumpy.kernel_testing` is not imported by `import cunumpy`, and it imports pytest only when one of its pytest objects is used. ## Run a test on both backends @@ -26,7 +26,7 @@ when one of its pytest objects is used. import pytest import cunumpy as xp -from cunumpy.testing import BACKENDS +from cunumpy.kernel_testing import BACKENDS @pytest.mark.parametrize("backend", BACKENDS) @@ -40,7 +40,7 @@ whole test. Import it into `conftest.py` to make it available everywhere: ```python # conftest.py -from cunumpy.testing import backend # noqa: F401 +from cunumpy.kernel_testing import backend # noqa: F401 ``` ```python @@ -55,7 +55,7 @@ def test_energy_is_conserved(backend): `requires_cupy` is a plain skip marker for GPU-only tests: ```python -from cunumpy.testing import requires_cupy +from cunumpy.kernel_testing import requires_cupy @requires_cupy @@ -70,7 +70,7 @@ def test_kernel_compiles(): import numpy as np import cunumpy as xp -from cunumpy.testing import assert_kernels_agree +from cunumpy.kernel_testing import assert_kernels_agree def make_axpy_args(backend, seed): @@ -150,7 +150,7 @@ def make_args(backend, seed): test asks for them), and the parity test of the whole package becomes ```python -from cunumpy.testing import check_parity, parity_cases +from cunumpy.kernel_testing import check_parity, parity_cases @pytest.mark.parametrize("kernel", parity_cases(catalog)) @@ -176,7 +176,7 @@ Compare it with the host kernel: import numpy as np import pytest -from cunumpy.testing import emulate_cuda_kernel, emulation_compiler +from cunumpy.kernel_testing import emulate_cuda_kernel, emulation_compiler from my_sim.kernels import catalog @@ -219,7 +219,7 @@ from pathlib import Path import numpy as np import cunumpy as xp -from cunumpy.testing import device_function_kernel, requires_cupy +from cunumpy.kernel_testing import device_function_kernel, requires_cupy BSPLINES = Path("my_sim/kernels/common/bsplines.cuh").read_text() @@ -273,12 +273,12 @@ them), CuPy functions reject NumPy arrays and lists, mixing the two raises, reductions return 0-d arrays, and arrays have `data.ptr`, `device` and `__cuda_array_interface__`. Kernels cannot run: `RawKernel` raises `NotImplementedError`, `requires_cupy` skips and `assert_kernels_agree` -skips while the fake is active (`cunumpy.testing.fake_cupy_active()`). It +skips while the fake is active (`cunumpy.kernel_testing.fake_cupy_active()`). It can also be installed from code, before the first backend use: ```python # conftest.py -from cunumpy.testing import install_fake_cupy +from cunumpy.kernel_testing import install_fake_cupy install_fake_cupy() ``` @@ -296,7 +296,7 @@ def test_time_step_has_no_transfers(): with xp.use_backend("cupy"): state = make_state() step(state, 1e-3) # warm-up: compilation, allocations - with xp.assert_no_transfers(): + with xp.profiling.assert_no_transfers(): step(state, 1e-3) ``` diff --git a/docs/source/pyodide.md b/docs/source/pyodide.md index cab17fb..ae51d32 100644 --- a/docs/source/pyodide.md +++ b/docs/source/pyodide.md @@ -25,7 +25,7 @@ await pyodide.runPythonAsync(` return values values = xp.array([1.0, 2.0, 3.0]) - result = xp.PyccelKernel(scale)(values, 2.0) + result = xp.kernels.PyccelKernel(scale)(values, 2.0) assert result is values print(xp.to_numpy(result)) # [2. 4. 6.] `); diff --git a/docs/source/quickstart.md b/docs/source/quickstart.md index 3cd2e99..48e74e4 100644 --- a/docs/source/quickstart.md +++ b/docs/source/quickstart.md @@ -65,7 +65,7 @@ Keep the arrays on the GPU for the whole computation and transfer once. A test can check that a time step makes no transfer at all: ```python -with xp.assert_no_transfers(): +with xp.profiling.assert_no_transfers(): step(state, dt) ``` @@ -91,7 +91,7 @@ def axpy(a, x, y, n): # the host version y[:n] += a * x[:n] -kernel = xp.Kernel(axpy, xp.CudaKernel(AXPY, "axpy")) +kernel = xp.kernels.Kernel(axpy, xp.cuda.CudaKernel(AXPY, "axpy")) x = xp.arange(1000, dtype=xp.float64) y = xp.zeros(1000) diff --git a/docs/source/troubleshooting.md b/docs/source/troubleshooting.md index afbfa10..9abc881 100644 --- a/docs/source/troubleshooting.md +++ b/docs/source/troubleshooting.md @@ -26,14 +26,14 @@ or data loaded from disk without `to_cunumpy()`), or check inputs with `xp.assert_same_backend()` at function entry. **The GPU version is slower than the CPU version.** Usually transfers in the -loop or host synchronization. Run a step inside `xp.count_transfers()` and read +loop or host synchronization. Run a step inside `xp.profiling.count_transfers()` and read the report; look for `float()`, `.item()`, `print()` or `if` on device values; profile with `nsys` ([Timing and profiling](guides/profiling.md)). Also check that the problem is large enough: GPUs need many thousands of elements per operation to pay off. **GPU memory looks full although arrays were deleted.** CuPy's memory pool -keeps freed blocks for reuse. `xp.free_memory()` returns them to the driver. +keeps freed blocks for reuse. `xp.cuda.free_memory()` returns them to the driver. Memory that remains in use is referenced by live arrays. ## Kernels @@ -95,14 +95,14 @@ load from CuPy's disk cache. ## MPI -**All ranks use GPU 0.** Call `xp.bind_local_device()` before `from mpi4py +**All ranks use GPU 0.** Call `xp.cuda.bind_local_device()` before `from mpi4py import MPI`. **Segfault in the first MPI call with a CuPy array.** The MPI library is not -CUDA-aware. `xp.require_cuda_aware_mpi()` at start-up gives a clear message; +CUDA-aware. `xp.mpi.require_cuda_aware_mpi()` at start-up gives a clear message; load a CUDA-aware MPI module or rebuild `mpi4py` against one. **Occasionally wrong data after an exchange.** A kernel was still writing the -send buffer. Call `xp.synchronize_for_mpi(send, recv)` before the MPI call. +send buffer. Call `xp.mpi.synchronize_for_mpi(send, recv)` before the MPI call. See [Multi-GPU programs with MPI](guides/mpi.md). diff --git a/pyproject.toml b/pyproject.toml index 43a8bda..f6e8447 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,7 +5,7 @@ requires = [ "setuptools", "wheel" ] [project] name = "cunumpy" -version = "0.4.0" +version = "0.5.0" description = "Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`." readme = "README.md" keywords = [ "python" ] diff --git a/src/cunumpy/LLM_GUIDE.md b/src/cunumpy/LLM_GUIDE.md index cf61288..c82f348 100644 --- a/src/cunumpy/LLM_GUIDE.md +++ b/src/cunumpy/LLM_GUIDE.md @@ -19,7 +19,7 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository. * A kernel layer for porting compiled CPU kernels (Pyccel, Numba, Python loops) to CUDA one at a time: `PyccelKernel`, `CudaKernel`, `Kernel`, `KernelCatalog`, `KernelArguments`, `CudaStruct`, `DeviceMirror`, and test - helpers in `cunumpy.testing`. + helpers in `cunumpy.kernel_testing`. ## Hard rules @@ -40,7 +40,7 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository. explicit: `xp.to_numpy(a)`, `xp.to_cupy(a)`, `xp.to_cunumpy(a)`. 6. **No transfers or host syncs inside time loops.** Avoid `to_numpy`, `float(x)`, `x.item()`, `print(x)`, `if device_value:` in hot loops. Verify with - `xp.assert_no_transfers()` in tests. + `xp.profiling.assert_no_transfers()` in tests. 7. **`CudaKernel` never copies.** Pointer parameters require C-contiguous CuPy arrays of the exact dtype; NumPy arrays raise `TypeError`. Do not "fix" that error by disabling checks; convert once with `xp.to_cupy` / @@ -53,6 +53,13 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository. 10. **Declare `outputs` correctly on `PyccelKernel`** (and via `host_options`). An argument the kernel writes but that is not declared leaves stale device data, silently. If unsure, leave `outputs=None` (copies everything back). +11. **Helpers are in submodules; the top level is NumPy plus backend control.** + `xp.cuda.CudaKernel`, `xp.kernels.Kernel`, `xp.rng.random_streams`, + `xp.algorithms.morton_keys`, `xp.mpi.mpi_buffer`, `xp.profiling.timed_region`, + `xp.memory.HostStaging`, `xp.petsc.petsc_vec`; kernel test helpers in + `cunumpy.kernel_testing`. The old top-level names (`xp.CudaKernel`) and + `cunumpy.testing` are deprecated (removed in 0.6); do not write new code + with them. ## Decision guide @@ -62,39 +69,39 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository. | function receiving arrays from elsewhere | `array_xp = xp.get_array_module(a)`; `xp.assert_same_backend(a, b)` | | normalize inputs at an API boundary | `xp.to_cunumpy(a)` once | | hand data to SciPy/matplotlib/h5py | `xp.to_numpy(a)` | -| call an existing NumPy-only kernel with GPU arrays (slow, correct) | `xp.PyccelKernel(fn, outputs=(...))` | -| launch a hand-written CUDA C kernel | `xp.CudaKernel(source, "name")` / `CudaKernel.from_file(path)` | -| host kernel + CUDA port, chosen by backend | `xp.Kernel(host_fn, cuda_kernel_or_None)` | -| many kernels in a package, ported incrementally | `xp.KernelCatalog.from_package(__name__, missing_cuda="fallback")` | -| host kernels compiled at first call (your compile function), NumPy fallback | `from_package(..., host_suffix="_pyccel", compile_host=my_compile, host_fallback={...})` -> `xp.CompiledHostKernel` | +| call an existing NumPy-only kernel with GPU arrays (slow, correct) | `xp.kernels.PyccelKernel(fn, outputs=(...))` | +| launch a hand-written CUDA C kernel | `xp.cuda.CudaKernel(source, "name")` / `CudaKernel.from_file(path)` | +| host kernel + CUDA port, chosen by backend | `xp.kernels.Kernel(host_fn, cuda_kernel_or_None)` | +| many kernels in a package, ported incrementally | `xp.kernels.KernelCatalog.from_package(__name__, missing_cuda="fallback")` | +| host kernels compiled at first call (your compile function), NumPy fallback | `from_package(..., host_suffix="_pyccel", compile_host=my_compile, host_fallback={...})` -> `xp.kernels.CompiledHostKernel` | | host arrays reach kernels while CuPy is active | `Kernel(..., dispatch="arrays")` / `from_package(..., dispatch="arrays")`: CUDA only for device arguments | -| one kernel folder declares its kernel in its own `__init__.py` | `kernel = xp.Kernel.from_folder(__name__, host_suffix="_pyccel", compile_host=..., dispatch="arrays")`; `_numba.py`, `_numpy.py` in the folder are further host implementations | -| bring a `dispatch="arrays"` kernel's arguments to the side of the main array | `xp.as_kernel_array(a, like=grid, dtype=float)`; outputs: `with xp.kernel_output(out, like=grid, dtype=float) as buf:` | -| choose the host implementation (pyccel/numba/numpy/python) | `xp.set_kernel_implementation("numpy")`, `with xp.use_kernel_implementation(...)`, `CUNUMPY_KERNEL_IMPLEMENTATION=numpy`; default: first available of pyccel, numba, numpy; `kernel.selected()` | +| one kernel folder declares its kernel in its own `__init__.py` | `kernel = xp.kernels.Kernel.from_folder(__name__, host_suffix="_pyccel", compile_host=..., dispatch="arrays")`; `_numba.py`, `_numpy.py` in the folder are further host implementations | +| bring a `dispatch="arrays"` kernel's arguments to the side of the main array | `xp.kernels.as_kernel_array(a, like=grid, dtype=float)`; outputs: `with xp.kernels.kernel_output(out, like=grid, dtype=float) as buf:` | +| choose the host implementation (pyccel/numba/numpy/python) | `xp.kernels.set_kernel_implementation("numpy")`, `with xp.kernels.use_kernel_implementation(...)`, `CUNUMPY_KERNEL_IMPLEMENTATION=numpy`; default: first available of pyccel, numba, numpy; `kernel.selected()` | | check host and CUDA kernels take the same parameters | `catalog.check_signatures()` (in a unit test) | -| test a CUDA kernel's arithmetic without a GPU | `cunumpy.testing.emulate_cuda_kernel(kernel, *numpy_args, n_threads=n)` (C++ compiler; shared memory and __syncthreads ok, no warp ops; `shared_mem=` for extern shared) | -| shared-memory budget of a block | `xp.max_shared_memory_per_block()` (48 KiB without a GPU) | -| random numbers inside a kernel, equal on the host | `#include `: `cunumpy_uniform(seed, particle_id, step)`; host: `xp.philox_uniform(seed, ids, step)` | -| sort points along a Z-curve / quadtree or octree nodes as contiguous ranges | `keys = xp.morton_keys(pos, lower, upper, levels)`, `keys, order, pos = xp.sort_by_key(keys, pos)`; in a kernel `#include `: `cunumpy_morton_key2(x, y, x0, y0, sx, sy, levels)` with `xp.morton_scales(...)` | +| test a CUDA kernel's arithmetic without a GPU | `cunumpy.kernel_testing.emulate_cuda_kernel(kernel, *numpy_args, n_threads=n)` (C++ compiler; shared memory and __syncthreads ok, no warp ops; `shared_mem=` for extern shared) | +| shared-memory budget of a block | `xp.cuda.max_shared_memory_per_block()` (48 KiB without a GPU) | +| random numbers inside a kernel, equal on the host | `#include `: `cunumpy_uniform(seed, particle_id, step)`; host: `xp.rng.philox_uniform(seed, ids, step)` | +| sort points along a Z-curve / quadtree or octree nodes as contiguous ranges | `keys = xp.algorithms.morton_keys(pos, lower, upper, levels)`, `keys, order, pos = xp.algorithms.sort_by_key(keys, pos)`; in a kernel `#include `: `cunumpy_morton_key2(x, y, x0, y0, sx, sy, levels)` with `xp.algorithms.morton_scales(...)` | | one thread per marker without passing n_threads | `CudaKernel(..., n_threads_from="first_array")` | -| copy device arrays to the host for output without stalling | `xp.HostStaging(shape, dtype)`: `c = staging.copy(a)` ... `c.result()` | +| copy device arrays to the host for output without stalling | `xp.memory.HostStaging(shape, dtype)`: `c = staging.copy(a)` ... `c.result()` | | PIC recipes (compaction, sort by cell, MPI exchange, graphs) | docs guide "Particle codes" | -| reproducible random numbers per MPI rank | `xp.random_streams.seed(seed, rank=rank)`, then `xp.random_streams.normal(...)` / `.generator()` | -| group arrays/scalars into one kernel argument | `xp.CudaArguments` (device only), `xp.KernelArguments` (host object + device tuple), `xp.CudaStruct` (C struct), `xp.CudaStructArguments` (C struct as a class) | -| CUDA struct from a Pyccel argument class | `xp.CudaStruct.from_signature(Cls.__init__, "Name")`, `xp.write_cuda_header(...)` | +| reproducible random numbers per MPI rank | `xp.rng.random_streams.seed(seed, rank=rank)`, then `xp.rng.random_streams.normal(...)` / `.generator()` | +| group arrays/scalars into one kernel argument | `xp.cuda.CudaArguments` (device only), `xp.kernels.KernelArguments` (host object + device tuple), `xp.cuda.CudaStruct` (C struct), `xp.cuda.CudaStructArguments` (C struct as a class) | +| CUDA struct from a Pyccel argument class | `xp.cuda.CudaStruct.from_signature(Cls.__init__, "Name")`, `xp.cuda.write_cuda_header(...)` | | SciPy (sparse, sparse.linalg, fft, special, ndimage, ...) on either backend | `xp.scipy..` (SciPy or `cupyx.scipy`); `xp.scipy.special.available(name)` | -| chain of elementwise operations as one GPU kernel | `@xp.fuse` (`cupy.fuse` for CuPy arrays, plain call otherwise) | -| PETSc solve on device arrays without copies | `xp.petsc_vec(array)` (CUDA/HIP petsc4py for CuPy arrays); `xp.synchronize()` around PETSc calls | +| chain of elementwise operations as one GPU kernel | `@xp.kernels.fuse` (`cupy.fuse` for CuPy arrays, plain call otherwise) | +| PETSc solve on device arrays without copies | `xp.petsc.petsc_vec(array)` (CUDA/HIP petsc4py for CuPy arrays); `xp.synchronize()` around PETSc calls | | reduction inside a CUDA kernel (energy, max velocity) | ``: `cunumpy_block_sum_to(out, v)`, `cunumpy_block_min/max`, `cunumpy_warp_sum` | -| kernel writes into a host buffer owned by another library | `xp.DeviceMirror(host_array)` + `` | +| kernel writes into a host buffer owned by another library | `xp.memory.DeviceMirror(host_array)` + `` | | N-D indexing in CUDA, non-contiguous arrays | `Array1D`..`Array4D` params from `` | | one MPI rank per GPU | `bind_local_device()` → `from mpi4py import MPI` → `require_cuda_aware_mpi()` → `synchronize_for_mpi(...)` before each call | -| timing GPU code | `with xp.timed_region("name") as t:` → `t.elapsed` | -| profiler markers | `xp.nvtx_range("name")` (context manager or decorator) | -| find transfers | `with xp.count_transfers() as c: ...; print(c.report())` | +| timing GPU code | `with xp.profiling.timed_region("name") as t:` → `t.elapsed` | +| profiler markers | `xp.profiling.nvtx_range("name")` (context manager or decorator) | +| find transfers | `with xp.profiling.count_transfers() as c: ...; print(c.report())` | | debug an illegal memory access | `CUNUMPY_CUDA_DEBUG=1`, then `compute-sanitizer` | -| test on both backends | `cunumpy.testing.BACKENDS`, `backend` fixture, `requires_cupy` | -| test CUDA vs host kernel | `cunumpy.testing.assert_kernels_agree(kernel, make_args, n_threads=...)` | +| test on both backends | `cunumpy.kernel_testing.BACKENDS`, `backend` fixture, `requires_cupy` | +| test CUDA vs host kernel | `cunumpy.kernel_testing.assert_kernels_agree(kernel, make_args, n_threads=...)` | ## API cheat sheet @@ -123,9 +130,9 @@ xp.to_cunumpy(a) # -> array of the active backend xp.as_device_array(value, dtype=None, ndim=None, *, name=None) # contiguous CuPy array of dtype: returned as is; else one copy; # RuntimeError on the NumPy backend -with xp.count_transfers() as c: ... # c.total, c.to_host, c.to_device, +with xp.profiling.count_transfers() as c: ... # c.total, c.to_host, c.to_device, # c.kernel_conversions, c.fallbacks, c.events, c.report() -with xp.assert_no_transfers(): ... # AssertionError with report if anything copied +with xp.profiling.assert_no_transfers(): ... # AssertionError with report if anything copied ``` Only transfers through cunumpy are counted (not raw `cupy.asarray`, `.get()`, @@ -134,19 +141,19 @@ Only transfers through cunumpy are counted (not raw `cupy.asarray`, `.get()`, MPI, accumulation and versions: ```python -xp.mpi_is_cuda_aware(comm) # collective, once at startup; remembered -with xp.mpi_buffer(a) as buf: comm.Send(buf, ...) # host array, CUDA-aware device -with xp.mpi_buffer(a, send=False, recv=True) as buf: ... # array, or pinned staging copy -xp.set_mpi_cuda_aware(True | False | None), xp.get_mpi_cuda_aware() -xp.segment_sum(values, keys, n_segments) # out[k] = sum(values[keys == k]); keys < 0 dropped -keys, order, a, b = xp.sort_by_key(keys, a, b) # stable argsort applied to every array +xp.mpi.mpi_is_cuda_aware(comm) # collective, once at startup; remembered +with xp.mpi.mpi_buffer(a) as buf: comm.Send(buf, ...) # host array, CUDA-aware device +with xp.mpi.mpi_buffer(a, send=False, recv=True) as buf: ... # array, or pinned staging copy +xp.mpi.set_mpi_cuda_aware(True | False | None), xp.mpi.get_mpi_cuda_aware() +xp.algorithms.segment_sum(values, keys, n_segments) # out[k] = sum(values[keys == k]); keys < 0 dropped +keys, order, a, b = xp.algorithms.sort_by_key(keys, a, b) # stable argsort applied to every array xp.require_version("0.4.0") # ImportError if cunumpy is older ``` Random numbers and dtypes: ```python -rng = xp.get_rng(seed=None) # numpy or cupy Generator for the active backend +rng = xp.rng.get_rng(seed=None) # numpy or cupy Generator for the active backend xp.default_float_dtype() # float64 of the active backend ``` @@ -156,38 +163,38 @@ identical data on both backends: `xp.to_cunumpy(np.random.default_rng(s).random( Devices, memory, streams (all safe on NumPy: no-ops / neutral values): ```python -xp.device_count() -> int # visible GPUs, 0 without CuPy -xp.set_device(i) -xp.memory_info() -> (free, total) | None -xp.free_memory() # release CuPy pool cached blocks +xp.cuda.device_count() -> int # visible GPUs, 0 without CuPy +xp.cuda.set_device(i) +xp.cuda.memory_info() -> (free, total) | None +xp.cuda.free_memory() # release CuPy pool cached blocks xp.synchronize() -with xp.stream() as s: ... # s is None on NumPy; prefer xp.synchronize() -xp.pin_memory(host_array) # pinned copy; needs CuPy +with xp.cuda.stream() as s: ... # s is None on NumPy; prefer xp.synchronize() +xp.cuda.pin_memory(host_array) # pinned copy; needs CuPy ``` MPI: ```python xp.set_backend("cupy") -xp.bind_local_device() # BEFORE `from mpi4py import MPI`; uses local_rank() +xp.cuda.bind_local_device() # BEFORE `from mpi4py import MPI`; uses local_rank() from mpi4py import MPI -xp.require_cuda_aware_mpi() # collective; RuntimeError if MPI can't take device buffers -xp.mpi_is_cuda_aware(comm=None) -> bool -xp.local_rank() -> int # node-local rank from launcher env vars -xp.synchronize_for_mpi(*buffers) # before every MPI call that touches device buffers +xp.mpi.require_cuda_aware_mpi() # collective; RuntimeError if MPI can't take device buffers +xp.mpi.mpi_is_cuda_aware(comm=None) -> bool +xp.mpi.local_rank() -> int # node-local rank from launcher env vars +xp.mpi.synchronize_for_mpi(*buffers) # before every MPI call that touches device buffers ``` Profiling: ```python -with xp.timed_region("name", sync=True) as t: ... # t.name, t.elapsed (s), t.synced -with xp.nvtx_range("name", color=None): ... # also usable as @decorator +with xp.profiling.timed_region("name", sync=True) as t: ... # t.name, t.elapsed (s), t.synced +with xp.profiling.nvtx_range("name", color=None): ... # also usable as @decorator ``` `PyccelKernel`: ```python -k = xp.PyccelKernel(fn, use_cupy=None, object_modules=(), is_array=None, outputs=None) +k = xp.kernels.PyccelKernel(fn, use_cupy=None, object_modules=(), is_array=None, outputs=None) k(*args, **kwargs) ``` @@ -200,18 +207,18 @@ CuPy. Does not compile anything. `CudaKernel`: ```python -k = xp.CudaKernel(source, name, *, block_size=128, options=(), include_dirs=(), +k = xp.cuda.CudaKernel(source, name, *, block_size=128, options=(), include_dirs=(), source_dir=None, structs=(), template_args=None, check_signature=True, debug=None) -k = xp.CudaKernel.from_file("push/push_cuda.cu") # name "push" -ks = xp.CudaKernel.all_from_file("ops.cu") # dict name -> kernel +k = xp.cuda.CudaKernel.from_file("push/push_cuda.cu") # name "push" +ks = xp.cuda.CudaKernel.all_from_file("ops.cu") # dict name -> kernel k(*args, n_threads=None, grid=None, block=None, shared_mem=0, stream=None) k.compile(); k.is_compiled; k.launch_shape(n_threads) -> (grid, block) k.included_headers; k.compile_options(); k.debug_active() -xp.CudaKernelVariants(factory).get(*key); .compile_all(keys, jobs=1) -xp.ctype_of(np.float64) == "double" -xp.cuda_kernel_names(source); xp.parse_cuda_signature(source, name) -xp.cuda_include_dir() +xp.cuda.CudaKernelVariants(factory).get(*key); .compile_all(keys, jobs=1) +xp.cuda.ctype_of(np.float64) == "double" +xp.cuda.cuda_kernel_names(source); xp.cuda.parse_cuda_signature(source, name) +xp.cuda.cuda_include_dir() ``` * Source must declare `extern "C" __global__` (templates: plain `__global__` @@ -241,15 +248,15 @@ Shipped CUDA headers (always on the include path): `Kernel` and `KernelCatalog`: ```python -k = xp.Kernel(host_kernel, cuda_kernel=None, *, name=None, +k = xp.kernels.Kernel(host_kernel, cuda_kernel=None, *, name=None, missing_cuda="raise" | "fallback", cuda_path=None, host_options=None) k(*args, n_threads=..., grid=None, block=None, shared_mem=0, stream=None) k.get_kernel(); k.compile(); k.has_cuda -catalog = xp.KernelCatalog.from_package(__name__, *, host_suffix="_kernels", +catalog = xp.kernels.KernelCatalog.from_package(__name__, *, host_suffix="_kernels", cuda_suffix="_cuda.cu", missing_cuda="raise", host_options=None, include_dirs=None, **cuda_options) -kernel = xp.Kernel.from_folder(__name__, *, host_suffix="_kernels", compile_host=None, +kernel = xp.kernels.Kernel.from_folder(__name__, *, host_suffix="_kernels", compile_host=None, dispatch="backend", extra_implementations=None, **same options as from_package) kernel.implementations; kernel.selected(device=False) catalog["push"]; catalog.summary(); catalog.with_cuda; catalog.without_cuda @@ -265,27 +272,27 @@ function `` (host); optional `pkg//_cuda.cu` defines Argument objects: ```python -class Dev(xp.CudaArguments): # flattened into several CUDA params +class Dev(xp.cuda.CudaArguments): # flattened into several CUDA params def __init__(self, x, n): super().__init__(x, n) -class Args(xp.KernelArguments): # one object, host form + device form +class Args(xp.kernels.KernelArguments): # one object, host form + device form def __host_args__(self): return host_object # host kernel gets this def __cuda_args__(self): return (arr, n, ...) # CUDA kernel gets these, flattened -S = xp.CudaStruct("S", [("x", "double*"), ("n", "long long"), ("a", "Array2D")]) +S = xp.cuda.CudaStruct("S", [("x", "double*"), ("n", "long long"), ("a", "Array2D")]) S.declaration; S.dtype; S.to_header(path); value = S(x=..., n=..., a=...) S.verify_layout() # GPU test: compiler layout == S.dtype (also verify_layout("hdr.cuh")) -S = xp.CudaStruct.from_signature(Cls.__init__, "S", int_type="long long") -xp.write_cuda_header("args.cuh", [S1, S2]) +S = xp.cuda.CudaStruct.from_signature(Cls.__init__, "S", int_type="long long") +xp.cuda.write_cuda_header("args.cuh", [S1, S2]) -class A(xp.CudaStructArguments): # the struct as a class; A.struct is the CudaStruct +class A(xp.cuda.CudaStructArguments): # the struct as a class; A.struct is the CudaStruct struct_name = "A" fields = (("x", "double*"), ("n", "int")) def __init__(self, x): self.x, self.n = x, x.shape[0] self.pack() # repacks itself when a field changes; copies repack -xp.CudaKernel(S.declaration + src, "k", structs=[S]) -xp.resolve_host_args(args, kwargs) +xp.cuda.CudaKernel(S.declaration + src, "k", structs=[S]) +xp.kernels.resolve_host_args(args, kwargs) ``` Only top-level arguments are resolved. Cache both forms lazily and invalidate @@ -296,7 +303,7 @@ itself at the next launch; make its fields properties to follow an owner's array `DeviceMirror`: ```python -m = xp.DeviceMirror(host_numpy_array) # TypeError if not numpy.ndarray +m = xp.memory.DeviceMirror(host_numpy_array) # TypeError if not numpy.ndarray m.device # CuPy copy (lazy) on CuPy; the host array itself on NumPy m.zero(); m.to_device(); m.to_host() # to_host copies in place; no-ops on NumPy m.rebind(new_host_array) # after the owner reallocates @@ -305,8 +312,8 @@ m.rebind(new_host_array) # after the owner reallocates Debugging: ```python -xp.set_cuda_debug(True); xp.get_cuda_debug(); with xp.cuda_debug(): ... -xp.CudaKernel(..., debug=True) # env: CUNUMPY_CUDA_DEBUG=1 +xp.cuda.set_cuda_debug(True); xp.cuda.get_cuda_debug(); with xp.cuda.cuda_debug(): ... +xp.cuda.CudaKernel(..., debug=True) # env: CUNUMPY_CUDA_DEBUG=1 ``` Debug mode adds `-lineinfo -DCUNUMPY_BOUNDS_CHECK` at compile time and @@ -314,10 +321,10 @@ synchronizes after each launch (errors become `RuntimeError` naming the kernel). Enable it before kernels compile. After an illegal memory access, the process must be restarted. -Testing (`import cunumpy.testing`; not imported by `import cunumpy`): +Testing (`import cunumpy.kernel_testing`; not imported by `import cunumpy`): ```python -from cunumpy.testing import BACKENDS, backend, requires_cupy, \ +from cunumpy.kernel_testing import BACKENDS, backend, requires_cupy, \ assert_kernels_agree, device_function_kernel @pytest.mark.parametrize("backend", BACKENDS) # "numpy" always, "cupy" if GPU @@ -333,7 +340,7 @@ k(t, p_array, x_array, out, n, n_threads=n) # scalars become per-thread array k = device_function_kernel(src, "double g(const DomainArgs& d, double x)", structs=[DomainArgs]) # /_test_args.py: make_args(backend, seed) + N_THREADS (or GRID), RTOL, ... -from cunumpy.testing import parity_cases, check_parity +from cunumpy.kernel_testing import parity_cases, check_parity @pytest.mark.parametrize("kernel", parity_cases(catalog)) # skip-marked if no test args def test_parity(kernel): check_parity(kernel) @@ -341,11 +348,11 @@ def test_parity(kernel): check_parity(kernel) # stand-in, no kernel launches; fake_cupy_active(); requires_cupy skips) # argument classes with a pyccel host class: one object on both backends -class MarkerArguments(xp.PyccelStructArguments): +class MarkerArguments(xp.kernels.PyccelStructArguments): struct_name = "MarkerArgs"; fields = (("markers", "Array2D"), ("Np", "long long")) host_class = pusher_args_kernels.MarkerArguments # pyccel class; cannot inherit host_fields = ("markers", "Np") # its constructor args, in order -MarkerArgs = xp.CudaStruct.from_pyccel_class("pusher_args_kernels.py", "MarkerArguments", "MarkerArgs") +MarkerArgs = xp.cuda.CudaStruct.from_pyccel_class("pusher_args_kernels.py", "MarkerArguments", "MarkerArgs") kernel.n_threads_from = lambda args: args[0].n_markers # launch size from an argument kernel.check_finite = True # NaN/inf after each launch (debug) ``` @@ -390,7 +397,7 @@ def scale_host(x: "float[:]", a: float, n: int): for i in range(n): x[i] *= a -scale = xp.Kernel(scale_host, xp.CudaKernel(SRC, "scale"), +scale = xp.kernels.Kernel(scale_host, xp.cuda.CudaKernel(SRC, "scale"), host_options={"outputs": (0,)}) scale(x, 2.0, x.size, n_threads=x.size) ``` @@ -420,9 +427,9 @@ def test_scale(): * Forgetting `if (i >= n) return;` (or `CUNUMPY_THREAD_1D`) in a kernel. * Plain `+=` from many threads into one cell; use `cunumpy_atomic_add`. * Timing GPU code with `time.perf_counter()` without synchronizing; use - `xp.timed_region()`. -* Importing `mpi4py.MPI` before `xp.bind_local_device()`; skipping - `xp.synchronize_for_mpi()` before MPI calls on device buffers. + `xp.profiling.timed_region()`. +* Importing `mpi4py.MPI` before `xp.cuda.bind_local_device()`; skipping + `xp.mpi.synchronize_for_mpi()` before MPI calls on device buffers. * Comparing CPU and GPU results with exact equality where the GPU uses atomics or a different reduction order; use a tolerance. * Assuming `xp.random.seed(s)` gives the same numbers on both backends. diff --git a/src/cunumpy/__init__.py b/src/cunumpy/__init__.py index 40ac479..e372fdb 100644 --- a/src/cunumpy/__init__.py +++ b/src/cunumpy/__init__.py @@ -1,111 +1,57 @@ # cunumpy/__init__.py import re as _re +import warnings as _warnings from importlib.metadata import PackageNotFoundError, version -from . import xp -from .cuda_kernel import ( - DEBUG_OPTIONS, - CudaArguments, - CudaKernel, - CudaKernelVariants, - CudaParameter, - CudaStruct, - CudaStructArguments, - CudaStructValue, - PyccelStructArguments, - ctype_of, - cuda_include_dir, - cuda_kernel_names, - include_hash, - parse_cuda_signature, - resolve_includes, - write_cuda_header, +from . import ( + algorithms, + cuda, + kernels, + memory, + mpi, + petsc, + profiling, + rng, + xp, ) -from .dispatch import Kernel, KernelCatalog -from .fusion import fuse -from .kernel import ( - HOST_IMPLEMENTATIONS, - CompiledHostKernel, - HostImplementations, - KernelArguments, - PyccelKernel, - get_kernel_implementation, - resolve_host_args, - set_kernel_implementation, - use_kernel_implementation, -) -from .mirror import DeviceMirror -from .morton import ( - MAX_MORTON_LEVELS, - morton_decode, - morton_encode, - morton_keys, - morton_scales, -) -from .petsc import petsc_vec -from .philox import ( - philox4x32_10, - philox_normal, - philox_normal2, - philox_uniform, - philox_uniform2, -) -from .random_streams import RandomStreams, random_streams from .scipy_backend import scipy -from .staging import HostStaging, StagedCopy -from .transfers import ( - TransferCounter, - TransferEvent, - assert_no_transfers, - count_transfers, -) from .xp import ( - DEFAULT_SHARED_MEMORY_PER_BLOCK, - Timing, as_device_array, - as_kernel_array, assert_same_backend, - bind_local_device, - cuda_debug, cupy_available, default_float_dtype, - device_count, - free_memory, get_array_backend, get_array_module, get_backend, - get_cuda_debug, - get_mpi_cuda_aware, - get_rng, is_cpu, is_gpu, - kernel_output, - local_rank, - max_shared_memory_per_block, - memory_info, - mpi_buffer, - mpi_is_cuda_aware, - nvtx_range, - pin_memory, - require_cuda_aware_mpi, same_backend, - segment_sum, set_backend, - set_cuda_debug, - set_device, - set_device_for_rank, - set_mpi_cuda_aware, - sort_by_key, - stream, synchronize, - synchronize_for_mpi, - timed_region, to_cunumpy, to_cupy, to_numpy, use_backend, ) +# Names that were at the top level before cunumpy 0.5, and the submodule each +# moved to. They still resolve (with a DeprecationWarning) until cunumpy 0.6. +_MOVED = { + **dict.fromkeys(cuda.__all__, "cuda"), + **dict.fromkeys(kernels.__all__, "kernels"), + **dict.fromkeys(rng.__all__, "rng"), + **dict.fromkeys(algorithms.__all__, "algorithms"), + **dict.fromkeys(mpi.__all__, "mpi"), + **dict.fromkeys(profiling.__all__, "profiling"), + **dict.fromkeys(memory.__all__, "memory"), + "petsc_vec": "petsc", +} +_MOVED.pop("BIT_GENERATORS") # never was at the top level + +# Importing cunumpy.rng loads the module cunumpy.random_streams, which would +# hide the deprecated top-level name random_streams (the generator). +globals().pop("random_streams", None) + try: __version__ = version("cunumpy") except PackageNotFoundError: @@ -140,111 +86,56 @@ def require_version(minimum: str) -> None: __all__ = [ - "DEBUG_OPTIONS", - "DEFAULT_SHARED_MEMORY_PER_BLOCK", - "HOST_IMPLEMENTATIONS", - "MAX_MORTON_LEVELS", - "CompiledHostKernel", - "CudaArguments", - "CudaKernel", - "CudaKernelVariants", - "CudaParameter", - "CudaStruct", - "CudaStructArguments", - "CudaStructValue", - "DeviceMirror", - "HostImplementations", - "HostStaging", - "Kernel", - "KernelArguments", - "KernelCatalog", - "PyccelKernel", - "PyccelStructArguments", - "RandomStreams", - "StagedCopy", - "Timing", - "TransferCounter", - "TransferEvent", "__version__", + "algorithms", "as_device_array", - "as_kernel_array", - "assert_no_transfers", "assert_same_backend", - "bind_local_device", - "count_transfers", - "ctype_of", - "cuda_debug", - "cuda_include_dir", - "cuda_kernel_names", + "cuda", "cupy_available", "cupy_backend", "default_float_dtype", - "device_count", - "free_memory", - "fuse", "get_array_backend", "get_array_module", "get_backend", - "get_cuda_debug", - "get_kernel_implementation", - "get_mpi_cuda_aware", - "get_rng", - "include_hash", "is_cpu", "is_gpu", - "kernel_output", - "local_rank", - "max_shared_memory_per_block", - "memory_info", - "morton_decode", - "morton_encode", - "morton_keys", - "morton_scales", - "mpi_buffer", - "mpi_is_cuda_aware", + "kernels", + "memory", + "mpi", "numpy_backend", - "nvtx_range", - "parse_cuda_signature", - "petsc_vec", - "philox4x32_10", - "philox_normal", - "philox_normal2", - "philox_uniform", - "philox_uniform2", - "pin_memory", - "random_streams", - "require_cuda_aware_mpi", + "petsc", + "profiling", "require_version", - "resolve_host_args", - "resolve_includes", + "rng", "same_backend", "scipy", - "segment_sum", "set_backend", - "set_cuda_debug", - "set_device", - "set_device_for_rank", - "set_kernel_implementation", - "set_mpi_cuda_aware", - "sort_by_key", - "stream", "synchronize", - "synchronize_for_mpi", - "timed_region", "to_cunumpy", "to_cupy", "to_numpy", "use_backend", - "use_kernel_implementation", - "write_cuda_header", "xp", ] def __getattr__(name: str): - """Set cunumpy. to cunumpy.xp. (NumPy/CuPy).""" + """Set cunumpy. to cunumpy.xp. (NumPy/CuPy). + + Names moved to a submodule in cunumpy 0.5 (see ``_MOVED``) still resolve, + with a ``DeprecationWarning``. + """ if name == "numpy_backend": return xp.numpy_backend if name == "cupy_backend": return xp.cupy_backend + submodule = _MOVED.get(name) + if submodule is not None: + _warnings.warn( + f"cunumpy.{name} moved to cunumpy.{submodule}.{name}; the top-level " + "name is deprecated and will be removed in cunumpy 0.6", + DeprecationWarning, + stacklevel=2, + ) + return getattr(globals()[submodule], name) return getattr(xp.xp, name) diff --git a/src/cunumpy/__init__.pyi b/src/cunumpy/__init__.pyi index c5b0239..659f3ea 100644 --- a/src/cunumpy/__init__.pyi +++ b/src/cunumpy/__init__.pyi @@ -8,48 +8,16 @@ from typing import Any import numpy as np from numpy import * +from . import algorithms as algorithms +from . import cuda as cuda +from . import kernels as kernels +from . import memory as memory +from . import mpi as mpi +from . import petsc as petsc +from . import profiling as profiling +from . import rng as rng from . import xp as xp -from .cuda_kernel import CudaArguments as CudaArguments -from .cuda_kernel import CudaKernel as CudaKernel -from .cuda_kernel import CudaKernelVariants as CudaKernelVariants -from .cuda_kernel import CudaParameter as CudaParameter -from .cuda_kernel import CudaStruct as CudaStruct -from .cuda_kernel import CudaStructArguments as CudaStructArguments -from .cuda_kernel import CudaStructValue as CudaStructValue -from .cuda_kernel import PyccelStructArguments as PyccelStructArguments -from .cuda_kernel import ctype_of as ctype_of -from .cuda_kernel import cuda_include_dir as cuda_include_dir -from .cuda_kernel import parse_cuda_signature as parse_cuda_signature -from .cuda_kernel import write_cuda_header as write_cuda_header -from .dispatch import Kernel as Kernel -from .dispatch import KernelCatalog as KernelCatalog -from .fusion import fuse as fuse -from .kernel import HOST_IMPLEMENTATIONS as HOST_IMPLEMENTATIONS -from .kernel import CompiledHostKernel as CompiledHostKernel -from .kernel import HostImplementations as HostImplementations -from .kernel import PyccelKernel as PyccelKernel -from .kernel import get_kernel_implementation as get_kernel_implementation -from .kernel import set_kernel_implementation as set_kernel_implementation -from .kernel import use_kernel_implementation as use_kernel_implementation -from .mirror import DeviceMirror as DeviceMirror -from .morton import MAX_MORTON_LEVELS as MAX_MORTON_LEVELS -from .morton import morton_decode as morton_decode -from .morton import morton_encode as morton_encode -from .morton import morton_keys as morton_keys -from .morton import morton_scales as morton_scales -from .petsc import petsc_vec as petsc_vec -from .philox import philox4x32_10 as philox4x32_10 -from .philox import philox_normal as philox_normal -from .philox import philox_normal2 as philox_normal2 -from .philox import philox_uniform as philox_uniform -from .philox import philox_uniform2 as philox_uniform2 -from .random_streams import RandomStreams as RandomStreams -from .random_streams import random_streams as random_streams from .scipy_backend import scipy as scipy -from .staging import HostStaging as HostStaging -from .staging import StagedCopy as StagedCopy -from .transfers import TransferCounter as TransferCounter -from .transfers import TransferEvent as TransferEvent def to_numpy(array: Any) -> np.ndarray: ... def to_cupy(array: Any) -> Any: ... @@ -65,44 +33,12 @@ def assert_same_backend(*arrays: Any) -> None: ... @contextmanager def use_backend(backend: str) -> Generator[None]: ... def set_backend(backend: str) -> None: ... -def set_device(device_id: int) -> None: ... -def set_device_for_rank(rank: int, devices_per_node: int | None = ...) -> int: ... -def local_rank() -> int: ... -def bind_local_device() -> int | None: ... -def synchronize_for_mpi(*arrays: Any) -> None: ... -def mpi_is_cuda_aware(comm: Any = ..., *, method: str = ...) -> bool: ... -def require_cuda_aware_mpi(comm: Any = ...) -> None: ... -def set_mpi_cuda_aware(value: bool | None) -> None: ... -def get_mpi_cuda_aware() -> bool | None: ... -@contextmanager -def mpi_buffer( - array: Any, *, send: bool = ..., recv: bool = ..., cuda_aware: bool | None = ... -) -> Generator[Any]: ... -def segment_sum(values: Any, keys: Any, n_segments: int) -> Any: ... -def sort_by_key(keys: Any, *arrays: Any) -> tuple[Any, ...]: ... -def as_kernel_array(value: Any, like: Any, dtype: Any = ...) -> Any: ... -@contextmanager -def kernel_output(out: Any, like: Any, dtype: Any = ...) -> Generator[Any]: ... def require_version(minimum: str) -> None: ... -def device_count() -> int: ... -def memory_info() -> tuple[int, int] | None: ... -def free_memory() -> None: ... -def max_shared_memory_per_block( - device: int | None = ..., *, opt_in: bool = ... -) -> int: ... - -DEFAULT_SHARED_MEMORY_PER_BLOCK: int - -def pin_memory(array: Any) -> Any: ... -@contextmanager -def stream() -> Generator[Any]: ... -def get_rng(seed: int | None = ...) -> Any: ... def default_float_dtype() -> Any: ... def synchronize() -> None: ... -@contextmanager -def count_transfers() -> Generator[TransferCounter]: ... -@contextmanager -def assert_no_transfers() -> Generator[TransferCounter]: ... +def as_device_array( + value: Any, dtype: Any = ..., ndim: int | None = ..., *, name: str | None = ... +) -> Any: ... numpy_backend: bool cupy_backend: bool diff --git a/src/cunumpy/_fake_cupy.py b/src/cunumpy/_fake_cupy.py index d8eb8e5..c1a58ee 100644 --- a/src/cunumpy/_fake_cupy.py +++ b/src/cunumpy/_fake_cupy.py @@ -13,10 +13,10 @@ like CuPy does; mixing CuPy and NumPy arrays in arithmetic raises; * reductions and scalar indexing return 0-d arrays, not Python scalars; * arrays have ``data.ptr``, ``device`` and ``__cuda_array_interface__``, so - :class:`~cunumpy.CudaStruct` packing and the argument checks of - :class:`~cunumpy.CudaKernel` work; + :class:`~cunumpy.cuda.CudaStruct` packing and the argument checks of + :class:`~cunumpy.cuda.CudaKernel` work; * CUDA kernels cannot run: ``RawKernel`` and friends raise - ``NotImplementedError`` when called, and :func:`cunumpy.testing.requires_cupy` + ``NotImplementedError`` when called, and :func:`cunumpy.kernel_testing.requires_cupy` skips tests while the fake is active. Activate it before CuPy or cunumpy's backend is first used, either with the diff --git a/src/cunumpy/algorithms.py b/src/cunumpy/algorithms.py new file mode 100644 index 0000000..361361c --- /dev/null +++ b/src/cunumpy/algorithms.py @@ -0,0 +1,30 @@ +"""Array algorithms missing from NumPy/CuPy, on either backend. + +Morton (Z-order) keys (the same as ``cunumpy/morton.cuh`` computes in a +kernel), a stable sort of several arrays by one key, and sums per key:: + + import cunumpy as xp + + keys = xp.algorithms.morton_keys(positions, lower, upper, levels) + keys, order, positions = xp.algorithms.sort_by_key(keys, positions) + charge = xp.algorithms.segment_sum(q, cell, n_cells) +""" + +from .morton import ( + MAX_MORTON_LEVELS, + morton_decode, + morton_encode, + morton_keys, + morton_scales, +) +from .xp import segment_sum, sort_by_key + +__all__ = [ + "MAX_MORTON_LEVELS", + "morton_decode", + "morton_encode", + "morton_keys", + "morton_scales", + "segment_sum", + "sort_by_key", +] diff --git a/src/cunumpy/cuda/__init__.py b/src/cunumpy/cuda/__init__.py new file mode 100644 index 0000000..8b21ee1 --- /dev/null +++ b/src/cunumpy/cuda/__init__.py @@ -0,0 +1,86 @@ +"""CUDA-only parts of cunumpy: writing and launching CUDA kernels, and the GPU. + +Everything here is only useful with CuPy and a CUDA device. The kernel classes +(:class:`CudaKernel`, :class:`CudaStruct`, ...) need CuPy to launch; the device +functions (:func:`set_device`, :func:`memory_info`, :func:`stream`, ...) do +nothing (or return ``None``/``0``) on the NumPy backend:: + + import cunumpy as xp + + kernel = xp.cuda.CudaKernel(source, "push") + with xp.cuda.stream(): + kernel(positions, velocities, dt, n_threads=n) + +The CUDA headers shipped with cunumpy (``cunumpy/atomic.cuh``, +``cunumpy/random.cuh``, ...) are in :func:`cuda_include_dir`. + +Backend-neutral kernel tools (:class:`~cunumpy.kernels.Kernel`, +:class:`~cunumpy.kernels.PyccelKernel`, ...) are in :mod:`cunumpy.kernels`. + +Importing this module makes ``xp.cuda`` refer to it instead of ``cupy.cuda``; +use ``import cupy; cupy.cuda`` for CuPy's module. +""" + +from ..cuda_kernel import ( + DEBUG_OPTIONS, + CudaArguments, + CudaKernel, + CudaKernelVariants, + CudaParameter, + CudaStruct, + CudaStructArguments, + CudaStructValue, + ctype_of, + cuda_include_dir, + cuda_kernel_names, + include_hash, + parse_cuda_signature, + resolve_includes, + write_cuda_header, +) +from ..xp import ( + DEFAULT_SHARED_MEMORY_PER_BLOCK, + bind_local_device, + cuda_debug, + device_count, + free_memory, + get_cuda_debug, + max_shared_memory_per_block, + memory_info, + pin_memory, + set_cuda_debug, + set_device, + set_device_for_rank, + stream, +) + +__all__ = [ + "DEBUG_OPTIONS", + "DEFAULT_SHARED_MEMORY_PER_BLOCK", + "CudaArguments", + "CudaKernel", + "CudaKernelVariants", + "CudaParameter", + "CudaStruct", + "CudaStructArguments", + "CudaStructValue", + "bind_local_device", + "ctype_of", + "cuda_debug", + "cuda_include_dir", + "cuda_kernel_names", + "device_count", + "free_memory", + "get_cuda_debug", + "include_hash", + "max_shared_memory_per_block", + "memory_info", + "parse_cuda_signature", + "pin_memory", + "resolve_includes", + "set_cuda_debug", + "set_device", + "set_device_for_rank", + "stream", + "write_cuda_header", +] diff --git a/src/cunumpy/cuda/include/cunumpy/morton.cuh b/src/cunumpy/cuda/include/cunumpy/morton.cuh index 82feb74..30495ed 100644 --- a/src/cunumpy/cuda/include/cunumpy/morton.cuh +++ b/src/cunumpy/cuda/include/cunumpy/morton.cuh @@ -25,7 +25,7 @@ // so the keys are bit-identical. Positions must be finite. // // Plain integer and double arithmetic only, so the header also compiles as -// C++ (see cunumpy.testing.emulate_cuda_kernel). +// C++ (see cunumpy.kernel_testing.emulate_cuda_kernel). #ifndef CUNUMPY_MORTON_CUH #define CUNUMPY_MORTON_CUH diff --git a/src/cunumpy/cuda/include/cunumpy/random.cuh b/src/cunumpy/cuda/include/cunumpy/random.cuh index 3aebf35..1892db1 100644 --- a/src/cunumpy/cuda/include/cunumpy/random.cuh +++ b/src/cunumpy/cuda/include/cunumpy/random.cuh @@ -29,7 +29,7 @@ // math functions are not the host's. // // Everything is plain integer arithmetic (no CUDA vector types or intrinsics), -// so the header also compiles as C++ (see cunumpy.testing.emulate_cuda_kernel). +// so the header also compiles as C++ (see cunumpy.kernel_testing.emulate_cuda_kernel). #ifndef CUNUMPY_RANDOM_CUH #define CUNUMPY_RANDOM_CUH diff --git a/src/cunumpy/cuda_kernel.py b/src/cunumpy/cuda_kernel.py index 6b6a8a3..982bae0 100644 --- a/src/cunumpy/cuda_kernel.py +++ b/src/cunumpy/cuda_kernel.py @@ -34,7 +34,7 @@ (:meth:`CudaStruct.to_header`, :func:`write_cuda_header`), so that the Python class is the one definition of the arguments. -In debug mode (``debug=True``, ``xp.set_cuda_debug(True)`` or the environment +In debug mode (``debug=True``, ``xp.cuda.set_cuda_debug(True)`` or the environment variable ``CUNUMPY_CUDA_DEBUG=1``) kernels are compiled with ``-lineinfo`` and ``-DCUNUMPY_BOUNDS_CHECK``, and every launch is synchronized so that an asynchronous CUDA error is raised, as a ``RuntimeError`` naming the kernel, at @@ -639,7 +639,7 @@ def _check_device_array(param: CudaParameter, index: int, value: Any) -> None: raise ValueError( f"{_describe(param, index)} is on CUDA device {device_id}, but the " f"current device is {current}; kernels only take arrays of the " - "current device (see cunumpy.bind_local_device)" + "current device (see cunumpy.cuda.bind_local_device)" ) @@ -855,7 +855,7 @@ def _header_source( if any(f.view_ndim is not None for s in structs for f in s.fields): includes.insert(0, _ARRAY_VIEW_INCLUDE) lines = [ - "// Generated by cunumpy.CudaStruct from the Python definition; do not edit.", + "// Generated by cunumpy.cuda.CudaStruct from the Python definition; do not edit.", f"#ifndef {guard}", f"#define {guard}", "", @@ -1666,8 +1666,8 @@ def _host_state(value: Any) -> Any: class PyccelStructArguments(CudaStructArguments): """Argument object with a pyccel host class and a C struct for CUDA kernels. - The :class:`~cunumpy.KernelArguments` form of :class:`CudaStructArguments`: - the same object is passed to a :class:`~cunumpy.Kernel` on both backends. + The :class:`~cunumpy.kernels.KernelArguments` form of :class:`CudaStructArguments`: + the same object is passed to a :class:`~cunumpy.kernels.Kernel` on both backends. On the device path it arrives as the packed struct (``__cuda_args__()``); on the host path, ``__host_args__()`` builds an instance of :attr:`host_class` (typically a pyccel-compiled argument class, which @@ -1684,7 +1684,7 @@ class PyccelStructArguments(CudaStructArguments): On the CuPy backend there is no host form: the arrays are device arrays, and a host kernel would have to copy them. ``__host_args__()`` raises then, unless :attr:`host_copies` is True, in which case the host object is - built from host copies (counted by :func:`~cunumpy.count_transfers`) and + built from host copies (counted by :func:`~cunumpy.profiling.count_transfers`) and what the host kernel writes is **not** copied back; use it for read-only evaluations only. @@ -1875,7 +1875,7 @@ class CudaKernel: ``-DCUNUMPY_BOUNDS_CHECK``) and synchronize after every launch, so that an asynchronous CUDA error is raised as a ``RuntimeError`` naming this kernel. None (the default) follows the global setting - (:func:`cunumpy.set_cuda_debug`, ``CUNUMPY_CUDA_DEBUG``) at every + (:func:`cunumpy.cuda.set_cuda_debug`, ``CUNUMPY_CUDA_DEBUG``) at every launch; True or False fix it for this kernel. The compile options are fixed when the kernel is compiled. @@ -2072,7 +2072,7 @@ def debug_active(self) -> bool: """Whether debug mode applies to this kernel now. The kernel's own setting if it was created with ``debug=True`` or - ``debug=False``, else the global setting (:func:`cunumpy.get_cuda_debug`), + ``debug=False``, else the global setting (:func:`cunumpy.cuda.get_cuda_debug`), read at the time of the call. """ if self._debug is not None: @@ -2116,7 +2116,7 @@ def compile_options(self) -> tuple[str, ...]: """The NVRTC options a compilation now would use. :attr:`options`, followed by ``-I`` for cunumpy's own header directory - (:func:`cunumpy.cuda_include_dir`) unless already present, then + (:func:`cunumpy.cuda.cuda_include_dir`) unless already present, then :data:`DEBUG_OPTIONS` (``-lineinfo`` and ``-DCUNUMPY_BOUNDS_CHECK``) if :meth:`debug_active` and they are not already among the options (``-G`` is not added: NVRTC does not support @@ -2153,7 +2153,7 @@ def n_threads_from(self) -> Callable[[tuple[Any, ...]], Any] | None: third argument is a struct argument object with the marker count. Set it to ``"first_array"`` for the most common case, one thread per row: the length of the first array argument (its first axis). - Settable, also on the ``cuda_kernel`` of a :class:`~cunumpy.Kernel`. + Settable, also on the ``cuda_kernel`` of a :class:`~cunumpy.kernels.Kernel`. """ return self._n_threads_from @@ -2379,7 +2379,7 @@ def __call__( def _opt_in_shared_memory(self, kernel: Any, shared_mem: int) -> None: """Allow `shared_mem` bytes of dynamic shared memory above the default limit. - Up to :data:`~cunumpy.DEFAULT_SHARED_MEMORY_PER_BLOCK` (48 KiB) every + Up to :data:`~cunumpy.cuda.DEFAULT_SHARED_MEMORY_PER_BLOCK` (48 KiB) every device launches without setup. Above it, newer GPUs need the kernel attribute ``max_dynamic_shared_size_bytes``; it is set once (and again for a larger request) up to the device's opt-in limit. @@ -2517,7 +2517,7 @@ def compile_all( jobs : int | None Number of variants compiled at a time, in threads (NVRTC releases the GIL); None for the number of CPUs. See - :meth:`KernelCatalog.compile_all `. + :meth:`KernelCatalog.compile_all `. """ for key in keys: self.get(*key) diff --git a/src/cunumpy/dispatch.py b/src/cunumpy/dispatch.py index 796f2f8..05b0fef 100644 --- a/src/cunumpy/dispatch.py +++ b/src/cunumpy/dispatch.py @@ -1,8 +1,8 @@ """Pairs of host and CUDA kernels, chosen by the backend or by the arguments. -A :class:`Kernel` holds a host kernel (a :class:`~cunumpy.PyccelKernel`, e.g. a +A :class:`Kernel` holds a host kernel (a :class:`~cunumpy.kernels.PyccelKernel`, e.g. a Pyccel-compiled function) and, optionally, its 1:1 corresponding CUDA kernel -(:class:`~cunumpy.CudaKernel`). It calls the host kernel on the NumPy backend and +(:class:`~cunumpy.cuda.CudaKernel`). It calls the host kernel on the NumPy backend and the CUDA kernel on the CuPy backend (or, with ``dispatch="arrays"``, the CUDA kernel for device arguments and the host kernel for host arguments), so a code base can port its kernels to CUDA one by one. @@ -58,8 +58,8 @@ def _on_device(arg: Any) -> bool: """Whether a kernel argument lives on the GPU. A CuPy array, or a device-only argument object (one with ``__cuda_args__`` - but no ``__host_args__``, e.g. a :class:`~cunumpy.CudaArguments` or a struct - value). A :class:`~cunumpy.KernelArguments` object has both forms and does + but no ``__host_args__``, e.g. a :class:`~cunumpy.cuda.CudaArguments` or a struct + value). A :class:`~cunumpy.kernels.KernelArguments` object has both forms and does not decide. """ if _is_device_array(arg): @@ -155,7 +155,7 @@ class Kernel: ---------- host_kernel : PyccelKernel | callable The host kernel, called on the NumPy backend. A plain callable is - wrapped in a :class:`~cunumpy.PyccelKernel`. + wrapped in a :class:`~cunumpy.kernels.PyccelKernel`. cuda_kernel : CudaKernel | None The CUDA kernel, called on the CuPy backend; None if it has not been written yet. @@ -164,12 +164,12 @@ class Kernel: missing_cuda : {"raise", "fallback"} What happens on the CuPy backend if there is no CUDA kernel: ``"raise"`` raises ``NotImplementedError``; ``"fallback"`` calls the host kernel - through :class:`~cunumpy.PyccelKernel`, which copies the arrays to the + through :class:`~cunumpy.kernels.PyccelKernel`, which copies the arrays to the host and back at every call (a warning is emitted once). cuda_path : str | Path | None Where the CUDA kernel is expected, for the error message if it is missing. host_options : Mapping[str, Any] | None - Keyword arguments for the :class:`~cunumpy.PyccelKernel` that wraps a + Keyword arguments for the :class:`~cunumpy.kernels.PyccelKernel` that wraps a plain callable `host_kernel`, e.g. ``{"object_modules": ("my_pkg.",), "outputs": (2,)}``; they matter for the fallback on the CuPy backend. Not allowed if `host_kernel` already is a ``PyccelKernel``. @@ -178,7 +178,7 @@ class Kernel: the CUDA kernel on the CuPy backend, the host kernel on the NumPy backend. ``"arrays"``: the CUDA kernel if any top-level argument lives on the GPU (a CuPy array, or a device-only argument object such as a - :class:`~cunumpy.CudaArguments` or a struct value), else the host + :class:`~cunumpy.cuda.CudaArguments` or a struct value), else the host kernel, whatever the backend. Use ``"arrays"`` when a code deliberately hands host arrays to kernels while CuPy is active (diagnostics, MPI staging, CPU fallbacks): the host kernel then runs on the host arrays @@ -188,8 +188,8 @@ class Kernel: ----- Both kernels take the same arguments, except that the CUDA kernel gets the launch shape (``n_threads`` or ``grid``) and argument objects in their CUDA - form (see :class:`~cunumpy.CudaArguments` and :class:`~cunumpy.CudaStruct`). - An argument object implementing :class:`~cunumpy.KernelArguments` is + form (see :class:`~cunumpy.cuda.CudaArguments` and :class:`~cunumpy.cuda.CudaStruct`). + An argument object implementing :class:`~cunumpy.kernels.KernelArguments` is replaced by its ``__host_args__()`` on the host path and flattened via ``__cuda_args__()`` on the CUDA path, so the call site is the same on both backends. @@ -267,14 +267,14 @@ def from_folder( * ``_numpy.py``: ``"numpy"``; * ````: the CUDA kernel. - The host implementations form a :class:`~cunumpy.HostImplementations`: - a call runs the one set with :func:`~cunumpy.set_kernel_implementation`, + The host implementations form a :class:`~cunumpy.kernels.HostImplementations`: + a call runs the one set with :func:`~cunumpy.kernels.set_kernel_implementation`, or by default the first available of pyccel, numba and NumPy. The folder's own ``__init__.py`` can declare its kernel with this method, so that the kernel is imported from where it is written:: # my_sim/kernels/push/__init__.py - kernel = xp.Kernel.from_folder( + kernel = xp.kernels.Kernel.from_folder( __name__, host_suffix="_pyccel", compile_host=compile, dispatch="arrays" ) @@ -425,7 +425,7 @@ def test_args_module(self) -> str | None: """Dotted name of the module with the test arguments of this kernel, or None. Set by :meth:`KernelCatalog.from_package` for a kernel folder that - contains ``_test_args.py``; see :func:`cunumpy.testing.check_parity`. + contains ``_test_args.py``; see :func:`cunumpy.kernel_testing.check_parity`. """ return self._test_args_module @@ -439,7 +439,7 @@ def test_args(self) -> ModuleType | None: def host_parameters(self) -> list[str] | None: """The parameter names of the host kernel, or None if they are unknown. - Read from the Python function (for a :class:`~cunumpy.CompiledHostKernel`, + Read from the Python function (for a :class:`~cunumpy.kernels.CompiledHostKernel`, its uncompiled Python version), or, for a pyccel-compiled function without a Python signature, from the ``__pyccel__/.pyi`` stub pyccel writes next to the extension module. None if neither is @@ -458,9 +458,9 @@ def check_signature(self) -> None: Compares the parameter names of the host function with those of the parsed ``__global__`` signature, with those of the other host - implementations of a :class:`~cunumpy.HostImplementations` (numba, + implementations of a :class:`~cunumpy.kernels.HostImplementations` (numba, NumPy; nothing is compiled, an implementation that fails to import is - skipped) and with the fallback of a :class:`~cunumpy.CompiledHostKernel`. + skipped) and with the fallback of a :class:`~cunumpy.kernels.CompiledHostKernel`. A side is skipped without a CUDA kernel, without a parsed CUDA signature (``check_signature=False``), or without a Python signature. @@ -502,7 +502,7 @@ def check_signature(self) -> None: def implementations(self) -> tuple[str, ...]: """Names of the implementations, e.g. ``("pyccel", "numpy", "python", "cuda")``. - A host kernel that is not a :class:`~cunumpy.HostImplementations` counts + A host kernel that is not a :class:`~cunumpy.kernels.HostImplementations` counts as ``"host"``. """ function = self._host_kernel.kernel @@ -515,9 +515,9 @@ def selected(self, device: bool = False) -> str: """The implementation a call with host (or `device`) arguments runs now. For host arguments: the setting of - :func:`~cunumpy.set_kernel_implementation` or the default (loads it), or + :func:`~cunumpy.kernels.set_kernel_implementation` or the default (loads it), or ``"host"`` for a host kernel that is not a - :class:`~cunumpy.HostImplementations`. For device arguments ``"cuda"``, + :class:`~cunumpy.kernels.HostImplementations`. For device arguments ``"cuda"``, or ``"host"`` if there is no CUDA kernel and ``missing_cuda="fallback"``. Useful to check that a run does not use a slow path. """ @@ -592,11 +592,11 @@ def __call__( ---------- *args Kernel arguments. Objects implementing - :class:`~cunumpy.KernelArguments` are resolved per backend (see - :func:`~cunumpy.resolve_host_args`). + :class:`~cunumpy.kernels.KernelArguments` are resolved per backend (see + :func:`~cunumpy.kernels.resolve_host_args`). n_threads, grid, block, shared_mem, stream Launch configuration of the CUDA kernel, see - :meth:`CudaKernel.__call__ `; + :meth:`CudaKernel.__call__ `; `n_threads` (or `grid`) is required when the CUDA kernel is called. Ignored by the host kernel. """ @@ -686,7 +686,7 @@ def from_package( is the CUDA kernel (with a ``__global__`` function ````). Other ``__global__`` functions in that file are ignored by the catalog; load them with :meth:`CudaKernel.all_from_file - `. + `. Parameters ---------- @@ -700,7 +700,7 @@ def from_package( Module name suffix of the test arguments: ``.py`` in the kernel's folder, if present, is recorded as :attr:`Kernel.test_args_module` (imported only when a test asks for - it, see :func:`cunumpy.testing.check_parity`). None disables this. + it, see :func:`cunumpy.kernel_testing.check_parity`). None disables this. check_name_length : bool Warn about a kernel whose module name is too long for the Fortran backend of pyccel: the wrapper module ``bind_c_`` @@ -709,7 +709,7 @@ def from_package( missing_cuda : {"raise", "fallback"} Passed on to every :class:`Kernel`. host_options : Mapping | Callable[[str], Mapping] | None - Keyword arguments for the :class:`~cunumpy.PyccelKernel` wrapping each + Keyword arguments for the :class:`~cunumpy.kernels.PyccelKernel` wrapping each host kernel (see :class:`Kernel`): the same for all kernels, or a function of the kernel name, e.g. to declare per-kernel ``outputs``. include_dirs : Sequence[str | Path] | None @@ -724,7 +724,7 @@ def from_package( Compiles a host kernel module, e.g. a function wrapping ``pyccel.epyccel`` (cunumpy does not compile anything itself). Each host kernel then is a - :class:`~cunumpy.CompiledHostKernel`: compiled on its first + :class:`~cunumpy.kernels.CompiledHostKernel`: compiled on its first call, falling back to `host_fallback`, or to the uncompiled Python function with a warning, if compilation fails. By default the Python function is @@ -829,7 +829,7 @@ def parity_cases(self) -> list[tuple[str, Kernel]]: """The ``(name, kernel)`` pairs of the kernels that have a CUDA kernel. For a parametrised parity test of the whole catalog with - :func:`cunumpy.testing.assert_kernels_agree`:: + :func:`cunumpy.kernel_testing.assert_kernels_agree`:: @pytest.mark.parametrize("name, kernel", catalog.parity_cases()) def test_parity(name, kernel): diff --git a/src/cunumpy/emulation.py b/src/cunumpy/emulation.py index 453a7b7..eb83560 100644 --- a/src/cunumpy/emulation.py +++ b/src/cunumpy/emulation.py @@ -1,7 +1,7 @@ """Run a CUDA kernel on the CPU, one thread after another, for tests without a GPU. A ported kernel is usually checked against its host version on a GPU -(:func:`cunumpy.testing.assert_kernels_agree`). Without one, CI cannot run that +(:func:`cunumpy.kernel_testing.assert_kernels_agree`). Without one, CI cannot run that check, and the kernel's index and weight arithmetic go untested. :func:`emulate_cuda_kernel` closes that gap: it compiles the kernel source as C++ with the CUDA built-ins replaced by plain C++ (``threadIdx``, ``blockIdx``, @@ -9,7 +9,7 @@ thread, serially, on copies of the NumPy arguments, and copies the arrays back, so the call looks like a launch:: - from cunumpy.testing import emulate_cuda_kernel + from cunumpy.kernel_testing import emulate_cuda_kernel y = np.zeros(1000) emulate_cuda_kernel(axpy, 2.0, x, y, 1000, n_threads=1000) diff --git a/src/cunumpy/fusion.py b/src/cunumpy/fusion.py index af72edf..7c4cd23 100644 --- a/src/cunumpy/fusion.py +++ b/src/cunumpy/fusion.py @@ -1,4 +1,4 @@ -"""``xp.fuse``: elementwise functions as one GPU kernel, plain calls on the host. +"""``xp.kernels.fuse``: elementwise functions as one GPU kernel, plain calls on the host. A chain of elementwise operations (a pressure from density and temperature, fluxes and limiters of a fluid update, a Maxwellian at many velocities) runs @@ -8,7 +8,7 @@ and calls the function as it is otherwise, so the same code runs on both backends:: - @xp.fuse + @xp.kernels.fuse def pressure(rho, T, gamma): return (gamma - 1.0) * rho * T @@ -73,7 +73,7 @@ def fuse( ) -> F | Callable[[F], F]: """Fuse an elementwise function into one kernel when called with CuPy arrays. - Usable as ``@xp.fuse`` or ``@xp.fuse(kernel_name="pressure")``. + Usable as ``@xp.kernels.fuse`` or ``@xp.kernels.fuse(kernel_name="pressure")``. Parameters ---------- diff --git a/src/cunumpy/kernel.py b/src/cunumpy/kernel.py index 68b9151..ccc9932 100644 --- a/src/cunumpy/kernel.py +++ b/src/cunumpy/kernel.py @@ -15,8 +15,8 @@ Argument objects that have a host and a device form implement the :class:`KernelArguments` protocol: ``__host_args__()`` returns what the host kernel receives in that position, ``__cuda_args__()`` what a -:class:`~cunumpy.CudaKernel` receives. :class:`PyccelKernel` and -:class:`~cunumpy.Kernel` resolve ``__host_args__()`` with +:class:`~cunumpy.cuda.CudaKernel` receives. :class:`PyccelKernel` and +:class:`~cunumpy.kernels.Kernel` resolve ``__host_args__()`` with :func:`resolve_host_args` before calling the host kernel. """ @@ -48,15 +48,15 @@ class KernelArguments: of a particle species, usually exists twice: as an object holding NumPy arrays for the host kernel (e.g. a Pyccel class) and as device arrays for the CUDA kernel. This protocol lets one object represent both, so that a - call to a :class:`~cunumpy.Kernel` never branches on the backend: + call to a :class:`~cunumpy.kernels.Kernel` never branches on the backend: * ``__host_args__()`` returns the single object that the host kernel receives in that position; :class:`PyccelKernel` and - :class:`~cunumpy.Kernel` resolve it (top-level positional and keyword + :class:`~cunumpy.kernels.Kernel` resolve it (top-level positional and keyword arguments only) before calling the host kernel; * ``__cuda_args__()`` returns the tuple of CUDA kernel arguments the object - stands for (the :class:`~cunumpy.CudaArguments` protocol); - :class:`~cunumpy.CudaKernel` flattens it into the kernel parameters. + stands for (the :class:`~cunumpy.cuda.CudaArguments` protocol); + :class:`~cunumpy.cuda.CudaKernel` flattens it into the kernel parameters. Subclassing is optional: any object whose *type* defines a callable ``__host_args__`` is resolved (an instance attribute of that name is not). @@ -79,7 +79,7 @@ class KernelArguments: ... self._kernel_args = ParticleArguments(self) ... return self._kernel_args ... - >>> class ParticleArguments(xp.KernelArguments): + >>> class ParticleArguments(xp.kernels.KernelArguments): ... def __init__(self, particles): ... self._particles = particles ... self._host = None @@ -206,7 +206,7 @@ class PyccelKernel: Top-level arguments implementing :class:`KernelArguments` (a ``__host_args__()`` method on their type) are replaced by their host form before anything else, so the same argument objects can be passed to a - ``PyccelKernel`` and to a :class:`~cunumpy.CudaKernel`. + ``PyccelKernel`` and to a :class:`~cunumpy.cuda.CudaKernel`. Examples -------- @@ -556,7 +556,7 @@ def get_kernel_implementation() -> str | None: def use_kernel_implementation(name: str | None) -> Iterator[None]: """Temporarily choose the host implementation, like :func:`use_backend`. - For tests and benchmarks, e.g. ``with xp.use_kernel_implementation("numpy"):`` + For tests and benchmarks, e.g. ``with xp.kernels.use_kernel_implementation("numpy"):`` to run the code path of a machine without pyccel. The setting is global, not per thread. """ diff --git a/src/cunumpy/kernel_testing.py b/src/cunumpy/kernel_testing.py new file mode 100644 index 0000000..a4da2d8 --- /dev/null +++ b/src/cunumpy/kernel_testing.py @@ -0,0 +1,634 @@ +"""Helpers for testing host/CUDA kernel pairs with pytest. + +A code base that ports its kernels to CUDA one by one needs the same test for +every kernel: build the arguments on both backends, run the host kernel and the +CUDA kernel, and compare what they wrote. This module provides that test +(:func:`assert_kernels_agree`), the pytest markers to parametrize tests over +the backends (:data:`BACKENDS`, :data:`requires_cupy`, the :func:`backend` +fixture), :func:`device_function_kernel`, which wraps a ``__device__`` +function in an elementwise ``__global__`` kernel so that device helpers can be +tested from Python without a hand-written test kernel, and +:func:`emulate_cuda_kernel` (from :mod:`cunumpy.emulation`), which runs a CUDA +kernel on the CPU, one thread after another, so that its arithmetic can be +checked against the host kernel in CI without a GPU. Without a GPU, the CuPy +code paths of a program (argument objects, conversions, backend branches) can +still run on the fake CuPy of :mod:`cunumpy._fake_cupy` (:func:`install_fake_cupy`, +or ``CUNUMPY_FAKE_CUPY=1``); :func:`fake_cupy_active` tells whether it is in +use, and ``requires_cupy`` skips the tests that launch kernels then. + +A catalog's parity tests need no code per kernel when each kernel folder +holds ``_test_args.py`` with ``make_args(backend, seed)`` (and +``N_THREADS``): :func:`parity_cases` and :func:`check_parity` drive +:func:`assert_kernels_agree` from these modules. + +The module imports pytest only when one of its pytest objects is used, so it +can be imported (e.g. for :func:`device_function_kernel`) without pytest, and +``import cunumpy`` never imports pytest. + +Examples +-------- +One parametrised test covers every kernel of a catalog that has a CUDA +version:: + + import pytest + from cunumpy.kernel_testing import assert_kernels_agree + + from my_kernels import catalog + + + def make_args(backend, seed): + rng = xp.rng.get_rng(seed) # the backend is active: arrays land on it + x = rng.random(1000) + return (x, 2.0, x.size) + + + @pytest.mark.parametrize("name, kernel", catalog.parity_cases()) + def test_parity(name, kernel): + assert_kernels_agree(kernel, make_args, n_threads=1000) +""" + +from __future__ import annotations + +import re +from collections.abc import Callable, Sequence +from typing import Any + +import array_api_compat +import numpy as np + +from . import _fake_cupy +from .cuda_kernel import ( + CudaKernel, + CudaParameter, + CudaStructArguments, + CudaStructValue, + _parse_parameter, + _split_top_level, + _strip_comments, +) +from .dispatch import Kernel +from .emulation import emulate_cuda_kernel, emulation_compiler +from .xp import cupy_available, get_backend, to_numpy, use_backend + +# the pytest objects are created on first access, see __getattr__ +__all__ = [ + "BACKENDS", # noqa: F822 + "assert_kernels_agree", + "backend", # noqa: F822 + "check_parity", + "device_function_kernel", + "emulate_cuda_kernel", + "emulation_compiler", + "fake_cupy_active", + "install_fake_cupy", + "parity_cases", + "requires_cupy", # noqa: F822 +] + +SKIP_REASON = "CuPy/GPU not available" +FAKE_SKIP_REASON = "the fake CuPy cannot run CUDA kernels" + + +def fake_cupy_active() -> bool: + """Whether the fake CuPy (:mod:`cunumpy._fake_cupy`) stands in for CuPy.""" + return _fake_cupy.is_active() + + +def install_fake_cupy() -> Any: + """Install the fake CuPy for this process; see :mod:`cunumpy._fake_cupy`. + + Call it before the first backend use (e.g. at the top of ``conftest.py``), + or set ``CUNUMPY_FAKE_CUPY=1`` in the environment instead. Returns the + fake ``cupy`` module. + """ + return _fake_cupy.install() + + +def _can_launch() -> bool: + """Whether CUDA kernels can run: a functional CuPy that is not the fake.""" + return cupy_available() and not fake_cupy_active() + + +# pytest objects, built on first use so that importing this module does not +# import pytest (see __getattr__ below) +_LAZY: dict[str, Any] = {} + + +def _pytest() -> Any: + try: + import pytest + except ImportError: # pragma: no cover - pytest is installed in the test suite + raise ImportError( + "cunumpy.kernel_testing needs pytest for this feature: pip install pytest" + ) from None + return pytest + + +def _build_lazy() -> None: + pytest = _pytest() + requires_cupy = pytest.mark.skipif( + not _can_launch(), + reason=FAKE_SKIP_REASON if fake_cupy_active() else SKIP_REASON, + ) + backends = ["numpy", pytest.param("cupy", marks=requires_cupy)] + + @pytest.fixture(params=backends) + def backend(request): + """Run the test once per backend, with that backend active.""" + with use_backend(request.param): + yield request.param + + _LAZY.update(requires_cupy=requires_cupy, BACKENDS=backends, backend=backend) + + +def __getattr__(name: str) -> Any: + if name in ("requires_cupy", "BACKENDS", "backend"): + if not _LAZY: + _build_lazy() + return _LAZY[name] + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") + + +# --------------------------------------------------------------------------- +# assert_kernels_agree +# --------------------------------------------------------------------------- + + +def _is_array(value: Any) -> bool: + return array_api_compat.is_numpy_array(value) or array_api_compat.is_cupy_array( + value + ) + + +def _arrays_in(value: Any, name: str, found: dict[str, Any], depth: int) -> None: + """Record the arrays in `value` under `name`, looking `depth` levels deep.""" + if _is_array(value): + found[name] = value + elif depth == 0: + return + elif isinstance(value, (CudaStructArguments, CudaStructValue)) and hasattr( + value, "struct" + ): + # by field name, like the attributes of the host argument object; the + # fields of a CudaStructArguments may be properties (not in vars()) + for field in value.struct.fields: + item = ( + value[field.name] + if isinstance(value, CudaStructValue) + else getattr(value, field.name) + ) + _arrays_in(item, f"{name}.{field.name}", found, depth - 1) + elif isinstance(value, (tuple, list)): + for i, item in enumerate(value): + _arrays_in(item, f"{name}[{i}]", found, depth - 1) + elif isinstance(value, dict): + for key, item in value.items(): + _arrays_in(item, f"{name}[{key!r}]", found, depth - 1) + elif hasattr(value, "__dict__"): + for attr, item in vars(value).items(): + _arrays_in(item, f"{name}.{attr}", found, depth - 1) + + +def _collect_arrays( + args: Sequence[Any], outputs: Sequence[int] | None = None +) -> dict[str, Any]: + """The arrays among `args` (or among the arguments `outputs`), by name. + + An argument that is an array is named ``"argument "``; arrays found one + level deep, in a tuple, list or dict argument or in the attributes of an + argument object (e.g. a ``CudaArguments`` object), are named + ``"argument []"`` or ``"argument ."``, and arrays in a + container attribute of an object ``"argument .[]"``. A + :class:`~cunumpy.cuda.CudaStructArguments` object or a struct value is read + through its struct fields, ``"argument ."``, so that its arrays + get the names of the attributes of the host argument object it mirrors, + also when the fields are properties. + """ + indices = range(len(args)) if outputs is None else outputs + found: dict[str, Any] = {} + for entry in indices: + if not isinstance(entry, int) or isinstance(entry, bool): + raise TypeError( + "outputs must be positional argument indices (kernels take " + f"positional arguments only), got {entry!r}" + ) + index = entry + len(args) if entry < 0 else entry + if not 0 <= index < len(args): + raise IndexError( + f"output argument {entry} does not exist: there are {len(args)} " + "arguments" + ) + _arrays_in(args[index], f"argument {index}", found, depth=2) + return found + + +def _compare_results( + host: dict[str, Any], + device: dict[str, Any], + rtol: float, + atol: float, + kernel_name: str = "kernel", +) -> None: + """Compare the arrays of `device` with those of `host`, by name. + + Raises + ------ + AssertionError + If the two do not hold the same names, or an array differs (the + message names the argument). + """ + if host.keys() != device.keys(): + raise AssertionError( + f"{kernel_name}: the host and CUDA calls do not have the same array " + f"arguments: host {sorted(host)}, CUDA {sorted(device)}" + ) + for name, expected in host.items(): + np.testing.assert_allclose( + to_numpy(device[name]), + to_numpy(expected), + rtol=rtol, + atol=atol, + err_msg=f"{kernel_name}: {name} differs between the host and CUDA kernels", + ) + + +def assert_kernels_agree( + kernel: Kernel, + make_args: Callable[[str, int], Sequence[Any]], + *, + n_threads: int | Sequence[int] | Callable[[tuple[Any, ...]], Any] | None = None, + grid: int | Sequence[int] | None = None, + block: int | Sequence[int] | None = None, + rtol: float = 1e-12, + atol: float = 0.0, + n_calls: int = 1, + outputs: Sequence[int] | None = None, + seed: int = 0, +) -> dict[str, np.ndarray]: + """Check that the host and CUDA versions of `kernel` compute the same. + + For each backend, ``"numpy"`` then ``"cupy"``, the backend is activated + with :func:`~cunumpy.use_backend`, the arguments are built with + ``make_args(backend, seed)``, the kernel is called `n_calls` times, and + the arrays among the arguments are collected. The arrays written by the + CUDA kernel are then copied to the host and compared with those of the host + kernel using ``numpy.testing.assert_allclose``. + + Parameters + ---------- + kernel : Kernel + A kernel with a CUDA version (``kernel.has_cuda``). + make_args : Callable[[str, int], Sequence] + ``make_args(backend, seed)`` returns the positional arguments of the + kernel, as a tuple or list. It is called with the backend (``"numpy"`` + or ``"cupy"``) active, so arrays created through ``cunumpy`` (e.g. with + ``xp.zeros`` or ``xp.rng.get_rng(seed)``) land on that backend; NumPy and + CuPy random generators do not produce the same sequence from one seed, + so build random data on the host with ``numpy.random.default_rng(seed)`` + and convert it with :func:`~cunumpy.to_cunumpy`. Kernels take positional + arguments only. + n_threads, grid, block + Launch configuration of the CUDA kernel, see + :meth:`CudaKernel.__call__ `. `n_threads` + may also be a function of the tuple of arguments, e.g. + ``lambda args: args[0].shape[0]``. One of `n_threads` and `grid` is + required unless the CUDA kernel has ``n_threads_from``. + rtol, atol : float + Tolerances of ``numpy.testing.assert_allclose``. + n_calls : int + How many times the kernel is called on each backend (e.g. to test a + kernel that accumulates). + outputs : Sequence[int] | None + Indices of the arguments to compare (negative indices count from the + end), like ``PyccelKernel(outputs=...)``. By default the ``outputs`` + declared by the host kernel are used, and if it declares none, every + argument. An argument that is an array is compared; for a tuple, list, + dict or object argument (e.g. a ``CudaArguments`` object), the arrays + it holds are compared (one level deep, plus arrays in a container + attribute of an object). + seed : int + Passed to `make_args` on both backends. + + Returns + ------- + dict[str, numpy.ndarray] + The arrays of the host call by argument name (``"argument 0"``, + ``"argument 1.x"``), for further checks. + + Raises + ------ + ValueError + If `kernel` has no CUDA version. + AssertionError + If an array differs; the message names the argument. + + Notes + ----- + The test is skipped with ``pytest.skip`` if CuPy or a GPU is not available, + or if the fake CuPy is active. + """ + if not isinstance(kernel, Kernel): + raise TypeError(f"expected a Kernel, got {type(kernel).__name__}") + if not kernel.has_cuda: + raise ValueError(f"kernel {kernel.name!r} has no CUDA version") + if n_threads is None and grid is None and kernel.cuda_kernel.n_threads_from is None: + raise TypeError("n_threads (or grid) is required to launch the CUDA kernel") + if n_calls < 1: + raise ValueError(f"n_calls must be at least 1, got {n_calls}") + if outputs is None: + outputs = kernel.host_kernel.outputs + if fake_cupy_active(): + _pytest().skip(FAKE_SKIP_REASON) + if not cupy_available(): + _pytest().skip(SKIP_REASON) + + results = {} + for backend in ("numpy", "cupy"): + with use_backend(backend): + if get_backend() != backend: # pragma: no cover - cupy_available() lied + raise RuntimeError(f"could not activate the {backend} backend") + args = tuple(make_args(backend, seed)) + launch = n_threads(args) if callable(n_threads) else n_threads + for _ in range(n_calls): + kernel(*args, n_threads=launch, grid=grid, block=block) + results[backend] = _collect_arrays(args, outputs) + + host = {name: to_numpy(a) for name, a in results["numpy"].items()} + _compare_results(host, results["cupy"], rtol, atol, kernel.name) + return host + + +# --------------------------------------------------------------------------- +# parity tests from _test_args.py modules +# --------------------------------------------------------------------------- + +#: Module-level names a ``_test_args.py`` module may define, and the +#: keyword of :func:`assert_kernels_agree` each one sets. +TEST_ARGS_SETTINGS = { + "N_THREADS": "n_threads", + "GRID": "grid", + "BLOCK": "block", + "RTOL": "rtol", + "ATOL": "atol", + "N_CALLS": "n_calls", + "OUTPUTS": "outputs", + "SEED": "seed", +} + + +def parity_cases(catalog: Any) -> list[Any]: + """The kernels of a catalog with a CUDA version, as pytest parameters. + + One ``pytest.param(kernel, id=name)`` per kernel of + ``catalog.parity_cases()``. A kernel without a test-arguments module + (:attr:`Kernel.test_args_module `, from + ``_test_args.py`` in its folder) is marked ``skip`` with a reason + naming the missing file, so the report shows which kernels still lack + their parity test:: + + @pytest.mark.parametrize("kernel", parity_cases(catalog)) + def test_parity(kernel): + check_parity(kernel) + """ + pytest = _pytest() + cases = [] + for name, kernel in catalog.parity_cases(): + marks = () + if kernel.test_args_module is None: + marks = ( + pytest.mark.skip( + reason=f"no test arguments for {name!r}: add {name}_test_args.py " + "with make_args(backend, seed) and N_THREADS to its folder" + ), + ) + cases.append(pytest.param(kernel, id=name, marks=marks)) + return cases + + +def check_parity(kernel: Kernel, **overrides: Any) -> dict[str, np.ndarray]: + """Run :func:`assert_kernels_agree` with the kernel's test-arguments module. + + The module (``_test_args.py`` in the kernel's folder, see + :meth:`KernelCatalog.from_package `) + defines ``make_args(backend, seed)`` and, as module-level names, the + launch and comparison settings of :data:`TEST_ARGS_SETTINGS`: + ``N_THREADS`` (an integer, a tuple, or a function of the argument tuple), + or ``GRID``, plus optionally ``BLOCK``, ``RTOL``, ``ATOL``, ``N_CALLS``, + ``OUTPUTS`` and ``SEED``. Keyword arguments override them. + + Returns + ------- + dict[str, numpy.ndarray] + The host arrays, as :func:`assert_kernels_agree` returns them. + + Raises + ------ + ValueError + If the kernel has no test-arguments module. + TypeError + If the module has no callable ``make_args``. + """ + module = kernel.test_args + if module is None: + raise ValueError( + f"kernel {kernel.name!r} has no test-arguments module: add " + f"{kernel.name}_test_args.py with make_args(backend, seed) to its folder" + ) + make_args = getattr(module, "make_args", None) + if not callable(make_args): + raise TypeError(f"{module.__name__} must define make_args(backend, seed)") + settings = { + keyword: getattr(module, name) + for name, keyword in TEST_ARGS_SETTINGS.items() + if hasattr(module, name) + } + settings.update(overrides) + return assert_kernels_agree(kernel, make_args, **settings) + + +# --------------------------------------------------------------------------- +# device_function_kernel +# --------------------------------------------------------------------------- + +_PROTOTYPE = re.compile( + r"^\s*(?P.+?)\s*\b(?P[A-Za-z_]\w*)\s*\((?P.*)\)\s*;?\s*$", + re.DOTALL, +) +_FUNCTION_QUALIFIERS = {"__device__", "__host__", "__forceinline__", "inline", "static"} + + +def _parse_prototype( + signature: str, + structs: dict[str, Any] | None = None, +) -> tuple[CudaParameter | None, str, list[tuple[str, CudaParameter]]]: + """Parse a C function prototype into (result, name, [(text, parameter)]). + + The result is None for a ``void`` function; each parameter is its original + text together with its parsed form. A struct of `structs` may be taken by + value or by (const) reference. + """ + match = _PROTOTYPE.match(_strip_comments(signature)) + if match is None: + raise ValueError(f"cannot parse the function prototype {signature!r}") + name = match.group("name") + result_words = [ + w for w in match.group("result").split() if w not in _FUNCTION_QUALIFIERS + ] + result_text = " ".join(result_words) + if result_text == "void": + result = None + else: + try: + result = _parse_parameter(f"{result_text} result") + except ValueError: + raise ValueError( + f"unsupported return type {result_text!r} of {name!r}: a scalar " + "type (or void) is required" + ) from None + if result.pointer: + raise ValueError( + f"{name!r} returns a pointer; only scalar results can be collected" + ) + params_text = match.group("params").strip() + params = [] + if params_text not in ("", "void"): + for text in _split_top_level(params_text): + text = text.strip() + by_reference = "&" in text + param = _parse_parameter(text.replace("&", " "), structs or None) + if by_reference and param.struct is None: + raise ValueError( + f"unsupported parameter {text!r} of {name!r}: only structs can " + "be passed by reference" + ) + params.append((text, param)) + return result, name, params + + +def device_function_kernel( + header_source: str, + signature: str, + *, + name: str | None = None, + includes: Sequence[str] = (), + n_threads_param: str = "n", + out_param: str = "out", + **kwargs: Any, +) -> CudaKernel: + """Wrap a ``__device__`` function in an elementwise kernel, for testing. + + Generates an ``extern "C" __global__`` kernel that calls the device function + once per thread, so that a device helper (e.g. a B-spline evaluation) can be + run from Python on many inputs at once and compared with its host version. + + Parameters + ---------- + header_source : str + CUDA source defining the device function (typically the content of the + header, or ``#include`` directives, see `includes`). + signature : str + The C prototype of the device function, e.g. + ``"int find_span(const double* t, int p, double eta)"``. Pointer + parameters, scalar parameters of the types :class:`CudaKernel` supports, + struct parameters (by value or by ``const`` reference, for the structs + passed in ``structs``) and a scalar or ``void`` return type are + supported. + name : str | None + Name of the generated kernel; ``"_kernel"`` by default. + includes : Sequence[str] + Headers to include before `header_source`: ``"bsplines.cuh"`` becomes + ``#include "bsplines.cuh"``, ``""`` is included with + angle brackets. Pass ``include_dirs`` for the directories. + n_threads_param : str + Name of the generated kernel's last parameter, the number of elements. + out_param : str + Name of the generated output array parameter. + **kwargs + Passed on to :class:`CudaKernel`, e.g. ``include_dirs``, ``options``, + ``block_size`` or ``structs`` (the :class:`~cunumpy.cuda.CudaStruct` types + of struct parameters, whose definitions `header_source` or the + `includes` must provide). + + Returns + ------- + CudaKernel + The wrapper kernel. Its parameters are those of the device function, in + order, followed by the output array (unless the function returns + ``void``) and the number of elements: + + * a pointer parameter stays as it is and is passed through unchanged to + every call (an array shared by all threads); + * a struct parameter (``DomainArgs d`` or ``const DomainArgs& d``) is + taken by value and passed through unchanged to every call (pass a + :class:`~cunumpy.cuda.CudaStructArguments` object or a packed value); + * a scalar parameter ``T x`` becomes a device array ``const T* x`` of + length ``n``, and thread ``i`` calls the function with ``x[i]``; + * the return value of thread ``i`` is stored in ``out[i]``, an array + ``R* out`` of length ``n`` where ``R`` is the return type; + * ``int n`` is the number of elements (threads ``i >= n`` do nothing). + + Launch it with ``n_threads=n``. + + Raises + ------ + ValueError + If the prototype cannot be parsed, the return type is not a scalar type + or ``void``, a parameter has an unsupported type, or a parameter is + named like `out_param` or `n_threads_param`. + + Examples + -------- + >>> from cunumpy.kernel_testing import device_function_kernel + >>> sq = device_function_kernel( + ... "__device__ double sq(double x) { return x * x; }", + ... "double sq(double x)", + ... ) + >>> sq.signature # (const double* x, double* out, int n) + >>> x = cp.arange(10.0); out = cp.empty(10) # doctest: +SKIP + >>> sq(x, out, 10, n_threads=10) # doctest: +SKIP + """ + structs = {struct.name: struct for struct in kwargs.get("structs", ())} + result, function, params = _parse_prototype(signature, structs) + if name is None: + name = f"{function}_kernel" + reserved = {out_param, n_threads_param} + for _, param in params: + if param.name in reserved: + raise ValueError( + f"the parameter {param.name!r} of {function!r} clashes with the " + f"generated parameter of that name; pass another out_param or " + "n_threads_param" + ) + + wrapper_params, call_args = [], [] + for text, param in params: + if param.pointer: + wrapper_params.append(text) + call_args.append(param.name) + elif param.struct is not None: + wrapper_params.append(f"{param.ctype} {param.name}") # by value + call_args.append(param.name) + else: + wrapper_params.append(f"const {param.ctype}* {param.name}") + call_args.append(f"{param.name}[i]") + call = f"{function}({', '.join(call_args)})" + if result is None: + body = f"{call};" + else: + wrapper_params.append(f"{result.ctype}* {out_param}") + body = f"{out_param}[i] = {call};" + wrapper_params.append(f"int {n_threads_param}") + + include_lines = "".join( + f"#include {h}\n" if h.startswith("<") else f'#include "{h}"\n' + for h in includes + ) + source = ( + f"{include_lines}{header_source}\n\n" + f'extern "C" __global__ void {name}({", ".join(wrapper_params)}) {{\n' + f" int i = blockDim.x * blockIdx.x + threadIdx.x;\n" + f" if (i >= {n_threads_param}) return;\n" + f" {body}\n" + f"}}\n" + ) + return CudaKernel(source, name, **kwargs) diff --git a/src/cunumpy/kernels.py b/src/cunumpy/kernels.py new file mode 100644 index 0000000..0028e3c --- /dev/null +++ b/src/cunumpy/kernels.py @@ -0,0 +1,54 @@ +"""Kernels that run on either backend: dispatch, host implementations, fusion. + +A :class:`Kernel` pairs a host implementation (Pyccel, numba, NumPy or plain +Python) with an optional CUDA version and runs the one matching the arrays it +is given; a :class:`KernelCatalog` loads every kernel of a package:: + + import cunumpy as xp + + push = xp.kernels.Kernel.from_folder("my_code.kernels.push") + push(positions, velocities, dt) + +Which host implementation runs is set with :func:`set_kernel_implementation` +or :func:`use_kernel_implementation`. :func:`as_kernel_array` and +:func:`kernel_output` bring the arguments of a kernel to the side of its main +array. :func:`fuse` turns an elementwise function into one CuPy kernel. + +The CUDA-only classes (:class:`~cunumpy.cuda.CudaKernel`, ...) are in +:mod:`cunumpy.cuda`; the pytest helpers for kernel pairs are in +:mod:`cunumpy.kernel_testing`. +""" + +from .cuda_kernel import PyccelStructArguments +from .dispatch import Kernel, KernelCatalog +from .fusion import fuse +from .kernel import ( + HOST_IMPLEMENTATIONS, + CompiledHostKernel, + HostImplementations, + KernelArguments, + PyccelKernel, + get_kernel_implementation, + resolve_host_args, + set_kernel_implementation, + use_kernel_implementation, +) +from .xp import as_kernel_array, kernel_output + +__all__ = [ + "HOST_IMPLEMENTATIONS", + "CompiledHostKernel", + "HostImplementations", + "Kernel", + "KernelArguments", + "KernelCatalog", + "PyccelKernel", + "PyccelStructArguments", + "as_kernel_array", + "fuse", + "get_kernel_implementation", + "kernel_output", + "resolve_host_args", + "set_kernel_implementation", + "use_kernel_implementation", +] diff --git a/src/cunumpy/memory.py b/src/cunumpy/memory.py new file mode 100644 index 0000000..29582b6 --- /dev/null +++ b/src/cunumpy/memory.py @@ -0,0 +1,22 @@ +"""Moving data between host and device without stalling, on either backend. + +:class:`HostStaging` copies device arrays to the host in the background (for +output), and :class:`DeviceMirror` pairs a host buffer owned by another +library with a device copy:: + + import cunumpy as xp + + staging = xp.memory.HostStaging(rho.shape, rho.dtype) + copy = staging.copy(rho) # returns at once + ... + h5file["rho"] = copy.result() # a NumPy array + +On the NumPy backend both work on host arrays only, so the code is the same on +both backends. Page-locked host memory (:func:`cunumpy.cuda.pin_memory`) is in +:mod:`cunumpy.cuda`. +""" + +from .mirror import DeviceMirror +from .staging import HostStaging, StagedCopy + +__all__ = ["DeviceMirror", "HostStaging", "StagedCopy"] diff --git a/src/cunumpy/morton.py b/src/cunumpy/morton.py index 3d5e0be..36afa5c 100644 --- a/src/cunumpy/morton.py +++ b/src/cunumpy/morton.py @@ -7,9 +7,9 @@ points of every node of a quadtree (2D) or octree (3D) on the same box are a contiguous range of the sorted array:: - scales = xp.morton_scales(lower, upper, levels) - keys = xp.morton_keys(positions, lower, upper, levels) # uint64, one per point - keys, order, positions, charges = xp.sort_by_key(keys, positions, charges) + scales = xp.algorithms.morton_scales(lower, upper, levels) + keys = xp.algorithms.morton_keys(positions, lower, upper, levels) # uint64, one per point + keys, order, positions, charges = xp.algorithms.sort_by_key(keys, positions, charges) node = keys >> np.uint64(ndim * (levels - level)) # node index at `level` Bit layout: with ``levels`` bits per axis the key has ``ndim * levels`` bits; diff --git a/src/cunumpy/mpi.py b/src/cunumpy/mpi.py new file mode 100644 index 0000000..a12431e --- /dev/null +++ b/src/cunumpy/mpi.py @@ -0,0 +1,34 @@ +"""MPI with NumPy or CuPy arrays. + +:func:`mpi_buffer` hands an array to mpi4py: the device array itself when the +MPI library is CUDA-aware, a host copy otherwise; :func:`mpi_is_cuda_aware` +finds out which. :func:`local_rank` is the rank of this process on its node +(to pick a GPU, see :func:`cunumpy.cuda.bind_local_device`):: + + import cunumpy as xp + + with xp.mpi.mpi_buffer(rho, send=True, recv=True) as buf: + comm.Allreduce(MPI.IN_PLACE, buf, op=MPI.SUM) + +mpi4py is imported only by the functions that need it. +""" + +from .xp import ( + get_mpi_cuda_aware, + local_rank, + mpi_buffer, + mpi_is_cuda_aware, + require_cuda_aware_mpi, + set_mpi_cuda_aware, + synchronize_for_mpi, +) + +__all__ = [ + "get_mpi_cuda_aware", + "local_rank", + "mpi_buffer", + "mpi_is_cuda_aware", + "require_cuda_aware_mpi", + "set_mpi_cuda_aware", + "synchronize_for_mpi", +] diff --git a/src/cunumpy/petsc.py b/src/cunumpy/petsc.py index a5ca0a9..b4df2ce 100644 --- a/src/cunumpy/petsc.py +++ b/src/cunumpy/petsc.py @@ -9,7 +9,7 @@ b = xp.zeros(n) # filled by the deposit kernel phi = xp.zeros(n) # the solution, read by the gather kernel - b_vec, phi_vec = xp.petsc_vec(b), xp.petsc_vec(phi) + b_vec, phi_vec = xp.petsc.petsc_vec(b), xp.petsc.petsc_vec(phi) ... xp.synchronize() # CuPy work on b done before PETSc reads it ksp.solve(b_vec, phi_vec) # writes into phi @@ -43,7 +43,7 @@ def _petsc() -> Any: from petsc4py import PETSc except ImportError as error: raise ImportError( - "xp.petsc_vec needs petsc4py (pip install petsc4py)" + "xp.petsc.petsc_vec needs petsc4py (pip install petsc4py)" ) from error return PETSc diff --git a/src/cunumpy/philox.py b/src/cunumpy/philox.py index 6f0dbfc..c674c14 100644 --- a/src/cunumpy/philox.py +++ b/src/cunumpy/philox.py @@ -6,8 +6,8 @@ numbers can be compared with its host version element by element:: ids = xp.arange(n, dtype=xp.uint64) # one stream per particle - u0, u1 = xp.philox_uniform2(seed, ids, step) # what each GPU thread draws - z0, z1 = xp.philox_normal2(seed, ids, step) + u0, u1 = xp.rng.philox_uniform2(seed, ids, step) # what each GPU thread draws + z0, z1 = xp.rng.philox_normal2(seed, ids, step) The numbers are a pure function of the key (``seed``, 64 bit) and the 128-bit counter (``counter`` and ``stream``, 64 bit each): no state, independent of the diff --git a/src/cunumpy/profiling.py b/src/cunumpy/profiling.py new file mode 100644 index 0000000..02f9ace --- /dev/null +++ b/src/cunumpy/profiling.py @@ -0,0 +1,32 @@ +"""Timing, NVTX ranges and counting host/device transfers, on either backend. + +:func:`timed_region` times a block (synchronizing the device first and last), +:class:`nvtx_range` marks it for Nsight (a no-op without NVTX), and +:func:`count_transfers` / :func:`assert_no_transfers` count the copies between +host and device that cunumpy makes inside a block:: + + import cunumpy as xp + + with xp.profiling.timed_region("push") as t, xp.profiling.nvtx_range("push"): + push(positions, velocities, dt) + with xp.profiling.assert_no_transfers(): + step() +""" + +from .transfers import ( + TransferCounter, + TransferEvent, + assert_no_transfers, + count_transfers, +) +from .xp import Timing, nvtx_range, timed_region + +__all__ = [ + "Timing", + "TransferCounter", + "TransferEvent", + "assert_no_transfers", + "count_transfers", + "nvtx_range", + "timed_region", +] diff --git a/src/cunumpy/random_streams.py b/src/cunumpy/random_streams.py index f8dcb76..a030aec 100644 --- a/src/cunumpy/random_streams.py +++ b/src/cunumpy/random_streams.py @@ -7,9 +7,9 @@ import cunumpy as xp - xp.random_streams.seed(42, rank=comm.Get_rank()) # once, at start-up - v = xp.random_streams.normal(0.0, v_th, (n, 3)) # anywhere afterwards - rng = xp.random_streams.generator() # the Generator itself + xp.rng.random_streams.seed(42, rank=comm.Get_rank()) # once, at start-up + v = xp.rng.random_streams.normal(0.0, v_th, (n, 3)) # anywhere afterwards + rng = xp.rng.random_streams.generator() # the Generator itself Each rank draws the stream ``(seed, rank)`` (a NumPy ``SeedSequence`` with the rank as spawn key), so a run with the same seed and the same number of ranks diff --git a/src/cunumpy/rng.py b/src/cunumpy/rng.py new file mode 100644 index 0000000..d28d0b1 --- /dev/null +++ b/src/cunumpy/rng.py @@ -0,0 +1,39 @@ +"""Random numbers on either backend. + +:data:`random_streams` is one seeded generator per process and backend, with a +separate stream on every MPI rank; :func:`get_rng` returns a fresh NumPy or +CuPy ``Generator`` for the active backend; the ``philox_*`` functions return +the counter-based random numbers of ``cunumpy/random.cuh`` on the host, the +same as a CUDA kernel draws:: + + import cunumpy as xp + + xp.rng.random_streams.seed(42, rank=comm.Get_rank()) + v = xp.rng.random_streams.normal(0.0, v_th, (n, 3)) + u0, u1 = xp.rng.philox_uniform2(seed, ids, step) + +Named ``rng`` and not ``random`` so that ``xp.random`` stays NumPy's (CuPy's) +``random`` module. +""" + +from .philox import ( + philox4x32_10, + philox_normal, + philox_normal2, + philox_uniform, + philox_uniform2, +) +from .random_streams import BIT_GENERATORS, RandomStreams, random_streams +from .xp import get_rng + +__all__ = [ + "BIT_GENERATORS", + "RandomStreams", + "get_rng", + "philox4x32_10", + "philox_normal", + "philox_normal2", + "philox_uniform", + "philox_uniform2", + "random_streams", +] diff --git a/src/cunumpy/staging.py b/src/cunumpy/staging.py index f59ad7b..180d6c4 100644 --- a/src/cunumpy/staging.py +++ b/src/cunumpy/staging.py @@ -4,7 +4,7 @@ ``array.get()`` waits for the GPU and then for the copy, while no kernel runs. :class:`HostStaging` overlaps the copy with the next time steps:: - staging = xp.HostStaging(rho.shape, rho.dtype) # once + staging = xp.memory.HostStaging(rho.shape, rho.dtype) # once for step in range(n_steps): advance(...) if step % output_every == 0: diff --git a/src/cunumpy/testing.py b/src/cunumpy/testing.py index cbbd748..6be5b14 100644 --- a/src/cunumpy/testing.py +++ b/src/cunumpy/testing.py @@ -1,634 +1,21 @@ -"""Helpers for testing host/CUDA kernel pairs with pytest. +"""Deprecated alias of :mod:`cunumpy.kernel_testing`. -A code base that ports its kernels to CUDA one by one needs the same test for -every kernel: build the arguments on both backends, run the host kernel and the -CUDA kernel, and compare what they wrote. This module provides that test -(:func:`assert_kernels_agree`), the pytest markers to parametrize tests over -the backends (:data:`BACKENDS`, :data:`requires_cupy`, the :func:`backend` -fixture), :func:`device_function_kernel`, which wraps a ``__device__`` -function in an elementwise ``__global__`` kernel so that device helpers can be -tested from Python without a hand-written test kernel, and -:func:`emulate_cuda_kernel` (from :mod:`cunumpy.emulation`), which runs a CUDA -kernel on the CPU, one thread after another, so that its arithmetic can be -checked against the host kernel in CI without a GPU. Without a GPU, the CuPy -code paths of a program (argument objects, conversions, backend branches) can -still run on the fake CuPy of :mod:`cunumpy._fake_cupy` (:func:`install_fake_cupy`, -or ``CUNUMPY_FAKE_CUPY=1``); :func:`fake_cupy_active` tells whether it is in -use, and ``requires_cupy`` skips the tests that launch kernels then. - -A catalog's parity tests need no code per kernel when each kernel folder -holds ``_test_args.py`` with ``make_args(backend, seed)`` (and -``N_THREADS``): :func:`parity_cases` and :func:`check_parity` drive -:func:`assert_kernels_agree` from these modules. - -The module imports pytest only when one of its pytest objects is used, so it -can be imported (e.g. for :func:`device_function_kernel`) without pytest, and -``import cunumpy`` never imports pytest. - -Examples --------- -One parametrised test covers every kernel of a catalog that has a CUDA -version:: - - import pytest - from cunumpy.testing import assert_kernels_agree - - from my_kernels import catalog - - - def make_args(backend, seed): - rng = xp.get_rng(seed) # the backend is active: arrays land on it - x = rng.random(1000) - return (x, 2.0, x.size) - - - @pytest.mark.parametrize("name, kernel", catalog.parity_cases()) - def test_parity(name, kernel): - assert_kernels_agree(kernel, make_args, n_threads=1000) +A submodule named ``testing`` replaced NumPy's ``xp.testing`` once imported, so +``xp.testing.assert_allclose`` failed after any ``import cunumpy.testing``. +Import :mod:`cunumpy.kernel_testing` instead; this alias will be removed in +cunumpy 0.6. """ -from __future__ import annotations - -import re -from collections.abc import Callable, Sequence -from typing import Any +import sys as _sys +import warnings as _warnings -import array_api_compat -import numpy as np +from . import kernel_testing as _kernel_testing -from . import _fake_cupy -from .cuda_kernel import ( - CudaKernel, - CudaParameter, - CudaStructArguments, - CudaStructValue, - _parse_parameter, - _split_top_level, - _strip_comments, +_warnings.warn( + "cunumpy.testing is deprecated and will be removed in cunumpy 0.6; " + "import cunumpy.kernel_testing instead", + DeprecationWarning, + stacklevel=2, ) -from .dispatch import Kernel -from .emulation import emulate_cuda_kernel, emulation_compiler -from .xp import cupy_available, get_backend, to_numpy, use_backend - -# the pytest objects are created on first access, see __getattr__ -__all__ = [ - "BACKENDS", # noqa: F822 - "assert_kernels_agree", - "backend", # noqa: F822 - "check_parity", - "device_function_kernel", - "emulate_cuda_kernel", - "emulation_compiler", - "fake_cupy_active", - "install_fake_cupy", - "parity_cases", - "requires_cupy", # noqa: F822 -] - -SKIP_REASON = "CuPy/GPU not available" -FAKE_SKIP_REASON = "the fake CuPy cannot run CUDA kernels" - - -def fake_cupy_active() -> bool: - """Whether the fake CuPy (:mod:`cunumpy._fake_cupy`) stands in for CuPy.""" - return _fake_cupy.is_active() - - -def install_fake_cupy() -> Any: - """Install the fake CuPy for this process; see :mod:`cunumpy._fake_cupy`. - - Call it before the first backend use (e.g. at the top of ``conftest.py``), - or set ``CUNUMPY_FAKE_CUPY=1`` in the environment instead. Returns the - fake ``cupy`` module. - """ - return _fake_cupy.install() - - -def _can_launch() -> bool: - """Whether CUDA kernels can run: a functional CuPy that is not the fake.""" - return cupy_available() and not fake_cupy_active() - - -# pytest objects, built on first use so that importing this module does not -# import pytest (see __getattr__ below) -_LAZY: dict[str, Any] = {} - - -def _pytest() -> Any: - try: - import pytest - except ImportError: # pragma: no cover - pytest is installed in the test suite - raise ImportError( - "cunumpy.testing needs pytest for this feature: pip install pytest" - ) from None - return pytest - - -def _build_lazy() -> None: - pytest = _pytest() - requires_cupy = pytest.mark.skipif( - not _can_launch(), - reason=FAKE_SKIP_REASON if fake_cupy_active() else SKIP_REASON, - ) - backends = ["numpy", pytest.param("cupy", marks=requires_cupy)] - - @pytest.fixture(params=backends) - def backend(request): - """Run the test once per backend, with that backend active.""" - with use_backend(request.param): - yield request.param - - _LAZY.update(requires_cupy=requires_cupy, BACKENDS=backends, backend=backend) - - -def __getattr__(name: str) -> Any: - if name in ("requires_cupy", "BACKENDS", "backend"): - if not _LAZY: - _build_lazy() - return _LAZY[name] - raise AttributeError(f"module {__name__!r} has no attribute {name!r}") - - -# --------------------------------------------------------------------------- -# assert_kernels_agree -# --------------------------------------------------------------------------- - - -def _is_array(value: Any) -> bool: - return array_api_compat.is_numpy_array(value) or array_api_compat.is_cupy_array( - value - ) - - -def _arrays_in(value: Any, name: str, found: dict[str, Any], depth: int) -> None: - """Record the arrays in `value` under `name`, looking `depth` levels deep.""" - if _is_array(value): - found[name] = value - elif depth == 0: - return - elif isinstance(value, (CudaStructArguments, CudaStructValue)) and hasattr( - value, "struct" - ): - # by field name, like the attributes of the host argument object; the - # fields of a CudaStructArguments may be properties (not in vars()) - for field in value.struct.fields: - item = ( - value[field.name] - if isinstance(value, CudaStructValue) - else getattr(value, field.name) - ) - _arrays_in(item, f"{name}.{field.name}", found, depth - 1) - elif isinstance(value, (tuple, list)): - for i, item in enumerate(value): - _arrays_in(item, f"{name}[{i}]", found, depth - 1) - elif isinstance(value, dict): - for key, item in value.items(): - _arrays_in(item, f"{name}[{key!r}]", found, depth - 1) - elif hasattr(value, "__dict__"): - for attr, item in vars(value).items(): - _arrays_in(item, f"{name}.{attr}", found, depth - 1) - - -def _collect_arrays( - args: Sequence[Any], outputs: Sequence[int] | None = None -) -> dict[str, Any]: - """The arrays among `args` (or among the arguments `outputs`), by name. - - An argument that is an array is named ``"argument "``; arrays found one - level deep, in a tuple, list or dict argument or in the attributes of an - argument object (e.g. a ``CudaArguments`` object), are named - ``"argument []"`` or ``"argument ."``, and arrays in a - container attribute of an object ``"argument .[]"``. A - :class:`~cunumpy.CudaStructArguments` object or a struct value is read - through its struct fields, ``"argument ."``, so that its arrays - get the names of the attributes of the host argument object it mirrors, - also when the fields are properties. - """ - indices = range(len(args)) if outputs is None else outputs - found: dict[str, Any] = {} - for entry in indices: - if not isinstance(entry, int) or isinstance(entry, bool): - raise TypeError( - "outputs must be positional argument indices (kernels take " - f"positional arguments only), got {entry!r}" - ) - index = entry + len(args) if entry < 0 else entry - if not 0 <= index < len(args): - raise IndexError( - f"output argument {entry} does not exist: there are {len(args)} " - "arguments" - ) - _arrays_in(args[index], f"argument {index}", found, depth=2) - return found - - -def _compare_results( - host: dict[str, Any], - device: dict[str, Any], - rtol: float, - atol: float, - kernel_name: str = "kernel", -) -> None: - """Compare the arrays of `device` with those of `host`, by name. - - Raises - ------ - AssertionError - If the two do not hold the same names, or an array differs (the - message names the argument). - """ - if host.keys() != device.keys(): - raise AssertionError( - f"{kernel_name}: the host and CUDA calls do not have the same array " - f"arguments: host {sorted(host)}, CUDA {sorted(device)}" - ) - for name, expected in host.items(): - np.testing.assert_allclose( - to_numpy(device[name]), - to_numpy(expected), - rtol=rtol, - atol=atol, - err_msg=f"{kernel_name}: {name} differs between the host and CUDA kernels", - ) - - -def assert_kernels_agree( - kernel: Kernel, - make_args: Callable[[str, int], Sequence[Any]], - *, - n_threads: int | Sequence[int] | Callable[[tuple[Any, ...]], Any] | None = None, - grid: int | Sequence[int] | None = None, - block: int | Sequence[int] | None = None, - rtol: float = 1e-12, - atol: float = 0.0, - n_calls: int = 1, - outputs: Sequence[int] | None = None, - seed: int = 0, -) -> dict[str, np.ndarray]: - """Check that the host and CUDA versions of `kernel` compute the same. - - For each backend, ``"numpy"`` then ``"cupy"``, the backend is activated - with :func:`~cunumpy.use_backend`, the arguments are built with - ``make_args(backend, seed)``, the kernel is called `n_calls` times, and - the arrays among the arguments are collected. The arrays written by the - CUDA kernel are then copied to the host and compared with those of the host - kernel using ``numpy.testing.assert_allclose``. - - Parameters - ---------- - kernel : Kernel - A kernel with a CUDA version (``kernel.has_cuda``). - make_args : Callable[[str, int], Sequence] - ``make_args(backend, seed)`` returns the positional arguments of the - kernel, as a tuple or list. It is called with the backend (``"numpy"`` - or ``"cupy"``) active, so arrays created through ``cunumpy`` (e.g. with - ``xp.zeros`` or ``xp.get_rng(seed)``) land on that backend; NumPy and - CuPy random generators do not produce the same sequence from one seed, - so build random data on the host with ``numpy.random.default_rng(seed)`` - and convert it with :func:`~cunumpy.to_cunumpy`. Kernels take positional - arguments only. - n_threads, grid, block - Launch configuration of the CUDA kernel, see - :meth:`CudaKernel.__call__ `. `n_threads` - may also be a function of the tuple of arguments, e.g. - ``lambda args: args[0].shape[0]``. One of `n_threads` and `grid` is - required unless the CUDA kernel has ``n_threads_from``. - rtol, atol : float - Tolerances of ``numpy.testing.assert_allclose``. - n_calls : int - How many times the kernel is called on each backend (e.g. to test a - kernel that accumulates). - outputs : Sequence[int] | None - Indices of the arguments to compare (negative indices count from the - end), like ``PyccelKernel(outputs=...)``. By default the ``outputs`` - declared by the host kernel are used, and if it declares none, every - argument. An argument that is an array is compared; for a tuple, list, - dict or object argument (e.g. a ``CudaArguments`` object), the arrays - it holds are compared (one level deep, plus arrays in a container - attribute of an object). - seed : int - Passed to `make_args` on both backends. - - Returns - ------- - dict[str, numpy.ndarray] - The arrays of the host call by argument name (``"argument 0"``, - ``"argument 1.x"``), for further checks. - - Raises - ------ - ValueError - If `kernel` has no CUDA version. - AssertionError - If an array differs; the message names the argument. - - Notes - ----- - The test is skipped with ``pytest.skip`` if CuPy or a GPU is not available, - or if the fake CuPy is active. - """ - if not isinstance(kernel, Kernel): - raise TypeError(f"expected a Kernel, got {type(kernel).__name__}") - if not kernel.has_cuda: - raise ValueError(f"kernel {kernel.name!r} has no CUDA version") - if n_threads is None and grid is None and kernel.cuda_kernel.n_threads_from is None: - raise TypeError("n_threads (or grid) is required to launch the CUDA kernel") - if n_calls < 1: - raise ValueError(f"n_calls must be at least 1, got {n_calls}") - if outputs is None: - outputs = kernel.host_kernel.outputs - if fake_cupy_active(): - _pytest().skip(FAKE_SKIP_REASON) - if not cupy_available(): - _pytest().skip(SKIP_REASON) - - results = {} - for backend in ("numpy", "cupy"): - with use_backend(backend): - if get_backend() != backend: # pragma: no cover - cupy_available() lied - raise RuntimeError(f"could not activate the {backend} backend") - args = tuple(make_args(backend, seed)) - launch = n_threads(args) if callable(n_threads) else n_threads - for _ in range(n_calls): - kernel(*args, n_threads=launch, grid=grid, block=block) - results[backend] = _collect_arrays(args, outputs) - - host = {name: to_numpy(a) for name, a in results["numpy"].items()} - _compare_results(host, results["cupy"], rtol, atol, kernel.name) - return host - - -# --------------------------------------------------------------------------- -# parity tests from _test_args.py modules -# --------------------------------------------------------------------------- - -#: Module-level names a ``_test_args.py`` module may define, and the -#: keyword of :func:`assert_kernels_agree` each one sets. -TEST_ARGS_SETTINGS = { - "N_THREADS": "n_threads", - "GRID": "grid", - "BLOCK": "block", - "RTOL": "rtol", - "ATOL": "atol", - "N_CALLS": "n_calls", - "OUTPUTS": "outputs", - "SEED": "seed", -} - - -def parity_cases(catalog: Any) -> list[Any]: - """The kernels of a catalog with a CUDA version, as pytest parameters. - - One ``pytest.param(kernel, id=name)`` per kernel of - ``catalog.parity_cases()``. A kernel without a test-arguments module - (:attr:`Kernel.test_args_module `, from - ``_test_args.py`` in its folder) is marked ``skip`` with a reason - naming the missing file, so the report shows which kernels still lack - their parity test:: - - @pytest.mark.parametrize("kernel", parity_cases(catalog)) - def test_parity(kernel): - check_parity(kernel) - """ - pytest = _pytest() - cases = [] - for name, kernel in catalog.parity_cases(): - marks = () - if kernel.test_args_module is None: - marks = ( - pytest.mark.skip( - reason=f"no test arguments for {name!r}: add {name}_test_args.py " - "with make_args(backend, seed) and N_THREADS to its folder" - ), - ) - cases.append(pytest.param(kernel, id=name, marks=marks)) - return cases - - -def check_parity(kernel: Kernel, **overrides: Any) -> dict[str, np.ndarray]: - """Run :func:`assert_kernels_agree` with the kernel's test-arguments module. - - The module (``_test_args.py`` in the kernel's folder, see - :meth:`KernelCatalog.from_package `) - defines ``make_args(backend, seed)`` and, as module-level names, the - launch and comparison settings of :data:`TEST_ARGS_SETTINGS`: - ``N_THREADS`` (an integer, a tuple, or a function of the argument tuple), - or ``GRID``, plus optionally ``BLOCK``, ``RTOL``, ``ATOL``, ``N_CALLS``, - ``OUTPUTS`` and ``SEED``. Keyword arguments override them. - - Returns - ------- - dict[str, numpy.ndarray] - The host arrays, as :func:`assert_kernels_agree` returns them. - - Raises - ------ - ValueError - If the kernel has no test-arguments module. - TypeError - If the module has no callable ``make_args``. - """ - module = kernel.test_args - if module is None: - raise ValueError( - f"kernel {kernel.name!r} has no test-arguments module: add " - f"{kernel.name}_test_args.py with make_args(backend, seed) to its folder" - ) - make_args = getattr(module, "make_args", None) - if not callable(make_args): - raise TypeError(f"{module.__name__} must define make_args(backend, seed)") - settings = { - keyword: getattr(module, name) - for name, keyword in TEST_ARGS_SETTINGS.items() - if hasattr(module, name) - } - settings.update(overrides) - return assert_kernels_agree(kernel, make_args, **settings) - - -# --------------------------------------------------------------------------- -# device_function_kernel -# --------------------------------------------------------------------------- - -_PROTOTYPE = re.compile( - r"^\s*(?P.+?)\s*\b(?P[A-Za-z_]\w*)\s*\((?P.*)\)\s*;?\s*$", - re.DOTALL, -) -_FUNCTION_QUALIFIERS = {"__device__", "__host__", "__forceinline__", "inline", "static"} - - -def _parse_prototype( - signature: str, - structs: dict[str, Any] | None = None, -) -> tuple[CudaParameter | None, str, list[tuple[str, CudaParameter]]]: - """Parse a C function prototype into (result, name, [(text, parameter)]). - - The result is None for a ``void`` function; each parameter is its original - text together with its parsed form. A struct of `structs` may be taken by - value or by (const) reference. - """ - match = _PROTOTYPE.match(_strip_comments(signature)) - if match is None: - raise ValueError(f"cannot parse the function prototype {signature!r}") - name = match.group("name") - result_words = [ - w for w in match.group("result").split() if w not in _FUNCTION_QUALIFIERS - ] - result_text = " ".join(result_words) - if result_text == "void": - result = None - else: - try: - result = _parse_parameter(f"{result_text} result") - except ValueError: - raise ValueError( - f"unsupported return type {result_text!r} of {name!r}: a scalar " - "type (or void) is required" - ) from None - if result.pointer: - raise ValueError( - f"{name!r} returns a pointer; only scalar results can be collected" - ) - params_text = match.group("params").strip() - params = [] - if params_text not in ("", "void"): - for text in _split_top_level(params_text): - text = text.strip() - by_reference = "&" in text - param = _parse_parameter(text.replace("&", " "), structs or None) - if by_reference and param.struct is None: - raise ValueError( - f"unsupported parameter {text!r} of {name!r}: only structs can " - "be passed by reference" - ) - params.append((text, param)) - return result, name, params - - -def device_function_kernel( - header_source: str, - signature: str, - *, - name: str | None = None, - includes: Sequence[str] = (), - n_threads_param: str = "n", - out_param: str = "out", - **kwargs: Any, -) -> CudaKernel: - """Wrap a ``__device__`` function in an elementwise kernel, for testing. - - Generates an ``extern "C" __global__`` kernel that calls the device function - once per thread, so that a device helper (e.g. a B-spline evaluation) can be - run from Python on many inputs at once and compared with its host version. - - Parameters - ---------- - header_source : str - CUDA source defining the device function (typically the content of the - header, or ``#include`` directives, see `includes`). - signature : str - The C prototype of the device function, e.g. - ``"int find_span(const double* t, int p, double eta)"``. Pointer - parameters, scalar parameters of the types :class:`CudaKernel` supports, - struct parameters (by value or by ``const`` reference, for the structs - passed in ``structs``) and a scalar or ``void`` return type are - supported. - name : str | None - Name of the generated kernel; ``"_kernel"`` by default. - includes : Sequence[str] - Headers to include before `header_source`: ``"bsplines.cuh"`` becomes - ``#include "bsplines.cuh"``, ``""`` is included with - angle brackets. Pass ``include_dirs`` for the directories. - n_threads_param : str - Name of the generated kernel's last parameter, the number of elements. - out_param : str - Name of the generated output array parameter. - **kwargs - Passed on to :class:`CudaKernel`, e.g. ``include_dirs``, ``options``, - ``block_size`` or ``structs`` (the :class:`~cunumpy.CudaStruct` types - of struct parameters, whose definitions `header_source` or the - `includes` must provide). - - Returns - ------- - CudaKernel - The wrapper kernel. Its parameters are those of the device function, in - order, followed by the output array (unless the function returns - ``void``) and the number of elements: - - * a pointer parameter stays as it is and is passed through unchanged to - every call (an array shared by all threads); - * a struct parameter (``DomainArgs d`` or ``const DomainArgs& d``) is - taken by value and passed through unchanged to every call (pass a - :class:`~cunumpy.CudaStructArguments` object or a packed value); - * a scalar parameter ``T x`` becomes a device array ``const T* x`` of - length ``n``, and thread ``i`` calls the function with ``x[i]``; - * the return value of thread ``i`` is stored in ``out[i]``, an array - ``R* out`` of length ``n`` where ``R`` is the return type; - * ``int n`` is the number of elements (threads ``i >= n`` do nothing). - - Launch it with ``n_threads=n``. - - Raises - ------ - ValueError - If the prototype cannot be parsed, the return type is not a scalar type - or ``void``, a parameter has an unsupported type, or a parameter is - named like `out_param` or `n_threads_param`. - - Examples - -------- - >>> from cunumpy.testing import device_function_kernel - >>> sq = device_function_kernel( - ... "__device__ double sq(double x) { return x * x; }", - ... "double sq(double x)", - ... ) - >>> sq.signature # (const double* x, double* out, int n) - >>> x = cp.arange(10.0); out = cp.empty(10) # doctest: +SKIP - >>> sq(x, out, 10, n_threads=10) # doctest: +SKIP - """ - structs = {struct.name: struct for struct in kwargs.get("structs", ())} - result, function, params = _parse_prototype(signature, structs) - if name is None: - name = f"{function}_kernel" - reserved = {out_param, n_threads_param} - for _, param in params: - if param.name in reserved: - raise ValueError( - f"the parameter {param.name!r} of {function!r} clashes with the " - f"generated parameter of that name; pass another out_param or " - "n_threads_param" - ) - - wrapper_params, call_args = [], [] - for text, param in params: - if param.pointer: - wrapper_params.append(text) - call_args.append(param.name) - elif param.struct is not None: - wrapper_params.append(f"{param.ctype} {param.name}") # by value - call_args.append(param.name) - else: - wrapper_params.append(f"const {param.ctype}* {param.name}") - call_args.append(f"{param.name}[i]") - call = f"{function}({', '.join(call_args)})" - if result is None: - body = f"{call};" - else: - wrapper_params.append(f"{result.ctype}* {out_param}") - body = f"{out_param}[i] = {call};" - wrapper_params.append(f"int {n_threads_param}") - include_lines = "".join( - f"#include {h}\n" if h.startswith("<") else f'#include "{h}"\n' - for h in includes - ) - source = ( - f"{include_lines}{header_source}\n\n" - f'extern "C" __global__ void {name}({", ".join(wrapper_params)}) {{\n' - f" int i = blockDim.x * blockIdx.x + threadIdx.x;\n" - f" if (i >= {n_threads_param}) return;\n" - f" {body}\n" - f"}}\n" - ) - return CudaKernel(source, name, **kwargs) +_sys.modules[__name__] = _kernel_testing diff --git a/src/cunumpy/transfers.py b/src/cunumpy/transfers.py index 2aaaf62..14c1fcb 100644 --- a/src/cunumpy/transfers.py +++ b/src/cunumpy/transfers.py @@ -7,13 +7,13 @@ that caused it, so a test can verify that a time step does not transfer at all:: - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: propagator(dt) assert counter.total == 0, counter.report() or, equivalently:: - with xp.assert_no_transfers(): + with xp.profiling.assert_no_transfers(): propagator(dt) Counted are @@ -22,9 +22,9 @@ called with a device array; * ``to_device``: :func:`~cunumpy.to_cupy` (and :func:`~cunumpy.to_cunumpy`) called with anything that is not already a device array; -* ``kernel_conversion``: a :class:`~cunumpy.PyccelKernel` call that copied +* ``kernel_conversion``: a :class:`~cunumpy.kernels.PyccelKernel` call that copied device arrays to the host (and back), one event per call; -* ``fallback``: a :class:`~cunumpy.Kernel` without CUDA kernel calling its host +* ``fallback``: a :class:`~cunumpy.kernels.Kernel` without CUDA kernel calling its host kernel on the CuPy backend (``missing_cuda="fallback"``), one event per call. Limitations @@ -196,8 +196,8 @@ def count_transfers() -> Generator[TransferCounter, None, None]: Yields a :class:`TransferCounter` that records every ``to_numpy``, ``to_cupy`` and ``to_cunumpy`` call that actually copies, every - :class:`~cunumpy.PyccelKernel` call that converts device arrays and every - :class:`~cunumpy.Kernel` fallback to the host kernel, with the call site of + :class:`~cunumpy.kernels.PyccelKernel` call that converts device arrays and every + :class:`~cunumpy.kernels.Kernel` fallback to the host kernel, with the call site of each. Nothing is counted for calls that do not copy, e.g. ``to_numpy`` of a NumPy array. @@ -208,7 +208,7 @@ def count_transfers() -> Generator[TransferCounter, None, None]: Examples -------- - >>> with xp.count_transfers() as counter: + >>> with xp.profiling.count_transfers() as counter: ... propagator(dt) >>> assert counter.total == 0, counter.report() """ @@ -230,7 +230,7 @@ def assert_no_transfers() -> Generator[TransferCounter, None, None]: Examples -------- - >>> with xp.assert_no_transfers(): + >>> with xp.profiling.assert_no_transfers(): ... propagator(dt) """ with count_transfers() as counter: diff --git a/src/cunumpy/xp.py b/src/cunumpy/xp.py index 155f3f5..99c3b0e 100644 --- a/src/cunumpy/xp.py +++ b/src/cunumpy/xp.py @@ -334,7 +334,7 @@ def mpi_buffer( One MPI call site for both backends and both kinds of MPI builds:: - with xp.mpi_buffer(markers_out) as sendbuf, xp.mpi_buffer( + with xp.mpi.mpi_buffer(markers_out) as sendbuf, xp.mpi.mpi_buffer( markers_in, send=False, recv=True ) as recvbuf: comm.Sendrecv(sendbuf, dest, recvbuf=recvbuf, source=source) @@ -377,8 +377,8 @@ def mpi_buffer( if cuda_aware is None: raise RuntimeError( "mpi_buffer(): it is not known whether MPI can take device buffers; " - "call xp.mpi_is_cuda_aware(comm) once at startup (every rank), or " - "xp.set_mpi_cuda_aware(True/False), or pass cuda_aware=" + "call xp.mpi.mpi_is_cuda_aware(comm) once at startup (every rank), or " + "xp.mpi.set_mpi_cuda_aware(True/False), or pass cuda_aware=" ) if cuda_aware: synchronize_for_mpi(array) @@ -714,10 +714,10 @@ class nvtx_range(ContextDecorator): Examples -------- - >>> with xp.nvtx_range("push markers"): + >>> with xp.profiling.nvtx_range("push markers"): ... kernel(markers, dt, n_threads=n) - >>> @xp.nvtx_range("accumulate") + >>> @xp.profiling.nvtx_range("accumulate") ... def accumulate(...): ... ... """ @@ -796,7 +796,7 @@ def timed_region(name: str, *, sync: bool = True) -> Generator[Timing, None, Non Examples -------- - >>> with xp.timed_region("push markers") as timing: + >>> with xp.profiling.timed_region("push markers") as timing: ... kernel(markers, dt, n_threads=n) >>> print(f"{timing.name}: {timing.elapsed:.3f} s (synced={timing.synced})") """ @@ -1043,11 +1043,11 @@ def sort_by_key(keys: Any, *arrays: Any) -> tuple[Any, ...]: """Sort `keys` and reorder every array the same way, in one stable argsort. The usual first step of a particle code on the GPU: sort the particles by - cell index or Morton key (:func:`cunumpy.morton_keys`), then work on + cell index or Morton key (:func:`cunumpy.algorithms.morton_keys`), then work on contiguous ranges. The sort is stable, so equal keys keep their order and the result is reproducible:: - keys, order, positions, charges = xp.sort_by_key(keys, positions, charges) + keys, order, positions, charges = xp.algorithms.sort_by_key(keys, positions, charges) Parameters ---------- @@ -1124,7 +1124,7 @@ def as_kernel_array(value: Any, like: Any, dtype: Any = None) -> Any: pass the main array (e.g. the grid) as `like`, and the kernel gets arguments all on one side:: - convert = functools.partial(xp.as_kernel_array, like=grid, dtype=float) + convert = functools.partial(xp.kernels.as_kernel_array, like=grid, dtype=float) deposit(convert(positions), convert(weights), grid, ...) For an array the kernel writes, use :func:`kernel_output`, which copies a @@ -1148,7 +1148,7 @@ def kernel_output(out: Any, like: Any, dtype: Any = None) -> Generator[Any]: written into `out` when the block ends without an error (moved back to the side of `out`):: - with xp.kernel_output(result, like=grid, dtype=float) as buffer: + with xp.kernels.kernel_output(result, like=grid, dtype=float) as buffer: gather(convert(positions), grid, buffer, ...) """ buffer = as_kernel_array(out, like, dtype) diff --git a/tests/portable/test_numpy_runtime.py b/tests/portable/test_numpy_runtime.py index df174c4..56123fe 100644 --- a/tests/portable/test_numpy_runtime.py +++ b/tests/portable/test_numpy_runtime.py @@ -23,7 +23,7 @@ def test_array_operations(): assert xp.same_backend(a, xp.ones(2)) xp.assert_same_backend(a, xp.ones(2)) xp.synchronize() - xp.set_device(0) + xp.cuda.set_device(0) def test_conversions_preserve_identity_and_views(): @@ -66,7 +66,7 @@ def kernel(values, alias, *, out, scale): out[...] *= scale return values, out, float(values.sum()), {"array": values} - wrapped = xp.PyccelKernel(kernel, use_cupy=use_cupy, outputs=("out",)) + wrapped = xp.kernels.PyccelKernel(kernel, use_cupy=use_cupy, outputs=("out",)) result = wrapped(a, a, out=view, scale=3) assert result[0] is a assert result[1] is view @@ -81,7 +81,7 @@ def test_python_kernel_none_return_and_exception(): def fill(out): out[...] = 4 - assert xp.PyccelKernel(fill)(a) is None + assert xp.kernels.PyccelKernel(fill)(a) is None np.testing.assert_array_equal(a, [4, 4]) def fail(out): @@ -89,5 +89,5 @@ def fail(out): raise ValueError("source kernel failed") with pytest.raises(ValueError, match="source kernel failed"): - xp.PyccelKernel(fail)(a) + xp.kernels.PyccelKernel(fail)(a) np.testing.assert_array_equal(a, [7, 4]) diff --git a/tests/unit/pyccel_kernels.py b/tests/unit/pyccel_kernels.py index 8ab2726..7e6ab09 100644 --- a/tests/unit/pyccel_kernels.py +++ b/tests/unit/pyccel_kernels.py @@ -4,7 +4,7 @@ be imported and run as-is, or compiled to C/Fortran with ``pyccel.epyccel(pyccel_kernels, language="c")``. The tests in `test_pyccel_kernel.py` compile it and drive the compiled kernels through -:class:`cunumpy.PyccelKernel`. +:class:`cunumpy.kernels.PyccelKernel`. """ diff --git a/tests/unit/test_array_api_compat_backend.py b/tests/unit/test_array_api_compat_backend.py index 175be61..5a3dab6 100644 --- a/tests/unit/test_array_api_compat_backend.py +++ b/tests/unit/test_array_api_compat_backend.py @@ -163,7 +163,7 @@ def test_set_device_and_synchronize_work_with_wrapped_backend(): import cupy as cp with xp.use_backend("cupy"): - xp.set_device(0) + xp.cuda.set_device(0) assert cp.cuda.Device().id == 0 arr = xp.asarray([1, 2, 3]) diff --git a/tests/unit/test_cuda_kernel.py b/tests/unit/test_cuda_kernel.py index 3042718..c038f67 100644 --- a/tests/unit/test_cuda_kernel.py +++ b/tests/unit/test_cuda_kernel.py @@ -1,4 +1,4 @@ -"""Tests for `cunumpy.CudaKernel` and `cunumpy.parse_cuda_signature`. +"""Tests for `cunumpy.cuda.CudaKernel` and `cunumpy.cuda.parse_cuda_signature`. Signature parsing and argument checking run everywhere: a small stand-in for a device array (`FakeDeviceArray`) takes the place of CuPy arrays. Launching @@ -13,13 +13,13 @@ import pytest import cunumpy as xp -from cunumpy import ( +from cunumpy import as_device_array +from cunumpy.cuda import ( CudaArguments, CudaKernel, CudaKernelVariants, CudaStruct, CudaStructValue, - as_device_array, ctype_of, cuda_include_dir, cuda_kernel_names, @@ -57,7 +57,7 @@ def _user_options(kernel): """`compile_options()` without cunumpy's own include directory.""" return tuple( - o for o in kernel.compile_options() if o != f"-I{xp.cuda_include_dir()}" + o for o in kernel.compile_options() if o != f"-I{xp.cuda.cuda_include_dir()}" ) @@ -850,7 +850,7 @@ def test_bounds_check_traps_on_gpu(): BOUNDS_TRAP = f""" import cupy as cp -from cunumpy import CudaKernel +from cunumpy.cuda import CudaKernel kernel = CudaKernel({SCALE_COLUMN!r}, "scale_column", options=("-DCUNUMPY_BOUNDS_CHECK",)) a = cp.ones((4, 3)) @@ -967,7 +967,7 @@ def test_to_header(tmp_path): struct = CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs") header = struct.to_header() assert header == ( - "// Generated by cunumpy.CudaStruct from the Python definition; do not edit.\n" + "// Generated by cunumpy.cuda.CudaStruct from the Python definition; do not edit.\n" "#ifndef MARKERARGS_CUH\n" "#define MARKERARGS_CUH\n" "\n" @@ -1002,7 +1002,7 @@ def test_write_cuda_header(tmp_path): ) assert path.read_text() == header assert header.startswith( - "// Generated by cunumpy.CudaStruct from the Python definition; do not edit.\n" + "// Generated by cunumpy.cuda.CudaStruct from the Python definition; do not edit.\n" "#ifndef PUSHER_ARGS_CUH\n#define PUSHER_ARGS_CUH\n\n" '#include "cunumpy/array_view.cuh"\n#include \n\n' ) @@ -1233,10 +1233,10 @@ def _run_python(code, env=None): @pytest.fixture def debug_off(): """Global debug mode off during the test, restored afterwards.""" - previous = xp.get_cuda_debug() - xp.set_cuda_debug(False) + previous = xp.cuda.get_cuda_debug() + xp.cuda.set_cuda_debug(False) yield - xp.set_cuda_debug(previous) + xp.cuda.set_cuda_debug(previous) @pytest.mark.parametrize( @@ -1272,30 +1272,30 @@ def test_debug_from_env_at_import(monkeypatch, value, expected): monkeypatch.delenv("CUNUMPY_CUDA_DEBUG", raising=False) else: monkeypatch.setenv("CUNUMPY_CUDA_DEBUG", value) - code = "import cunumpy as xp; print(xp.get_cuda_debug())" + code = "import cunumpy as xp; print(xp.cuda.get_cuda_debug())" stdout, _ = _run_python(code) assert stdout.strip() == expected def test_set_and_get_cuda_debug(debug_off): - assert xp.get_cuda_debug() is False - xp.set_cuda_debug(True) - assert xp.get_cuda_debug() is True - xp.set_cuda_debug(0) - assert xp.get_cuda_debug() is False + assert xp.cuda.get_cuda_debug() is False + xp.cuda.set_cuda_debug(True) + assert xp.cuda.get_cuda_debug() is True + xp.cuda.set_cuda_debug(0) + assert xp.cuda.get_cuda_debug() is False def test_cuda_debug_context_restores(debug_off): - with xp.cuda_debug(): - assert xp.get_cuda_debug() is True - with xp.cuda_debug(False): - assert xp.get_cuda_debug() is False - assert xp.get_cuda_debug() is True - assert xp.get_cuda_debug() is False - - with pytest.raises(ValueError), xp.cuda_debug(): + with xp.cuda.cuda_debug(): + assert xp.cuda.get_cuda_debug() is True + with xp.cuda.cuda_debug(False): + assert xp.cuda.get_cuda_debug() is False + assert xp.cuda.get_cuda_debug() is True + assert xp.cuda.get_cuda_debug() is False + + with pytest.raises(ValueError), xp.cuda.cuda_debug(): raise ValueError - assert xp.get_cuda_debug() is False # restored after an exception too + assert xp.cuda.get_cuda_debug() is False # restored after an exception too def test_compile_options_follow_the_global_setting(debug_off): @@ -1305,7 +1305,7 @@ def test_compile_options_follow_the_global_setting(debug_off): assert _user_options(kernel) == ("-std=c++17",) assert "-lineinfo" not in kernel.compile_options() - with xp.cuda_debug(): + with xp.cuda.cuda_debug(): # decided at call time: the kernel created before is affected assert kernel.debug_active() is True assert _user_options(kernel) == ( @@ -1320,8 +1320,8 @@ def test_compile_options_follow_the_global_setting(debug_off): def test_compile_options_no_duplicates(debug_off): kernel = CudaKernel(AXPY, "axpy", options=("-lineinfo",), debug=True) assert _user_options(kernel) == ("-lineinfo", "-DCUNUMPY_BOUNDS_CHECK") - kernel = CudaKernel(AXPY, "axpy", options=xp.DEBUG_OPTIONS[::-1], debug=True) - assert _user_options(kernel) == xp.DEBUG_OPTIONS[::-1] + kernel = CudaKernel(AXPY, "axpy", options=xp.cuda.DEBUG_OPTIONS[::-1], debug=True) + assert _user_options(kernel) == xp.cuda.DEBUG_OPTIONS[::-1] def test_explicit_debug_overrides_the_global_setting(debug_off): @@ -1330,23 +1330,23 @@ def test_explicit_debug_overrides_the_global_setting(debug_off): assert on.debug is True and off.debug is False assert on.debug_active() is True assert off.debug_active() is False - assert set(xp.DEBUG_OPTIONS) <= set(on.compile_options()) + assert set(xp.cuda.DEBUG_OPTIONS) <= set(on.compile_options()) - with xp.cuda_debug(): + with xp.cuda.cuda_debug(): assert off.debug_active() is False assert _user_options(off) == () - assert _user_options(on) == xp.DEBUG_OPTIONS + assert _user_options(on) == xp.cuda.DEBUG_OPTIONS def test_debug_options_include_dirs_and_from_file(tmp_path, debug_off): (tmp_path / "axpy_cuda.cu").write_text(AXPY) kernel = CudaKernel.from_file(tmp_path / "axpy_cuda.cu", debug=True) - assert _user_options(kernel) == (f"-I{tmp_path}", *xp.DEBUG_OPTIONS) + assert _user_options(kernel) == (f"-I{tmp_path}", *xp.cuda.DEBUG_OPTIONS) def test_debug_option_is_not_G(): """NVRTC does not support -G; it must not be added.""" - assert "-G" not in xp.DEBUG_OPTIONS + assert "-G" not in xp.cuda.DEBUG_OPTIONS def test_debug_on_gpu_compiles_and_synchronizes(): @@ -1375,7 +1375,7 @@ def test_debug_on_gpu_compiles_and_synchronizes(): if (i < n) y[((long long)i + 1) << 36] = 1.0; // 512 GB and more past y } ''' -kernel = xp.CudaKernel(SOURCE, "smash", debug=DEBUG) +kernel = xp.cuda.CudaKernel(SOURCE, "smash", debug=DEBUG) y = cp.zeros(64) try: kernel(y, 64, n_threads=64) @@ -1664,7 +1664,7 @@ def test_as_device_array_checks_ndim(): # --------------------------------------------------------------------------- -class ParticleArguments(xp.CudaStructArguments): +class ParticleArguments(xp.cuda.CudaStructArguments): """The class form of PARTICLES.""" struct_name = "Particles" @@ -1727,7 +1727,7 @@ def test_struct_arguments_check_their_fields(): def test_struct_arguments_need_every_field_attribute(): - class Incomplete(xp.CudaStructArguments): + class Incomplete(xp.cuda.CudaStructArguments): struct_name = "Incomplete" fields = (("x", "double*"), ("n", "int")) @@ -1744,16 +1744,16 @@ def __init__(self, x): def test_struct_arguments_class_definition(): with pytest.raises(TypeError, match="must define both struct_name and fields"): - class OnlyName(xp.CudaStructArguments): + class OnlyName(xp.cuda.CudaStructArguments): struct_name = "OnlyName" with pytest.raises(ValueError, match="unsupported type"): - class BadField(xp.CudaStructArguments): + class BadField(xp.cuda.CudaStructArguments): struct_name = "BadField" fields = (("a", "Other"),) - class Base(xp.CudaStructArguments): # intermediate base: no struct + class Base(xp.cuda.CudaStructArguments): # intermediate base: no struct def __init__(self): self.pack() @@ -1795,7 +1795,7 @@ def grow(self, n, ptr): self.markers = FakeDeviceArray(np.float64, ptr=ptr, shape=(n, 4)) -class OwnerArguments(xp.CudaStructArguments): +class OwnerArguments(xp.cuda.CudaStructArguments): struct_name = "OwnerArgs" fields = (("markers", "Array2D"), ("n_markers", "int")) @@ -1858,7 +1858,7 @@ def test_struct_arguments_as_kernel_arguments(): packed, dt, _, _ = kernel.prepare_args(args, 1, out, size) assert packed is args.packed and type(dt) is np.float64 - class Other(xp.CudaStructArguments): + class Other(xp.cuda.CudaStructArguments): struct_name = "Other" fields = (("x", "double*"),) diff --git a/tests/unit/test_cunumpy.py b/tests/unit/test_cunumpy.py index 481224b..4240789 100644 --- a/tests/unit/test_cunumpy.py +++ b/tests/unit/test_cunumpy.py @@ -147,7 +147,7 @@ def test_set_backend(): def test_device_count_is_zero_without_cupy(): if xp.cupy_available(): pytest.skip("CuPy is installed/functional; device_count() may be > 0") - assert xp.device_count() == 0 + assert xp.cuda.device_count() == 0 def test_device_count_matches_cupy_when_available(): @@ -155,39 +155,39 @@ def test_device_count_matches_cupy_when_available(): pytest.skip("CuPy not installed or not functional") import cupy as cp - assert xp.device_count() == cp.cuda.runtime.getDeviceCount() + assert xp.cuda.device_count() == cp.cuda.runtime.getDeviceCount() def test_set_device_for_rank_is_noop_without_gpus(): if xp.cupy_available(): pytest.skip("CuPy is installed/functional") - assert xp.set_device_for_rank(3) == 0 + assert xp.cuda.set_device_for_rank(3) == 0 def test_set_device_for_rank_wraps_around_devices_per_node(): if not xp.cupy_available(): pytest.skip("CuPy not installed or not functional") - n = xp.device_count() - assert xp.set_device_for_rank(n, devices_per_node=n) == 0 - assert xp.set_device_for_rank(n + 1, devices_per_node=n) == 1 % n + n = xp.cuda.device_count() + assert xp.cuda.set_device_for_rank(n, devices_per_node=n) == 0 + assert xp.cuda.set_device_for_rank(n + 1, devices_per_node=n) == 1 % n def test_memory_info_is_none_on_numpy_backend(): with xp.use_backend("numpy"): - assert xp.memory_info() is None + assert xp.cuda.memory_info() is None def test_memory_info_returns_free_and_total_on_cupy(): if not xp.cupy_available(): pytest.skip("CuPy not installed or not functional") with xp.use_backend("cupy"): - free, total = xp.memory_info() + free, total = xp.cuda.memory_info() assert 0 <= free <= total def test_free_memory_is_noop_on_numpy_backend(): with xp.use_backend("numpy"): - xp.free_memory() # must not raise + xp.cuda.free_memory() # must not raise def test_free_memory_does_not_increase_cupy_pool_cache(): @@ -200,7 +200,7 @@ def test_free_memory_does_not_increase_cupy_pool_cache(): del array pool = cp.get_default_memory_pool() cached_before = pool.free_bytes() - xp.free_memory() # must not raise + xp.cuda.free_memory() # must not raise # CuPy can retain split blocks even after free_all_blocks(). assert pool.free_bytes() <= cached_before @@ -209,19 +209,19 @@ def test_pin_memory_requires_cupy(): if xp.cupy_available(): pytest.skip("CuPy is installed/functional") with pytest.raises(ImportError): - xp.pin_memory(np.ones(3)) + xp.cuda.pin_memory(np.ones(3)) def test_pin_memory_round_trips_values(): if not xp.cupy_available(): pytest.skip("CuPy not installed or not functional") arr = np.array([1.0, 2.0, 3.0]) - pinned = xp.pin_memory(arr) + pinned = xp.cuda.pin_memory(arr) assert np.array_equal(pinned, arr) def test_stream_is_noop_on_numpy_backend(): - with xp.use_backend("numpy"), xp.stream() as s: + with xp.use_backend("numpy"), xp.cuda.stream() as s: assert s is None @@ -231,7 +231,7 @@ def test_stream_yields_a_cupy_stream_on_cupy_backend(): import cupy as cp with xp.use_backend("cupy"): - with xp.stream() as s: + with xp.cuda.stream() as s: assert isinstance(s, cp.cuda.Stream) arr = xp.zeros(10) assert xp.is_gpu(arr) @@ -240,7 +240,7 @@ def test_stream_yields_a_cupy_stream_on_cupy_backend(): def test_get_rng_returns_numpy_generator_on_numpy_backend(): with xp.use_backend("numpy"): - rng = xp.get_rng(42) + rng = xp.rng.get_rng(42) assert isinstance(rng, np.random.Generator) assert rng.random(3).shape == (3,) @@ -251,14 +251,14 @@ def test_get_rng_returns_cupy_generator_on_cupy_backend(): import cupy as cp with xp.use_backend("cupy"): - rng = xp.get_rng(42) + rng = xp.rng.get_rng(42) assert isinstance(rng, cp.random.Generator) def test_get_rng_is_reproducible_given_a_seed(): with xp.use_backend("numpy"): - a = xp.get_rng(123).random(5) - b = xp.get_rng(123).random(5) + a = xp.rng.get_rng(123).random(5) + b = xp.rng.get_rng(123).random(5) assert np.array_equal(a, b) @@ -354,7 +354,7 @@ def test_invalid_backend_raises_value_error(): def test_set_device_is_noop_on_numpy(): with xp.use_backend("numpy"): # Must not raise even though there's no GPU to select on the CPU backend. - xp.set_device(0) + xp.cuda.set_device(0) def test_set_device_selects_cuda_device(): @@ -364,7 +364,7 @@ def test_set_device_selects_cuda_device(): pytest.skip("CuPy not installed") with xp.use_backend("cupy"): - xp.set_device(0) + xp.cuda.set_device(0) assert cp.cuda.Device().id == 0 @@ -398,8 +398,10 @@ class FakeCupy: def test_max_shared_memory_per_block_without_a_gpu(monkeypatch): monkeypatch.setattr(xp.xp, "cupy_available", lambda: False) - assert xp.max_shared_memory_per_block() == xp.DEFAULT_SHARED_MEMORY_PER_BLOCK - assert xp.max_shared_memory_per_block(opt_in=True) == 48 * 1024 + assert ( + xp.cuda.max_shared_memory_per_block() == xp.cuda.DEFAULT_SHARED_MEMORY_PER_BLOCK + ) + assert xp.cuda.max_shared_memory_per_block(opt_in=True) == 48 * 1024 def test_max_shared_memory_per_block_reads_the_device(monkeypatch): @@ -417,15 +419,16 @@ def __init__(self, device_id=0): cupy.cuda = types.SimpleNamespace(Device=Device) monkeypatch.setitem(sys.modules, "cupy", cupy) monkeypatch.setattr(xp.xp, "cupy_available", lambda: True) - assert xp.max_shared_memory_per_block() == 49152 - assert xp.max_shared_memory_per_block(1) == 49153 - assert xp.max_shared_memory_per_block(opt_in=True) == 232448 + assert xp.cuda.max_shared_memory_per_block() == 49152 + assert xp.cuda.max_shared_memory_per_block(1) == 49153 + assert xp.cuda.max_shared_memory_per_block(opt_in=True) == 232448 def test_max_shared_memory_per_block_on_gpu(): if not xp.cupy_available(): pytest.skip("CuPy not installed or not functional") - assert xp.max_shared_memory_per_block() >= 48 * 1024 + assert xp.cuda.max_shared_memory_per_block() >= 48 * 1024 assert ( - xp.max_shared_memory_per_block(opt_in=True) >= xp.max_shared_memory_per_block() + xp.cuda.max_shared_memory_per_block(opt_in=True) + >= xp.cuda.max_shared_memory_per_block() ) diff --git a/tests/unit/test_device_binding.py b/tests/unit/test_device_binding.py index 19a1a2e..0b75abf 100644 --- a/tests/unit/test_device_binding.py +++ b/tests/unit/test_device_binding.py @@ -17,27 +17,27 @@ def clean_env(monkeypatch): def test_local_rank_default(clean_env): - assert xp.local_rank() == 0 + assert xp.mpi.local_rank() == 0 @pytest.mark.parametrize("variable", _LOCAL_RANK_VARIABLES) def test_local_rank_from_launcher(clean_env, variable): clean_env.setenv(variable, "3") - assert xp.local_rank() == 3 + assert xp.mpi.local_rank() == 3 def test_local_rank_order_and_invalid_values(clean_env): clean_env.setenv("SLURM_LOCALID", "5") clean_env.setenv("OMPI_COMM_WORLD_LOCAL_RANK", "2") - assert xp.local_rank() == 2 # the MPI launcher's value wins over Slurm's + assert xp.mpi.local_rank() == 2 # the MPI launcher's value wins over Slurm's clean_env.setenv("OMPI_COMM_WORLD_LOCAL_RANK", "not a number") - assert xp.local_rank() == 5 # invalid values are skipped + assert xp.mpi.local_rank() == 5 # invalid values are skipped def test_bind_local_device_on_numpy(clean_env): with xp.use_backend("numpy"): - assert xp.bind_local_device() is None + assert xp.cuda.bind_local_device() is None def test_bind_local_device_on_cupy(clean_env): @@ -45,12 +45,12 @@ def test_bind_local_device_on_cupy(clean_env): pytest.skip("CuPy not installed or not functional") import cupy as cp - count = xp.device_count() + count = xp.cuda.device_count() previous = cp.cuda.runtime.getDevice() clean_env.setenv("OMPI_COMM_WORLD_LOCAL_RANK", str(count + 1)) try: with xp.use_backend("cupy"): - device = xp.bind_local_device() + device = xp.cuda.bind_local_device() assert device == (count + 1) % count assert cp.cuda.runtime.getDevice() == device finally: @@ -59,8 +59,8 @@ def test_bind_local_device_on_cupy(clean_env): def test_synchronize_for_mpi_host_arrays(): # host buffers and None: nothing to wait for (and no CuPy needed) - xp.synchronize_for_mpi(np.zeros(3), None) - xp.synchronize_for_mpi() + xp.mpi.synchronize_for_mpi(np.zeros(3), None) + xp.mpi.synchronize_for_mpi() def test_synchronize_for_mpi_device_arrays(): @@ -70,6 +70,6 @@ def test_synchronize_for_mpi_device_arrays(): x = cp.zeros(1_000_000) x += 1 # a kernel still running on the current stream - xp.synchronize_for_mpi(np.zeros(1), x) + xp.mpi.synchronize_for_mpi(np.zeros(1), x) assert cp.cuda.get_current_stream().done # nothing pending any more assert float(x[0]) == 1.0 diff --git a/tests/unit/test_emulation.py b/tests/unit/test_emulation.py index bc15a23..b6bfb64 100644 --- a/tests/unit/test_emulation.py +++ b/tests/unit/test_emulation.py @@ -1,10 +1,10 @@ -"""Tests for `cunumpy.testing.emulate_cuda_kernel`: CUDA kernels run on the CPU.""" +"""Tests for `cunumpy.kernel_testing.emulate_cuda_kernel`: CUDA kernels run on the CPU.""" import numpy as np import pytest -from cunumpy import CudaKernel -from cunumpy.testing import emulate_cuda_kernel, emulation_compiler +from cunumpy.cuda import CudaKernel +from cunumpy.kernel_testing import emulate_cuda_kernel, emulation_compiler pytestmark = pytest.mark.skipif( emulation_compiler() is None, reason="no C++ compiler for the emulation" diff --git a/tests/unit/test_fusion.py b/tests/unit/test_fusion.py index 717ffe9..61443d3 100644 --- a/tests/unit/test_fusion.py +++ b/tests/unit/test_fusion.py @@ -1,4 +1,4 @@ -"""Tests for `xp.fuse`: cupy.fuse for CuPy arrays, a plain call otherwise.""" +"""Tests for `xp.kernels.fuse`: cupy.fuse for CuPy arrays, a plain call otherwise.""" import numpy as np import pytest @@ -12,19 +12,19 @@ def pressure(rho, T, gamma): def test_host_arrays_call_the_function(): - fused = xp.fuse(pressure) + fused = xp.kernels.fuse(pressure) rho, T = np.full(4, 2.0), np.zeros(4) np.testing.assert_array_equal(fused(rho, T, 3.0), pressure(rho, T, 3.0)) assert fused.__name__ == "pressure" and fused.__wrapped__ is pressure - assert "fuse" in xp.__all__ + assert "fuse" in xp.kernels.__all__ def test_decorator_forms(): - @xp.fuse + @xp.kernels.fuse def double(x): return 2.0 * x - @xp.fuse(kernel_name="triple_kernel") + @xp.kernels.fuse(kernel_name="triple_kernel") def triple(x): return 3.0 * x @@ -54,7 +54,7 @@ def fused(*args, **kwargs): fusion, "_is_device_array", lambda a: isinstance(a, FakeDeviceArray) ) - @xp.fuse(kernel_name="p") + @xp.kernels.fuse(kernel_name="p") def p(rho, T, gamma=1.0): raise AssertionError("not called with device arrays") @@ -75,7 +75,7 @@ def test_python_scalars_take_the_dtype_of_the_arrays(monkeypatch): ) monkeypatch.setattr(fusion, "_is_device_array", lambda a: hasattr(a, "dtype")) - p = xp.fuse(pressure) + p = xp.kernels.fuse(pressure) p(np.ones(2), 0.5, 5.0 / 3.0) p(np.ones(2, dtype=np.float32), 0.5, gamma=2) p(np.ones(2, dtype=np.int64), True, 3) @@ -90,7 +90,7 @@ def test_fuse_on_gpu(): pytest.skip("CuPy not installed or not functional") import cupy as cp - fused = xp.fuse(pressure) + fused = xp.kernels.fuse(pressure) rho, T = cp.full(1000, 2.0), cp.linspace(0.0, 1.0, 1000) with xp.use_backend("cupy"): expected = pressure(rho, T, 5.0 / 3.0) diff --git a/tests/unit/test_kernel_dispatch.py b/tests/unit/test_kernel_dispatch.py index a34c8d2..3444f85 100644 --- a/tests/unit/test_kernel_dispatch.py +++ b/tests/unit/test_kernel_dispatch.py @@ -1,4 +1,4 @@ -"""Tests for `cunumpy.Kernel` (host/CUDA pairs) and `cunumpy.KernelCatalog`. +"""Tests for `cunumpy.kernels.Kernel` (host/CUDA pairs) and `cunumpy.kernels.KernelCatalog`. The host kernels here are plain Python functions (wrapped in `PyccelKernel`), so the NumPy-side tests run everywhere; the CuPy-side tests need a GPU. @@ -14,8 +14,8 @@ import pytest import cunumpy as xp -from cunumpy import ( - CudaKernel, +from cunumpy.cuda import CudaKernel +from cunumpy.kernels import ( Kernel, KernelArguments, KernelCatalog, @@ -34,7 +34,7 @@ def _user_options(kernel): """`compile_options()` without cunumpy's own include directory.""" return tuple( - o for o in kernel.compile_options() if o != f"-I{xp.cuda_include_dir()}" + o for o in kernel.compile_options() if o != f"-I{xp.cuda.cuda_include_dir()}" ) @@ -137,7 +137,7 @@ def {name}(x, a, n): (root / "scale" / "scale_cuda.cu").write_text(SCALE_CUDA) (root / "not_a_kernel").mkdir() (root / "__init__.py").write_text( - "from cunumpy import KernelCatalog\n\n" + "from cunumpy.kernels import KernelCatalog\n\n" "catalog = KernelCatalog.from_package(__name__)\n" ) monkeypatch.syspath_prepend(str(tmp_path)) diff --git a/tests/unit/test_kernel_dispatch_arrays.py b/tests/unit/test_kernel_dispatch_arrays.py index d3c18cb..a044b26 100644 --- a/tests/unit/test_kernel_dispatch_arrays.py +++ b/tests/unit/test_kernel_dispatch_arrays.py @@ -10,16 +10,15 @@ import pytest import cunumpy as xp -from cunumpy import ( +from cunumpy import dispatch as dispatch_module +from cunumpy.cuda import CudaArguments, CudaKernel +from cunumpy.kernels import ( CompiledHostKernel, - CudaArguments, - CudaKernel, HostImplementations, Kernel, KernelArguments, KernelCatalog, ) -from cunumpy import dispatch as dispatch_module SCALE_CUDA = r""" extern "C" __global__ void scale(double* x, double factor, int n) { @@ -331,7 +330,7 @@ def self_declaring_package(tmp_path, monkeypatch): "def _compile(module):\n" " COMPILED.append(module.__name__)\n" " return module\n\n" - "kernel = xp.Kernel.from_folder(\n" + "kernel = xp.kernels.Kernel.from_folder(\n" " __name__, host_suffix='_pyccel', dispatch='arrays',\n" " compile_host=_compile, n_threads_from='first_array',\n" ")\n" @@ -370,27 +369,27 @@ def test_from_folder_needs_a_kernel_folder(self_declaring_package): def test_kernel_implementation_setting(self_declaring_package): kernel = self_declaring_package.kernel calls = importlib.import_module("demo_folder_pkg.scale.scale_numpy").CALLS - assert xp.get_kernel_implementation() is None - with xp.use_kernel_implementation("numpy"): - assert xp.get_kernel_implementation() == "numpy" + assert xp.kernels.get_kernel_implementation() is None + with xp.kernels.use_kernel_implementation("numpy"): + assert xp.kernels.get_kernel_implementation() == "numpy" assert kernel.selected() == "numpy" x = np.ones(2) kernel(x, 3.0, 2) assert x.tolist() == [3.0, 3.0] and calls == [2] - with xp.use_kernel_implementation("python"): + with xp.kernels.use_kernel_implementation("python"): kernel(x, 2.0, 2) # the uncompiled pyccel source assert calls == [2] and x.tolist() == [6.0, 6.0] - assert xp.get_kernel_implementation() is None + assert xp.kernels.get_kernel_implementation() is None assert self_declaring_package.COMPILED == [] # pyccel never needed # a chosen implementation that cannot run raises instead of running another - xp.set_kernel_implementation("numba") + xp.kernels.set_kernel_implementation("numba") try: with pytest.raises(LookupError, match="'numba' implementation .* unavailable"): kernel(np.ones(1), 2.0, 1) finally: - xp.set_kernel_implementation(None) + xp.kernels.set_kernel_implementation(None) with pytest.raises(ValueError, match="kernel implementation must be one of"): - xp.set_kernel_implementation("fortran") + xp.kernels.set_kernel_implementation("fortran") def test_default_skips_unavailable_implementations(): @@ -430,7 +429,7 @@ def test_kernel_implementation_environment_variable(): import os import subprocess - code = "import cunumpy as xp; print(xp.get_kernel_implementation())" + code = "import cunumpy as xp; print(xp.kernels.get_kernel_implementation())" env = {**os.environ, "CUNUMPY_KERNEL_IMPLEMENTATION": "numpy"} printed = subprocess.run( [sys.executable, "-c", code], @@ -459,14 +458,14 @@ def test_arrays_dispatch_calls_the_host_kernel_without_conversion( def no_conversion(self, args, kwargs): raise AssertionError("conversion checked") - monkeypatch.setattr(xp.PyccelKernel, "_needs_conversion", no_conversion) + monkeypatch.setattr(xp.kernels.PyccelKernel, "_needs_conversion", no_conversion) kernel = Kernel(scale, CudaKernel(SCALE_CUDA, "scale"), dispatch="arrays") x = np.ones(2) kernel(x, 2.0, 2) assert x.tolist() == [2.0, 2.0] # a host kernel that is told to convert keeps doing so forced = Kernel( - xp.PyccelKernel(scale, use_cupy=True), + xp.kernels.PyccelKernel(scale, use_cupy=True), CudaKernel(SCALE_CUDA, "scale"), dispatch="arrays", ) @@ -500,26 +499,29 @@ def swapped(x, n, factor): def test_as_kernel_array_on_the_host(): grid = np.zeros((4, 3)) fits = np.ones(5) - assert xp.as_kernel_array(fits, like=grid, dtype=float) is fits # no copy + assert xp.kernels.as_kernel_array(fits, like=grid, dtype=float) is fits # no copy column = np.ones((5, 2))[:, 1] - converted = xp.as_kernel_array(column, like=grid, dtype=float) + converted = xp.kernels.as_kernel_array(column, like=grid, dtype=float) assert converted.flags.c_contiguous and converted is not column - ints = xp.as_kernel_array([1, 2], like=grid, dtype=float) + ints = xp.kernels.as_kernel_array([1, 2], like=grid, dtype=float) assert ints.dtype == np.float64 and isinstance(ints, np.ndarray) def test_kernel_output_writes_into_its_target(): grid = np.zeros(3) out = np.zeros(4) - with xp.kernel_output(out, like=grid, dtype=float) as buffer: + with xp.kernels.kernel_output(out, like=grid, dtype=float) as buffer: assert buffer is out # written directly buffer += 1.0 strided = np.zeros((4, 2))[:, 0] - with xp.kernel_output(strided, like=grid, dtype=float) as buffer: + with xp.kernels.kernel_output(strided, like=grid, dtype=float) as buffer: assert buffer is not strided buffer[...] = 7.0 assert strided.tolist() == [7.0] * 4 - with pytest.raises(RuntimeError), xp.kernel_output(strided, like=grid) as buffer: + with ( + pytest.raises(RuntimeError), + xp.kernels.kernel_output(strided, like=grid) as buffer, + ): buffer[...] = 1.0 raise RuntimeError("kernel failed") assert strided.tolist() == [7.0] * 4 # not copied back after an error @@ -531,10 +533,10 @@ def test_kernel_arrays_follow_a_device_grid(): import cupy as cp grid = cp.zeros(3) - on_device = xp.as_kernel_array(np.ones(4), like=grid, dtype=float) + on_device = xp.kernels.as_kernel_array(np.ones(4), like=grid, dtype=float) assert xp.is_gpu(on_device) host_out = np.zeros(4) - with xp.kernel_output(host_out, like=grid, dtype=float) as buffer: + with xp.kernels.kernel_output(host_out, like=grid, dtype=float) as buffer: assert xp.is_gpu(buffer) buffer[...] = 2.0 assert host_out.tolist() == [2.0] * 4 diff --git a/tests/unit/test_testing.py b/tests/unit/test_kernel_testing.py similarity index 94% rename from tests/unit/test_testing.py rename to tests/unit/test_kernel_testing.py index b3a496a..cd2cf39 100644 --- a/tests/unit/test_testing.py +++ b/tests/unit/test_kernel_testing.py @@ -1,4 +1,4 @@ -"""Tests for `cunumpy.testing`. +"""Tests for `cunumpy.kernel_testing`. The pytest markers, the array collection and comparison of `assert_kernels_agree` and the source generation of `device_function_kernel` run everywhere; running @@ -14,17 +14,18 @@ import pytest import cunumpy as xp -import cunumpy.testing -from cunumpy import CudaArguments, CudaKernel, Kernel, parse_cuda_signature -from cunumpy.testing import backend # noqa: F401 - the fixture is used by name -from cunumpy.testing import ( +import cunumpy.kernel_testing +from cunumpy.cuda import CudaArguments, CudaKernel, parse_cuda_signature +from cunumpy.kernel_testing import ( BACKENDS, _collect_arrays, _compare_results, assert_kernels_agree, + backend, # noqa: F401 - the fixture is used by name device_function_kernel, requires_cupy, ) +from cunumpy.kernels import Kernel SCALE_CUDA = r""" extern "C" __global__ void scale(double* x, double factor, int n) { @@ -55,11 +56,11 @@ def make_scale_args(backend_name, seed): def test_import_does_not_need_pytest(): - """cunumpy.testing (and cunumpy) import without importing pytest.""" + """cunumpy.kernel_testing (and cunumpy) import without importing pytest.""" code = ( - "import sys, cunumpy, cunumpy.testing\n" + "import sys, cunumpy, cunumpy.kernel_testing\n" "assert 'pytest' not in sys.modules\n" - "assert 'device_function_kernel' in dir(cunumpy.testing)\n" + "assert 'device_function_kernel' in dir(cunumpy.kernel_testing)\n" ) # the same source tree as this test, whether or not cunumpy is installed source_root = Path(cunumpy.__file__).parents[1] @@ -74,7 +75,7 @@ def test_backends_and_marker(): assert requires_cupy.args == (not xp.cupy_available(),) assert requires_cupy.kwargs["reason"] == "CuPy/GPU not available" with pytest.raises(AttributeError): - _ = cunumpy.testing.no_such_thing + _ = cunumpy.kernel_testing.no_such_thing @pytest.mark.parametrize("backend_name", BACKENDS) @@ -308,7 +309,7 @@ def test_device_function_kernel_on_gpu(): def test_struct_arguments_are_compared_by_field_name(): """A CudaStructArguments object (fields may be properties) gets the host names.""" - from cunumpy.testing import _collect_arrays + from cunumpy.kernel_testing import _collect_arrays class Owner: def __init__(self): @@ -321,7 +322,7 @@ def __init__(self, owner): self.weights = owner.weights self.n = 3 - class DeviceArguments(xp.CudaStructArguments): + class DeviceArguments(xp.cuda.CudaStructArguments): struct_name = "OwnerArgs" fields = (("markers", "Array2D"), ("weights", "double*"), ("n", "int")) @@ -347,7 +348,7 @@ def weights(self): assert device["argument 1.markers"] is owner.markers struct = DeviceArguments.struct - value = xp.CudaStructValue( + value = xp.cuda.CudaStructValue( struct, np.zeros((), struct.dtype)[()], vars(owner) | {"n": 3} ) assert sorted(_collect_arrays((value,))) == [ diff --git a/tests/unit/test_mirror.py b/tests/unit/test_mirror.py index 1a97892..5daa24b 100644 --- a/tests/unit/test_mirror.py +++ b/tests/unit/test_mirror.py @@ -1,4 +1,4 @@ -"""Tests for `cunumpy.DeviceMirror` and the shipped `cunumpy/atomic.cuh` header. +"""Tests for `cunumpy.memory.DeviceMirror` and the shipped `cunumpy/atomic.cuh` header. On the NumPy backend a mirror is transparent: `device` is the host array and transfers are no-ops. Device copies and the atomic kernel need a GPU and are @@ -11,7 +11,8 @@ import pytest import cunumpy as xp -from cunumpy import CudaKernel, DeviceMirror +from cunumpy.cuda import CudaKernel +from cunumpy.memory import DeviceMirror requires_gpu = pytest.mark.skipif( not xp.cupy_available(), reason="CuPy/GPU not available or not functional" @@ -125,7 +126,7 @@ def test_rebind(): def test_cuda_include_dir_contains_atomic_header(): - include_dir = xp.cuda_include_dir() + include_dir = xp.cuda.cuda_include_dir() assert isinstance(include_dir, str) assert include_dir.endswith(os.path.join("cuda", "include")) header = os.path.join(include_dir, "cunumpy", "atomic.cuh") @@ -139,7 +140,7 @@ def test_cuda_include_dir_contains_atomic_header(): def test_cuda_kernel_options_include_cunumpy_headers(): - flag = f"-I{xp.cuda_include_dir()}" + flag = f"-I{xp.cuda.cuda_include_dir()}" kernel = CudaKernel(BIN_ADD, "bin_add") assert flag not in kernel.options # options are as given assert flag in kernel.compile_options() diff --git a/tests/unit/test_morton.py b/tests/unit/test_morton.py index de15868..318fad4 100644 --- a/tests/unit/test_morton.py +++ b/tests/unit/test_morton.py @@ -6,8 +6,8 @@ import pytest import cunumpy as xp -from cunumpy import CudaKernel, cuda_include_dir -from cunumpy.testing import emulate_cuda_kernel, emulation_compiler +from cunumpy.cuda import CudaKernel, cuda_include_dir +from cunumpy.kernel_testing import emulate_cuda_kernel, emulation_compiler def interleave(cells, levels): @@ -22,12 +22,12 @@ def interleave(cells, levels): @pytest.mark.parametrize("ndim", [2, 3]) def test_encode_matches_bitwise_reference(ndim): - levels = xp.MAX_MORTON_LEVELS[ndim] + levels = xp.algorithms.MAX_MORTON_LEVELS[ndim] rng = np.random.default_rng(0) cells = rng.integers(0, 2**levels, size=(ndim, 500), dtype=np.uint64) cells[:, 0] = 2**levels - 1 # all bits set cells[:, 1] = 0 - keys = xp.morton_encode(*cells) + keys = xp.algorithms.morton_encode(*cells) assert keys.dtype == np.uint64 expected = [interleave(cells[:, i], levels) for i in range(cells.shape[1])] assert keys.tolist() == expected @@ -35,20 +35,20 @@ def test_encode_matches_bitwise_reference(ndim): @pytest.mark.parametrize("ndim", [2, 3]) def test_decode_inverts_encode(ndim): - levels = xp.MAX_MORTON_LEVELS[ndim] + levels = xp.algorithms.MAX_MORTON_LEVELS[ndim] rng = np.random.default_rng(1) cells = rng.integers(0, 2**levels, size=(ndim, 1000), dtype=np.uint64) - decoded = xp.morton_decode(xp.morton_encode(*cells), ndim) + decoded = xp.algorithms.morton_decode(xp.algorithms.morton_encode(*cells), ndim) for axis in range(ndim): np.testing.assert_array_equal(decoded[axis], cells[axis]) def test_encode_broadcasts_and_rejects_bad_dimensions(): - keys = xp.morton_encode(np.arange(4)[:, None], np.arange(3)) + keys = xp.algorithms.morton_encode(np.arange(4)[:, None], np.arange(3)) assert keys.shape == (4, 3) assert keys[2, 1] == interleave((2, 1), 2) with pytest.raises(ValueError, match="2 or 3 dimensions"): - xp.morton_encode(np.arange(3)) + xp.algorithms.morton_encode(np.arange(3)) def test_keys_cells_and_clipping(): @@ -62,7 +62,7 @@ def test_keys_cells_and_clipping(): [-5.0, 7.0], # outside: the nearest face ] ) - keys = xp.morton_keys(positions, [0.0, 0.0], [1.0, 1.0], levels) + keys = xp.algorithms.morton_keys(positions, [0.0, 0.0], [1.0, 1.0], levels) cells = [(0, 0), (1, 0), (7, 4), (7, 7), (0, 7)] assert keys.tolist() == [interleave(c, levels) for c in cells] @@ -70,36 +70,38 @@ def test_keys_cells_and_clipping(): def test_reversed_axis(): # lower > upper on y: cell 0 at the top, like a quadtree with y < mid as # its second quadrant bit - keys = xp.morton_keys(np.array([[0.2, 0.9], [0.2, 0.1]]), [0, 1], [1, 0], 1) + keys = xp.algorithms.morton_keys( + np.array([[0.2, 0.9], [0.2, 0.1]]), [0, 1], [1, 0], 1 + ) assert keys.tolist() == [0, 2] def test_keys_validate_arguments(): with pytest.raises(ValueError, match="levels"): - xp.morton_keys(np.zeros((3, 2)), [0, 0], [1, 1], 33) + xp.algorithms.morton_keys(np.zeros((3, 2)), [0, 0], [1, 1], 33) with pytest.raises(ValueError, match="levels"): - xp.morton_keys(np.zeros((3, 3)), [0, 0, 0], [1, 1, 1], 22) + xp.algorithms.morton_keys(np.zeros((3, 3)), [0, 0, 0], [1, 1, 1], 22) with pytest.raises(ValueError, match="differ"): - xp.morton_keys(np.zeros((3, 2)), [0, 0], [1, 0], 4) + xp.algorithms.morton_keys(np.zeros((3, 2)), [0, 0], [1, 0], 4) with pytest.raises(ValueError, match="need 2 values"): - xp.morton_keys(np.zeros((3, 2)), [0, 0, 0], [1, 1, 1], 4) + xp.algorithms.morton_keys(np.zeros((3, 2)), [0, 0, 0], [1, 1, 1], 4) with pytest.raises(ValueError, match=r"\(n, ndim\)"): - xp.morton_keys(np.zeros(3), [0, 0], [1, 1], 4) + xp.algorithms.morton_keys(np.zeros(3), [0, 0], [1, 1], 4) def test_sorted_keys_make_tree_nodes_contiguous(): rng = np.random.default_rng(2) positions = rng.random((2000, 2)) levels = 10 - keys, _, sorted_positions = xp.sort_by_key( - xp.morton_keys(positions, [0, 0], [1, 1], levels), positions + keys, _, sorted_positions = xp.algorithms.sort_by_key( + xp.algorithms.morton_keys(positions, [0, 0], [1, 1], levels), positions ) for level in (1, 2, 3): node = keys >> np.uint64(2 * (levels - level)) assert np.all(node[1:] >= node[:-1]) # every point of a node lies in that node's square size = 0.5**level - cx, cy = xp.morton_decode(node, 2) + cx, cy = xp.algorithms.morton_decode(node, 2) assert np.all(np.floor(sorted_positions[:, 0] / size) == cx) assert np.all(np.floor(sorted_positions[:, 1] / size) == cy) @@ -108,20 +110,22 @@ def test_sort_by_key_is_stable_and_reorders_all_arrays(): keys = np.array([2, 0, 1, 0, 2], dtype=np.uint64) ids = np.arange(5) rows = np.arange(10.0).reshape(5, 2) - sorted_keys, order, sorted_ids, sorted_rows = xp.sort_by_key(keys, ids, rows) + sorted_keys, order, sorted_ids, sorted_rows = xp.algorithms.sort_by_key( + keys, ids, rows + ) assert order.dtype == np.int64 assert order.tolist() == [1, 3, 2, 0, 4] assert sorted_keys.tolist() == [0, 0, 1, 2, 2] np.testing.assert_array_equal(sorted_ids, ids[order]) np.testing.assert_array_equal(sorted_rows, rows[order]) - assert len(xp.sort_by_key(keys)) == 2 + assert len(xp.algorithms.sort_by_key(keys)) == 2 def test_sort_by_key_validates_shapes(): with pytest.raises(ValueError, match="1D"): - xp.sort_by_key(np.zeros((2, 2))) + xp.algorithms.sort_by_key(np.zeros((2, 2))) with pytest.raises(ValueError, match="3 rows"): - xp.sort_by_key(np.zeros(3), np.zeros(4)) + xp.algorithms.sort_by_key(np.zeros(3), np.zeros(4)) def test_header_is_shipped(): @@ -160,9 +164,9 @@ def _cases(ndim): lower = [-1.5, 0.25, 2.0][:ndim] upper = [2.5, -0.75, 3.0][:ndim] # the second axis is reversed positions = rng.uniform(-2.0, 3.5, size=(997, ndim)) - levels = xp.MAX_MORTON_LEVELS[ndim] + levels = xp.algorithms.MAX_MORTON_LEVELS[ndim] # points on cell boundaries, where rounding would show - edges = np.array(lower) + np.arange(5)[:, None] / xp.morton_scales( + edges = np.array(lower) + np.arange(5)[:, None] / xp.algorithms.morton_scales( lower, upper, levels ) positions = np.ascontiguousarray(np.vstack([positions, edges])) @@ -173,10 +177,10 @@ def _device_keys(run, ndim): positions, lower, upper, levels = _cases(ndim) n = positions.shape[0] keys = np.zeros(n, dtype=np.uint64) - scales = xp.morton_scales(lower, upper, levels) + scales = xp.algorithms.morton_scales(lower, upper, levels) args = (*lower, *scales.tolist(), levels) run(CudaKernel(KEYS, f"keys{ndim}"), positions, keys, n, *args, n_threads=n) - return keys, xp.morton_keys(positions, lower, upper, levels) + return keys, xp.algorithms.morton_keys(positions, lower, upper, levels) @pytest.mark.skipif(emulation_compiler() is None, reason="no C++ compiler") @@ -210,13 +214,13 @@ def test_cupy_arrays_stay_on_the_device(): import cupy as cp positions, lower, upper, levels = _cases(2) - keys = xp.morton_keys(cp.asarray(positions), lower, upper, levels) + keys = xp.algorithms.morton_keys(cp.asarray(positions), lower, upper, levels) assert isinstance(keys, cp.ndarray) np.testing.assert_array_equal( - cp.asnumpy(keys), xp.morton_keys(positions, lower, upper, levels) + cp.asnumpy(keys), xp.algorithms.morton_keys(positions, lower, upper, levels) ) - sorted_keys, order, _ = xp.sort_by_key(keys, cp.asarray(positions)) + sorted_keys, order, _ = xp.algorithms.sort_by_key(keys, cp.asarray(positions)) assert isinstance(order, cp.ndarray) assert bool((sorted_keys[1:] >= sorted_keys[:-1]).all()) - cx, _ = xp.morton_decode(sorted_keys, 2) + cx, _ = xp.algorithms.morton_decode(sorted_keys, 2) assert isinstance(cx, cp.ndarray) diff --git a/tests/unit/test_mpi_cuda_aware.py b/tests/unit/test_mpi_cuda_aware.py index aaec610..6231a0c 100644 --- a/tests/unit/test_mpi_cuda_aware.py +++ b/tests/unit/test_mpi_cuda_aware.py @@ -72,55 +72,55 @@ def buffers(rank, n=4): def test_numpy_backend_returns_false_without_mpi(no_mpi4py): with xp.use_backend("numpy"): - assert xp.mpi_is_cuda_aware() is False - assert xp.mpi_is_cuda_aware(FakeComm()) is False - assert xp.require_cuda_aware_mpi() is None # no-op, mpi4py not imported + assert xp.mpi.mpi_is_cuda_aware() is False + assert xp.mpi.mpi_is_cuda_aware(FakeComm()) is False + assert xp.mpi.require_cuda_aware_mpi() is None # no-op, mpi4py not imported def test_unknown_method(): with pytest.raises(ValueError, match="probe"): - xp.mpi_is_cuda_aware(FakeComm(), method="query") + xp.mpi.mpi_is_cuda_aware(FakeComm(), method="query") def test_probe_succeeds(device_buffers, fake_mpi): comm = FakeComm() - assert xp.mpi_is_cuda_aware(comm) is True + assert xp.mpi.mpi_is_cuda_aware(comm) is True assert comm.sendrecv_calls == [(0, 0)] # size 1: to and from itself assert comm.allreduce_calls == [(True, "LAND")] def test_probe_uses_comm_world_by_default(device_buffers, fake_mpi): - assert xp.mpi_is_cuda_aware() is True + assert xp.mpi.mpi_is_cuda_aware() is True assert fake_mpi.sendrecv_calls == [(0, 0)] def test_probe_neighbours(device_buffers, fake_mpi): comm = FakeComm(rank=3, size=4) - assert xp.mpi_is_cuda_aware(comm) is True + assert xp.mpi.mpi_is_cuda_aware(comm) is True assert comm.sendrecv_calls == [(0, 2)] # to the next rank, from the previous def test_probe_exception_gives_false(device_buffers, fake_mpi): comm = FakeComm(fail=RuntimeError("MPI_ERR_BUFFER")) - assert xp.mpi_is_cuda_aware(comm) is False + assert xp.mpi.mpi_is_cuda_aware(comm) is False assert comm.allreduce_calls == [(False, "LAND")] def test_probe_wrong_values_give_false(device_buffers, fake_mpi): comm = FakeComm(corrupt=True) - assert xp.mpi_is_cuda_aware(comm) is False + assert xp.mpi.mpi_is_cuda_aware(comm) is False def test_probe_other_rank_failed(device_buffers, fake_mpi): comm = FakeComm(allreduce_result=False) # this rank ok, another one not - assert xp.mpi_is_cuda_aware(comm) is False + assert xp.mpi.mpi_is_cuda_aware(comm) is False def test_require_raises(device_buffers, fake_mpi): comm = FakeComm(fail=RuntimeError("MPI_ERR_BUFFER")) with pytest.raises(RuntimeError, match="CUDA-aware"): - xp.require_cuda_aware_mpi(comm) - xp.require_cuda_aware_mpi(FakeComm()) # succeeds silently + xp.mpi.require_cuda_aware_mpi(comm) + xp.mpi.require_cuda_aware_mpi(FakeComm()) # succeeds silently def test_probe_on_comm_world(): @@ -128,4 +128,4 @@ def test_probe_on_comm_world(): pytest.skip("CuPy not installed or not functional") pytest.importorskip("mpi4py") with xp.use_backend("cupy"): - assert isinstance(xp.mpi_is_cuda_aware(), bool) + assert isinstance(xp.mpi.mpi_is_cuda_aware(), bool) diff --git a/tests/unit/test_namespaces.py b/tests/unit/test_namespaces.py new file mode 100644 index 0000000..1bb7723 --- /dev/null +++ b/tests/unit/test_namespaces.py @@ -0,0 +1,89 @@ +"""Tests for the submodule layout of cunumpy (0.5) and the deprecated top-level names.""" + +import importlib +import subprocess +import sys +import warnings + +import numpy as np +import pytest + +import cunumpy as xp + +SUBMODULES = ( + "algorithms", + "cuda", + "kernels", + "memory", + "mpi", + "petsc", + "profiling", + "rng", +) + + +@pytest.mark.parametrize("name", SUBMODULES) +def test_submodule_names_do_not_shadow_numpy(name): + # `import cunumpy as xp` stands in for numpy: a submodule must not hide a numpy name + assert not hasattr(np, name) + assert getattr(xp, name) is importlib.import_module(f"cunumpy.{name}") + + +@pytest.mark.parametrize("name", SUBMODULES) +def test_submodule_exports_resolve(name): + module = getattr(xp, name) + for attr in getattr(module, "__all__", ()): + assert hasattr(module, attr), f"cunumpy.{name}.{attr}" + + +def test_top_level_is_backend_and_numpy_only(): + moved = set(xp._MOVED) + assert not moved & set(xp.__all__) + for name in xp.__all__: + assert hasattr(xp, name) + + +@pytest.mark.parametrize("name", sorted(xp._MOVED)) +def test_moved_names_warn_and_resolve(name): + submodule = xp._MOVED[name] + with pytest.warns(DeprecationWarning, match=f"cunumpy.{submodule}.{name}"): + value = getattr(xp, name) + assert value is getattr(getattr(xp, submodule), name) + + +def test_numpy_names_do_not_warn(): + with warnings.catch_warnings(): + warnings.simplefilter("error") + assert xp.zeros(2).shape == (2,) + assert xp.random is not None + assert xp.testing.assert_allclose is not None + + +def test_unknown_name_raises(): + with pytest.raises(AttributeError): + _ = xp.no_such_function_in_cunumpy + + +def test_kernel_testing_keeps_numpy_testing(): + import cunumpy.kernel_testing # noqa: F401 + + assert xp.testing.assert_allclose is not None + + +def test_testing_alias_is_deprecated(): + # a fresh process: importing cunumpy.testing rebinds xp.testing for the rest of it + code = ( + "import warnings\n" + "warnings.simplefilter('error')\n" + "try:\n" + " import cunumpy.testing\n" + "except DeprecationWarning as w:\n" + " assert 'cunumpy.kernel_testing' in str(w)\n" + "else:\n" + " raise SystemExit('no warning')\n" + "warnings.simplefilter('ignore')\n" + "import cunumpy.testing, cunumpy.kernel_testing\n" + "assert cunumpy.testing.assert_kernels_agree is " + "cunumpy.kernel_testing.assert_kernels_agree\n" + ) + subprocess.run([sys.executable, "-c", code], check=True) diff --git a/tests/unit/test_petsc.py b/tests/unit/test_petsc.py index b2a6eb9..7a328bc 100644 --- a/tests/unit/test_petsc.py +++ b/tests/unit/test_petsc.py @@ -1,4 +1,4 @@ -"""Tests for `xp.petsc_vec`: PETSc vectors sharing the memory of an array.""" +"""Tests for `xp.petsc.petsc_vec`: PETSc vectors sharing the memory of an array.""" import sys import types @@ -21,7 +21,7 @@ def _petsc(): def test_host_vector_shares_memory(): PETSc = _petsc() a = np.zeros(6, dtype=PETSc.ScalarType) - vec = xp.petsc_vec(a, comm=PETSc.COMM_SELF) + vec = xp.petsc.petsc_vec(a, comm=PETSc.COMM_SELF) assert vec.getSize() == 6 and vec.getType() == "seq" vec.set(3.0) assert a.tolist() == [3.0] * 6 # PETSc wrote into the array @@ -33,7 +33,7 @@ def test_host_vector_shares_memory(): def test_multidimensional_arrays_are_unrolled(): PETSc = _petsc() a = np.arange(6, dtype=PETSc.ScalarType).reshape(2, 3) - vec = xp.petsc_vec(a, comm=PETSc.COMM_SELF) + vec = xp.petsc.petsc_vec(a, comm=PETSc.COMM_SELF) assert vec.getArray().tolist() == a.ravel().tolist() @@ -55,7 +55,8 @@ def test_ksp_solve_writes_into_the_array(): ksp.getPC().setType("none") ksp.setTolerances(rtol=1e-12) ksp.solve( - xp.petsc_vec(b, comm=PETSc.COMM_SELF), xp.petsc_vec(x, comm=PETSc.COMM_SELF) + xp.petsc.petsc_vec(b, comm=PETSc.COMM_SELF), + xp.petsc.petsc_vec(x, comm=PETSc.COMM_SELF), ) residual = 2 * x - np.r_[0.0, x[:-1]] - np.r_[x[1:], 0.0] - 1.0 assert np.abs(residual).max() < 1e-9 @@ -65,11 +66,11 @@ def test_rejects_arrays_that_would_need_a_copy(): PETSc = _petsc() other = np.float32 if np.dtype(PETSc.ScalarType) != np.float32 else np.float64 with pytest.raises(TypeError, match="scalar type"): - xp.petsc_vec(np.zeros(4, dtype=other)) + xp.petsc.petsc_vec(np.zeros(4, dtype=other)) with pytest.raises(ValueError, match="C-contiguous"): - xp.petsc_vec(np.zeros((4, 4))[:, 0]) + xp.petsc.petsc_vec(np.zeros((4, 4))[:, 0]) with pytest.raises(TypeError, match="NumPy or CuPy array"): - xp.petsc_vec([0.0, 1.0]) + xp.petsc.petsc_vec([0.0, 1.0]) class FakeDeviceArray: @@ -113,21 +114,21 @@ def setAttr(self, name, value): def test_device_arrays_give_device_vectors(monkeypatch, vec_type): _fake_petsc(monkeypatch, vec_type) array = FakeDeviceArray() - vec = xp.petsc_vec(array) + vec = xp.petsc.petsc_vec(array) assert vec.attr == ("cunumpy_array", array) def test_device_arrays_need_a_gpu_petsc(monkeypatch): Vec = _fake_petsc(monkeypatch, "seq") # PETSc without CUDA made a host vector with pytest.raises(RuntimeError, match="created a 'seq' vector for a CuPy array"): - xp.petsc_vec(FakeDeviceArray()) + xp.petsc.petsc_vec(FakeDeviceArray()) assert Vec.destroyed _fake_petsc(monkeypatch, error=True) with pytest.raises(RuntimeError, match="built with CUDA or HIP support"): - xp.petsc_vec(FakeDeviceArray()) + xp.petsc.petsc_vec(FakeDeviceArray()) def test_missing_petsc4py(monkeypatch): monkeypatch.setitem(sys.modules, "petsc4py", None) with pytest.raises(ImportError, match="needs petsc4py"): - xp.petsc_vec(np.zeros(3)) + xp.petsc.petsc_vec(np.zeros(3)) diff --git a/tests/unit/test_philox.py b/tests/unit/test_philox.py index 04c1ff7..9a91fba 100644 --- a/tests/unit/test_philox.py +++ b/tests/unit/test_philox.py @@ -6,8 +6,8 @@ import pytest import cunumpy as xp -from cunumpy import CudaKernel, cuda_include_dir -from cunumpy.testing import emulate_cuda_kernel, emulation_compiler +from cunumpy.cuda import CudaKernel, cuda_include_dir +from cunumpy.kernel_testing import emulate_cuda_kernel, emulation_compiler # Known-answer vectors of Philox4x32-10 (Random123, kat_vectors) KAT = [ @@ -27,7 +27,7 @@ @pytest.mark.parametrize(("counter", "key", "expected"), KAT) def test_known_answers(counter, key, expected): - out = xp.philox4x32_10(np.array(counter, dtype=np.uint32), *key) + out = xp.rng.philox4x32_10(np.array(counter, dtype=np.uint32), *key) assert out.dtype == np.uint32 assert out.tolist() == list(expected) @@ -35,41 +35,41 @@ def test_known_answers(counter, key, expected): def test_vectorized_over_counters(): counters = np.array([k[0] for k in KAT], dtype=np.uint32) keys = np.array([k[1] for k in KAT], dtype=np.uint32) - out = xp.philox4x32_10(counters, keys[:, 0], keys[:, 1]) + out = xp.rng.philox4x32_10(counters, keys[:, 0], keys[:, 1]) assert out.tolist() == [list(k[2]) for k in KAT] def test_uniforms_and_normals(): ids = np.arange(200_000, dtype=np.uint64) - u0, u1 = xp.philox_uniform2(42, ids, 7) + u0, u1 = xp.rng.philox_uniform2(42, ids, 7) assert u0.shape == u1.shape == (200_000,) assert u0.min() >= 0.0 and u0.max() < 1.0 assert abs(u0.mean() - 0.5) < 3e-3 and abs(u1.var() - 1 / 12) < 1e-3 assert abs(np.corrcoef(u0, u1)[0, 1]) < 1e-2 - np.testing.assert_array_equal(xp.philox_uniform(42, ids, 7), u0) - z0, z1 = xp.philox_normal2(42, ids, 7) + np.testing.assert_array_equal(xp.rng.philox_uniform(42, ids, 7), u0) + z0, z1 = xp.rng.philox_normal2(42, ids, 7) assert abs(z0.mean()) < 1e-2 and abs(z1.std() - 1.0) < 1e-2 - np.testing.assert_array_equal(xp.philox_normal(42, ids, 7), z0) + np.testing.assert_array_equal(xp.rng.philox_normal(42, ids, 7), z0) def test_streams_counters_and_seeds_differ(): - base = xp.philox_uniform(1, 5, 9) - assert xp.philox_uniform(1, 5, 9) == base # no state - assert xp.philox_uniform(1, 6, 9) != base - assert xp.philox_uniform(1, 5, 10) != base - assert xp.philox_uniform(2, 5, 9) != base + base = xp.rng.philox_uniform(1, 5, 9) + assert xp.rng.philox_uniform(1, 5, 9) == base # no state + assert xp.rng.philox_uniform(1, 6, 9) != base + assert xp.rng.philox_uniform(1, 5, 10) != base + assert xp.rng.philox_uniform(2, 5, 9) != base # 64-bit stream and counter: the high words matter - assert xp.philox_uniform(1, 5 + 2**32, 9) != base - assert xp.philox_uniform(1, 5, 9 + 2**32) != base - assert xp.philox_uniform(1 + 2**32, 5, 9) != base + assert xp.rng.philox_uniform(1, 5 + 2**32, 9) != base + assert xp.rng.philox_uniform(1, 5, 9 + 2**32) != base + assert xp.rng.philox_uniform(1 + 2**32, 5, 9) != base def test_broadcasting(): - u = xp.philox_uniform( + u = xp.rng.philox_uniform( np.uint64(3), np.arange(4, dtype=np.uint64)[:, None], np.arange(5) ) assert u.shape == (4, 5) - assert u[2, 3] == xp.philox_uniform(3, 2, 3) + assert u[2, 3] == xp.rng.philox_uniform(3, 2, 3) def test_header_is_shipped(): @@ -96,8 +96,8 @@ def _device_samples(run, n=1000, seed=2**40 + 17, counter=2**33 + 5): run(CudaKernel(SAMPLE, "sample"), u0, u1, z0, z1, n, seed, counter, n_threads=n) return ( (u0, u1, z0, z1), - xp.philox_uniform2(seed, np.arange(n, dtype=np.uint64), counter), - xp.philox_normal2(seed, np.arange(n, dtype=np.uint64), counter), + xp.rng.philox_uniform2(seed, np.arange(n, dtype=np.uint64), counter), + xp.rng.philox_normal2(seed, np.arange(n, dtype=np.uint64), counter), ) @@ -128,8 +128,8 @@ def run(kernel, *args, n_threads): np.testing.assert_allclose(z0, n0, rtol=1e-13, atol=1e-13) np.testing.assert_allclose(z1, n1, rtol=1e-13, atol=1e-13) # the device-side generator on device arrays, too - du0, _ = xp.philox_uniform2(7, cp.arange(10, dtype=cp.uint64), 1) + du0, _ = xp.rng.philox_uniform2(7, cp.arange(10, dtype=cp.uint64), 1) assert isinstance(du0, cp.ndarray) np.testing.assert_array_equal( - du0.get(), xp.philox_uniform2(7, np.arange(10, dtype=np.uint64), 1)[0] + du0.get(), xp.rng.philox_uniform2(7, np.arange(10, dtype=np.uint64), 1)[0] ) diff --git a/tests/unit/test_porting_helpers.py b/tests/unit/test_porting_helpers.py index a51f592..098a43e 100644 --- a/tests/unit/test_porting_helpers.py +++ b/tests/unit/test_porting_helpers.py @@ -23,16 +23,11 @@ import cunumpy as xp import cunumpy.cuda_kernel as cuda_kernel_module -import cunumpy.testing -from cunumpy import ( - CudaKernel, - CudaStruct, - Kernel, - KernelCatalog, - PyccelStructArguments, -) +import cunumpy.kernel_testing +from cunumpy.cuda import CudaKernel, CudaStruct from cunumpy.dispatch import FORTRAN_NAME_LIMIT, _pyccel_stub_parameters -from cunumpy.testing import check_parity, device_function_kernel, parity_cases +from cunumpy.kernel_testing import check_parity, device_function_kernel, parity_cases +from cunumpy.kernels import Kernel, KernelCatalog, PyccelStructArguments class FakeDeviceArray: @@ -378,7 +373,7 @@ def fake_agree(kernel, make_args, **settings): calls["kernel"], calls["settings"] = kernel, settings return {"argument 0": np.zeros(1)} - monkeypatch.setattr(cunumpy.testing, "assert_kernels_agree", fake_agree) + monkeypatch.setattr(cunumpy.kernel_testing, "assert_kernels_agree", fake_agree) check_parity(catalog["scale"], atol=1e-14) assert calls["kernel"] is catalog["scale"] assert calls["settings"] == {"n_threads": 300, "rtol": 1e-10, "atol": 1e-14} @@ -486,34 +481,42 @@ def test_device_function_kernel_struct_parameters(): def test_segment_sum(): keys = np.array([0, 2, 0, -1, 2]) values = np.array([1.0, 2.0, 3.0, 100.0, 4.0]) - np.testing.assert_array_equal(xp.segment_sum(values, keys, 4), [4.0, 0.0, 6.0, 0.0]) + np.testing.assert_array_equal( + xp.algorithms.segment_sum(values, keys, 4), [4.0, 0.0, 6.0, 0.0] + ) columns = np.stack([values, -values], axis=1) - out = xp.segment_sum(columns, keys, 3) + out = xp.algorithms.segment_sum(columns, keys, 3) np.testing.assert_array_equal(out, [[4.0, -4.0], [0.0, 0.0], [6.0, -6.0]]) - assert xp.segment_sum(np.array([1, 2]), np.array([1, 1]), 2).dtype == np.float64 - assert xp.segment_sum(values.astype(np.float32), keys, 3).dtype == np.float32 - complex_sum = xp.segment_sum(values * (1 + 1j), keys, 3) + assert ( + xp.algorithms.segment_sum(np.array([1, 2]), np.array([1, 1]), 2).dtype + == np.float64 + ) + assert ( + xp.algorithms.segment_sum(values.astype(np.float32), keys, 3).dtype + == np.float32 + ) + complex_sum = xp.algorithms.segment_sum(values * (1 + 1j), keys, 3) np.testing.assert_allclose(complex_sum, [4 + 4j, 0, 6 + 6j]) with pytest.raises(ValueError, match="smaller than n_segments"): - xp.segment_sum(values, keys, 2) + xp.algorithms.segment_sum(values, keys, 2) with pytest.raises(ValueError, match="one entry per value"): - xp.segment_sum(values, keys[:2], 3) + xp.algorithms.segment_sum(values, keys[:2], 3) def test_mpi_buffer_on_host_arrays(): x = np.arange(3.0) - with xp.mpi_buffer(x) as buf: + with xp.mpi.mpi_buffer(x) as buf: assert buf is x - with xp.mpi_buffer(x, send=False, recv=True) as buf: + with xp.mpi.mpi_buffer(x, send=False, recv=True) as buf: assert buf is x def test_mpi_cuda_aware_setting(): - xp.set_mpi_cuda_aware(None) - assert xp.get_mpi_cuda_aware() is None - xp.set_mpi_cuda_aware(True) - assert xp.get_mpi_cuda_aware() is True - xp.set_mpi_cuda_aware(None) + xp.mpi.set_mpi_cuda_aware(None) + assert xp.mpi.get_mpi_cuda_aware() is None + xp.mpi.set_mpi_cuda_aware(True) + assert xp.mpi.get_mpi_cuda_aware() is True + xp.mpi.set_mpi_cuda_aware(None) def test_require_version(monkeypatch): @@ -535,8 +538,9 @@ def test_require_version(monkeypatch): import numpy as np import pytest import cunumpy as xp -import cunumpy.testing as testing -from cunumpy import CudaKernel, CudaStruct, KernelArguments +import cunumpy.kernel_testing as testing +from cunumpy.cuda import CudaKernel, CudaStruct +from cunumpy.kernels import KernelArguments assert testing.fake_cupy_active() assert xp.cupy_available() and xp.get_backend() == "cupy", xp.get_backend() @@ -570,15 +574,15 @@ def test_require_version(monkeypatch): d = xp.as_device_array([1.0, 2.0], dtype=np.float64) assert xp.is_gpu(d) with pytest.raises(RuntimeError, match="not known whether MPI"): - with xp.mpi_buffer(d): + with xp.mpi.mpi_buffer(d): pass -with xp.count_transfers() as counter: - with xp.mpi_buffer(d, recv=True, cuda_aware=False) as buf: +with xp.profiling.count_transfers() as counter: + with xp.mpi.mpi_buffer(d, recv=True, cuda_aware=False) as buf: assert isinstance(buf, np.ndarray) and buf.tolist() == [1.0, 2.0] buf[:] = [5.0, 6.0] assert xp.to_numpy(d).tolist() == [5.0, 6.0] assert sorted(e.kind for e in counter.events) == ["to_device", "to_host"] -with xp.mpi_buffer(d, cuda_aware=True) as buf: +with xp.mpi.mpi_buffer(d, cuda_aware=True) as buf: assert buf is d print("fake cupy OK") """ @@ -606,5 +610,5 @@ def test_fake_cupy_in_subprocess(): def test_install_fake_cupy_refuses_a_real_cupy(monkeypatch): monkeypatch.setitem(sys.modules, "cupy", ModuleType("cupy")) with pytest.raises(RuntimeError, match="real CuPy is already imported"): - cunumpy.testing.install_fake_cupy() - assert not cunumpy.testing.fake_cupy_active() + cunumpy.kernel_testing.install_fake_cupy() + assert not cunumpy.kernel_testing.fake_cupy_active() diff --git a/tests/unit/test_profiling.py b/tests/unit/test_profiling.py index 6d01102..54029a9 100644 --- a/tests/unit/test_profiling.py +++ b/tests/unit/test_profiling.py @@ -43,14 +43,14 @@ def fake_nvtx(monkeypatch): def test_nvtx_range_is_noop_on_numpy(): with xp.use_backend("numpy"): - with xp.nvtx_range("region") as r: + with xp.profiling.nvtx_range("region") as r: assert r.name == "region" - with xp.nvtx_range("colored", color=3): + with xp.profiling.nvtx_range("colored", color=3): pass def test_nvtx_range_as_decorator(): - @xp.nvtx_range("decorated") + @xp.profiling.nvtx_range("decorated") def add(a, b): return a + b @@ -60,23 +60,28 @@ def add(a, b): def test_nvtx_range_repr(): - assert repr(xp.nvtx_range("r", color=1)) == "nvtx_range(name='r', color=1)" + assert ( + repr(xp.profiling.nvtx_range("r", color=1)) == "nvtx_range(name='r', color=1)" + ) def test_timed_region_on_numpy(): - with xp.use_backend("numpy"), xp.timed_region("sleep") as timing: + with xp.use_backend("numpy"), xp.profiling.timed_region("sleep") as timing: assert timing.name == "sleep" assert timing.elapsed is None time.sleep(0.02) - assert isinstance(timing, xp.Timing) + assert isinstance(timing, xp.profiling.Timing) assert timing.elapsed >= 0.02 assert timing.elapsed < 5.0 assert timing.synced is False def test_timed_region_without_sync_on_numpy(): - with xp.use_backend("numpy"), xp.timed_region("no sync", sync=False) as timing: + with ( + xp.use_backend("numpy"), + xp.profiling.timed_region("no sync", sync=False) as timing, + ): pass assert timing.elapsed >= 0.0 assert timing.synced is False @@ -86,7 +91,7 @@ def test_timed_region_records_time_on_exception(): with ( xp.use_backend("numpy"), pytest.raises(RuntimeError, match="boom"), - xp.timed_region("failing") as timing, + xp.profiling.timed_region("failing") as timing, ): raise RuntimeError("boom") assert timing.elapsed is not None @@ -97,21 +102,21 @@ def test_timed_region_records_time_on_exception(): def test_nvtx_range_pushes_and_pops(fake_nvtx): - with xp.nvtx_range("outer"): + with xp.profiling.nvtx_range("outer"): fake_nvtx.append(("body",)) assert fake_nvtx == [("push", "outer", -1), ("body",), ("pop",)] def test_nvtx_range_color(fake_nvtx): - with xp.nvtx_range("colored", color=5): + with xp.profiling.nvtx_range("colored", color=5): pass assert fake_nvtx == [("push", "colored", 5), ("pop",)] def test_nvtx_range_nested_and_reentrant(fake_nvtx): - outer = xp.nvtx_range("outer") + outer = xp.profiling.nvtx_range("outer") with outer: - with xp.nvtx_range("inner"): + with xp.profiling.nvtx_range("inner"): pass with outer: # same instance re-entered pass @@ -126,13 +131,13 @@ def test_nvtx_range_nested_and_reentrant(fake_nvtx): def test_nvtx_range_pops_on_exception(fake_nvtx): - with pytest.raises(ValueError), xp.nvtx_range("failing"): + with pytest.raises(ValueError), xp.profiling.nvtx_range("failing"): raise ValueError assert fake_nvtx == [("push", "failing", -1), ("pop",)] def test_nvtx_range_decorator_pushes_and_pops(fake_nvtx): - @xp.nvtx_range("decorated") + @xp.profiling.nvtx_range("decorated") def work(): fake_nvtx.append(("body",)) @@ -142,14 +147,14 @@ def work(): def test_timed_region_pushes_nvtx_range_and_syncs(fake_nvtx): - with xp.timed_region("timed") as timing: + with xp.profiling.timed_region("timed") as timing: fake_nvtx.append(("body",)) assert fake_nvtx == [("push", "timed", -1), ("body",), ("pop",)] assert timing.synced is True assert timing.elapsed >= 0.0 del fake_nvtx[:] - with xp.timed_region("host only", sync=False) as timing: + with xp.profiling.timed_region("host only", sync=False) as timing: pass assert fake_nvtx == [("push", "host only", -1), ("pop",)] assert timing.synced is False @@ -159,7 +164,7 @@ def test_nvtx_range_without_nvtx_module(monkeypatch): # CuPy backend but no NVTX (e.g. a build without it): still a no-op monkeypatch.setitem(sys.modules, "cupy.cuda.nvtx", None) # import fails monkeypatch.setattr(xp_module.array_backend, "_backend", "cupy") - with xp.nvtx_range("no nvtx"): + with xp.profiling.nvtx_range("no nvtx"): pass @@ -171,7 +176,7 @@ def test_nvtx_range_on_gpu(): import cupy as cp with xp.use_backend("cupy"): - with xp.nvtx_range("gpu region", color=2): + with xp.profiling.nvtx_range("gpu region", color=2): x = cp.ones(1000) x += 1 xp.synchronize() @@ -198,7 +203,7 @@ def work(): xp.synchronize() reference = time.perf_counter() - start - with xp.timed_region("work") as timing: + with xp.profiling.timed_region("work") as timing: work() assert timing.synced is True assert cp.cuda.get_current_stream().done diff --git a/tests/unit/test_pyccel_kernel.py b/tests/unit/test_pyccel_kernel.py index e25abc5..8fd39f7 100644 --- a/tests/unit/test_pyccel_kernel.py +++ b/tests/unit/test_pyccel_kernel.py @@ -1,4 +1,4 @@ -"""Tests for `cunumpy.PyccelKernel`. +"""Tests for `cunumpy.kernels.PyccelKernel`. Two groups of tests live here: @@ -19,7 +19,7 @@ import pytest import cunumpy as xp -from cunumpy import PyccelKernel +from cunumpy.kernels import PyccelKernel KERNEL_SOURCE = Path(__file__).parent / "pyccel_kernels.py" diff --git a/tests/unit/test_random_streams.py b/tests/unit/test_random_streams.py index 396d9bb..47fdeb5 100644 --- a/tests/unit/test_random_streams.py +++ b/tests/unit/test_random_streams.py @@ -1,11 +1,10 @@ -"""Tests for `xp.random_streams`: one seeded generator per process and backend.""" +"""Tests for `xp.rng.random_streams`: one seeded generator per process and backend.""" import numpy as np import pytest import cunumpy as xp -from cunumpy import RandomStreams, random_streams -from cunumpy.random_streams import BIT_GENERATORS +from cunumpy.rng import BIT_GENERATORS, RandomStreams, random_streams @pytest.fixture @@ -24,7 +23,7 @@ def _draws(streams): def test_exported(): assert isinstance(random_streams, RandomStreams) - assert "random_streams" in xp.__all__ + assert "random_streams" in xp.rng.__all__ assert repr(RandomStreams()) == "RandomStreams(not seeded)" diff --git a/tests/unit/test_staging.py b/tests/unit/test_staging.py index fe982ad..11aee97 100644 --- a/tests/unit/test_staging.py +++ b/tests/unit/test_staging.py @@ -1,4 +1,4 @@ -"""Tests for `xp.HostStaging`: background copies of device arrays to the host.""" +"""Tests for `xp.memory.HostStaging`: background copies of device arrays to the host.""" import types @@ -6,8 +6,8 @@ import pytest import cunumpy as xp -from cunumpy import HostStaging from cunumpy import staging as staging_module +from cunumpy.memory import HostStaging def test_host_arrays_are_copied_at_once(): @@ -138,7 +138,7 @@ def test_a_buffer_is_reused_only_after_its_copy_finished(fake_device): def test_copies_are_counted_as_transfers(fake_device): staging = HostStaging((2,), np.float64) - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: staging.copy(np.zeros(2).view(DeviceArray)) assert counter.total == 1 diff --git a/tests/unit/test_transfers.py b/tests/unit/test_transfers.py index 00cb46c..bb3f3c6 100644 --- a/tests/unit/test_transfers.py +++ b/tests/unit/test_transfers.py @@ -1,4 +1,4 @@ -"""Tests for `cunumpy.count_transfers` and `cunumpy.assert_no_transfers`. +"""Tests for `cunumpy.profiling.count_transfers` and `cunumpy.profiling.assert_no_transfers`. Without a GPU there are no real device arrays, so most tests simulate one: either by monkeypatching `cunumpy.xp.get_array_backend` (for `to_numpy` and @@ -13,11 +13,12 @@ import pytest import cunumpy as xp -from cunumpy import Kernel, PyccelKernel, TransferCounter, TransferEvent from cunumpy import dispatch as dispatch_module from cunumpy import kernel as kernel_module from cunumpy import transfers as transfers_module from cunumpy import xp as xp_module +from cunumpy.kernels import Kernel, PyccelKernel +from cunumpy.profiling import TransferCounter, TransferEvent THIS_FILE = str(Path(__file__)) @@ -70,7 +71,7 @@ def get_array_backend(array): def test_empty_counter(): - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: pass assert isinstance(counter, TransferCounter) @@ -85,7 +86,7 @@ def test_empty_counter(): def test_to_numpy_of_numpy_array_is_not_a_transfer(): - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: xp.to_numpy(np.zeros(3)) xp.to_numpy([1, 2, 3]) with xp.use_backend("numpy"): @@ -99,13 +100,13 @@ def test_no_counter_active_records_nothing(fake_device): xp.to_numpy(fake_device(np.zeros(3))) assert transfers_module._ACTIVE == [] - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: pass assert counter.total == 0 def test_to_numpy_of_device_array_is_counted(fake_device): - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: host = xp.to_numpy(fake_device(np.arange(3.0))) assert np.array_equal(host, np.arange(3.0)) @@ -117,7 +118,7 @@ def test_to_numpy_of_device_array_is_counted(fake_device): def test_to_cupy_of_host_array_is_counted(fake_device): - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: xp.to_cupy(np.zeros((2, 2))) xp.to_cupy([1, 2]) @@ -127,7 +128,7 @@ def test_to_cupy_of_host_array_is_counted(fake_device): def test_to_cupy_of_device_array_is_not_counted(fake_device): - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: xp.to_cupy(fake_device(np.zeros(2))) assert counter.total == 0 @@ -135,7 +136,7 @@ def test_to_cupy_of_device_array_is_not_counted(fake_device): def test_to_cunumpy_counts_the_direction_it_delegates_to(fake_device, monkeypatch): device = fake_device(np.zeros(2)) - with xp.count_transfers() as counter, xp.use_backend("numpy"): + with xp.profiling.count_transfers() as counter, xp.use_backend("numpy"): xp.to_cunumpy(device) # device -> host xp.to_cunumpy(np.zeros(2)) # already on the host @@ -143,7 +144,7 @@ def test_to_cunumpy_counts_the_direction_it_delegates_to(fake_device, monkeypatc monkeypatch.setattr(xp_module.array_backend, "_backend", "cupy") monkeypatch.setattr(xp_module, "cupy_available", lambda: True) - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: xp.to_cunumpy(np.zeros(2)) # host -> device xp.to_cunumpy(device) # already on the device @@ -156,7 +157,7 @@ def test_to_cunumpy_counts_the_direction_it_delegates_to(fake_device, monkeypatc def test_where_points_at_the_caller_outside_cunumpy(fake_device): - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: xp.to_numpy(fake_device(np.zeros(1))) line = _current_line() - 1 @@ -167,7 +168,7 @@ def test_where_points_at_the_caller_outside_cunumpy(fake_device): def test_where_skips_frames_inside_cunumpy(fake_device): """`to_cunumpy` calls `to_numpy`; the call site is still the test.""" - with xp.count_transfers() as counter, xp.use_backend("numpy"): + with xp.profiling.count_transfers() as counter, xp.use_backend("numpy"): xp.to_cunumpy(fake_device(np.zeros(1))) (event,) = counter.events @@ -176,7 +177,7 @@ def test_where_skips_frames_inside_cunumpy(fake_device): def test_report_groups_events_by_kind_and_call_site(fake_device): device = fake_device(np.zeros(4)) - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: for _ in range(3): xp.to_numpy(device) xp.to_cupy(np.zeros(4)) @@ -209,9 +210,9 @@ def test_report_groups_events_by_kind_and_call_site(fake_device): def test_nested_counters_each_see_their_own_block(fake_device): device = fake_device(np.zeros(1)) - with xp.count_transfers() as outer: + with xp.profiling.count_transfers() as outer: xp.to_numpy(device) - with xp.count_transfers() as inner: + with xp.profiling.count_transfers() as inner: xp.to_numpy(device) xp.to_numpy(device) @@ -222,21 +223,21 @@ def test_nested_counters_each_see_their_own_block(fake_device): def test_counter_is_removed_when_the_block_raises(fake_device): - with pytest.raises(RuntimeError), xp.count_transfers(): + with pytest.raises(RuntimeError), xp.profiling.count_transfers(): raise RuntimeError assert transfers_module._ACTIVE == [] def test_assert_no_transfers_passes_without_transfers(): - with xp.assert_no_transfers() as counter: + with xp.profiling.assert_no_transfers() as counter: xp.to_numpy(np.zeros(3)) assert counter.total == 0 def test_assert_no_transfers_raises_with_report(fake_device): - with pytest.raises(AssertionError) as info, xp.assert_no_transfers(): + with pytest.raises(AssertionError) as info, xp.profiling.assert_no_transfers(): xp.to_numpy(fake_device(np.zeros(3))) message = str(info.value) @@ -247,7 +248,7 @@ def test_assert_no_transfers_raises_with_report(fake_device): def test_assert_no_transfers_lets_exceptions_through(fake_device): - with pytest.raises(ValueError, match="inside"), xp.assert_no_transfers(): + with pytest.raises(ValueError, match="inside"), xp.profiling.assert_no_transfers(): xp.to_numpy(fake_device(np.zeros(3))) raise ValueError("inside") @@ -266,7 +267,7 @@ def scale(x, y, factor): y = fake_device(np.ones(3)) wrapped = PyccelKernel(scale, use_cupy=True) - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: wrapped(x, y, 2.0) line = _current_line() - 1 wrapped(x, x, 3.0) # one array passed twice: converted once @@ -289,7 +290,7 @@ def scale(x, y, factor): def test_pyccel_kernel_without_device_arrays_is_not_counted(fake_device): wrapped = PyccelKernel(lambda x: None, use_cupy=True) - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: wrapped(np.zeros(3)) # conversion path, but nothing to convert with xp.use_backend("numpy"): PyccelKernel(lambda x: None)(np.zeros(3)) @@ -309,7 +310,7 @@ def shift(x, n): kernel = Kernel(shift, missing_cuda="fallback") x = np.zeros(2) - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: line = _current_line() + 2 with pytest.warns(RuntimeWarning, match="copies its arrays"): kernel(x, 2) @@ -331,7 +332,7 @@ def shift(x, n): x[:n] += 1.0 kernel = Kernel(shift, missing_cuda="fallback") - with xp.count_transfers() as counter, xp.use_backend("numpy"): + with xp.profiling.count_transfers() as counter, xp.use_backend("numpy"): kernel(np.zeros(2), 2) assert counter.total == 0 @@ -346,7 +347,7 @@ def shift(x, n): def test_real_transfers_are_counted(): import cupy as cp - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: device = xp.to_cupy(np.arange(3.0)) xp.to_cupy(device) # already on the device host = xp.to_numpy(device) @@ -369,7 +370,7 @@ def scale(x, factor): x[:] *= factor x = cp.ones(4) - with xp.count_transfers() as counter: + with xp.profiling.count_transfers() as counter: PyccelKernel(scale)(x, 2.0) assert cp.all(x == 2.0) @@ -390,7 +391,7 @@ def scale(x, factor, n): kernel = Kernel(scale, missing_cuda="fallback") x = cp.ones(3) with ( - xp.count_transfers() as counter, + xp.profiling.count_transfers() as counter, xp.use_backend("cupy"), pytest.warns(RuntimeWarning), ): @@ -404,7 +405,7 @@ def scale(x, factor, n): @requires_cupy def test_assert_no_transfers_on_device_only_work(): - with xp.use_backend("cupy"), xp.assert_no_transfers(): + with xp.use_backend("cupy"), xp.profiling.assert_no_transfers(): values = xp.arange(10, dtype=xp.float64) values = values * 2 + 1 xp.synchronize() From a1f2bcdd6e0a6d85daebd1cb5148ff94e97d617c Mon Sep 17 00:00:00 2001 From: Max Date: Sat, 3 Oct 2026 15:10:01 +0200 Subject: [PATCH 2/2] formatting --- tests/unit/test_kernel_testing.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/unit/test_kernel_testing.py b/tests/unit/test_kernel_testing.py index cd2cf39..4451a03 100644 --- a/tests/unit/test_kernel_testing.py +++ b/tests/unit/test_kernel_testing.py @@ -16,12 +16,12 @@ import cunumpy as xp import cunumpy.kernel_testing from cunumpy.cuda import CudaArguments, CudaKernel, parse_cuda_signature +from cunumpy.kernel_testing import backend # noqa: F401 - the fixture is used by name from cunumpy.kernel_testing import ( BACKENDS, _collect_arrays, _compare_results, assert_kernels_agree, - backend, # noqa: F401 - the fixture is used by name device_function_kernel, requires_cupy, )