From 1e2709dba3f67330fe0f19b3a92e2cc382827970 Mon Sep 17 00:00:00 2001 From: Max Date: Sat, 3 Oct 2026 12:52:21 +0200 Subject: [PATCH 1/3] Build Kernel from folder --- CHANGELOG.md | 5 + docs/source/api.md | 53 ++++- docs/source/kernels/dispatch.md | 57 +++++- src/cunumpy/LLM_GUIDE.md | 5 + src/cunumpy/__init__.py | 13 +- src/cunumpy/__init__.pyi | 4 + src/cunumpy/dispatch.py | 235 ++++++++++++++++------ src/cunumpy/kernel.py | 52 ++++- src/cunumpy/xp.py | 45 +++++ tests/unit/test_kernel_dispatch_arrays.py | 190 +++++++++++++++++ 10 files changed, 590 insertions(+), 69 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f41ad41..9bb227f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -15,6 +15,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Support for Python 3.8 and 3.9 (both end-of-life); `cunumpy` now requires Python 3.10 or newer. ### Changed +- `Kernel(..., dispatch="arrays")` calls the host kernel directly for host arguments, without `PyccelKernel`'s check for device arrays to convert, which converted nothing but ran while CuPy was the active backend (unless the `PyccelKernel` was built with `use_cupy=True`). +- `Kernel.check_signature()` (and `KernelCatalog.check_signatures()`) also compares the parameters of a `CompiledHostKernel`'s fallback with those of the host kernel. - `CudaKernel` pointer and array view parameters and `CudaStruct` pointer and view fields reject an array that lives on another CUDA device than the current one (`ValueError`; `cupy.RawKernel` would read the foreign address silently). Checked only when CuPy is imported and the array reports a `device`. - `assert_kernels_agree(..., n_threads=...)` also takes a function of the argument tuple, and `n_threads` may be omitted when the CUDA kernel has `n_threads_from`. - NumPy integer scalars passed to integer kernel parameters (and struct fields) are checked by value, like Python ints: `np.int64(5)` fits an `int` parameter; before, they raised `TypeError` unless their dtype was exactly the parameter's. @@ -26,6 +28,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - `KernelCatalog.from_package(..., include_dirs=None)` is now an explicit keyword; by default the source root of the top-level package (the directory containing it) is an include directory of every CUDA kernel, in addition to the kernel's own folder, so kernels can `#include "my_pkg/common.cuh"`. ### Added +- `Kernel.from_folder(package, ...)`: the kernel of one kernel folder, with the options of `KernelCatalog.from_package` (`host_suffix`, `compile_host`, `fallback`, `dispatch`, `include_dirs`, CUDA options...). A folder's own `__init__.py` can declare `kernel = xp.Kernel.from_folder(__name__, ...)`, so code imports the kernel from where it is written; `from_package` now builds each kernel with it. +- `xp.as_kernel_array(value, like, dtype=None)` and `xp.kernel_output(out, like, dtype=None)`: bring the arguments of a `dispatch="arrays"` kernel to the side of the main array `like` (CuPy or NumPy, C-contiguous, `dtype`), without a copy when they already fit; `kernel_output` yields the buffer the kernel writes and copies it back into `out` if it had to be converted. +- `xp.force_host_fallback(enabled=True)` and the environment variable `CUNUMPY_HOST_FALLBACK=1`: every `CompiledHostKernel` runs its fallback as if compilation had failed (`compiled` is False), to test the code path of a machine without the compiler. `CompiledHostKernel.fallback` returns the fallback. - `cunumpy/random.cuh` and `xp.philox_uniform`, `philox_uniform2`, `philox_normal`, `philox_normal2`, `philox4x32_10`: counter-based random numbers (Philox4x32-10, passing the Random123 known-answer tests) as a pure function of `(seed, stream, counter)`, the same in a kernel and on the host (uniform numbers bit for bit, normal numbers up to the last bits of the math functions), so kernels that draw random numbers can be compared with their host versions. - `CudaKernel(..., n_threads_from="first_array")`: one thread per row of the first array argument when a launch gives neither `n_threads` nor `grid`. - `CudaKernel` launches with `shared_mem` above 48 KiB set the kernel's `max_dynamic_shared_size_bytes` once, up to the device's opt-in limit, and raise `ValueError` beyond it. diff --git a/docs/source/api.md b/docs/source/api.md index 06dee24..cabecf5 100644 --- a/docs/source/api.md +++ b/docs/source/api.md @@ -288,6 +288,23 @@ class DeviceParticles(xp.CudaArguments): super().__init__(self.markers, self.degree, self.markers.shape[0]) ``` +### `as_kernel_array(value, like, dtype=None)`, `kernel_output(out, like, dtype=None)` + +For the arguments of a `Kernel` with `dispatch="arrays"`, whose choice follows +the arrays. `as_kernel_array` returns `value` on the side of `like` (a CuPy +array if `like` is one, a NumPy array otherwise), C-contiguous and with `dtype` +(any if `None`): `value` itself if it already is such an array, else one copy, +moved across if needed (counted by `count_transfers()`). `kernel_output` is a +context manager yielding the buffer for an array the kernel writes: `out` +itself if `as_kernel_array` takes it unchanged, else a converted copy whose +contents are written into `out` (on its own side) when the block ends without +an error. + +```python +with xp.kernel_output(result, like=field, dtype=float) as buffer: + gather(xp.as_kernel_array(positions, like=field, dtype=float), field, buffer) +``` + ## Random numbers and dtype ### `get_rng(seed=None)` @@ -1350,11 +1367,12 @@ array, or a device-only argument object: one with `__cuda_args__()` but no `__host_args__()`, such as a `CudaArguments` or a struct value), else the host kernel, whatever the backend. Use `"arrays"` in codes that hand host arrays to kernels while CuPy is active (diagnostics, MPI staging, CPU fallbacks): those -calls then run the host kernel instead of failing in the CUDA argument checks. +calls then run the host kernel instead of failing in the CUDA argument checks, +calling the host function directly (no device-array conversion). `missing_cuda` applies to device arguments without a CUDA kernel. -`kernel.check_signature()` checks that the host and CUDA kernels take the same -parameters in the same order (the names of the Python host function, or of the +`kernel.check_signature()` checks that the host and CUDA kernels (and the +fallback of a `CompiledHostKernel`) take the same parameters in the same order (the names of the Python host function, or of the uncompiled Python version of a `CompiledHostKernel`, against the parsed `__global__` signature) and raises `ValueError` showing both lists otherwise. It does nothing without a CUDA kernel, with `check_signature=False`, or when @@ -1471,6 +1489,24 @@ that have a CUDA kernel, for a parametrised parity test (see "Testing utilities"). `KernelCatalog(kernels)` and `catalog.register(kernel, name=None)` build a catalog by hand. +## `Kernel.from_folder` + +```python +# my_sim/kernels/push/__init__.py +kernel = xp.Kernel.from_folder(__name__, host_suffix="_pyccel", dispatch="arrays", + compile_host=compile_kernels, fallback=push_numpy) +``` + +The kernel of one kernel folder `package` (its dotted name, `__name__` in its +`__init__.py`): the function `` of `.py` is the host +kernel, `` in the folder, if present, the CUDA kernel. Takes +the options of `KernelCatalog.from_package` for one kernel: `host_suffix`, +`cuda_suffix`, `test_args_suffix`, `check_name_length`, `missing_cuda`, +`host_options`, `include_dirs`, `dispatch`, `compile_host`, `fallback` (one +callable, used with `compile_host`) and CUDA options such as `block_size` or +`n_threads_from`. Raises `FileNotFoundError` if the folder has no host kernel +module and `ModuleNotFoundError` if `package` is not a package. + ## `CompiledHostKernel` ```python @@ -1487,7 +1523,16 @@ on-disk cache, or one that imports modules compiled ahead of time with the the uncompiled Python function with a `RuntimeWarning`. `kernel.compiled` builds and reports whether that worked (so that callers can choose another path), `kernel.error` is the exception of a failed build, `kernel.python` the -uncompiled function and `kernel.build()` compiles now. +uncompiled function, `kernel.fallback` the fallback and `kernel.build()` +compiles now. + +`with xp.force_host_fallback():` makes every `CompiledHostKernel` run its +fallback as if compilation had failed (`compiled` is False inside the block; +already compiled versions are kept for after it), to test the path of a +machine without the compiler; `force_host_fallback(False)` switches it off +inside such a block. The environment variable `CUNUMPY_HOST_FALLBACK=1`, read +when cunumpy is imported, does the same for a whole run. The switch is global, +not per thread. ## Testing utilities diff --git a/docs/source/kernels/dispatch.md b/docs/source/kernels/dispatch.md index 5ea0151..a767aad 100644 --- a/docs/source/kernels/dispatch.md +++ b/docs/source/kernels/dispatch.md @@ -123,6 +123,39 @@ A catalog is a read-only mapping: `catalog["push"]`, `"push" in catalog`, `len(catalog)`, iteration over names. `KernelCatalog()` plus `catalog.register(kernel)` builds one by hand. +### One folder, declared in its own `__init__.py` + +`Kernel.from_folder()` builds the kernel of one folder, with the same options +as `from_package` (`from_package` calls it for every folder). The folder's +`__init__.py` can then declare its kernel, and code imports the kernel from +where it is written: + +```python +# my_sim/kernels/push/__init__.py +import cunumpy as xp + +from my_sim.kernels.push.push_numpy import push as _numpy_version + +kernel = xp.Kernel.from_folder( + __name__, + host_suffix="_pyccel", + compile_host=compile_kernels, # see below + fallback=_numpy_version, + dispatch="arrays", + n_threads_from="first_array", +) +``` + +```python +# anywhere in the code +from my_sim.kernels.push import kernel as push + +push(positions, velocities, dt) # CUDA for CuPy arrays, host kernel otherwise +``` + +Per-kernel options (a fallback, `missing_cuda`, launch defaults) then sit next +to the kernel instead of in mappings keyed by name. + ## Compiled Pyccel host kernels By default the host kernel is the Python function itself, which is fine for @@ -152,7 +185,9 @@ catalog = xp.KernelCatalog.from_package( Each host kernel is a `cunumpy.CompiledHostKernel`: `catalog["push"].host_kernel.kernel.compiled` reports whether the compiled -version is available. Note that `epyccel` compiles again on every call; a +version is available. To test the path of a machine without Pyccel, run the +code inside `with xp.force_host_fallback():` (or set `CUNUMPY_HOST_FALLBACK=1` +for a whole run): every compiled host kernel then runs its fallback. Note that `epyccel` compiles again on every call; a code that compiles at run time usually keeps the builds in an on-disk cache keyed on the module source, so that only the first run after an edit compiles. @@ -176,7 +211,22 @@ catalog["gather"](host_positions, host_field, host_result) # host kernel, also The CUDA kernel runs if any top-level argument is on the GPU: a CuPy array, or a device-only argument object (`CudaArguments`, a struct value). A -`KernelArguments` object has both forms and does not decide on its own. +`KernelArguments` object has both forms and does not decide on its own. Host +arguments go to the host function directly, without conversion. + +The arguments of one call must then all be on one side, with the dtype and +layout the kernels take. `xp.as_kernel_array(value, like, dtype)` brings an +input to the side of the main array `like` (no copy if it already fits), and +`xp.kernel_output(out, like, dtype)` gives the buffer for an output, copied +back into `out` after the block if it had to be converted: + +```python +from functools import partial + +convert = partial(xp.as_kernel_array, like=field, dtype=float) +with xp.kernel_output(result, like=field, dtype=float) as buffer: + catalog["gather"](convert(positions), convert(field), buffer, n_threads=n) +``` ## Same parameters on both sides @@ -191,7 +241,8 @@ def test_kernel_signatures(): It compares the parameter names of each host function (for a compiled host kernel, its Python source, or the `__pyccel__/.pyi` stub that pyccel writes next to a compiled extension module) with the parsed `__global__` -signature, and lists every kernel that differs, e.g. +signature, and the parameters of a compiled host kernel's fallback with the host +function's, and lists every kernel that differs, e.g. `kernel 'push': the host kernel takes (x, v, dt, n), the CUDA kernel (x, v, n, dt)`. ## Porting status and setup diff --git a/src/cunumpy/LLM_GUIDE.md b/src/cunumpy/LLM_GUIDE.md index 72a80cf..f5cfa0a 100644 --- a/src/cunumpy/LLM_GUIDE.md +++ b/src/cunumpy/LLM_GUIDE.md @@ -68,6 +68,9 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository. | many kernels in a package, ported incrementally | `xp.KernelCatalog.from_package(__name__, missing_cuda="fallback")` | | host kernels compiled at first call (your compile function), NumPy fallback | `from_package(..., host_suffix="_pyccel", compile_host=my_compile, host_fallback={...})` -> `xp.CompiledHostKernel` | | host arrays reach kernels while CuPy is active | `Kernel(..., dispatch="arrays")` / `from_package(..., dispatch="arrays")`: CUDA only for device arguments | +| one kernel folder declares its kernel in its own `__init__.py` | `kernel = xp.Kernel.from_folder(__name__, host_suffix="_pyccel", compile_host=..., fallback=..., dispatch="arrays")` | +| bring a `dispatch="arrays"` kernel's arguments to the side of the main array | `xp.as_kernel_array(a, like=grid, dtype=float)`; outputs: `with xp.kernel_output(out, like=grid, dtype=float) as buf:` | +| test the path without the host compiler | `with xp.force_host_fallback():` or `CUNUMPY_HOST_FALLBACK=1` | | check host and CUDA kernels take the same parameters | `catalog.check_signatures()` (in a unit test) | | test a CUDA kernel's arithmetic without a GPU | `cunumpy.testing.emulate_cuda_kernel(kernel, *numpy_args, n_threads=n)` (C++ compiler; shared memory and __syncthreads ok, no warp ops; `shared_mem=` for extern shared) | | shared-memory budget of a block | `xp.max_shared_memory_per_block()` (48 KiB without a GPU) | @@ -244,6 +247,8 @@ k.get_kernel(); k.compile(); k.has_cuda catalog = xp.KernelCatalog.from_package(__name__, *, host_suffix="_kernels", cuda_suffix="_cuda.cu", missing_cuda="raise", host_options=None, include_dirs=None, **cuda_options) +kernel = xp.Kernel.from_folder(__name__, *, host_suffix="_kernels", compile_host=None, + fallback=None, dispatch="backend", **same options as from_package) catalog["push"]; catalog.summary(); catalog.with_cuda; catalog.without_cuda catalog.compile_all(jobs=1); catalog.parity_cases(); catalog.register(kernel) ``` diff --git a/src/cunumpy/__init__.py b/src/cunumpy/__init__.py index 4aa6692..32e44b2 100644 --- a/src/cunumpy/__init__.py +++ b/src/cunumpy/__init__.py @@ -23,7 +23,13 @@ ) from .dispatch import Kernel, KernelCatalog from .fusion import fuse -from .kernel import CompiledHostKernel, KernelArguments, PyccelKernel, resolve_host_args +from .kernel import ( + CompiledHostKernel, + KernelArguments, + PyccelKernel, + force_host_fallback, + resolve_host_args, +) from .mirror import DeviceMirror from .petsc import petsc_vec from .philox import ( @@ -46,6 +52,7 @@ DEFAULT_SHARED_MEMORY_PER_BLOCK, Timing, as_device_array, + as_kernel_array, assert_same_backend, bind_local_device, cuda_debug, @@ -61,6 +68,7 @@ get_rng, is_cpu, is_gpu, + kernel_output, local_rank, max_shared_memory_per_block, memory_info, @@ -144,6 +152,7 @@ def require_version(minimum: str) -> None: "TransferEvent", "__version__", "as_device_array", + "as_kernel_array", "assert_no_transfers", "assert_same_backend", "bind_local_device", @@ -156,6 +165,7 @@ def require_version(minimum: str) -> None: "cupy_backend", "default_float_dtype", "device_count", + "force_host_fallback", "free_memory", "fuse", "get_array_backend", @@ -167,6 +177,7 @@ def require_version(minimum: str) -> None: "include_hash", "is_cpu", "is_gpu", + "kernel_output", "local_rank", "max_shared_memory_per_block", "memory_info", diff --git a/src/cunumpy/__init__.pyi b/src/cunumpy/__init__.pyi index f54f7ad..d7cf4d4 100644 --- a/src/cunumpy/__init__.pyi +++ b/src/cunumpy/__init__.pyi @@ -26,6 +26,7 @@ from .dispatch import KernelCatalog as KernelCatalog from .fusion import fuse as fuse from .kernel import CompiledHostKernel as CompiledHostKernel from .kernel import PyccelKernel as PyccelKernel +from .kernel import force_host_fallback as force_host_fallback from .mirror import DeviceMirror as DeviceMirror from .petsc import petsc_vec as petsc_vec from .philox import philox4x32_10 as philox4x32_10 @@ -69,6 +70,9 @@ def mpi_buffer( array: Any, *, send: bool = ..., recv: bool = ..., cuda_aware: bool | None = ... ) -> Generator[Any]: ... def segment_sum(values: Any, keys: Any, n_segments: int) -> Any: ... +def as_kernel_array(value: Any, like: Any, dtype: Any = ...) -> Any: ... +@contextmanager +def kernel_output(out: Any, like: Any, dtype: Any = ...) -> Generator[Any]: ... def require_version(minimum: str) -> None: ... def device_count() -> int: ... def memory_info() -> tuple[int, int] | None: ... diff --git a/src/cunumpy/dispatch.py b/src/cunumpy/dispatch.py index 37742f3..070b208 100644 --- a/src/cunumpy/dispatch.py +++ b/src/cunumpy/dispatch.py @@ -23,6 +23,7 @@ import ast import importlib +import importlib.util import inspect import sys import warnings @@ -99,6 +100,19 @@ def _pyccel_stub_parameters(function: Any) -> list[str] | None: return None +def _positional_parameters(function: Any) -> list[str] | None: + """Names of the positional parameters of `function`, or None without a signature.""" + try: + signature = inspect.signature(function) + except (TypeError, ValueError): + return None + positional = ( + inspect.Parameter.POSITIONAL_ONLY, + inspect.Parameter.POSITIONAL_OR_KEYWORD, + ) + return [p.name for p in signature.parameters.values() if p.kind in positional] + + def _source_root(package: str) -> Path: """The directory containing the top-level package of `package`.""" top = importlib.import_module(package.partition(".")[0]) @@ -194,6 +208,118 @@ def __init__( self._test_args: ModuleType | None = None self._warned = False + @classmethod + def from_folder( + cls, + package: str, + *, + host_suffix: str = "_kernels", + cuda_suffix: str = "_cuda.cu", + test_args_suffix: str | None = "_test_args", + check_name_length: bool = True, + missing_cuda: str = "raise", + host_options: Mapping[str, Any] | None = None, + include_dirs: Sequence[str | Path] | None = None, + dispatch: str = "backend", + compile_host: Callable[[Any], Any] | None = None, + fallback: Callable[..., Any] | None = None, + **cuda_options: Any, + ) -> Kernel: + """The kernel of one kernel folder, the unit of :meth:`KernelCatalog.from_package`. + + `package` is the dotted name of the folder ````; the function + ```` of its module ```` is the host kernel, and + ```` in the folder, if present, the CUDA kernel. The + folder's own ``__init__.py`` can declare its kernel with it, so that + the kernel is imported from where it is written:: + + # my_sim/kernels/push/__init__.py + kernel = xp.Kernel.from_folder(__name__, dispatch="arrays") + + # anywhere in the code + from my_sim.kernels.push import kernel as push + + Nothing is compiled here: the CUDA source is parsed, the host module + imported. + + Parameters + ---------- + package : str + Dotted name of the kernel folder (``__name__`` in its ``__init__.py``). + host_suffix, cuda_suffix, test_args_suffix, check_name_length + As for :meth:`KernelCatalog.from_package`. + missing_cuda, host_options, dispatch + Passed on to :class:`Kernel`. + include_dirs : Sequence[str | Path] | None + Include directories of the CUDA kernel, in addition to the folder + itself; by default the source root of the top-level package (the + directory containing it). + compile_host : Callable | None + Compiles the host kernel module (see + :class:`~cunumpy.CompiledHostKernel`); by default the Python + function is called as it is. + fallback : Callable | None + With `compile_host`: called when compilation fails, e.g. a + vectorized NumPy version with the same arguments. + **cuda_options + Passed on to :meth:`CudaKernel.from_file`, e.g. ``block_size`` or + ``n_threads_from``. + + Raises + ------ + FileNotFoundError + If the folder has no host kernel module. + """ + spec = importlib.util.find_spec(package) + if spec is None or not spec.submodule_search_locations: + raise ModuleNotFoundError(f"{package!r} is not a package (kernel folder)") + folder = Path(next(iter(spec.submodule_search_locations))) + name = package.rpartition(".")[2] + if not (folder / f"{name}{host_suffix}.py").is_file(): + raise FileNotFoundError( + f"kernel folder {folder} has no host kernel {name}{host_suffix}.py" + ) + wrapper = f"bind_c_{name}{host_suffix}" + if check_name_length and len(wrapper) > FORTRAN_NAME_LIMIT: + warnings.warn( + f"kernel {name!r}: the module name {name}{host_suffix} gives the " + f"pyccel Fortran wrapper module {wrapper!r} ({len(wrapper)} " + f"characters), longer than Fortran's limit of " + f"{FORTRAN_NAME_LIMIT}; shorten the kernel name to at most " + f"{FORTRAN_NAME_LIMIT - len('bind_c_') - len(host_suffix)} " + "characters or compile with the C backend", + stacklevel=2, + ) + test_args = None + if ( + test_args_suffix is not None + and (folder / f"{name}{test_args_suffix}.py").is_file() + ): + test_args = f"{package}.{name}{test_args_suffix}" + module = importlib.import_module(f"{package}.{name}{host_suffix}") + host: Callable[..., Any] = getattr(module, name) + if compile_host is not None: + host = CompiledHostKernel(module, name, compile_host, fallback) + if include_dirs is None: + include_dirs = (_source_root(package),) + cuda_options["include_dirs"] = tuple(include_dirs) + cuda_path = folder / f"{name}{cuda_suffix}" + cuda_kernel = ( + CudaKernel.from_file(cuda_path, name, **cuda_options) + if cuda_path.is_file() + else None + ) + return cls( + host, + cuda_kernel, + name=name, + missing_cuda=missing_cuda, + cuda_path=cuda_path, + host_options=host_options, + dispatch=dispatch, + test_args=test_args, + ) + def __repr__(self) -> str: return ( f"Kernel(name={self._name!r}, cuda={self._cuda_kernel is not None}, " @@ -266,40 +392,48 @@ def host_parameters(self) -> list[str] | None: function = self._host_kernel.kernel if isinstance(function, CompiledHostKernel): function = function.python - try: - signature = inspect.signature(function) - except (TypeError, ValueError): - return _pyccel_stub_parameters(function) - positional = ( - inspect.Parameter.POSITIONAL_ONLY, - inspect.Parameter.POSITIONAL_OR_KEYWORD, + parameters = _positional_parameters(function) + return ( + parameters if parameters is not None else _pyccel_stub_parameters(function) ) - return [p.name for p in signature.parameters.values() if p.kind in positional] def check_signature(self) -> None: - """Check that the host and CUDA kernels take the same parameters, in order. + """Check that the host kernel, its fallback and the CUDA kernel agree. Compares the parameter names of the host function with those of the - parsed ``__global__`` signature. Nothing is checked without a CUDA - kernel, without a parsed CUDA signature (``check_signature=False``) or - without a Python host signature. + parsed ``__global__`` signature and, for a + :class:`~cunumpy.CompiledHostKernel` with a fallback, with those of the + fallback. A side is skipped without a CUDA kernel, without a parsed CUDA + signature (``check_signature=False``), without a fallback, or without a + Python signature (of the host function or the fallback). Raises ------ ValueError If the names or their order differ; the message shows both lists. """ - if self._cuda_kernel is None or self._cuda_kernel.signature is None: - return host = self.host_parameters() if host is None: return - cuda = [param.name for param in self._cuda_kernel.signature] - if host != cuda: - raise ValueError( - f"kernel {self._name!r}: the host kernel takes ({', '.join(host)}), " - f"the CUDA kernel ({', '.join(cuda)})" - ) + if self._cuda_kernel is not None and self._cuda_kernel.signature is not None: + cuda = [param.name for param in self._cuda_kernel.signature] + if host != cuda: + raise ValueError( + f"kernel {self._name!r}: the host kernel takes " + f"({', '.join(host)}), the CUDA kernel ({', '.join(cuda)})" + ) + function = self._host_kernel.kernel + fallback = ( + function.fallback if isinstance(function, CompiledHostKernel) else None + ) + if fallback is not None: + fallback_params = _positional_parameters(fallback) + if fallback_params is not None and fallback_params != host: + raise ValueError( + f"kernel {self._name!r}: the host kernel takes " + f"({', '.join(host)}), its fallback " + f"({', '.join(fallback_params)})" + ) def get_kernel(self) -> PyccelKernel | CudaKernel: """The kernel for the active backend. @@ -387,6 +521,14 @@ def __call__( "called on the CuPy backend", ) args, _ = resolve_host_args(args) + if ( + self._dispatch == "arrays" + and not on_device + and kernel._use_cupy is not True + ): + # only host arguments, by the choice above: nothing to convert, + # even while CuPy is the active backend + return kernel.kernel(*args) return kernel(*args) if n_threads is None and grid is None and kernel.n_threads_from is None: raise ValueError( @@ -506,55 +648,30 @@ def from_package( root = Path(importlib.import_module(package).__file__).parent if include_dirs is None: include_dirs = (_source_root(package),) - cuda_options["include_dirs"] = tuple(include_dirs) kernels = {} for folder in sorted(p for p in root.iterdir() if p.is_dir()): name = folder.name if not (folder / f"{name}{host_suffix}.py").is_file(): continue - wrapper = f"bind_c_{name}{host_suffix}" - if check_name_length and len(wrapper) > FORTRAN_NAME_LIMIT: - warnings.warn( - f"kernel {name!r}: the module name {name}{host_suffix} gives the " - f"pyccel Fortran wrapper module {wrapper!r} ({len(wrapper)} " - f"characters), longer than Fortran's limit of " - f"{FORTRAN_NAME_LIMIT}; shorten the kernel name to at most " - f"{FORTRAN_NAME_LIMIT - len('bind_c_') - len(host_suffix)} " - "characters or compile with the C backend", - stacklevel=2, - ) - test_args = None - if ( - test_args_suffix is not None - and (folder / f"{name}{test_args_suffix}.py").is_file() - ): - test_args = f"{package}.{name}.{name}{test_args_suffix}" - module = importlib.import_module(f"{package}.{name}.{name}{host_suffix}") - host: Callable[..., Any] = getattr(module, name) - if compile_host is not None: - fallback = ( - host_fallback(name) - if callable(host_fallback) - else (host_fallback or {}).get(name) - ) - host = CompiledHostKernel(module, name, compile_host, fallback) - cuda_path = folder / f"{name}{cuda_suffix}" - cuda_kernel = ( - CudaKernel.from_file(cuda_path, name, **cuda_options) - if cuda_path.is_file() - else None - ) - kernels[name] = Kernel( - host, - cuda_kernel, - name=name, + kernels[name] = Kernel.from_folder( + f"{package}.{name}", + host_suffix=host_suffix, + cuda_suffix=cuda_suffix, + test_args_suffix=test_args_suffix, + check_name_length=check_name_length, missing_cuda=missing_cuda, - cuda_path=cuda_path, host_options=( host_options(name) if callable(host_options) else host_options ), + include_dirs=include_dirs, dispatch=dispatch, - test_args=test_args, + compile_host=compile_host, + fallback=( + host_fallback(name) + if callable(host_fallback) + else (host_fallback or {}).get(name) + ), + **cuda_options, ) return cls(kernels) diff --git a/src/cunumpy/kernel.py b/src/cunumpy/kernel.py index 463103a..a0ca15f 100644 --- a/src/cunumpy/kernel.py +++ b/src/cunumpy/kernel.py @@ -24,8 +24,10 @@ import copy import importlib +import os import warnings -from collections.abc import Callable, Mapping, Sequence +from collections.abc import Callable, Iterator, Mapping, Sequence +from contextlib import contextmanager from types import ModuleType from typing import Any @@ -511,6 +513,42 @@ def outputs(self) -> tuple[int | str, ...] | None: return self._outputs +#: Whether every CompiledHostKernel runs its fallback instead of compiling, +#: set by ``CUNUMPY_HOST_FALLBACK=1`` or `force_host_fallback`. +_FORCE_FALLBACK = os.environ.get("CUNUMPY_HOST_FALLBACK", "").strip().lower() in ( + "1", + "true", + "yes", +) + + +@contextmanager +def force_host_fallback(enabled: bool = True) -> Iterator[None]: + """Run every :class:`CompiledHostKernel` as if its compilation had failed. + + Inside the block, a compiled host kernel calls its fallback (or the + uncompiled Python function, with a warning) without compiling, and + :attr:`CompiledHostKernel.compiled` is False; already compiled kernels are + kept for after the block. This tests the code path of a machine without the + compiler (e.g. without pyccel). The environment variable + ``CUNUMPY_HOST_FALLBACK=1`` (read when cunumpy is imported) does the same + for a whole run. Not thread-safe: the switch is global. + + Parameters + ---------- + enabled : bool + False switches it off inside the block, e.g. to compare with the + compiled version in a run with ``CUNUMPY_HOST_FALLBACK=1``. + """ + global _FORCE_FALLBACK + previous = _FORCE_FALLBACK + _FORCE_FALLBACK = enabled + try: + yield + finally: + _FORCE_FALLBACK = previous + + class CompiledHostKernel: """A host kernel compiled on its first call, with a fallback. @@ -575,8 +613,18 @@ def python(self) -> Callable[..., Any]: """The uncompiled Python function (for signatures and reference results).""" return getattr(self.module, self.__name__) + @property + def fallback(self) -> Callable[..., Any] | None: + """What runs when compilation fails, or None for the Python function.""" + return self._fallback + def build(self) -> Callable[..., Any] | None: - """Compile now (once); the compiled function, or None if that failed.""" + """Compile now (once); the compiled function, or None if that failed. + + None also inside :func:`force_host_fallback`, without compiling. + """ + if _FORCE_FALLBACK: + return None if not self._built: self._built = True try: diff --git a/src/cunumpy/xp.py b/src/cunumpy/xp.py index a039a35..3ed2762 100644 --- a/src/cunumpy/xp.py +++ b/src/cunumpy/xp.py @@ -13,6 +13,7 @@ import array_api_compat import array_api_compat.numpy as np +import numpy as _numpy from .transfers import _ACTIVE as _COUNTERS from .transfers import _describe, _record @@ -1071,6 +1072,50 @@ def get_array_module(array: Any) -> ModuleType: return np +def as_kernel_array(value: Any, like: Any, dtype: Any = None) -> Any: + """`value` as an array the kernel chosen for `like` takes. + + On the device of `like` (a CuPy array if `like` is one, a NumPy array + otherwise), C-contiguous and with `dtype` (any dtype if None): `value` + itself when it already is such an array, else one copy, moved to the other + side if needed (counted by `count_transfers()`). For input arrays of a + :class:`Kernel` with ``dispatch="arrays"``, whose choice follows the arrays: + pass the main array (e.g. the grid) as `like`, and the kernel gets + arguments all on one side:: + + convert = functools.partial(xp.as_kernel_array, like=grid, dtype=float) + deposit(convert(positions), convert(weights), grid, ...) + + For an array the kernel writes, use :func:`kernel_output`, which copies a + converted array back. + """ + if is_gpu(like): + import cupy + + if not is_gpu(value): + value = to_cupy(value) + return cupy.ascontiguousarray(value, dtype=dtype) + return _numpy.ascontiguousarray(to_numpy(value), dtype=dtype) + + +@contextmanager +def kernel_output(out: Any, like: Any, dtype: Any = None) -> Generator[Any]: + """The buffer a kernel chosen for `like` writes, copied back into `out` after it. + + Yields `out` itself if :func:`as_kernel_array` takes it unchanged (then the + kernel writes into it directly), else a converted copy whose contents are + written into `out` when the block ends without an error (moved back to the + side of `out`):: + + with xp.kernel_output(result, like=grid, dtype=float) as buffer: + gather(convert(positions), grid, buffer, ...) + """ + buffer = as_kernel_array(out, like, dtype) + yield buffer + if buffer is not out: + out[...] = buffer if is_gpu(out) or not is_gpu(buffer) else to_numpy(buffer) + + def is_gpu(array: Any) -> bool: """Check if the array is stored on a GPU (CuPy).""" return get_array_backend(array) == "cupy" diff --git a/tests/unit/test_kernel_dispatch_arrays.py b/tests/unit/test_kernel_dispatch_arrays.py index 27d9ac0..3e0563a 100644 --- a/tests/unit/test_kernel_dispatch_arrays.py +++ b/tests/unit/test_kernel_dispatch_arrays.py @@ -290,3 +290,193 @@ def test_arrays_dispatch_on_gpu(): kernel(device, 3.0, 5, n_threads=5) assert host.tolist() == [2.0] * 5 assert device.get().tolist() == [3.0] * 5 + + +# --------------------------------------------------------------------------- +# one kernel folder, declared in its own __init__.py +# --------------------------------------------------------------------------- + + +@pytest.fixture +def self_declaring_package(tmp_path, monkeypatch): + """`scale/__init__.py` declares its kernel with Kernel.from_folder.""" + root = tmp_path / "demo_folder_pkg" + (root / "scale").mkdir(parents=True) + (root / "__init__.py").write_text("") + (root / "scale" / "scale_pyccel.py").write_text( + "def scale(x: 'float[:]', factor: float, n: int):\n" + " for i in range(n):\n" + " x[i] *= factor\n" + ) + (root / "scale" / "scale_numpy.py").write_text( + "def scale(x, factor, n):\n x[:n] *= factor\n" + ) + (root / "scale" / "scale_cuda.cu").write_text(SCALE_CUDA) + (root / "scale" / "__init__.py").write_text( + "import cunumpy as xp\n" + "from demo_folder_pkg.scale.scale_numpy import scale as _numpy\n\n" + "COMPILED = []\n\n" + "def _compile(module):\n" + " COMPILED.append(module.__name__)\n" + " return module\n\n" + "kernel = xp.Kernel.from_folder(\n" + " __name__, host_suffix='_pyccel', dispatch='arrays',\n" + " compile_host=_compile, fallback=_numpy, n_threads_from='first_array',\n" + ")\n" + ) + monkeypatch.syspath_prepend(str(tmp_path)) + yield "demo_folder_pkg" + for module in [m for m in sys.modules if m.startswith("demo_folder_pkg")]: + del sys.modules[module] + + +def test_from_folder_in_the_folders_own_init(self_declaring_package): + import importlib + + folder = importlib.import_module(f"{self_declaring_package}.scale") + kernel = folder.kernel + assert kernel.name == "scale" and kernel.dispatch == "arrays" + assert kernel.cuda_kernel.n_threads_from is not None + host = kernel.host_kernel.kernel + assert isinstance(host, CompiledHostKernel) + assert ( + host.fallback + is sys.modules[f"{self_declaring_package}.scale.scale_numpy"].scale + ) + kernel.check_signature() # host, fallback and CUDA kernel: x, factor, n + x = np.ones(3) + kernel(x, 2.0, 3) + assert x.tolist() == [2.0] * 3 + assert folder.COMPILED == [f"{self_declaring_package}.scale.scale_pyccel"] + # the catalog of the parent package builds the same kernel + catalog = KernelCatalog.from_package(self_declaring_package, host_suffix="_pyccel") + assert catalog["scale"].host_parameters() == kernel.host_parameters() + + +def test_from_folder_needs_a_kernel_folder(self_declaring_package): + with pytest.raises(FileNotFoundError, match="no host kernel scale_kernels.py"): + Kernel.from_folder(f"{self_declaring_package}.scale") # default suffix + with pytest.raises(ModuleNotFoundError, match="not a package"): + Kernel.from_folder(f"{self_declaring_package}.scale.scale_numpy") + + +def test_arrays_dispatch_calls_the_host_kernel_without_conversion( + fake_gpu, monkeypatch +): + # CuPy is active, but host arguments never go through PyccelKernel's + # device-to-host conversion: the choice already says they are host arrays + def no_conversion(self, args, kwargs): + raise AssertionError("conversion checked") + + monkeypatch.setattr(xp.PyccelKernel, "_needs_conversion", no_conversion) + kernel = Kernel(scale, CudaKernel(SCALE_CUDA, "scale"), dispatch="arrays") + x = np.ones(2) + kernel(x, 2.0, 2) + assert x.tolist() == [2.0, 2.0] + # a host kernel that is told to convert keeps doing so + forced = Kernel( + xp.PyccelKernel(scale, use_cupy=True), + CudaKernel(SCALE_CUDA, "scale"), + dispatch="arrays", + ) + with pytest.raises(AssertionError, match="conversion checked"): + forced(x, 2.0, 2) + + +def test_check_signature_covers_the_fallback(): + module = _module("m", "def scale(x, factor, n):\n x *= factor\n") + + def fallback(x, n, factor): + pass + + kernel = Kernel(CompiledHostKernel(module, "scale", lambda m: m, fallback)) + with pytest.raises(ValueError, match=r"its fallback \(x, n, factor\)"): + kernel.check_signature() + Kernel( + CompiledHostKernel(module, "scale", lambda m: m, lambda x, factor, n: None) + ).check_signature() + + +def test_force_host_fallback(): + module = _module("m", "def double(x):\n x *= 2\n") + compiled = [] + + def compiler(mod): + compiled.append(mod) + return mod + + used = [] + kernel = CompiledHostKernel(module, "double", compiler, fallback=used.append) + with xp.force_host_fallback(): + assert not kernel.compiled + kernel("x") + with xp.force_host_fallback(False): + assert kernel.compiled # compiles once, kept for later + assert used == ["x"] and compiled == [module] + with xp.force_host_fallback(): + assert not kernel.compiled # the compiled version is kept, not used + assert kernel.compiled and compiled == [module] + + +def test_host_fallback_environment_variable(): + import os + import subprocess + + code = "import cunumpy.kernel as k; print(k._FORCE_FALLBACK)" + env = {**os.environ, "CUNUMPY_HOST_FALLBACK": "1"} + printed = subprocess.run( + [sys.executable, "-c", code], + env=env, + capture_output=True, + text=True, + check=True, + ).stdout + assert printed.strip() == "True" + + +# --------------------------------------------------------------------------- +# kernel arguments on the side of the arrays +# --------------------------------------------------------------------------- + + +def test_as_kernel_array_on_the_host(): + grid = np.zeros((4, 3)) + fits = np.ones(5) + assert xp.as_kernel_array(fits, like=grid, dtype=float) is fits # no copy + column = np.ones((5, 2))[:, 1] + converted = xp.as_kernel_array(column, like=grid, dtype=float) + assert converted.flags.c_contiguous and converted is not column + ints = xp.as_kernel_array([1, 2], like=grid, dtype=float) + assert ints.dtype == np.float64 and isinstance(ints, np.ndarray) + + +def test_kernel_output_writes_into_its_target(): + grid = np.zeros(3) + out = np.zeros(4) + with xp.kernel_output(out, like=grid, dtype=float) as buffer: + assert buffer is out # written directly + buffer += 1.0 + strided = np.zeros((4, 2))[:, 0] + with xp.kernel_output(strided, like=grid, dtype=float) as buffer: + assert buffer is not strided + buffer[...] = 7.0 + assert strided.tolist() == [7.0] * 4 + with pytest.raises(RuntimeError), xp.kernel_output(strided, like=grid) as buffer: + buffer[...] = 1.0 + raise RuntimeError("kernel failed") + assert strided.tolist() == [7.0] * 4 # not copied back after an error + + +def test_kernel_arrays_follow_a_device_grid(): + if not xp.cupy_available(): + pytest.skip("CuPy not installed or not functional") + import cupy as cp + + grid = cp.zeros(3) + on_device = xp.as_kernel_array(np.ones(4), like=grid, dtype=float) + assert xp.is_gpu(on_device) + host_out = np.zeros(4) + with xp.kernel_output(host_out, like=grid, dtype=float) as buffer: + assert xp.is_gpu(buffer) + buffer[...] = 2.0 + assert host_out.tolist() == [2.0] * 4 From 753559f8f2b1369423e6f1af60aec56dfcb59c51 Mon Sep 17 00:00:00 2001 From: Max Date: Sat, 3 Oct 2026 13:19:37 +0200 Subject: [PATCH 2/3] Removed the ordering and added setters and getters for the implementation istead --- CHANGELOG.md | 8 +- docs/source/api.md | 58 +++++-- docs/source/kernels/dispatch.md | 51 ++++-- src/cunumpy/LLM_GUIDE.md | 7 +- src/cunumpy/__init__.py | 12 +- src/cunumpy/__init__.pyi | 6 +- src/cunumpy/dispatch.py | 161 ++++++++++++++---- src/cunumpy/kernel.py | 198 ++++++++++++++++++---- tests/unit/test_kernel_dispatch_arrays.py | 196 +++++++++++++-------- 9 files changed, 524 insertions(+), 173 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9bb227f..25ab0ce 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,7 +16,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Changed - `Kernel(..., dispatch="arrays")` calls the host kernel directly for host arguments, without `PyccelKernel`'s check for device arrays to convert, which converted nothing but ran while CuPy was the active backend (unless the `PyccelKernel` was built with `use_cupy=True`). -- `Kernel.check_signature()` (and `KernelCatalog.check_signatures()`) also compares the parameters of a `CompiledHostKernel`'s fallback with those of the host kernel. +- `Kernel.check_signature()` (and `KernelCatalog.check_signatures()`) also compares the parameters of every host implementation (numba, NumPy; nothing is compiled) and of a `CompiledHostKernel`'s fallback with those of the host kernel. +- `KernelCatalog.from_package(..., compile_host=..., host_fallback=...)` builds `HostImplementations` instead of `CompiledHostKernel`s: `host_fallback` becomes the `"numpy"` implementation, and `_numba.py`/`_numpy.py` files in a kernel folder are picked up. - `CudaKernel` pointer and array view parameters and `CudaStruct` pointer and view fields reject an array that lives on another CUDA device than the current one (`ValueError`; `cupy.RawKernel` would read the foreign address silently). Checked only when CuPy is imported and the array reports a `device`. - `assert_kernels_agree(..., n_threads=...)` also takes a function of the argument tuple, and `n_threads` may be omitted when the CUDA kernel has `n_threads_from`. - NumPy integer scalars passed to integer kernel parameters (and struct fields) are checked by value, like Python ints: `np.int64(5)` fits an `int` parameter; before, they raised `TypeError` unless their dtype was exactly the parameter's. @@ -28,9 +29,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - `KernelCatalog.from_package(..., include_dirs=None)` is now an explicit keyword; by default the source root of the top-level package (the directory containing it) is an include directory of every CUDA kernel, in addition to the kernel's own folder, so kernels can `#include "my_pkg/common.cuh"`. ### Added -- `Kernel.from_folder(package, ...)`: the kernel of one kernel folder, with the options of `KernelCatalog.from_package` (`host_suffix`, `compile_host`, `fallback`, `dispatch`, `include_dirs`, CUDA options...). A folder's own `__init__.py` can declare `kernel = xp.Kernel.from_folder(__name__, ...)`, so code imports the kernel from where it is written; `from_package` now builds each kernel with it. +- `Kernel.from_folder(package, ...)`: the kernel of one kernel folder, with the options of `KernelCatalog.from_package` (`host_suffix`, `compile_host`, `dispatch`, `include_dirs`, CUDA options...). Every version in the folder is an implementation: `.py` (`"pyccel"`, compiled with `compile_host`, and `"python"`, uncompiled), `_numba.py`, `_numpy.py` and `_cuda.cu`. A folder's own `__init__.py` can declare `kernel = xp.Kernel.from_folder(__name__, ...)`, so code imports the kernel from where it is written; `from_package` now builds each kernel with it. `Kernel.implementations` lists them, `Kernel.selected(device=False)` tells which one a call runs. +- `xp.HostImplementations` and `xp.HOST_IMPLEMENTATIONS`: the host implementations of a kernel, loaded on first use; a call runs the default, the first available of pyccel, numba and NumPy (the uncompiled Python version, with a warning, if none is), or the one chosen with `xp.set_kernel_implementation(name)` / `with xp.use_kernel_implementation(name):` / `CUNUMPY_KERNEL_IMPLEMENTATION=name` (read at import), like `set_backend`/`use_backend`/`ARRAY_BACKEND`; a chosen implementation that a kernel lacks or cannot load raises instead of running another. `xp.get_kernel_implementation()` reads the setting. - `xp.as_kernel_array(value, like, dtype=None)` and `xp.kernel_output(out, like, dtype=None)`: bring the arguments of a `dispatch="arrays"` kernel to the side of the main array `like` (CuPy or NumPy, C-contiguous, `dtype`), without a copy when they already fit; `kernel_output` yields the buffer the kernel writes and copies it back into `out` if it had to be converted. -- `xp.force_host_fallback(enabled=True)` and the environment variable `CUNUMPY_HOST_FALLBACK=1`: every `CompiledHostKernel` runs its fallback as if compilation had failed (`compiled` is False), to test the code path of a machine without the compiler. `CompiledHostKernel.fallback` returns the fallback. +- `CompiledHostKernel.fallback` returns the fallback. - `cunumpy/random.cuh` and `xp.philox_uniform`, `philox_uniform2`, `philox_normal`, `philox_normal2`, `philox4x32_10`: counter-based random numbers (Philox4x32-10, passing the Random123 known-answer tests) as a pure function of `(seed, stream, counter)`, the same in a kernel and on the host (uniform numbers bit for bit, normal numbers up to the last bits of the math functions), so kernels that draw random numbers can be compared with their host versions. - `CudaKernel(..., n_threads_from="first_array")`: one thread per row of the first array argument when a launch gives neither `n_threads` nor `grid`. - `CudaKernel` launches with `shared_mem` above 48 KiB set the kernel's `max_dynamic_shared_size_bytes` once, up to the device's opt-in limit, and raise `ValueError` beyond it. diff --git a/docs/source/api.md b/docs/source/api.md index cabecf5..294acbb 100644 --- a/docs/source/api.md +++ b/docs/source/api.md @@ -1494,18 +1494,50 @@ build a catalog by hand. ```python # my_sim/kernels/push/__init__.py kernel = xp.Kernel.from_folder(__name__, host_suffix="_pyccel", dispatch="arrays", - compile_host=compile_kernels, fallback=push_numpy) + compile_host=compile_kernels) +kernel.implementations # ("pyccel", "numpy", "python", "cuda") +kernel.selected() # "pyccel": what a call with host arrays runs now ``` The kernel of one kernel folder `package` (its dotted name, `__name__` in its -`__init__.py`): the function `` of `.py` is the host -kernel, `` in the folder, if present, the CUDA kernel. Takes -the options of `KernelCatalog.from_package` for one kernel: `host_suffix`, -`cuda_suffix`, `test_args_suffix`, `check_name_length`, `missing_cuda`, -`host_options`, `include_dirs`, `dispatch`, `compile_host`, `fallback` (one -callable, used with `compile_host`) and CUDA options such as `block_size` or -`n_threads_from`. Raises `FileNotFoundError` if the folder has no host kernel -module and `ModuleNotFoundError` if `package` is not a package. +`__init__.py`). Each version of `` in the folder is an implementation: +`.py` gives `"pyccel"` (compiled with `compile_host` on +first use, as it is without one) and `"python"` (uncompiled), `_numba.py` +gives `"numba"`, `_numpy.py` gives `"numpy"`, and `` the +CUDA kernel. `extra_implementations` adds host implementations that are not +files, as loaders by name (`{"numpy": lambda: push_numpy}`). Takes the options of +`KernelCatalog.from_package` for one kernel: `host_suffix`, `cuda_suffix`, +`test_args_suffix`, `check_name_length`, `missing_cuda`, `host_options`, +`include_dirs`, `dispatch`, `compile_host` and CUDA options such as +`block_size` or `n_threads_from`. Raises `FileNotFoundError` if the folder has +no host kernel module and `ModuleNotFoundError` if `package` is not a package. +`kernel.selected(device=True)` names the implementation for device arguments. + +## `HostImplementations`, `set_kernel_implementation` + +```python +host = xp.HostImplementations("push", {"pyccel": load_compiled, "numpy": lambda: push_numpy, + "python": lambda: push}) +host(*args) # the default implementation +xp.set_kernel_implementation("numpy") # every kernel: like xp.set_backend +with xp.use_kernel_implementation("python"): # like xp.use_backend + host(*args) +``` + +The host implementations of one kernel (names from `xp.HOST_IMPLEMENTATIONS`: +`"pyccel"`, `"numba"`, `"numpy"`, `"python"`; `"python"` is required), each +given as a loader that returns the function or raises if it is unavailable. +Loaded on first use; `available(name)` loads and reports, `get(name)` returns it +or raises `LookupError` (missing, or failed to load with the error as cause), +`errors` maps names to load errors, `names` lists them, `python` is the +uncompiled function, `build()` loads the default now. A call runs +`selected()`: the implementation set with `set_kernel_implementation(name)` (or +`use_kernel_implementation`, or the environment variable +`CUNUMPY_KERNEL_IMPLEMENTATION` read at import), which raises if the kernel +lacks it or cannot load it, else the default: the first available of pyccel, +numba and NumPy, and else `"python"` with a `RuntimeWarning` (once). +`get_kernel_implementation()` reads the setting; `None` is the default. The +setting is global, not per thread, and applies to host calls only. ## `CompiledHostKernel` @@ -1526,14 +1558,6 @@ path), `kernel.error` is the exception of a failed build, `kernel.python` the uncompiled function, `kernel.fallback` the fallback and `kernel.build()` compiles now. -`with xp.force_host_fallback():` makes every `CompiledHostKernel` run its -fallback as if compilation had failed (`compiled` is False inside the block; -already compiled versions are kept for after it), to test the path of a -machine without the compiler; `force_host_fallback(False)` switches it off -inside such a block. The environment variable `CUNUMPY_HOST_FALLBACK=1`, read -when cunumpy is imported, does the same for a whole run. The switch is global, -not per thread. - ## Testing utilities ```python diff --git a/docs/source/kernels/dispatch.md b/docs/source/kernels/dispatch.md index a767aad..b1c3463 100644 --- a/docs/source/kernels/dispatch.md +++ b/docs/source/kernels/dispatch.md @@ -130,17 +130,23 @@ as `from_package` (`from_package` calls it for every folder). The folder's `__init__.py` can then declare its kernel, and code imports the kernel from where it is written: +```text +my_sim/kernels/push/ +├── __init__.py # kernel = xp.Kernel.from_folder(__name__, ...) +├── push_pyccel.py # "pyccel" (compiled with compile_host) and "python" (as is) +├── push_numba.py # "numba": def push(...) decorated with numba.njit +├── push_numpy.py # "numpy": vectorized +└── push_cuda.cu # the CUDA kernel +``` + ```python # my_sim/kernels/push/__init__.py import cunumpy as xp -from my_sim.kernels.push.push_numpy import push as _numpy_version - kernel = xp.Kernel.from_folder( __name__, host_suffix="_pyccel", compile_host=compile_kernels, # see below - fallback=_numpy_version, dispatch="arrays", n_threads_from="first_array", ) @@ -153,8 +159,32 @@ from my_sim.kernels.push import kernel as push push(positions, velocities, dt) # CUDA for CuPy arrays, host kernel otherwise ``` -Per-kernel options (a fallback, `missing_cuda`, launch defaults) then sit next -to the kernel instead of in mappings keyed by name. +Per-kernel options (`missing_cuda`, launch defaults) then sit next to the +kernel instead of in mappings keyed by name. + +### Several host implementations + +Every file of the folder is one implementation; only the host side has a +choice. A call with host arrays runs the first available of `"pyccel"`, +`"numba"` and `"numpy"` (an implementation is unavailable if it fails to +compile or import), and the uncompiled `"python"` version, with a warning, if +none is. Device arrays run the CUDA kernel. To choose, use the same pattern as +for the array backend: + +```python +xp.set_kernel_implementation("numpy") # like xp.set_backend +with xp.use_kernel_implementation("numba"): # like xp.use_backend + push(positions, velocities, dt) +xp.set_kernel_implementation(None) # back to the default +``` + +or `CUNUMPY_KERNEL_IMPLEMENTATION=numpy` for a whole run (read at import, like +`ARRAY_BACKEND`). A chosen implementation that a kernel does not have, or +cannot load, raises `LookupError` instead of running another one: a benchmark +of numba never silently measures NumPy. `kernel.implementations` lists the +implementations, `kernel.selected()` names the one a call with host arrays runs +now (`kernel.selected(device=True)` for device arrays), and +`kernel.host_kernel.kernel.errors` holds why an implementation failed to load. ## Compiled Pyccel host kernels @@ -183,11 +213,12 @@ catalog = xp.KernelCatalog.from_package( ) ``` -Each host kernel is a `cunumpy.CompiledHostKernel`: -`catalog["push"].host_kernel.kernel.compiled` reports whether the compiled -version is available. To test the path of a machine without Pyccel, run the -code inside `with xp.force_host_fallback():` (or set `CUNUMPY_HOST_FALLBACK=1` -for a whole run): every compiled host kernel then runs its fallback. Note that `epyccel` compiles again on every call; a +`host_fallback` becomes each kernel's `"numpy"` implementation (see "Several +host implementations" above); `catalog["push"].host_kernel.kernel.available("pyccel")` +reports whether the compiled version builds, and `catalog["push"].selected()` +which version runs. To test the path of a machine without Pyccel, run the code +inside `with xp.use_kernel_implementation("numpy"):` (or set +`CUNUMPY_KERNEL_IMPLEMENTATION=numpy` for a whole run). Note that `epyccel` compiles again on every call; a code that compiles at run time usually keeps the builds in an on-disk cache keyed on the module source, so that only the first run after an edit compiles. diff --git a/src/cunumpy/LLM_GUIDE.md b/src/cunumpy/LLM_GUIDE.md index f5cfa0a..54b8e41 100644 --- a/src/cunumpy/LLM_GUIDE.md +++ b/src/cunumpy/LLM_GUIDE.md @@ -68,9 +68,9 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository. | many kernels in a package, ported incrementally | `xp.KernelCatalog.from_package(__name__, missing_cuda="fallback")` | | host kernels compiled at first call (your compile function), NumPy fallback | `from_package(..., host_suffix="_pyccel", compile_host=my_compile, host_fallback={...})` -> `xp.CompiledHostKernel` | | host arrays reach kernels while CuPy is active | `Kernel(..., dispatch="arrays")` / `from_package(..., dispatch="arrays")`: CUDA only for device arguments | -| one kernel folder declares its kernel in its own `__init__.py` | `kernel = xp.Kernel.from_folder(__name__, host_suffix="_pyccel", compile_host=..., fallback=..., dispatch="arrays")` | +| one kernel folder declares its kernel in its own `__init__.py` | `kernel = xp.Kernel.from_folder(__name__, host_suffix="_pyccel", compile_host=..., dispatch="arrays")`; `_numba.py`, `_numpy.py` in the folder are further host implementations | | bring a `dispatch="arrays"` kernel's arguments to the side of the main array | `xp.as_kernel_array(a, like=grid, dtype=float)`; outputs: `with xp.kernel_output(out, like=grid, dtype=float) as buf:` | -| test the path without the host compiler | `with xp.force_host_fallback():` or `CUNUMPY_HOST_FALLBACK=1` | +| choose the host implementation (pyccel/numba/numpy/python) | `xp.set_kernel_implementation("numpy")`, `with xp.use_kernel_implementation(...)`, `CUNUMPY_KERNEL_IMPLEMENTATION=numpy`; default: first available of pyccel, numba, numpy; `kernel.selected()` | | check host and CUDA kernels take the same parameters | `catalog.check_signatures()` (in a unit test) | | test a CUDA kernel's arithmetic without a GPU | `cunumpy.testing.emulate_cuda_kernel(kernel, *numpy_args, n_threads=n)` (C++ compiler; shared memory and __syncthreads ok, no warp ops; `shared_mem=` for extern shared) | | shared-memory budget of a block | `xp.max_shared_memory_per_block()` (48 KiB without a GPU) | @@ -248,7 +248,8 @@ catalog = xp.KernelCatalog.from_package(__name__, *, host_suffix="_kernels", cuda_suffix="_cuda.cu", missing_cuda="raise", host_options=None, include_dirs=None, **cuda_options) kernel = xp.Kernel.from_folder(__name__, *, host_suffix="_kernels", compile_host=None, - fallback=None, dispatch="backend", **same options as from_package) + dispatch="backend", extra_implementations=None, **same options as from_package) +kernel.implementations; kernel.selected(device=False) catalog["push"]; catalog.summary(); catalog.with_cuda; catalog.without_cuda catalog.compile_all(jobs=1); catalog.parity_cases(); catalog.register(kernel) ``` diff --git a/src/cunumpy/__init__.py b/src/cunumpy/__init__.py index 32e44b2..5206d75 100644 --- a/src/cunumpy/__init__.py +++ b/src/cunumpy/__init__.py @@ -24,11 +24,15 @@ from .dispatch import Kernel, KernelCatalog from .fusion import fuse from .kernel import ( + HOST_IMPLEMENTATIONS, CompiledHostKernel, + HostImplementations, KernelArguments, PyccelKernel, - force_host_fallback, + get_kernel_implementation, resolve_host_args, + set_kernel_implementation, + use_kernel_implementation, ) from .mirror import DeviceMirror from .petsc import petsc_vec @@ -130,6 +134,7 @@ def require_version(minimum: str) -> None: __all__ = [ "DEBUG_OPTIONS", "DEFAULT_SHARED_MEMORY_PER_BLOCK", + "HOST_IMPLEMENTATIONS", "CompiledHostKernel", "CudaArguments", "CudaKernel", @@ -139,6 +144,7 @@ def require_version(minimum: str) -> None: "CudaStructArguments", "CudaStructValue", "DeviceMirror", + "HostImplementations", "HostStaging", "Kernel", "KernelArguments", @@ -165,13 +171,13 @@ def require_version(minimum: str) -> None: "cupy_backend", "default_float_dtype", "device_count", - "force_host_fallback", "free_memory", "fuse", "get_array_backend", "get_array_module", "get_backend", "get_cuda_debug", + "get_kernel_implementation", "get_mpi_cuda_aware", "get_rng", "include_hash", @@ -205,6 +211,7 @@ def require_version(minimum: str) -> None: "set_cuda_debug", "set_device", "set_device_for_rank", + "set_kernel_implementation", "set_mpi_cuda_aware", "stream", "synchronize", @@ -214,6 +221,7 @@ def require_version(minimum: str) -> None: "to_cupy", "to_numpy", "use_backend", + "use_kernel_implementation", "write_cuda_header", "xp", ] diff --git a/src/cunumpy/__init__.pyi b/src/cunumpy/__init__.pyi index d7cf4d4..5de4eec 100644 --- a/src/cunumpy/__init__.pyi +++ b/src/cunumpy/__init__.pyi @@ -24,9 +24,13 @@ from .cuda_kernel import write_cuda_header as write_cuda_header from .dispatch import Kernel as Kernel from .dispatch import KernelCatalog as KernelCatalog from .fusion import fuse as fuse +from .kernel import HOST_IMPLEMENTATIONS as HOST_IMPLEMENTATIONS from .kernel import CompiledHostKernel as CompiledHostKernel +from .kernel import HostImplementations as HostImplementations from .kernel import PyccelKernel as PyccelKernel -from .kernel import force_host_fallback as force_host_fallback +from .kernel import get_kernel_implementation as get_kernel_implementation +from .kernel import set_kernel_implementation as set_kernel_implementation +from .kernel import use_kernel_implementation as use_kernel_implementation from .mirror import DeviceMirror as DeviceMirror from .petsc import petsc_vec as petsc_vec from .philox import philox4x32_10 as philox4x32_10 diff --git a/src/cunumpy/dispatch.py b/src/cunumpy/dispatch.py index 070b208..6bdf744 100644 --- a/src/cunumpy/dispatch.py +++ b/src/cunumpy/dispatch.py @@ -35,7 +35,12 @@ import array_api_compat from .cuda_kernel import CudaKernel, _compile_in_threads -from .kernel import CompiledHostKernel, PyccelKernel, resolve_host_args +from .kernel import ( + CompiledHostKernel, + HostImplementations, + PyccelKernel, + resolve_host_args, +) from .transfers import _ACTIVE as _COUNTERS from .transfers import _record from .xp import get_backend @@ -100,6 +105,26 @@ def _pyccel_stub_parameters(function: Any) -> list[str] | None: return None +def _import_loader(module: str, name: str) -> Callable[[], Callable[..., Any]]: + """Loads function `name` of `module` on first use (the import may fail).""" + return lambda: getattr(importlib.import_module(module), name) + + +def _fallback_loader( + host_fallback: Mapping[str, Callable[..., Any]] + | Callable[[str], Callable[..., Any] | None] + | None, + name: str, +) -> dict[str, Callable[[], Callable[..., Any]]]: + """`from_package`'s `host_fallback` for kernel `name`, as its NumPy implementation.""" + fallback = ( + host_fallback(name) + if callable(host_fallback) + else (host_fallback or {}).get(name) + ) + return {} if fallback is None else {"numpy": lambda: fallback} + + def _positional_parameters(function: Any) -> list[str] | None: """Names of the positional parameters of `function`, or None without a signature.""" try: @@ -222,19 +247,34 @@ def from_folder( include_dirs: Sequence[str | Path] | None = None, dispatch: str = "backend", compile_host: Callable[[Any], Any] | None = None, - fallback: Callable[..., Any] | None = None, + extra_implementations: ( + Mapping[str, Callable[[], Callable[..., Any]]] | None + ) = None, **cuda_options: Any, ) -> Kernel: """The kernel of one kernel folder, the unit of :meth:`KernelCatalog.from_package`. - `package` is the dotted name of the folder ````; the function - ```` of its module ```` is the host kernel, and - ```` in the folder, if present, the CUDA kernel. The - folder's own ``__init__.py`` can declare its kernel with it, so that - the kernel is imported from where it is written:: + `package` is the dotted name of the folder ````. Every file of the + folder that defines a version of the kernel (a function ````, or + the ``__global__`` function ````) is one implementation: + + * ``.py``: ``"pyccel"``, compiled with `compile_host` + (as it is without one), and ``"python"``, the same file uncompiled; + * ``_numba.py``: ``"numba"`` (the function decorated there, e.g. + with ``numba.njit``; unavailable if the import fails); + * ``_numpy.py``: ``"numpy"``; + * ````: the CUDA kernel. + + The host implementations form a :class:`~cunumpy.HostImplementations`: + a call runs the one set with :func:`~cunumpy.set_kernel_implementation`, + or by default the first available of pyccel, numba and NumPy. The + folder's own ``__init__.py`` can declare its kernel with this method, so + that the kernel is imported from where it is written:: # my_sim/kernels/push/__init__.py - kernel = xp.Kernel.from_folder(__name__, dispatch="arrays") + kernel = xp.Kernel.from_folder( + __name__, host_suffix="_pyccel", compile_host=compile, dispatch="arrays" + ) # anywhere in the code from my_sim.kernels.push import kernel as push @@ -255,12 +295,13 @@ def from_folder( itself; by default the source root of the top-level package (the directory containing it). compile_host : Callable | None - Compiles the host kernel module (see - :class:`~cunumpy.CompiledHostKernel`); by default the Python - function is called as it is. - fallback : Callable | None - With `compile_host`: called when compilation fails, e.g. a - vectorized NumPy version with the same arguments. + Returns the compiled form of the ```` module + (cunumpy does not compile anything itself, e.g. a wrapper around + ``pyccel.epyccel`` with a cache), called on first use of the pyccel + implementation; if it raises, that implementation is unavailable. + extra_implementations : Mapping | None + Loaders of host implementations that are not files of the folder, + by implementation name, e.g. ``{"numpy": lambda: push_numpy}``. **cuda_options Passed on to :meth:`CudaKernel.from_file`, e.g. ``block_size`` or ``n_threads_from``. @@ -297,9 +338,22 @@ def from_folder( ): test_args = f"{package}.{name}{test_args_suffix}" module = importlib.import_module(f"{package}.{name}{host_suffix}") - host: Callable[..., Any] = getattr(module, name) - if compile_host is not None: - host = CompiledHostKernel(module, name, compile_host, fallback) + python = getattr(module, name) + loaders: dict[str, Callable[[], Callable[..., Any]]] = { + "python": lambda: python, + "pyccel": ( + (lambda: getattr(compile_host(module), name)) + if compile_host is not None + else (lambda: python) + ), + } + for implementation in ("numba", "numpy"): + if (folder / f"{name}_{implementation}.py").is_file(): + loaders[implementation] = _import_loader( + f"{package}.{name}_{implementation}", name + ) + loaders.update(extra_implementations or {}) + host = HostImplementations(name, loaders) if include_dirs is None: include_dirs = (_source_root(package),) cuda_options["include_dirs"] = tuple(include_dirs) @@ -390,7 +444,7 @@ def host_parameters(self) -> list[str] | None: available. """ function = self._host_kernel.kernel - if isinstance(function, CompiledHostKernel): + if isinstance(function, (CompiledHostKernel, HostImplementations)): function = function.python parameters = _positional_parameters(function) return ( @@ -398,14 +452,15 @@ def host_parameters(self) -> list[str] | None: ) def check_signature(self) -> None: - """Check that the host kernel, its fallback and the CUDA kernel agree. + """Check that all implementations of the kernel take the same parameters. Compares the parameter names of the host function with those of the - parsed ``__global__`` signature and, for a - :class:`~cunumpy.CompiledHostKernel` with a fallback, with those of the - fallback. A side is skipped without a CUDA kernel, without a parsed CUDA - signature (``check_signature=False``), without a fallback, or without a - Python signature (of the host function or the fallback). + parsed ``__global__`` signature, with those of the other host + implementations of a :class:`~cunumpy.HostImplementations` (numba, + NumPy; nothing is compiled, an implementation that fails to import is + skipped) and with the fallback of a :class:`~cunumpy.CompiledHostKernel`. + A side is skipped without a CUDA kernel, without a parsed CUDA signature + (``check_signature=False``), or without a Python signature. Raises ------ @@ -423,18 +478,54 @@ def check_signature(self) -> None: f"({', '.join(host)}), the CUDA kernel ({', '.join(cuda)})" ) function = self._host_kernel.kernel - fallback = ( - function.fallback if isinstance(function, CompiledHostKernel) else None - ) - if fallback is not None: - fallback_params = _positional_parameters(fallback) - if fallback_params is not None and fallback_params != host: + others: dict[str, Any] = {} + if isinstance(function, CompiledHostKernel) and function.fallback is not None: + others["fallback"] = function.fallback + elif isinstance(function, HostImplementations): + for name in ("numba", "numpy"): + if function.available(name): + # a numba dispatcher keeps the Python function in py_func + implementation = function.get(name) + others[name] = getattr(implementation, "py_func", implementation) + for name, implementation in others.items(): + parameters = _positional_parameters(implementation) + if parameters is not None and parameters != host: + which = "its fallback" if name == "fallback" else f"the {name} version" raise ValueError( f"kernel {self._name!r}: the host kernel takes " - f"({', '.join(host)}), its fallback " - f"({', '.join(fallback_params)})" + f"({', '.join(host)}), {which} ({', '.join(parameters)})" ) + @property + def implementations(self) -> tuple[str, ...]: + """Names of the implementations, e.g. ``("pyccel", "numpy", "python", "cuda")``. + + A host kernel that is not a :class:`~cunumpy.HostImplementations` counts + as ``"host"``. + """ + function = self._host_kernel.kernel + host = ( + function.names if isinstance(function, HostImplementations) else ("host",) + ) + return (*host, "cuda") if self._cuda_kernel is not None else host + + def selected(self, device: bool = False) -> str: + """The implementation a call with host (or `device`) arguments runs now. + + For host arguments: the setting of + :func:`~cunumpy.set_kernel_implementation` or the default (loads it), or + ``"host"`` for a host kernel that is not a + :class:`~cunumpy.HostImplementations`. For device arguments ``"cuda"``, + or ``"host"`` if there is no CUDA kernel and ``missing_cuda="fallback"``. + Useful to check that a run does not use a slow path. + """ + if device: + return "cuda" if self._device_kernel() is self._cuda_kernel else "host" + function = self._host_kernel.kernel + return ( + function.selected() if isinstance(function, HostImplementations) else "host" + ) + def get_kernel(self) -> PyccelKernel | CudaKernel: """The kernel for the active backend. @@ -666,11 +757,7 @@ def from_package( include_dirs=include_dirs, dispatch=dispatch, compile_host=compile_host, - fallback=( - host_fallback(name) - if callable(host_fallback) - else (host_fallback or {}).get(name) - ), + extra_implementations=_fallback_loader(host_fallback, name), **cuda_options, ) return cls(kernels) diff --git a/src/cunumpy/kernel.py b/src/cunumpy/kernel.py index a0ca15f..68b9151 100644 --- a/src/cunumpy/kernel.py +++ b/src/cunumpy/kernel.py @@ -513,40 +513,181 @@ def outputs(self) -> tuple[int | str, ...] | None: return self._outputs -#: Whether every CompiledHostKernel runs its fallback instead of compiling, -#: set by ``CUNUMPY_HOST_FALLBACK=1`` or `force_host_fallback`. -_FORCE_FALLBACK = os.environ.get("CUNUMPY_HOST_FALLBACK", "").strip().lower() in ( - "1", - "true", - "yes", +#: The host implementations a kernel folder can have, see `HostImplementations`. +HOST_IMPLEMENTATIONS = ("pyccel", "numba", "numpy", "python") + + +def _check_implementation(name: str | None) -> str | None: + if name is not None and name not in HOST_IMPLEMENTATIONS: + raise ValueError( + f"kernel implementation must be one of {HOST_IMPLEMENTATIONS} or None, " + f"got {name!r}" + ) + return name + + +_KERNEL_IMPLEMENTATION: str | None = _check_implementation( + os.environ.get("CUNUMPY_KERNEL_IMPLEMENTATION", "").strip().lower() or None ) -@contextmanager -def force_host_fallback(enabled: bool = True) -> Iterator[None]: - """Run every :class:`CompiledHostKernel` as if its compilation had failed. +def set_kernel_implementation(name: str | None) -> None: + """Choose the host implementation every kernel runs, like :func:`set_backend`. - Inside the block, a compiled host kernel calls its fallback (or the - uncompiled Python function, with a warning) without compiling, and - :attr:`CompiledHostKernel.compiled` is False; already compiled kernels are - kept for after the block. This tests the code path of a machine without the - compiler (e.g. without pyccel). The environment variable - ``CUNUMPY_HOST_FALLBACK=1`` (read when cunumpy is imported) does the same - for a whole run. Not thread-safe: the switch is global. + ``"pyccel"``, ``"numba"``, ``"numpy"`` or ``"python"`` (the uncompiled + source of the pyccel version); None restores the default (see + :class:`HostImplementations`). A kernel without the chosen implementation, + or whose chosen implementation is unavailable (e.g. pyccel failed to + compile), raises instead of running another one. CUDA kernels are not + affected: device arrays always run the CUDA version. The environment + variable ``CUNUMPY_KERNEL_IMPLEMENTATION`` (read when cunumpy is imported) + sets it for a whole run. + """ + global _KERNEL_IMPLEMENTATION + _KERNEL_IMPLEMENTATION = _check_implementation(name) - Parameters - ---------- - enabled : bool - False switches it off inside the block, e.g. to compare with the - compiled version in a run with ``CUNUMPY_HOST_FALLBACK=1``. + +def get_kernel_implementation() -> str | None: + """The host implementation set with :func:`set_kernel_implementation`, or None.""" + return _KERNEL_IMPLEMENTATION + + +@contextmanager +def use_kernel_implementation(name: str | None) -> Iterator[None]: + """Temporarily choose the host implementation, like :func:`use_backend`. + + For tests and benchmarks, e.g. ``with xp.use_kernel_implementation("numpy"):`` + to run the code path of a machine without pyccel. The setting is global, + not per thread. """ - global _FORCE_FALLBACK - previous = _FORCE_FALLBACK - _FORCE_FALLBACK = enabled + global _KERNEL_IMPLEMENTATION + previous = _KERNEL_IMPLEMENTATION + set_kernel_implementation(name) try: yield finally: - _FORCE_FALLBACK = previous + _KERNEL_IMPLEMENTATION = previous + + +class HostImplementations: + """The host implementations of one kernel, run by name or by the default rule. + + Each implementation is loaded on first use (a pyccel build, an import) and + may be unavailable (no compiler, numba not installed); a failed load is + remembered with its exception. A call runs the implementation chosen with + :func:`set_kernel_implementation`, which must exist and load, or else the + default: the first available of ``"pyccel"``, ``"numba"`` and ``"numpy"``, + and as a last resort ``"python"``, with a warning (correct, but slow). + Built by :meth:`Kernel.from_folder` from the files of a kernel folder. + + Parameters + ---------- + name : str + Name of the kernel. + loaders : Mapping[str, Callable[[], Callable]] + Per implementation name (one of ``HOST_IMPLEMENTATIONS``), a function + returning the implementation, or raising if it is unavailable. + """ + + def __init__( + self, name: str, loaders: Mapping[str, Callable[[], Callable[..., Any]]] + ) -> None: + unknown = set(loaders) - set(HOST_IMPLEMENTATIONS) + if unknown: + raise ValueError( + f"kernel {name!r}: unknown implementations {sorted(unknown)}, " + f"expected names from {HOST_IMPLEMENTATIONS}" + ) + if "python" not in loaders: + raise ValueError(f"kernel {name!r}: the 'python' implementation is needed") + self.__name__ = name + self._loaders = dict(loaders) + self._loaded: dict[str, Callable[..., Any]] = {} + self.errors: dict[str, BaseException] = {} + self._default: str | None = None + + def __repr__(self) -> str: + return f"HostImplementations({self.__name__!r}, {list(self.names)})" + + @property + def names(self) -> tuple[str, ...]: + """The implementations this kernel has, loaded or not.""" + return tuple(name for name in HOST_IMPLEMENTATIONS if name in self._loaders) + + @property + def python(self) -> Callable[..., Any]: + """The uncompiled Python function (for signatures and reference results).""" + return self.get("python") + + def available(self, name: str) -> bool: + """Whether implementation `name` exists and loads (loads it now).""" + if name not in self._loaders: + return False + try: + self.get(name) + except Exception: # noqa: BLE001 -- recorded in errors + return False + return True + + def get(self, name: str) -> Callable[..., Any]: + """Implementation `name`, loaded on first use. + + Raises + ------ + LookupError + If the kernel has no such implementation, or it failed to load + (the load error is the cause). + """ + function = self._loaded.get(name) + if function is not None: + return function + if name not in self._loaders: + raise LookupError( + f"kernel {self.__name__!r} has no {name!r} implementation " + f"(it has {', '.join(self.names)})" + ) + if name in self.errors: + raise LookupError( + f"the {name!r} implementation of kernel {self.__name__!r} is " + f"unavailable: {self.errors[name]!r}" + ) from self.errors[name] + try: + function = self._loaders[name]() + except Exception as error: + self.errors[name] = error + raise LookupError( + f"the {name!r} implementation of kernel {self.__name__!r} is " + f"unavailable: {error!r}" + ) from error + self._loaded[name] = function + return function + + def selected(self) -> str: + """The implementation a call runs now (loads the default on first use).""" + if _KERNEL_IMPLEMENTATION is not None: + return _KERNEL_IMPLEMENTATION + if self._default is None: + for name in ("pyccel", "numba", "numpy"): + if self.available(name): + self._default = name + break + else: + warnings.warn( + f"Kernel {self.__name__!r}: no compiled or NumPy implementation " + f"is available ({self.errors!r}); running the uncompiled " + "Python version.", + RuntimeWarning, + stacklevel=3, + ) + self._default = "python" + return self._default + + def build(self) -> str: + """Load the default implementation now, e.g. compile at setup; its name.""" + return self.selected() + + def __call__(self, *args: Any, **kwargs: Any) -> Any: + return self.get(self.selected())(*args, **kwargs) class CompiledHostKernel: @@ -619,12 +760,7 @@ def fallback(self) -> Callable[..., Any] | None: return self._fallback def build(self) -> Callable[..., Any] | None: - """Compile now (once); the compiled function, or None if that failed. - - None also inside :func:`force_host_fallback`, without compiling. - """ - if _FORCE_FALLBACK: - return None + """Compile now (once); the compiled function, or None if that failed.""" if not self._built: self._built = True try: diff --git a/tests/unit/test_kernel_dispatch_arrays.py b/tests/unit/test_kernel_dispatch_arrays.py index 3e0563a..d3c18cb 100644 --- a/tests/unit/test_kernel_dispatch_arrays.py +++ b/tests/unit/test_kernel_dispatch_arrays.py @@ -1,5 +1,6 @@ """Kernels chosen by where the arguments live, compiled host kernels, signature checks.""" +import importlib import sys import textwrap import warnings @@ -13,6 +14,7 @@ CompiledHostKernel, CudaArguments, CudaKernel, + HostImplementations, Kernel, KernelArguments, KernelCatalog, @@ -247,7 +249,8 @@ def compiler(module): ) kernel = catalog["scale"] assert kernel.dispatch == "arrays" - assert isinstance(kernel.host_kernel.kernel, CompiledHostKernel) + assert isinstance(kernel.host_kernel.kernel, HostImplementations) + assert kernel.implementations == ("pyccel", "python", "cuda") catalog.check_signatures() # x, factor, n on both sides x = np.ones(3) kernel(x, 2.0, 3) @@ -299,7 +302,11 @@ def test_arrays_dispatch_on_gpu(): @pytest.fixture def self_declaring_package(tmp_path, monkeypatch): - """`scale/__init__.py` declares its kernel with Kernel.from_folder.""" + """`scale/__init__.py` declares its kernel with Kernel.from_folder. + + The folder has a pyccel, a NumPy, a numba (whose import fails) and a CUDA + version of `scale`. + """ root = tmp_path / "demo_folder_pkg" (root / "scale").mkdir(parents=True) (root / "__init__.py").write_text("") @@ -309,55 +316,139 @@ def self_declaring_package(tmp_path, monkeypatch): " x[i] *= factor\n" ) (root / "scale" / "scale_numpy.py").write_text( - "def scale(x, factor, n):\n x[:n] *= factor\n" + "CALLS = []\n\n" + "def scale(x, factor, n):\n" + " CALLS.append(n)\n" + " x[:n] *= factor\n" + ) + (root / "scale" / "scale_numba.py").write_text( + "import a_jit_library_that_is_not_installed\n" ) (root / "scale" / "scale_cuda.cu").write_text(SCALE_CUDA) (root / "scale" / "__init__.py").write_text( - "import cunumpy as xp\n" - "from demo_folder_pkg.scale.scale_numpy import scale as _numpy\n\n" + "import cunumpy as xp\n\n" "COMPILED = []\n\n" "def _compile(module):\n" " COMPILED.append(module.__name__)\n" " return module\n\n" "kernel = xp.Kernel.from_folder(\n" " __name__, host_suffix='_pyccel', dispatch='arrays',\n" - " compile_host=_compile, fallback=_numpy, n_threads_from='first_array',\n" + " compile_host=_compile, n_threads_from='first_array',\n" ")\n" ) monkeypatch.syspath_prepend(str(tmp_path)) - yield "demo_folder_pkg" + yield importlib.import_module("demo_folder_pkg.scale") for module in [m for m in sys.modules if m.startswith("demo_folder_pkg")]: del sys.modules[module] -def test_from_folder_in_the_folders_own_init(self_declaring_package): - import importlib - - folder = importlib.import_module(f"{self_declaring_package}.scale") +def test_from_folder_finds_every_implementation(self_declaring_package): + folder = self_declaring_package kernel = folder.kernel assert kernel.name == "scale" and kernel.dispatch == "arrays" + assert kernel.implementations == ("pyccel", "numba", "numpy", "python", "cuda") assert kernel.cuda_kernel.n_threads_from is not None - host = kernel.host_kernel.kernel - assert isinstance(host, CompiledHostKernel) - assert ( - host.fallback - is sys.modules[f"{self_declaring_package}.scale.scale_numpy"].scale - ) - kernel.check_signature() # host, fallback and CUDA kernel: x, factor, n + kernel.check_signature() # pyccel, NumPy and CUDA: x, factor, n; numba skipped + assert kernel.selected() == "pyccel" # the compiled version by default + assert kernel.selected(device=True) == "cuda" x = np.ones(3) kernel(x, 2.0, 3) assert x.tolist() == [2.0] * 3 - assert folder.COMPILED == [f"{self_declaring_package}.scale.scale_pyccel"] + assert folder.COMPILED == ["demo_folder_pkg.scale.scale_pyccel"] # the catalog of the parent package builds the same kernel - catalog = KernelCatalog.from_package(self_declaring_package, host_suffix="_pyccel") - assert catalog["scale"].host_parameters() == kernel.host_parameters() + catalog = KernelCatalog.from_package("demo_folder_pkg", host_suffix="_pyccel") + assert catalog["scale"].implementations == kernel.implementations def test_from_folder_needs_a_kernel_folder(self_declaring_package): with pytest.raises(FileNotFoundError, match="no host kernel scale_kernels.py"): - Kernel.from_folder(f"{self_declaring_package}.scale") # default suffix + Kernel.from_folder("demo_folder_pkg.scale") # default suffix with pytest.raises(ModuleNotFoundError, match="not a package"): - Kernel.from_folder(f"{self_declaring_package}.scale.scale_numpy") + Kernel.from_folder("demo_folder_pkg.scale.scale_numpy") + + +def test_kernel_implementation_setting(self_declaring_package): + kernel = self_declaring_package.kernel + calls = importlib.import_module("demo_folder_pkg.scale.scale_numpy").CALLS + assert xp.get_kernel_implementation() is None + with xp.use_kernel_implementation("numpy"): + assert xp.get_kernel_implementation() == "numpy" + assert kernel.selected() == "numpy" + x = np.ones(2) + kernel(x, 3.0, 2) + assert x.tolist() == [3.0, 3.0] and calls == [2] + with xp.use_kernel_implementation("python"): + kernel(x, 2.0, 2) # the uncompiled pyccel source + assert calls == [2] and x.tolist() == [6.0, 6.0] + assert xp.get_kernel_implementation() is None + assert self_declaring_package.COMPILED == [] # pyccel never needed + # a chosen implementation that cannot run raises instead of running another + xp.set_kernel_implementation("numba") + try: + with pytest.raises(LookupError, match="'numba' implementation .* unavailable"): + kernel(np.ones(1), 2.0, 1) + finally: + xp.set_kernel_implementation(None) + with pytest.raises(ValueError, match="kernel implementation must be one of"): + xp.set_kernel_implementation("fortran") + + +def test_default_skips_unavailable_implementations(): + def no_pyccel(): + raise ImportError("pyccel missing") + + used = [] + with_numpy = HostImplementations( + "scale", + { + "pyccel": no_pyccel, + "numpy": lambda: used.append, + "python": lambda: scale, + }, + ) + assert with_numpy.selected() == "numpy" and not with_numpy.available("pyccel") + assert isinstance(with_numpy.errors["pyccel"], ImportError) + with_numpy("x") + assert used == ["x"] + only_python = HostImplementations( + "scale", {"pyccel": no_pyccel, "python": lambda: scale} + ) + x = np.ones(2) + with pytest.warns(RuntimeWarning, match="uncompiled Python version"): + only_python(x, 2.0, 2) + with warnings.catch_warnings(): + warnings.simplefilter("error") + only_python(x, 2.0, 2) # warned once + assert x.tolist() == [4.0, 4.0] + with pytest.raises(LookupError, match="has no 'numba' implementation"): + only_python.get("numba") + with pytest.raises(ValueError, match="unknown implementations"): + HostImplementations("scale", {"python": lambda: scale, "julia": lambda: scale}) + + +def test_kernel_implementation_environment_variable(): + import os + import subprocess + + code = "import cunumpy as xp; print(xp.get_kernel_implementation())" + env = {**os.environ, "CUNUMPY_KERNEL_IMPLEMENTATION": "numpy"} + printed = subprocess.run( + [sys.executable, "-c", code], + env=env, + capture_output=True, + text=True, + check=True, + ).stdout + assert printed.strip() == "numpy" + env["CUNUMPY_KERNEL_IMPLEMENTATION"] = "fortran" + failed = subprocess.run( + [sys.executable, "-c", code], + env=env, + capture_output=True, + text=True, + check=False, + ) + assert failed.returncode != 0 and "kernel implementation must be" in failed.stderr def test_arrays_dispatch_calls_the_host_kernel_without_conversion( @@ -383,55 +474,22 @@ def no_conversion(self, args, kwargs): forced(x, 2.0, 2) -def test_check_signature_covers_the_fallback(): - module = _module("m", "def scale(x, factor, n):\n x *= factor\n") - - def fallback(x, n, factor): +def test_check_signature_covers_every_host_implementation(): + def swapped(x, n, factor): pass - kernel = Kernel(CompiledHostKernel(module, "scale", lambda m: m, fallback)) - with pytest.raises(ValueError, match=r"its fallback \(x, n, factor\)"): + kernel = Kernel( + HostImplementations( + "scale", {"python": lambda: scale, "numpy": lambda: swapped} + ), + CudaKernel(SCALE_CUDA, "scale"), + ) + with pytest.raises(ValueError, match=r"the numpy version \(x, n, factor\)"): kernel.check_signature() - Kernel( - CompiledHostKernel(module, "scale", lambda m: m, lambda x, factor, n: None) - ).check_signature() - - -def test_force_host_fallback(): - module = _module("m", "def double(x):\n x *= 2\n") - compiled = [] - - def compiler(mod): - compiled.append(mod) - return mod - - used = [] - kernel = CompiledHostKernel(module, "double", compiler, fallback=used.append) - with xp.force_host_fallback(): - assert not kernel.compiled - kernel("x") - with xp.force_host_fallback(False): - assert kernel.compiled # compiles once, kept for later - assert used == ["x"] and compiled == [module] - with xp.force_host_fallback(): - assert not kernel.compiled # the compiled version is kept, not used - assert kernel.compiled and compiled == [module] - - -def test_host_fallback_environment_variable(): - import os - import subprocess - - code = "import cunumpy.kernel as k; print(k._FORCE_FALLBACK)" - env = {**os.environ, "CUNUMPY_HOST_FALLBACK": "1"} - printed = subprocess.run( - [sys.executable, "-c", code], - env=env, - capture_output=True, - text=True, - check=True, - ).stdout - assert printed.strip() == "True" + module = _module("m", "def scale(x, factor, n):\n x *= factor\n") + compiled = Kernel(CompiledHostKernel(module, "scale", lambda m: m, swapped)) + with pytest.raises(ValueError, match=r"its fallback \(x, n, factor\)"): + compiled.check_signature() # --------------------------------------------------------------------------- From c62a21d04af89342f3126fba5726d16a05c4f8fd Mon Sep 17 00:00:00 2001 From: Max Date: Sat, 3 Oct 2026 13:28:14 +0200 Subject: [PATCH 3/3] black --- src/cunumpy/dispatch.py | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/src/cunumpy/dispatch.py b/src/cunumpy/dispatch.py index 6bdf744..796f2f8 100644 --- a/src/cunumpy/dispatch.py +++ b/src/cunumpy/dispatch.py @@ -111,9 +111,11 @@ def _import_loader(module: str, name: str) -> Callable[[], Callable[..., Any]]: def _fallback_loader( - host_fallback: Mapping[str, Callable[..., Any]] - | Callable[[str], Callable[..., Any] | None] - | None, + host_fallback: ( + Mapping[str, Callable[..., Any]] + | Callable[[str], Callable[..., Any] | None] + | None + ), name: str, ) -> dict[str, Callable[[], Callable[..., Any]]]: """`from_package`'s `host_fallback` for kernel `name`, as its NumPy implementation."""