From bfae91e20cd53e8d93fdb9badd096574f93c6c89 Mon Sep 17 00:00:00 2001 From: Max Date: Mon, 5 Oct 2026 09:35:59 +0200 Subject: [PATCH 1/5] Renamed CUNUMPY_KERNEL_IMPLEMENTATION to CUNUMPY_HOST_KERNEL_IMPLEMENTATION --- CHANGELOG.md | 6 +++ docs/source/api.md | 2 +- docs/source/installation.md | 8 +++- docs/source/kernels/dispatch.md | 4 +- src/cunumpy/LLM_GUIDE.md | 2 +- src/cunumpy/_kernel.py | 4 +- tests/unit/test_kernel_dispatch_arrays.py | 56 ++++++++++++++++++----- 7 files changed, 64 insertions(+), 18 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 46cfb95..a88d646 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Changed +- Rename `CUNUMPY_KERNEL_IMPLEMENTATION` to + `CUNUMPY_HOST_KERNEL_IMPLEMENTATION` to make its host-only scope explicit. + The former environment variable is no longer read; update job scripts. + The Python selection functions retain their names and CUDA dispatch is unchanged. + ### Added - CUDA launches infer thread counts from the first array by default, including arrays in argument objects. 1D blocks use rows; multidimensional blocks use diff --git a/docs/source/api.md b/docs/source/api.md index c7c1c4c..72acc9e 100644 --- a/docs/source/api.md +++ b/docs/source/api.md @@ -1735,7 +1735,7 @@ or raises `LookupError` (missing, or failed to load with the error as cause), uncompiled function, `build()` loads the default now. A call runs `selected()`: the implementation set with `set_kernel_implementation(name)` (or `use_kernel_implementation`, or the environment variable -`CUNUMPY_KERNEL_IMPLEMENTATION` read at import), which raises if the kernel +`CUNUMPY_HOST_KERNEL_IMPLEMENTATION` read at import), which raises if the kernel lacks it or cannot load it, else the default: the first available of pyccel, numba and NumPy, and else `"python"` with a `RuntimeWarning` (once). `get_kernel_implementation()` reads the setting; `None` is the default. The diff --git a/docs/source/installation.md b/docs/source/installation.md index 5d06ee4..7c8d2ed 100644 --- a/docs/source/installation.md +++ b/docs/source/installation.md @@ -69,7 +69,7 @@ Tests that need a GPU are skipped automatically where CuPy is not functional. | --- | --- | | `CUNUMPY_BACKEND=cupy` | start with the CuPy backend instead of NumPy (read once, at import) | | `CUNUMPY_CUDA_DEBUG=1` | enable [CUDA debug mode](kernels/debugging.md) for all kernels | -| `CUNUMPY_KERNEL_IMPLEMENTATION=numpy` | choose the host kernel implementation (read at import) | +| `CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy` | choose the host kernel implementation (read at import) | | `CUNUMPY_MPI=1` / `0` | require MPI / use serial MPI regardless of launcher detection | | `CUNUMPY_FAKE_CUPY=1` | install the strict CPU stand-in for CuPy for tests | | `CUNUMPY_REQUIRE_CUDA=1` | require a real usable GPU when starting the test suite (CI guard) | @@ -78,6 +78,12 @@ Use `CUNUMPY_BACKEND` in job scripts; the former `ARRAY_BACKEND` setting is no longer read. Standard toolchain/device variables such as `CXX` and `CUDA_VISIBLE_DEVICES` keep their standard meanings. +Use `CUNUMPY_HOST_KERNEL_IMPLEMENTATION` instead of the former +`CUNUMPY_KERNEL_IMPLEMENTATION`, which is no longer read. This selects host +implementations only; CUDA dispatch is unaffected. The Python functions +`set_kernel_implementation`, `get_kernel_implementation`, and +`use_kernel_implementation` retain their names. + MPI launchers also export node-local rank variables (`OMPI_COMM_WORLD_LOCAL_RANK`, `SLURM_LOCALID`, ...), which `xp.mpi.local_rank()` reads to pick a GPU per process. diff --git a/docs/source/kernels/dispatch.md b/docs/source/kernels/dispatch.md index d7a19b0..e1fdf3f 100644 --- a/docs/source/kernels/dispatch.md +++ b/docs/source/kernels/dispatch.md @@ -179,7 +179,7 @@ with xp.kernels.use_kernel_implementation("numba"): # like xp.use_backend xp.kernels.set_kernel_implementation(None) # back to the default ``` -or `CUNUMPY_KERNEL_IMPLEMENTATION=numpy` for a whole run (read at import, like +or `CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy` for a whole run (read at import, like `CUNUMPY_BACKEND`). A chosen implementation that a kernel does not have, or cannot load, raises `LookupError` instead of running another one: a benchmark of numba never silently measures NumPy. `kernel.implementations` lists the @@ -219,7 +219,7 @@ host implementations" above); `catalog["push"].host_kernel.kernel.available("pyc reports whether the compiled version builds, and `catalog["push"].selected()` which version runs. To test the path of a machine without Pyccel, run the code inside `with xp.kernels.use_kernel_implementation("numpy"):` (or set -`CUNUMPY_KERNEL_IMPLEMENTATION=numpy` for a whole run). Note that `epyccel` compiles again on every call; a +`CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy` for a whole run). Note that `epyccel` compiles again on every call; a code that compiles at run time usually keeps the builds in an on-disk cache keyed on the module source, so that only the first run after an edit compiles. diff --git a/src/cunumpy/LLM_GUIDE.md b/src/cunumpy/LLM_GUIDE.md index e9a1b2d..0cea1cd 100644 --- a/src/cunumpy/LLM_GUIDE.md +++ b/src/cunumpy/LLM_GUIDE.md @@ -78,7 +78,7 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository. | host arrays reach kernels while CuPy is active | `Kernel(..., dispatch="arrays")` / `from_package(..., dispatch="arrays")`: CUDA only for device arguments | | one kernel folder declares its kernel in its own `__init__.py` | `kernel = xp.kernels.Kernel.from_folder(__name__, host_suffix="_pyccel", compile_host=..., dispatch="arrays")`; `_numba.py`, `_numpy.py` in the folder are further host implementations | | bring a `dispatch="arrays"` kernel's arguments to the side of the main array | `xp.kernels.as_kernel_array(a, like=grid, dtype=float)`; outputs: `with xp.kernels.kernel_output(out, like=grid, dtype=float) as buf:` | -| choose the host implementation (pyccel/numba/numpy/python) | `xp.kernels.set_kernel_implementation("numpy")`, `with xp.kernels.use_kernel_implementation(...)`, `CUNUMPY_KERNEL_IMPLEMENTATION=numpy`; default: first available of pyccel, numba, numpy; `kernel.selected()` | +| choose the host implementation (pyccel/numba/numpy/python) | `xp.kernels.set_kernel_implementation("numpy")`, `with xp.kernels.use_kernel_implementation(...)`, `CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy`; default: first available of pyccel, numba, numpy; `kernel.selected()` | | check host and CUDA kernels take the same parameters | `catalog.check_signatures()` (in a unit test) | | test a CUDA kernel's arithmetic without a GPU | `cunumpy.kernel_testing.emulate_cuda_kernel(kernel, *numpy_args, n_threads=n)` (C++ compiler; shared memory and __syncthreads ok, no warp ops; `shared_mem=` for extern shared) | | shared-memory budget of a block | `xp.cuda.max_shared_memory_per_block()` (48 KiB without a GPU) | diff --git a/src/cunumpy/_kernel.py b/src/cunumpy/_kernel.py index d626273..13291a3 100644 --- a/src/cunumpy/_kernel.py +++ b/src/cunumpy/_kernel.py @@ -546,7 +546,7 @@ def _check_implementation(name: str | None) -> str | None: _KERNEL_IMPLEMENTATION: str | None = _check_implementation( - os.environ.get("CUNUMPY_KERNEL_IMPLEMENTATION", "").strip().lower() or None, + os.environ.get("CUNUMPY_HOST_KERNEL_IMPLEMENTATION", "").strip().lower() or None, ) @@ -559,7 +559,7 @@ def set_kernel_implementation(name: str | None) -> None: or whose chosen implementation is unavailable (e.g. pyccel failed to compile), raises instead of running another one. CUDA kernels are not affected: device arrays always run the CUDA version. The environment - variable ``CUNUMPY_KERNEL_IMPLEMENTATION`` (read when cunumpy is imported) + variable ``CUNUMPY_HOST_KERNEL_IMPLEMENTATION`` (read when cunumpy is imported) sets it for a whole run. """ global _KERNEL_IMPLEMENTATION diff --git a/tests/unit/test_kernel_dispatch_arrays.py b/tests/unit/test_kernel_dispatch_arrays.py index e28f60e..8b3f427 100644 --- a/tests/unit/test_kernel_dispatch_arrays.py +++ b/tests/unit/test_kernel_dispatch_arrays.py @@ -4,6 +4,7 @@ import sys import textwrap import warnings +from pathlib import Path from types import SimpleNamespace import numpy as np @@ -437,21 +438,54 @@ def no_pyccel(): HostImplementations("scale", {"python": lambda: scale, "julia": lambda: scale}) -def test_kernel_implementation_environment_variable(): +@pytest.mark.parametrize( + "implementation,legacy,expected", + [ + (None, None, None), + ("", None, None), + ("pyccel", None, "pyccel"), + ("numba", None, "numba"), + (" NuMpY ", None, "numpy"), + ("python", None, "python"), + (None, "numpy", None), + ("numpy", "fortran", "numpy"), + ], +) +def test_host_kernel_implementation_environment_variable( + implementation, legacy, expected +): + import os + import subprocess + + env = dict(os.environ) + env["PYTHONPATH"] = str(Path(xp.__file__).parents[1]) + env.pop("CUNUMPY_HOST_KERNEL_IMPLEMENTATION", None) + env.pop("CUNUMPY_KERNEL_IMPLEMENTATION", None) + if implementation is not None: + env["CUNUMPY_HOST_KERNEL_IMPLEMENTATION"] = implementation + if legacy is not None: + env["CUNUMPY_KERNEL_IMPLEMENTATION"] = legacy + code = ( + "import os, cunumpy as xp\n" + f"assert xp.kernels.get_kernel_implementation() == {expected!r}\n" + "os.environ['CUNUMPY_HOST_KERNEL_IMPLEMENTATION'] = 'python'\n" + f"assert xp.kernels.get_kernel_implementation() == {expected!r}\n" + "with xp.kernels.use_kernel_implementation('numpy'):\n" + " assert xp.kernels.get_kernel_implementation() == 'numpy'\n" + f"assert xp.kernels.get_kernel_implementation() == {expected!r}\n" + "xp.kernels.set_kernel_implementation(None)\n" + "assert xp.kernels.get_kernel_implementation() is None\n" + ) + subprocess.run([sys.executable, "-c", code], env=env, check=True) + + +def test_invalid_host_kernel_implementation_environment_variable(): import os import subprocess code = "import cunumpy as xp; print(xp.kernels.get_kernel_implementation())" - env = {**os.environ, "CUNUMPY_KERNEL_IMPLEMENTATION": "numpy"} - printed = subprocess.run( - [sys.executable, "-c", code], - env=env, - capture_output=True, - text=True, - check=True, - ).stdout - assert printed.strip() == "numpy" - env["CUNUMPY_KERNEL_IMPLEMENTATION"] = "fortran" + env = {**os.environ, "CUNUMPY_HOST_KERNEL_IMPLEMENTATION": "fortran"} + env["PYTHONPATH"] = str(Path(xp.__file__).parents[1]) failed = subprocess.run( [sys.executable, "-c", code], env=env, From 102da5daddd18ac41bb8f22b20b2b8463fef8994 Mon Sep 17 00:00:00 2001 From: Max Date: Mon, 5 Oct 2026 09:44:03 +0200 Subject: [PATCH 2/5] Updated get/set for host kernel implementations --- CHANGELOG.md | 5 +++- docs/source/api.md | 12 ++++----- docs/source/installation.md | 4 +-- docs/source/kernels/dispatch.md | 8 +++--- src/cunumpy/LLM_GUIDE.md | 2 +- src/cunumpy/__init__.py | 9 ++++++- src/cunumpy/_dispatch.py | 4 +-- src/cunumpy/_kernel.py | 14 +++++----- src/cunumpy/kernels.py | 16 ++++++------ tests/unit/test_kernel_dispatch_arrays.py | 32 +++++++++++------------ 10 files changed, 58 insertions(+), 48 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a88d646..6fb1db9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,7 +11,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Rename `CUNUMPY_KERNEL_IMPLEMENTATION` to `CUNUMPY_HOST_KERNEL_IMPLEMENTATION` to make its host-only scope explicit. The former environment variable is no longer read; update job scripts. - The Python selection functions retain their names and CUDA dispatch is unchanged. + CUDA dispatch is unchanged. +- Rename the `kernels` selection functions to `set_host_kernel_implementation`, + `get_host_kernel_implementation`, and `use_host_kernel_implementation`. + The former function names are removed without compatibility aliases. ### Added - CUDA launches infer thread counts from the first array by default, including diff --git a/docs/source/api.md b/docs/source/api.md index 72acc9e..79d63f5 100644 --- a/docs/source/api.md +++ b/docs/source/api.md @@ -1713,7 +1713,7 @@ files, as loaders by name (`{"numpy": lambda: push_numpy}`). Takes the options o no host kernel module and `ModuleNotFoundError` if `package` is not a package. `kernel.selected(device=True)` names the implementation for device arguments. -## `kernels.HostImplementations`, `kernels.set_kernel_implementation` +## `kernels.HostImplementations`, `kernels.set_host_kernel_implementation` ```python host = xp.kernels.HostImplementations( @@ -1721,8 +1721,8 @@ host = xp.kernels.HostImplementations( {"pyccel": load_compiled, "numpy": lambda: push_numpy, "python": lambda: push}, ) host(*args) # the default implementation -xp.kernels.set_kernel_implementation("numpy") # every kernel: like xp.set_backend -with xp.kernels.use_kernel_implementation("python"): # like xp.use_backend +xp.kernels.set_host_kernel_implementation("numpy") # every kernel: like xp.set_backend +with xp.kernels.use_host_kernel_implementation("python"): # like xp.use_backend host(*args) ``` @@ -1733,12 +1733,12 @@ Loaded on first use; `available(name)` loads and reports, `get(name)` returns it or raises `LookupError` (missing, or failed to load with the error as cause), `errors` maps names to load errors, `names` lists them, `python` is the uncompiled function, `build()` loads the default now. A call runs -`selected()`: the implementation set with `set_kernel_implementation(name)` (or -`use_kernel_implementation`, or the environment variable +`selected()`: the implementation set with `set_host_kernel_implementation(name)` (or +`use_host_kernel_implementation`, or the environment variable `CUNUMPY_HOST_KERNEL_IMPLEMENTATION` read at import), which raises if the kernel lacks it or cannot load it, else the default: the first available of pyccel, numba and NumPy, and else `"python"` with a `RuntimeWarning` (once). -`get_kernel_implementation()` reads the setting; `None` is the default. The +`get_host_kernel_implementation()` reads the setting; `None` is the default. The setting is global, not per thread, and applies to host calls only. ## `kernels.CompiledHostKernel` diff --git a/docs/source/installation.md b/docs/source/installation.md index 7c8d2ed..3ccf82a 100644 --- a/docs/source/installation.md +++ b/docs/source/installation.md @@ -81,8 +81,8 @@ longer read. Standard toolchain/device variables such as `CXX` and Use `CUNUMPY_HOST_KERNEL_IMPLEMENTATION` instead of the former `CUNUMPY_KERNEL_IMPLEMENTATION`, which is no longer read. This selects host implementations only; CUDA dispatch is unaffected. The Python functions -`set_kernel_implementation`, `get_kernel_implementation`, and -`use_kernel_implementation` retain their names. +`set_host_kernel_implementation`, `get_host_kernel_implementation`, and +`use_host_kernel_implementation` provide runtime selection under `xp.kernels`. MPI launchers also export node-local rank variables (`OMPI_COMM_WORLD_LOCAL_RANK`, `SLURM_LOCALID`, ...), which `xp.mpi.local_rank()` reads to pick a GPU per process. diff --git a/docs/source/kernels/dispatch.md b/docs/source/kernels/dispatch.md index e1fdf3f..b4d1a10 100644 --- a/docs/source/kernels/dispatch.md +++ b/docs/source/kernels/dispatch.md @@ -173,10 +173,10 @@ none is. Device arrays run the CUDA kernel. To choose, use the same pattern as for the array backend: ```python -xp.kernels.set_kernel_implementation("numpy") # like xp.set_backend -with xp.kernels.use_kernel_implementation("numba"): # like xp.use_backend +xp.kernels.set_host_kernel_implementation("numpy") # like xp.set_backend +with xp.kernels.use_host_kernel_implementation("numba"): # like xp.use_backend push(positions, velocities, dt) -xp.kernels.set_kernel_implementation(None) # back to the default +xp.kernels.set_host_kernel_implementation(None) # back to the default ``` or `CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy` for a whole run (read at import, like @@ -218,7 +218,7 @@ catalog = xp.kernels.KernelCatalog.from_package( host implementations" above); `catalog["push"].host_kernel.kernel.available("pyccel")` reports whether the compiled version builds, and `catalog["push"].selected()` which version runs. To test the path of a machine without Pyccel, run the code -inside `with xp.kernels.use_kernel_implementation("numpy"):` (or set +inside `with xp.kernels.use_host_kernel_implementation("numpy"):` (or set `CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy` for a whole run). Note that `epyccel` compiles again on every call; a code that compiles at run time usually keeps the builds in an on-disk cache keyed on the module source, so that only the first run after an edit compiles. diff --git a/src/cunumpy/LLM_GUIDE.md b/src/cunumpy/LLM_GUIDE.md index 0cea1cd..041c2db 100644 --- a/src/cunumpy/LLM_GUIDE.md +++ b/src/cunumpy/LLM_GUIDE.md @@ -78,7 +78,7 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository. | host arrays reach kernels while CuPy is active | `Kernel(..., dispatch="arrays")` / `from_package(..., dispatch="arrays")`: CUDA only for device arguments | | one kernel folder declares its kernel in its own `__init__.py` | `kernel = xp.kernels.Kernel.from_folder(__name__, host_suffix="_pyccel", compile_host=..., dispatch="arrays")`; `_numba.py`, `_numpy.py` in the folder are further host implementations | | bring a `dispatch="arrays"` kernel's arguments to the side of the main array | `xp.kernels.as_kernel_array(a, like=grid, dtype=float)`; outputs: `with xp.kernels.kernel_output(out, like=grid, dtype=float) as buf:` | -| choose the host implementation (pyccel/numba/numpy/python) | `xp.kernels.set_kernel_implementation("numpy")`, `with xp.kernels.use_kernel_implementation(...)`, `CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy`; default: first available of pyccel, numba, numpy; `kernel.selected()` | +| choose the host implementation (pyccel/numba/numpy/python) | `xp.kernels.set_host_kernel_implementation("numpy")`, `with xp.kernels.use_host_kernel_implementation(...)`, `CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy`; default: first available of pyccel, numba, numpy; `kernel.selected()` | | check host and CUDA kernels take the same parameters | `catalog.check_signatures()` (in a unit test) | | test a CUDA kernel's arithmetic without a GPU | `cunumpy.kernel_testing.emulate_cuda_kernel(kernel, *numpy_args, n_threads=n)` (C++ compiler; shared memory and __syncthreads ok, no warp ops; `shared_mem=` for extern shared) | | shared-memory budget of a block | `xp.cuda.max_shared_memory_per_block()` (48 KiB without a GPU) | diff --git a/src/cunumpy/__init__.py b/src/cunumpy/__init__.py index 24297ab..c0d8471 100644 --- a/src/cunumpy/__init__.py +++ b/src/cunumpy/__init__.py @@ -29,7 +29,14 @@ # moved to. They still resolve (with a DeprecationWarning) until cunumpy 0.6. _MOVED = { **dict.fromkeys(cuda.__all__, "cuda"), - **dict.fromkeys(kernels.__all__, "kernels"), + **dict.fromkeys( + ( + name + for name in kernels.__all__ + if not name.endswith("_host_kernel_implementation") + ), + "kernels", + ), **dict.fromkeys(rng.__all__, "rng"), **dict.fromkeys(algorithms.__all__, "algorithms"), # the names of cunumpy.mpi that were at the top level (not the later ones) diff --git a/src/cunumpy/_dispatch.py b/src/cunumpy/_dispatch.py index 2fdcfa6..a8ec481 100644 --- a/src/cunumpy/_dispatch.py +++ b/src/cunumpy/_dispatch.py @@ -268,7 +268,7 @@ def from_folder( * ````: the CUDA kernel. The host implementations form a :class:`~cunumpy.kernels.HostImplementations`: - a call runs the one set with :func:`~cunumpy.kernels.set_kernel_implementation`, + a call runs the one set with :func:`~cunumpy.kernels.set_host_kernel_implementation`, or by default the first available of pyccel, numba and NumPy. The folder's own ``__init__.py`` can declare its kernel with this method, so that the kernel is imported from where it is written:: @@ -516,7 +516,7 @@ def selected(self, device: bool = False) -> str: """The implementation a call with host (or `device`) arguments runs now. For host arguments: the setting of - :func:`~cunumpy.kernels.set_kernel_implementation` or the default (loads it), or + :func:`~cunumpy.kernels.set_host_kernel_implementation` or the default (loads it), or ``"host"`` for a host kernel that is not a :class:`~cunumpy.kernels.HostImplementations`. For device arguments ``"cuda"``, or ``"host"`` if there is no CUDA kernel and ``missing_cuda="fallback"``. diff --git a/src/cunumpy/_kernel.py b/src/cunumpy/_kernel.py index 13291a3..d924a76 100644 --- a/src/cunumpy/_kernel.py +++ b/src/cunumpy/_kernel.py @@ -550,7 +550,7 @@ def _check_implementation(name: str | None) -> str | None: ) -def set_kernel_implementation(name: str | None) -> None: +def set_host_kernel_implementation(name: str | None) -> None: """Choose the host implementation every kernel runs, like :func:`set_backend`. ``"pyccel"``, ``"numba"``, ``"numpy"`` or ``"python"`` (the uncompiled @@ -566,22 +566,22 @@ def set_kernel_implementation(name: str | None) -> None: _KERNEL_IMPLEMENTATION = _check_implementation(name) -def get_kernel_implementation() -> str | None: - """The host implementation set with :func:`set_kernel_implementation`, or None.""" +def get_host_kernel_implementation() -> str | None: + """The host implementation set with :func:`set_host_kernel_implementation`, or None.""" return _KERNEL_IMPLEMENTATION @contextmanager -def use_kernel_implementation(name: str | None) -> Iterator[None]: +def use_host_kernel_implementation(name: str | None) -> Iterator[None]: """Temporarily choose the host implementation, like :func:`use_backend`. - For tests and benchmarks, e.g. ``with xp.kernels.use_kernel_implementation("numpy"):`` + For tests and benchmarks, e.g. ``with xp.kernels.use_host_kernel_implementation("numpy"):`` to run the code path of a machine without pyccel. The setting is global, not per thread. """ global _KERNEL_IMPLEMENTATION previous = _KERNEL_IMPLEMENTATION - set_kernel_implementation(name) + set_host_kernel_implementation(name) try: yield finally: @@ -594,7 +594,7 @@ class HostImplementations: Each implementation is loaded on first use (a pyccel build, an import) and may be unavailable (no compiler, numba not installed); a failed load is remembered with its exception. A call runs the implementation chosen with - :func:`set_kernel_implementation`, which must exist and load, or else the + :func:`set_host_kernel_implementation`, which must exist and load, or else the default: the first available of ``"pyccel"``, ``"numba"`` and ``"numpy"``, and as a last resort ``"python"``, with a warning (correct, but slow). Built by :meth:`Kernel.from_folder` from the files of a kernel folder. diff --git a/src/cunumpy/kernels.py b/src/cunumpy/kernels.py index 7f276c4..de3b20f 100644 --- a/src/cunumpy/kernels.py +++ b/src/cunumpy/kernels.py @@ -9,8 +9,8 @@ push = xp.kernels.Kernel.from_folder("my_code.kernels.push") push(positions, velocities, dt) -Which host implementation runs is set with :func:`set_kernel_implementation` -or :func:`use_kernel_implementation`. :func:`as_kernel_array` and +Which host implementation runs is set with :func:`set_host_kernel_implementation` +or :func:`use_host_kernel_implementation`. :func:`as_kernel_array` and :func:`kernel_output` bring the arguments of a kernel to the side of its main array. :func:`fuse` turns an elementwise function into one CuPy kernel. @@ -29,11 +29,11 @@ KernelArguments, PyccelKernel, as_kernel_array, - get_kernel_implementation, + get_host_kernel_implementation, kernel_output, resolve_host_args, - set_kernel_implementation, - use_kernel_implementation, + set_host_kernel_implementation, + use_host_kernel_implementation, ) __all__ = [ @@ -47,9 +47,9 @@ "PyccelStructArguments", "as_kernel_array", "fuse", - "get_kernel_implementation", + "get_host_kernel_implementation", "kernel_output", "resolve_host_args", - "set_kernel_implementation", - "use_kernel_implementation", + "set_host_kernel_implementation", + "use_host_kernel_implementation", ] diff --git a/tests/unit/test_kernel_dispatch_arrays.py b/tests/unit/test_kernel_dispatch_arrays.py index 8b3f427..8eb3515 100644 --- a/tests/unit/test_kernel_dispatch_arrays.py +++ b/tests/unit/test_kernel_dispatch_arrays.py @@ -381,27 +381,27 @@ def test_from_folder_needs_a_kernel_folder(self_declaring_package): def test_kernel_implementation_setting(self_declaring_package): kernel = self_declaring_package.kernel calls = importlib.import_module("demo_folder_pkg.scale.scale_numpy").CALLS - assert xp.kernels.get_kernel_implementation() is None - with xp.kernels.use_kernel_implementation("numpy"): - assert xp.kernels.get_kernel_implementation() == "numpy" + assert xp.kernels.get_host_kernel_implementation() is None + with xp.kernels.use_host_kernel_implementation("numpy"): + assert xp.kernels.get_host_kernel_implementation() == "numpy" assert kernel.selected() == "numpy" x = np.ones(2) kernel(x, 3.0, 2) assert x.tolist() == [3.0, 3.0] and calls == [2] - with xp.kernels.use_kernel_implementation("python"): + with xp.kernels.use_host_kernel_implementation("python"): kernel(x, 2.0, 2) # the uncompiled pyccel source assert calls == [2] and x.tolist() == [6.0, 6.0] - assert xp.kernels.get_kernel_implementation() is None + assert xp.kernels.get_host_kernel_implementation() is None assert self_declaring_package.COMPILED == [] # pyccel never needed # a chosen implementation that cannot run raises instead of running another - xp.kernels.set_kernel_implementation("numba") + xp.kernels.set_host_kernel_implementation("numba") try: with pytest.raises(LookupError, match="'numba' implementation .* unavailable"): kernel(np.ones(1), 2.0, 1) finally: - xp.kernels.set_kernel_implementation(None) + xp.kernels.set_host_kernel_implementation(None) with pytest.raises(ValueError, match="kernel implementation must be one of"): - xp.kernels.set_kernel_implementation("fortran") + xp.kernels.set_host_kernel_implementation("fortran") def test_default_skips_unavailable_implementations(): @@ -467,14 +467,14 @@ def test_host_kernel_implementation_environment_variable( env["CUNUMPY_KERNEL_IMPLEMENTATION"] = legacy code = ( "import os, cunumpy as xp\n" - f"assert xp.kernels.get_kernel_implementation() == {expected!r}\n" + f"assert xp.kernels.get_host_kernel_implementation() == {expected!r}\n" "os.environ['CUNUMPY_HOST_KERNEL_IMPLEMENTATION'] = 'python'\n" - f"assert xp.kernels.get_kernel_implementation() == {expected!r}\n" - "with xp.kernels.use_kernel_implementation('numpy'):\n" - " assert xp.kernels.get_kernel_implementation() == 'numpy'\n" - f"assert xp.kernels.get_kernel_implementation() == {expected!r}\n" - "xp.kernels.set_kernel_implementation(None)\n" - "assert xp.kernels.get_kernel_implementation() is None\n" + f"assert xp.kernels.get_host_kernel_implementation() == {expected!r}\n" + "with xp.kernels.use_host_kernel_implementation('numpy'):\n" + " assert xp.kernels.get_host_kernel_implementation() == 'numpy'\n" + f"assert xp.kernels.get_host_kernel_implementation() == {expected!r}\n" + "xp.kernels.set_host_kernel_implementation(None)\n" + "assert xp.kernels.get_host_kernel_implementation() is None\n" ) subprocess.run([sys.executable, "-c", code], env=env, check=True) @@ -483,7 +483,7 @@ def test_invalid_host_kernel_implementation_environment_variable(): import os import subprocess - code = "import cunumpy as xp; print(xp.kernels.get_kernel_implementation())" + code = "import cunumpy as xp; print(xp.kernels.get_host_kernel_implementation())" env = {**os.environ, "CUNUMPY_HOST_KERNEL_IMPLEMENTATION": "fortran"} env["PYTHONPATH"] = str(Path(xp.__file__).parents[1]) failed = subprocess.run( From 20e728974bf72850e0135cc0e685ef1c818a9b86 Mon Sep 17 00:00:00 2001 From: Max Date: Mon, 5 Oct 2026 10:00:20 +0200 Subject: [PATCH 3/5] Added CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION --- CHANGELOG.md | 5 ++ docs/source/api.md | 26 ++++++++ docs/source/installation.md | 1 + docs/source/kernels/dispatch.md | 19 ++++++ src/cunumpy/LLM_GUIDE.md | 1 + src/cunumpy/__init__.py | 5 +- src/cunumpy/_dispatch.py | 12 +++- src/cunumpy/_kernel.py | 47 +++++++++++++ src/cunumpy/kernels.py | 10 +++ tests/unit/test_device_implementation.py | 80 +++++++++++++++++++++++ tests/unit/test_kernel_dispatch_arrays.py | 34 ++++++++++ 11 files changed, 238 insertions(+), 2 deletions(-) create mode 100644 tests/unit/test_device_implementation.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 6fb1db9..94dd615 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -17,6 +17,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 The former function names are removed without compatibility aliases. ### Added +- Device implementation selection via `kernels.set_device_kernel_implementation`, + `get_device_kernel_implementation`, `use_device_kernel_implementation`, and + `CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION`. Accept `"cuda"` or automatic selection + (`None`); explicit CUDA selection rejects missing CUDA kernels instead of + falling back to the host. Unsupported implementation names raise. - CUDA launches infer thread counts from the first array by default, including arrays in argument objects. 1D blocks use rows; multidimensional blocks use matching leading shape axes. Explicit sizes and callbacks override inference. diff --git a/docs/source/api.md b/docs/source/api.md index 79d63f5..b172027 100644 --- a/docs/source/api.md +++ b/docs/source/api.md @@ -1741,6 +1741,32 @@ numba and NumPy, and else `"python"` with a `RuntimeWarning` (once). `get_host_kernel_implementation()` reads the setting; `None` is the default. The setting is global, not per thread, and applies to host calls only. +## Device kernel implementation selection + +```python +xp.kernels.set_device_kernel_implementation("cuda") +xp.kernels.get_device_kernel_implementation() # "cuda" +with xp.kernels.use_device_kernel_implementation(None): + push(positions, velocities, dt) # automatic selection +xp.kernels.set_device_kernel_implementation(None) +``` + +`xp.kernels.DEVICE_IMPLEMENTATIONS` is currently `("cuda",)`. The setter accepts +`"cuda"` or `None` (automatic selection); unsupported values raise `ValueError` +without changing the setting. The getter reports the requested setting, so +it returns `None` in automatic mode even when a call would use CUDA. +`CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION` initializes the setting at import; +an unset or empty value means automatic selection. Environment values are +case-insensitive and stripped of whitespace; unsupported values fail at import. + +Explicit `"cuda"` requires a CUDA implementation for device dispatch: missing +implementations raise `LookupError`, even with `missing_cuda="fallback"`. +Automatic mode preserves that per-kernel fallback policy. This affects +`Kernel` and `KernelCatalog` device dispatch; it does not switch the array +backend, alter host calls, or affect direct `CudaKernel`/CuPy RawKernel calls. +The context manager restores the previous setting even after an exception. +The setting is global, not per thread. + ## `kernels.CompiledHostKernel` ```python diff --git a/docs/source/installation.md b/docs/source/installation.md index 3ccf82a..9d06516 100644 --- a/docs/source/installation.md +++ b/docs/source/installation.md @@ -70,6 +70,7 @@ Tests that need a GPU are skipped automatically where CuPy is not functional. | `CUNUMPY_BACKEND=cupy` | start with the CuPy backend instead of NumPy (read once, at import) | | `CUNUMPY_CUDA_DEBUG=1` | enable [CUDA debug mode](kernels/debugging.md) for all kernels | | `CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy` | choose the host kernel implementation (read at import) | +| `CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION=cuda` | require CUDA for device kernel dispatch (read at import); unset allows the kernel's configured fallback | | `CUNUMPY_MPI=1` / `0` | require MPI / use serial MPI regardless of launcher detection | | `CUNUMPY_FAKE_CUPY=1` | install the strict CPU stand-in for CuPy for tests | | `CUNUMPY_REQUIRE_CUDA=1` | require a real usable GPU when starting the test suite (CI guard) | diff --git a/docs/source/kernels/dispatch.md b/docs/source/kernels/dispatch.md index b4d1a10..ff43991 100644 --- a/docs/source/kernels/dispatch.md +++ b/docs/source/kernels/dispatch.md @@ -187,6 +187,25 @@ implementations, `kernel.selected()` names the one a call with host arrays runs now (`kernel.selected(device=True)` for device arrays), and `kernel.host_kernel.kernel.errors` holds why an implementation failed to load. +### Device implementation selection + +Currently CUDA is the only device implementation. To require it explicitly: + +```python +xp.kernels.set_device_kernel_implementation("cuda") +assert xp.kernels.get_device_kernel_implementation() == "cuda" +with xp.kernels.use_device_kernel_implementation(None): + push(positions, velocities, dt) # automatic selection, normal fallback policy +xp.kernels.set_device_kernel_implementation(None) # restore automatic selection +``` + +`CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION=cuda` sets the same choice at import. +Unsupported values raise `ValueError`. An explicit CUDA choice raises +`LookupError` for a missing device implementation, including kernels configured +with `missing_cuda="fallback"`. The default (`None`, or an unset/empty environment +variable) preserves that fallback policy. This setting controls device dispatch; +the array backend and host implementation are selected independently. + ## Compiled Pyccel host kernels By default the host kernel is the Python function itself, which is fine for diff --git a/src/cunumpy/LLM_GUIDE.md b/src/cunumpy/LLM_GUIDE.md index 041c2db..630a4e3 100644 --- a/src/cunumpy/LLM_GUIDE.md +++ b/src/cunumpy/LLM_GUIDE.md @@ -79,6 +79,7 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository. | one kernel folder declares its kernel in its own `__init__.py` | `kernel = xp.kernels.Kernel.from_folder(__name__, host_suffix="_pyccel", compile_host=..., dispatch="arrays")`; `_numba.py`, `_numpy.py` in the folder are further host implementations | | bring a `dispatch="arrays"` kernel's arguments to the side of the main array | `xp.kernels.as_kernel_array(a, like=grid, dtype=float)`; outputs: `with xp.kernels.kernel_output(out, like=grid, dtype=float) as buf:` | | choose the host implementation (pyccel/numba/numpy/python) | `xp.kernels.set_host_kernel_implementation("numpy")`, `with xp.kernels.use_host_kernel_implementation(...)`, `CUNUMPY_HOST_KERNEL_IMPLEMENTATION=numpy`; default: first available of pyccel, numba, numpy; `kernel.selected()` | +| require CUDA for device kernel dispatch | `xp.kernels.set_device_kernel_implementation("cuda")`, `get_device_kernel_implementation()`, `with xp.kernels.use_device_kernel_implementation(...)`, `CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION=cuda`; default `None` preserves `missing_cuda` policy; explicit CUDA rejects host fallback | | check host and CUDA kernels take the same parameters | `catalog.check_signatures()` (in a unit test) | | test a CUDA kernel's arithmetic without a GPU | `cunumpy.kernel_testing.emulate_cuda_kernel(kernel, *numpy_args, n_threads=n)` (C++ compiler; shared memory and __syncthreads ok, no warp ops; `shared_mem=` for extern shared) | | shared-memory budget of a block | `xp.cuda.max_shared_memory_per_block()` (48 KiB without a GPU) | diff --git a/src/cunumpy/__init__.py b/src/cunumpy/__init__.py index c0d8471..935ac6d 100644 --- a/src/cunumpy/__init__.py +++ b/src/cunumpy/__init__.py @@ -33,7 +33,10 @@ ( name for name in kernels.__all__ - if not name.endswith("_host_kernel_implementation") + if not name.endswith( + ("_host_kernel_implementation", "_device_kernel_implementation") + ) + and name != "DEVICE_IMPLEMENTATIONS" ), "kernels", ), diff --git a/src/cunumpy/_dispatch.py b/src/cunumpy/_dispatch.py index a8ec481..2dfc41b 100644 --- a/src/cunumpy/_dispatch.py +++ b/src/cunumpy/_dispatch.py @@ -39,6 +39,7 @@ CompiledHostKernel, HostImplementations, PyccelKernel, + get_device_kernel_implementation, resolve_host_args, ) from cunumpy._transfers import _ACTIVE as _COUNTERS @@ -520,6 +521,8 @@ def selected(self, device: bool = False) -> str: ``"host"`` for a host kernel that is not a :class:`~cunumpy.kernels.HostImplementations`. For device arguments ``"cuda"``, or ``"host"`` if there is no CUDA kernel and ``missing_cuda="fallback"``. + Explicit CUDA selection rejects missing CUDA implementations with + ``LookupError``, including when host fallback is configured. Useful to check that a run does not use a slow path. """ if device: @@ -536,6 +539,8 @@ def get_kernel(self) -> PyccelKernel | CudaKernel: Raises ------ + LookupError + On the CuPy backend, if CUDA is explicitly selected but missing. NotImplementedError On the CuPy backend, if there is no CUDA kernel and ``missing_cuda="raise"``. @@ -545,9 +550,14 @@ def get_kernel(self) -> PyccelKernel | CudaKernel: return self._device_kernel() def _device_kernel(self) -> PyccelKernel | CudaKernel: - """The CUDA kernel, or what ``missing_cuda`` says without one.""" + """Resolve the device selection, honoring fallback only in automatic mode.""" if self._cuda_kernel is not None: return self._cuda_kernel + if get_device_kernel_implementation() == "cuda": + raise LookupError( + f"kernel {self._name!r} has no 'cuda' implementation " + "(explicitly selected device implementation)", + ) if self._missing_cuda == "raise": expected = ( "" if self._cuda_path is None else f" (expected {self._cuda_path})" diff --git a/src/cunumpy/_kernel.py b/src/cunumpy/_kernel.py index d924a76..ba34a33 100644 --- a/src/cunumpy/_kernel.py +++ b/src/cunumpy/_kernel.py @@ -588,6 +588,53 @@ def use_host_kernel_implementation(name: str | None) -> Iterator[None]: _KERNEL_IMPLEMENTATION = previous +DEVICE_IMPLEMENTATIONS = ("cuda",) + + +def _check_device_implementation(name: str | None) -> str | None: + if name is not None and name not in DEVICE_IMPLEMENTATIONS: + raise ValueError( + f"device kernel implementation must be one of {DEVICE_IMPLEMENTATIONS} " + f"or None, got {name!r}", + ) + return name + + +_DEVICE_KERNEL_IMPLEMENTATION = _check_device_implementation( + os.environ.get("CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION", "").strip().lower() or None, +) + + +def set_device_kernel_implementation(name: str | None) -> None: + """Choose the device implementation for dispatched kernels. + + Currently only ``"cuda"`` is supported; ``None`` restores automatic selection. + Explicit CUDA selection raises if a kernel has no CUDA implementation, even + with ``missing_cuda="fallback"``. This does not switch the array backend or + affect host calls or direct CudaKernel calls. The import-time environment + variable ``CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION`` initializes this setting. + """ + global _DEVICE_KERNEL_IMPLEMENTATION + _DEVICE_KERNEL_IMPLEMENTATION = _check_device_implementation(name) + + +def get_device_kernel_implementation() -> str | None: + """Return the requested device implementation, or None for automatic selection.""" + return _DEVICE_KERNEL_IMPLEMENTATION + + +@contextmanager +def use_device_kernel_implementation(name: str | None) -> Iterator[None]: + """Temporarily choose the device implementation; global, not per thread.""" + global _DEVICE_KERNEL_IMPLEMENTATION + previous = _DEVICE_KERNEL_IMPLEMENTATION + set_device_kernel_implementation(name) + try: + yield + finally: + _DEVICE_KERNEL_IMPLEMENTATION = previous + + class HostImplementations: """The host implementations of one kernel, run by name or by the default rule. diff --git a/src/cunumpy/kernels.py b/src/cunumpy/kernels.py index de3b20f..7b3b27c 100644 --- a/src/cunumpy/kernels.py +++ b/src/cunumpy/kernels.py @@ -13,6 +13,8 @@ or :func:`use_host_kernel_implementation`. :func:`as_kernel_array` and :func:`kernel_output` bring the arguments of a kernel to the side of its main array. :func:`fuse` turns an elementwise function into one CuPy kernel. +Device dispatch can require CUDA with :func:`set_device_kernel_implementation` +or temporarily with :func:`use_device_kernel_implementation`. The CUDA-only classes (:class:`~cunumpy.cuda.CudaKernel`, ...) are in :mod:`cunumpy.cuda`; the pytest helpers for kernel pairs are in @@ -23,20 +25,25 @@ from cunumpy._dispatch import Kernel, KernelCatalog from cunumpy._fusion import fuse from cunumpy._kernel import ( + DEVICE_IMPLEMENTATIONS, HOST_IMPLEMENTATIONS, CompiledHostKernel, HostImplementations, KernelArguments, PyccelKernel, as_kernel_array, + get_device_kernel_implementation, get_host_kernel_implementation, kernel_output, resolve_host_args, + set_device_kernel_implementation, set_host_kernel_implementation, + use_device_kernel_implementation, use_host_kernel_implementation, ) __all__ = [ + "DEVICE_IMPLEMENTATIONS", "HOST_IMPLEMENTATIONS", "CompiledHostKernel", "HostImplementations", @@ -47,9 +54,12 @@ "PyccelStructArguments", "as_kernel_array", "fuse", + "get_device_kernel_implementation", "get_host_kernel_implementation", "kernel_output", "resolve_host_args", + "set_device_kernel_implementation", "set_host_kernel_implementation", + "use_device_kernel_implementation", "use_host_kernel_implementation", ] diff --git a/tests/unit/test_device_implementation.py b/tests/unit/test_device_implementation.py new file mode 100644 index 0000000..92a589a --- /dev/null +++ b/tests/unit/test_device_implementation.py @@ -0,0 +1,80 @@ +"""Device implementation configuration without requiring a CUDA installation.""" + +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +import cunumpy as xp + + +def test_device_setting_validation_and_context_restoration(): + previous = xp.kernels.get_device_kernel_implementation() + assert xp.kernels.DEVICE_IMPLEMENTATIONS == ("cuda",) + with xp.kernels.use_device_kernel_implementation(None): + assert xp.kernels.get_device_kernel_implementation() is None + xp.kernels.set_device_kernel_implementation("cuda") + for invalid in ("cupy", "numba", "", "CUDA"): + with pytest.raises(ValueError, match="device kernel implementation"): + xp.kernels.set_device_kernel_implementation(invalid) + assert xp.kernels.get_device_kernel_implementation() == "cuda" + with ( + pytest.raises(ValueError, match="device kernel implementation"), + xp.kernels.use_device_kernel_implementation("unsupported"), + ): + pytest.fail("invalid context was entered") + assert xp.kernels.get_device_kernel_implementation() == "cuda" + with ( + pytest.raises(RuntimeError, match="body failed"), + xp.kernels.use_device_kernel_implementation(None), + ): + assert xp.kernels.get_device_kernel_implementation() is None + raise RuntimeError("body failed") + assert xp.kernels.get_device_kernel_implementation() == "cuda" + assert xp.kernels.get_device_kernel_implementation() == previous + + +@pytest.mark.parametrize("value", [None, "", "cuda", " CuDa ", "cupy"]) +def test_device_environment_is_read_at_import(value): + env = { + **os.environ, + "PYTHONPATH": str(Path(xp.__file__).parents[1]), + "CUNUMPY_BACKEND": "numpy", + "CUNUMPY_HOST_KERNEL_IMPLEMENTATION": "numpy", + } + env.pop("CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION", None) + if value is not None: + env["CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION"] = value + expected = value.strip().lower() if value else None + code = ( + "import os, cunumpy as xp\n" + f"assert xp.kernels.get_device_kernel_implementation() == {expected!r}\n" + "assert xp.kernels.get_host_kernel_implementation() == 'numpy'\n" + "assert xp.get_backend() == 'numpy'\n" + "os.environ['CUNUMPY_DEVICE_KERNEL_IMPLEMENTATION'] = 'unsupported'\n" + f"assert xp.kernels.get_device_kernel_implementation() == {expected!r}\n" + "kernel = xp.kernels.Kernel(lambda: None, missing_cuda='fallback')\n" + "if xp.kernels.get_device_kernel_implementation() == 'cuda':\n" + " try:\n" + " kernel.selected(device=True)\n" + " except LookupError:\n" + " pass\n" + " else:\n" + " raise AssertionError('explicit CUDA allowed host fallback')\n" + "xp.kernels.set_device_kernel_implementation(None)\n" + "assert xp.kernels.get_device_kernel_implementation() is None\n" + ) + result = subprocess.run( + [sys.executable, "-c", code], + env=env, + capture_output=True, + text=True, + check=False, + ) + if value == "cupy": + assert result.returncode != 0 + assert "ValueError: device kernel implementation" in result.stderr + else: + assert result.returncode == 0, result.stderr diff --git a/tests/unit/test_kernel_dispatch_arrays.py b/tests/unit/test_kernel_dispatch_arrays.py index 8eb3515..7e74d07 100644 --- a/tests/unit/test_kernel_dispatch_arrays.py +++ b/tests/unit/test_kernel_dispatch_arrays.py @@ -81,6 +81,40 @@ def test_backend_dispatch_sends_everything_to_cuda_on_cupy(fake_gpu): assert x.tolist() == [1.0] * 4 +@pytest.mark.parametrize("dispatch", ["backend", "arrays"]) +def test_explicit_device_implementation_dispatch(fake_gpu, dispatch): + kernel = Kernel(scale, CudaKernel(SCALE_CUDA, "scale"), dispatch=dispatch) + missing = Kernel(scale, dispatch=dispatch, missing_cuda="fallback") + x = FakeDeviceArray(4) + with xp.kernels.use_device_kernel_implementation("cuda"): + assert kernel.selected(device=True) == "cuda" + kernel(x, 3.0, 4) + assert [name for name, *_ in fake_gpu] == ["scale"] + with pytest.raises(LookupError, match="has no 'cuda' implementation"): + missing(x, 3.0, 4) + with pytest.raises(LookupError, match="has no 'cuda' implementation"): + missing.selected(device=True) + with ( + xp.kernels.use_device_kernel_implementation(None), + pytest.warns(RuntimeWarning, match="No CUDA version"), + ): + assert missing.selected(device=True) == "host" + + +def test_device_implementation_does_not_change_host_selection(fake_gpu): + with ( + xp.use_backend("numpy"), + xp.kernels.use_host_kernel_implementation("python"), + xp.kernels.use_device_kernel_implementation("cuda"), + ): + kernel = Kernel(scale, dispatch="arrays") + x = np.ones(4) + kernel(x, 3.0, 4) + assert x.tolist() == [3.0] * 4 and fake_gpu == [] + assert xp.get_backend() == "numpy" + assert xp.kernels.get_host_kernel_implementation() == "python" + + def test_arrays_dispatch_runs_device_arrays_on_the_gpu(fake_gpu): kernel = Kernel(scale, CudaKernel(SCALE_CUDA, "scale"), dispatch="arrays") x = FakeDeviceArray(4) From 0569a604b62ffd2064c5843547659fe2c2ad3e54 Mon Sep 17 00:00:00 2001 From: Max Date: Mon, 5 Oct 2026 10:01:36 +0200 Subject: [PATCH 4/5] formatting --- src/cunumpy/_mpi.py | 4 +--- tests/unit/test_kernel_testing.py | 2 +- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/src/cunumpy/_mpi.py b/src/cunumpy/_mpi.py index 1ed6387..3804744 100644 --- a/src/cunumpy/_mpi.py +++ b/src/cunumpy/_mpi.py @@ -11,9 +11,7 @@ import array_api_compat import array_api_compat.numpy as np -from cunumpy._mpi_serial import ( - _LOCAL_RANK_VARIABLES, # noqa: F401 - re-exported -) +from cunumpy._mpi_serial import _LOCAL_RANK_VARIABLES # noqa: F401 - re-exported from cunumpy._transfers import _ACTIVE as _COUNTERS from cunumpy._transfers import _describe, _nbytes, _record from cunumpy.xp import array_backend, cupy_available, to_numpy diff --git a/tests/unit/test_kernel_testing.py b/tests/unit/test_kernel_testing.py index c5147ef..21d51d5 100644 --- a/tests/unit/test_kernel_testing.py +++ b/tests/unit/test_kernel_testing.py @@ -16,12 +16,12 @@ import cunumpy as xp import cunumpy.kernel_testing from cunumpy.cuda import CudaArguments, CudaKernel, parse_cuda_signature +from cunumpy.kernel_testing import backend # noqa: F401 - the fixture is used by name from cunumpy.kernel_testing import ( BACKENDS, _collect_arrays, _compare_results, assert_kernels_agree, - backend, # noqa: F401 - the fixture is used by name device_function_kernel, requires_cupy, ) From 1f9e2dbedb1a1d4531de4a6c69158d0016935b30 Mon Sep 17 00:00:00 2001 From: Max Date: Mon, 5 Oct 2026 10:02:13 +0200 Subject: [PATCH 5/5] ruff check --fix --- tests/unit/test_kernel_testing.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/unit/test_kernel_testing.py b/tests/unit/test_kernel_testing.py index 21d51d5..c5147ef 100644 --- a/tests/unit/test_kernel_testing.py +++ b/tests/unit/test_kernel_testing.py @@ -16,12 +16,12 @@ import cunumpy as xp import cunumpy.kernel_testing from cunumpy.cuda import CudaArguments, CudaKernel, parse_cuda_signature -from cunumpy.kernel_testing import backend # noqa: F401 - the fixture is used by name from cunumpy.kernel_testing import ( BACKENDS, _collect_arrays, _compare_results, assert_kernels_agree, + backend, # noqa: F401 - the fixture is used by name device_function_kernel, requires_cupy, )