diff --git a/.github/workflows/static_analysis.yml b/.github/workflows/static_analysis.yml index 151b4bb..d9df172 100644 --- a/.github/workflows/static_analysis.yml +++ b/.github/workflows/static_analysis.yml @@ -38,7 +38,6 @@ jobs: run: | pip install black "black[jupyter]" black --check src/ - black --check tutorials/ isort: runs-on: ubuntu-latest @@ -50,7 +49,6 @@ jobs: run: | pip install isort isort --check src/ - isort --check tutorials/ mypy: runs-on: ubuntu-latest diff --git a/.github/workflows/testing.yml b/.github/workflows/testing.yml index 2b15a25..aa01052 100644 --- a/.github/workflows/testing.yml +++ b/.github/workflows/testing.yml @@ -53,10 +53,6 @@ jobs: run: | pytest . - - name: Test tutorials - run: | - jupyter nbconvert --to notebook --execute tutorials/*.ipynb --output-dir=/tmp --ExecutePreprocessor.timeout=300 - - name: Test docs build run: | pip install ".[docs]" @@ -64,4 +60,4 @@ jobs: make clean make html cd .. - ls docs/build/html/index.html \ No newline at end of file + ls docs/build/html/index.html diff --git a/CHANGELOG.md b/CHANGELOG.md index 26eaec4..3093847 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,7 +7,28 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +## [0.2.0] - 2026-09-28 + +### Changed +- `xp.get_backend()` now returns the active global backend, mirroring `xp.set_backend()`. Use the new `xp.get_array_backend(array)` to inspect an individual array. Calls to the former `xp.get_backend(array)` must be updated. +- Expanded the README and documentation with backend selection, array conversion, GPU controls, kernel adaptation, and Pyodide examples. + +### Fixed +- GPU shape round-trip coverage now creates a zero-dimensional NumPy array for the scalar case instead of calling `.astype()` on a Python float. +- GPU tests now account for CuPy's retained split memory blocks and provide an array conversion method on the custom host-array test fixture. +- `set_backend()` and `use_backend()` now reject unsupported backend names with `ValueError` without changing the active selection. + ### Added +- `xp.get_array_module(array)`: Return the array-api-compat module (`numpy`/`cupy`) matching a given array's own backend, regardless of the process-wide active backend. Mirrors `cupy.get_array_module` and works in Pyodide. +- `xp.get_rng(seed=None)`: Return a `numpy.random.Generator`/`cupy.random.Generator` matching the active backend, without having to branch on the backend yourself. +- `xp.device_count()`: Number of visible CUDA devices (`0` without a functional CuPy/CUDA install), independent of the currently active backend. +- `xp.set_device_for_rank(rank, devices_per_node=None)`: Convenience for one-MPI-rank-per-GPU codes; selects `rank % devices_per_node` (defaulting `devices_per_node` to `device_count()`) via `set_device()` and returns the chosen device id. +- `xp.memory_info()`: `(free, total)` bytes of memory on the active CUDA device, or `None` on the NumPy backend. +- `xp.free_memory()`: Release all free blocks held by CuPy's device and pinned-host memory pools (no-op on the NumPy backend). +- `xp.default_float_dtype()`: Return the active backend's `float64` dtype object, for pinning a portable float precision instead of the backend/platform-dependent `dtype=float`. +- `xp.stream()`: Context manager for a CUDA stream, to overlap host/device transfers with compute (no-op, yielding `None`, on the NumPy backend). +- `xp.pin_memory(array)`: Copy a host array into pinned (page-locked) CUDA host memory for faster transfers. +- `PyccelKernel(..., is_array=...)`: Extension point overriding the default `isinstance(value, np.ndarray)` check used to decide which host values returned by (or reachable from a declared output of) the wrapped kernel are converted back to the device -- for kernels that return/mutate a NumPy subclass or other custom host array type. - Pyodide NumPy support documentation and CI that installs the built wheel in Pyodide's WebAssembly runtime and runs compiler-free tests for arrays, conversions, contexts, and Python kernels without CuPy or Pyccel imports. - `test-compiled` extra for native Pyccel tests. The `test` extra is now compiler-free; `dev` continues to include compiled-test dependencies. - `xp.same_backend(*arrays)`: Return `True` if all given arrays live on the same backend. diff --git a/README.md b/README.md index e339eb4..3df254f 100644 --- a/README.md +++ b/README.md @@ -1,115 +1,227 @@ # CuNumpy -Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`. +CuNumpy lets a Python program use a NumPy-like API while choosing NumPy arrays +on the CPU or CuPy arrays on an NVIDIA GPU. In the simplest case, replace +`import numpy as np` with `import cunumpy as xp`; the array operations you +already know then run on the selected backend. -# Install +```python +import cunumpy as xp -```bash -pip install cunumpy +values = xp.arange(5, dtype=xp.float64) +print(values * 2) +print(xp.get_backend()) # 'numpy' by default ``` -Example usage: +CuNumpy selects an array library for newly requested operations. It does not +move existing arrays just because the selected backend changes. This guide +covers backend selection, array movement, mixed CPU/GPU workflows, and the +helper APIs CuNumpy provides around NumPy and CuPy. +## Install + +```bash +python -m pip install cunumpy ``` -export ARRAY_BACKEND=cupy -``` + +NumPy and `array-api-compat` are installed as dependencies. To use a GPU, +install a CuPy package compatible with your CUDA environment as well. CuPy +installation depends on the CUDA version and platform; follow the CuPy +installation instructions for your system. CuNumpy does not install CUDA. + +`array-api-compat` supplies NumPy and CuPy compatibility modules with more +consistent behavior for shared array operations. CuNumpy uses them internally; +your arrays remain ordinary NumPy or CuPy arrays. See [why CuNumpy uses +`array-api-compat`](docs/source/array-api-compat.md) for a plain-language +explanation and examples. + +## Choose a backend + +CuNumpy starts with NumPy unless `ARRAY_BACKEND=cupy` is set before import. +You can also choose at runtime: ```python import cunumpy as xp xp.set_backend("cupy") +print(xp.get_backend()) # 'cupy' if CuPy and CUDA are functional + +values = xp.arange(5) # created by the active backend +``` + +The accepted backend names are `"numpy"` and `"cupy"`. If CuPy is requested +but unavailable or not functional, CuNumpy falls back to NumPy. Always check +`get_backend()` when the effective backend matters, such as when reporting +configuration or deciding whether GPU-specific work will happen. -arr = xp.array([1, 2]) +Use `use_backend()` for a temporary selection. It restores the previous +selection when the block exits, including when an exception is raised: -print(f"{type(arr) = }") -print(f"{xp.__version__ = }") +```python +with xp.use_backend("numpy"): + cpu_values = xp.linspace(0, 1, 100) + assert xp.get_backend() == "numpy" -# Convert to NumPy -arr_np = xp.to_numpy(arr) +# The previous global backend is active again here. +``` -# Convert to active backend -arr_xp = xp.to_cunumpy(arr) +The backend selection is process-wide shared state. Do not switch it +independently from multiple threads or async tasks; those changes can +interfere. A context manager is useful for sequential code, tests, and +notebooks. -# Inspect backend -print(f"{xp.get_backend(arr) = }") -print(f"{xp.is_gpu(arr) = }") -print(f"{xp.is_cpu(arr) = }") +## Understand the two backend questions -# Temporarily switch backend -with xp.use_backend("numpy"): - # This code runs on CPU even if ARRAY_BACKEND=cupy - arr_cpu = xp.zeros(100) +The active backend controls which library CuNumpy exposes through its NumPy +like operations. The array backend reports where one particular array lives. +These can differ: changing the active backend does not convert arrays that +already exist. -# Set backend globally -xp.set_backend("cupy") +```python +xp.set_backend("numpy") +cpu_values = xp.arange(3) -# Synchronize GPU operations (no-op on CPU) -xp.synchronize() +gpu_values = xp.to_cupy(cpu_values) # explicit transfer +print(xp.get_backend()) # 'numpy' +print(xp.get_array_backend(gpu_values)) # 'cupy' ``` -Output: +Use `is_cpu(array)`, `is_gpu(array)`, or `get_array_backend(array)` when +dispatch should follow the array passed to a function. `get_array_module()` +returns the matching `array_api_compat` module, which is useful when writing +backend-generic functions: +```python +def vector_norm(values): + array_xp = xp.get_array_module(values) + return array_xp.sqrt(array_xp.sum(values * values)) ``` -type(arr) = -xp.__version__ = '0.1.4' -xp.get_backend(arr) = 'cupy' -xp.is_gpu(arr) = True -xp.is_cpu(arr) = False + +## Move data between CPU and GPU + +Transfers are explicit so it is clear when data crosses the CPU/GPU boundary: + +```python +host = xp.to_numpy(gpu_values) # CuPy -> NumPy (host) +device = xp.to_cupy(host) # NumPy/array-like -> CuPy (device) +active = xp.to_cunumpy(host) # convert to the currently selected backend ``` -# Pyodide +`to_numpy()` also accepts ordinary array-like values. `to_cupy()` raises +`ImportError` when CuPy or a functional CUDA runtime is unavailable. +`to_cunumpy()` is useful at API boundaries where the consumer expects the +currently selected backend. It does not change the original array. -cuNumPy supports Pyodide with the NumPy backend. In an initialized Pyodide -JavaScript runtime, install the package and run a Python-source kernel: +Avoid transferring data inside a tight loop. Keep intermediate arrays on one +backend and move only at boundaries such as file I/O, plotting, or a +CPU-only library call. For example: -```javascript -await pyodide.loadPackage("micropip"); -await pyodide.runPythonAsync(` - import micropip - await micropip.install("cunumpy") +```python +with xp.use_backend("cupy"): + signal = xp.asarray(host_signal) + filtered = xp.fft.rfft(signal) + result = xp.to_numpy(filtered) # one transfer for a CPU-only consumer +``` - import cunumpy as xp - xp.set_backend("numpy") +## Random numbers and dtypes - def scale(values, factor): - values[:] *= factor - return values +`get_rng(seed)` returns a random generator for the active backend. NumPy and +CuPy have similar generator APIs, though exact bit-for-bit sequences are not +guaranteed to match between libraries: - values = xp.array([1.0, 2.0, 3.0]) - result = xp.PyccelKernel(scale)(values, 2.0) - assert result is values - print(xp.to_numpy(result)) # [2. 4. 6.] -`); +```python +rng = xp.get_rng(seed=42) +samples = rng.normal(size=1000) ``` -NumPy is the default backend when `ARRAY_BACKEND` is unset. CuPy/CUDA and -native Pyccel compilation are not supported in Pyodide. Despite its name, -`PyccelKernel` accepts ordinary Python callables and neither imports Pyccel nor -compiles code. Applications must supply Python-source kernels with -Pyodide-compatible imports. +Use `default_float_dtype()` when code needs to explicitly request the active +backend's `float64` dtype rather than rely on Python scalar inference: -CI tests the built wheel in Pyodide's WebAssembly runtime under Node.js, including -array operations, conversions, mutation/aliasing, backend context restoration, -and Python kernels, while rejecting CuPy or Pyccel imports. This does not test -browser-specific integration such as page loading or workers. See the -[Pyodide guide](docs/source/pyodide.md) for details and local test commands. +```python +x = xp.asarray([1.0, 2.0], dtype=xp.default_float_dtype()) +``` -# Development tests +## GPU selection and memory helpers -```bash -pip install -e '.[test]' -pytest tests/portable +These helpers are useful for multi-GPU programs and for understanding CuPy's +memory behavior: + +```python +print("visible GPUs:", xp.device_count()) +xp.set_device(0) # selects CUDA device 0 when CuPy is active +print("memory (free, total):", xp.memory_info()) ``` -The `test` extra is compiler-free. For compiled-kernel tests, install -`'.[test-compiled]'` and a working native compiler, then run `pytest`. -The `dev` extra includes these compiled-test dependencies as before. +`set_device()` is a no-op on NumPy. `device_count()` checks visible CUDA +hardware even if the active backend is NumPy; it returns zero when CuPy/CUDA +cannot be used. `memory_info()` returns `(free_bytes, total_bytes)` on the +active CuPy device and `None` on NumPy. `set_device_for_rank(rank)` is a +round-robin convenience for MPI layouts where local ranks map contiguously to +GPUs. If your scheduler uses a different mapping, select the device directly. -# Build docs +CuPy caches released allocations in memory pools. This can make process-level +GPU memory appear occupied after arrays go out of scope. `free_memory()` asks +CuPy to release currently free cached blocks; it does not free memory still +referenced by live arrays. +`pin_memory(host_array)` makes a pinned host copy, which can improve transfer +throughput for workloads that explicitly manage asynchronous transfers. +`stream()` creates a non-blocking CuPy stream and yields it; it yields `None` +on NumPy. GPU work is asynchronous, so synchronize before reading results on +the host: +```python +with xp.stream(): + device = xp.to_cupy(host) + transformed = xp.fft.fft(device) + +xp.synchronize() +result = xp.to_numpy(transformed) ``` -make html -cd ../ -open docs/_build/html/index.html + +## Use NumPy-only kernels with CuPy arrays + +`PyccelKernel` adapts a callable that expects NumPy arrays. When conversion is +needed, CuNumpy copies CuPy inputs to the host, calls the wrapped function, +copies in-place output changes back to the device, and moves returned NumPy +arrays to CuPy. With NumPy inputs, the wrapper calls the function directly. +CuNumpy does not compile functions or import Pyccel for you. + +```python +import cunumpy as xp + + +def scale_in_place(values, factor): + values[:] *= factor + return values + + +scale = xp.PyccelKernel(scale_in_place, outputs=(0,)) + +with xp.use_backend("cupy"): + values = xp.arange(5, dtype=xp.float64) + returned = scale(values, 3.0) + xp.synchronize() ``` + +By default every converted argument is copied back, because the wrapper cannot +know which arguments the kernel changed. `outputs=(0,)` declares that +positional argument 0 is written, avoiding unnecessary copy-back for +read-only inputs. For a keyword call, declare the keyword name, such as +`outputs=("out",)`. A wrong declaration can leave GPU output values stale. +The wrapper can also traverse arrays nested in lists, tuples, dictionaries, +and selected application objects; see the full [API reference](docs/source/api.md) +for `object_modules`, `is_array`, aliasing, and output declarations. + +## Pyodide + +CuNumpy supports the NumPy backend in Pyodide. It does not provide CuPy/CUDA +there. Ordinary Python callables can be wrapped with `PyccelKernel` without +compilation. See the [Pyodide guide](docs/source/pyodide.md) for a complete +installation example and compatibility notes. + +## Documentation + +The [user guide](docs/source/quickstart.md) explains common workflows. The +[API reference](docs/source/api.md) documents each helper and its behavior. +The [Pyodide guide](docs/source/pyodide.md) covers WebAssembly usage. diff --git a/docs/source/api.md b/docs/source/api.md index 178d6df..f7d63e3 100644 --- a/docs/source/api.md +++ b/docs/source/api.md @@ -1,92 +1,360 @@ -# API +# API reference -The `cunumpy` package exposes the following helper functions in addition to the standard NumPy/CuPy API. +CuNumpy exports the active NumPy-like namespace and a set of helpers for +backend selection, array inspection and conversion, hardware control, and +host-only kernels. Most examples use `import cunumpy as xp`. -## Backend Management +## NumPy-like namespace + +```python +import cunumpy as xp + +values = xp.arange(5) +total = xp.sum(values) +``` + +At runtime, NumPy-like attributes such as `array`, `sum`, `fft`, and `linalg` +are forwarded to the currently selected `array-api-compat` NumPy or CuPy +module. CuNumpy does not wrap every operation individually. The available +operations and some details can therefore vary with the installed NumPy and +CuPy versions. In normal use, access those operations through the top-level +`cunumpy` namespace, commonly imported as `xp`. + +For an explanation of what the compatibility module does and why CuNumpy +uses it, read [Why CuNumpy uses `array-api-compat`](array-api-compat.md). + +NumPy and CuPy are not interchangeable for every function or object. A +function that needs to follow an input array's location should use +`get_array_module(array)` instead of assuming the global backend matches it. + +## Backend selection + +### `set_backend(backend)` + +Selects the process-wide backend used for new NumPy-like operations. Supported +values are `"numpy"` and `"cupy"`: + +```python +xp.set_backend("cupy") +values = xp.arange(10) +``` + +Requesting CuPy selects it only when CuPy and its CUDA runtime are functional; +otherwise CuNumpy falls back to NumPy. Check `get_backend()` to inspect the +effective selection. Changing the selection does not move arrays that have +already been created. + +The initial backend is NumPy unless `ARRAY_BACKEND=cupy` is set before CuNumpy +is imported. Other values of this environment variable result in the NumPy +default. + +Backend state is shared process-wide and `set_backend()` is not thread-safe. +Concurrent tasks that change the backend may race. + +### `get_backend()` + +Returns the active global backend name, either `"numpy"` or `"cupy"`. This is +the getter paired with `set_backend()`: + +```python +xp.set_backend("numpy") +assert xp.get_backend() == "numpy" +``` + +### `use_backend(backend)` + +Context manager that temporarily selects a backend and restores the previous +backend on exit, even if the block raises an exception: + +```python +with xp.use_backend("numpy"): + reference = xp.zeros(10) +``` + +As backend selection is shared process state, this context manager is suited +to sequential use rather than concurrent backend switching. + +### `numpy_backend`, `cupy_backend` + +Boolean properties indicating whether the currently selected global backend +is NumPy or CuPy: + +```python +if xp.cupy_backend: + print("new arrays are being created on the GPU") +``` + +For a string value, prefer `get_backend()`. + +## Inspect arrays and select operations + +### `get_array_backend(array)` + +Returns `"numpy"` or `"cupy"` according to the given array's type. It reports +the array's location, not the active global selection: + +```python +xp.set_backend("numpy") +values_gpu = xp.to_cupy([1, 2, 3]) +assert xp.get_backend() == "numpy" +assert xp.get_array_backend(values_gpu) == "cupy" +``` + +### `get_array_module(array)` + +Returns the `array-api-compat` module matching the array: the NumPy module for +a NumPy array or the CuPy module for a CuPy array. This supports functions +that dispatch based on their input rather than global state: + +```python +def standardize(values): + array_xp = xp.get_array_module(values) + mean = array_xp.mean(values) + scale = array_xp.std(values) + return (values - mean) / scale +``` + +The returned module is an `array-api-compat` module, consistent with the +active module used by CuNumpy. It is not necessarily identical to importing +raw `numpy` or raw `cupy`. + +### `is_cpu(array)`, `is_gpu(array)` + +Return booleans indicating whether an array is a NumPy (CPU) or CuPy (GPU) +array. They are convenience checks equivalent to comparing +`get_array_backend(array)` with `"numpy"` or `"cupy"`. + +### `same_backend(*arrays)` + +Returns `True` if all provided arrays have the same backend. Zero or one +argument is considered to match: + +```python +if xp.same_backend(position, velocity): + update(position, velocity) +``` + +### `assert_same_backend(*arrays)` + +Raises `TypeError` with the detected backend names if arrays do not all share +a backend. Use it at API boundaries to provide a clear error before a mixed +NumPy/CuPy operation fails deeper in a library: + +```python +def combine(left, right): + xp.assert_same_backend(left, right) + return left + right +``` + +## Convert arrays ### `to_numpy(array)` -Converts an array to a NumPy array on the CPU. + +Converts to a host-side NumPy array. CuPy arrays are copied from device to +host. Other array-like inputs are passed through `numpy.asarray`; NumPy arrays +may therefore be returned as-is rather than copied. ### `to_cupy(array)` -Converts an array to a CuPy array on the GPU. Raises `ImportError` if CuPy is not available. + +Converts an array-like input to a CuPy array. Raises `ImportError` if CuPy or +CUDA is unavailable or not functional. The source is not modified. ### `to_cunumpy(array)` -Converts an array to the currently active backend. -### `get_backend(array)` -Returns the name of the backend (`"numpy"` or `"cupy"`) for the given array. +Converts to the currently active backend. This is convenient at an API +boundary when an input should be normalized to the configured backend: -### `is_gpu(array)` -Returns `True` if the array is stored on a GPU (CuPy). +```python +normalized = xp.to_cunumpy(input_array) +assert xp.get_array_backend(normalized) == xp.get_backend() +``` -### `is_cpu(array)` -Returns `True` if the array is stored on a CPU (NumPy). +Each conversion returns a suitable array; it does not change the active +backend or mutate the source. -## Global Configuration +## Random numbers and dtype -### `numpy_backend` -Boolean property that returns `True` if the currently active global backend is NumPy. +### `get_rng(seed=None)` -### `cupy_backend` -Boolean property that returns `True` if the currently active global backend is CuPy. +Returns a NumPy or CuPy `Generator` matching the active backend: -### `set_backend(backend_name)` -Globally sets the active backend for all `cunumpy` operations. `backend_name` should be `"numpy"` or `"cupy"`. +```python +rng = xp.get_rng(seed=7) +samples = rng.uniform(size=100) +``` + +The generator APIs are similar, but seeds do not guarantee identical random +sequences across NumPy and CuPy. -### `use_backend(backend_name)` -A context manager that temporarily sets the active backend. +### `default_float_dtype()` + +Returns the active backend module's `float64` dtype object. Pass it to array +creation when code requires an explicit precision: ```python -with xp.use_backend("numpy"): - # Operations here use NumPy - pass +values = xp.asarray([0.1, 0.2], dtype=xp.default_float_dtype()) ``` -## Hardware Control +## CUDA devices and memory -### `synchronize()` -Blocks until all preceding GPU operations are complete. This is a no-op when using the NumPy backend. +### `cupy_available()` -## Compiled Kernels +Returns whether CuPy can be imported and reports itself functional. The result +is cached for the process. This checks availability, not whether every +GPU-specific operation will succeed later. -### `PyccelKernel(kernel, use_cupy=None, object_modules=(), outputs=None)` -Wraps a kernel compiled with [pyccel](https://github.com/pyccel/pyccel) — which only accepts NumPy arrays — so that it can be called with CuPy arrays as well. +### `device_count()` -On the CuPy backend the arguments are copied to the host before the call, any in-place updates the kernel makes are copied back to the device afterwards, and arrays returned by the kernel are moved back to the device. On the NumPy backend the kernel is called directly, without any conversion. +Returns the number of visible CUDA devices. Returns zero when CuPy/CUDA is +unavailable or querying the runtime fails. This is independent of the active +backend, so it may return a positive number while NumPy is selected. -```python -import cunumpy as xp -from my_package.kernels import axpy # pyccelized kernel +### `set_device(device_id)` -axpy = xp.PyccelKernel(axpy) +Selects a CUDA device when CuPy is active; it is a no-op on NumPy. The device +must be valid for the current CUDA process. -with xp.use_backend("cupy"): - x = xp.arange(10, dtype=xp.float64) - y = xp.ones(10, dtype=xp.float64) - out = xp.zeros(10, dtype=xp.float64) - axpy(2.0, x, y, out) # `out` is updated in place, on the GPU +### `set_device_for_rank(rank, devices_per_node=None)` + +Selects a device using `rank % devices_per_node` and returns its ID. If +`devices_per_node` is omitted, it uses `device_count()`. When no devices are +visible, it returns `0` without selecting a device. This helper assumes +contiguous rank-to-device mapping on each node, suitable for a common +one-rank-per-GPU MPI layout. Use `set_device()` directly when the scheduler's +mapping differs: + +```python +device_id = xp.set_device_for_rank(mpi_rank) ``` -Tuples, lists and dicts are traversed recursively. Pass `object_modules` to also traverse the attributes of your own objects, e.g. `object_modules=("struphy.", "feectools.")`; instances from other modules are handed to the kernel untouched. +### `memory_info()` + +Returns `(free_bytes, total_bytes)` reported by the CUDA runtime for the +active device, or `None` on NumPy. The values cover the device, not only +allocations owned by CuPy. + +### `free_memory()` -#### Declaring outputs +Releases currently free blocks in CuPy's device and pinned-host memory pools. +It is a no-op on NumPy. It does not release blocks still referenced by live +arrays. CuPy normally caches freed allocations for reuse, so cached memory +does not necessarily indicate a leak. -By default every array that was copied to the host is copied back afterwards, since the wrapper cannot know which ones the kernel wrote to. Most pyccel kernels write to one `out` argument and only read the rest, so `outputs` lets you skip the needless transfers: +### `pin_memory(array)` + +Copies a host array to page-locked (pinned) host memory. Pinned memory can +improve host/device transfer throughput in suitable asynchronous workloads. +Raises `ImportError` if CuPy is unavailable. If the input may be a CuPy array, +first transfer it with `to_numpy()`. + +### `synchronize()` + +Waits for queued work on the current CUDA device to finish. This is useful +before reading asynchronously computed results from host code. It is a no-op +on NumPy. + +### `stream()` + +Context manager that creates a non-blocking CuPy stream and yields it. Work +issued in the block is enqueued on that stream. On NumPy, yields `None` and +does nothing: ```python -interpolate = xp.PyccelKernel(some_interpolation_kernel, outputs=(5,)) +with xp.stream() as work_stream: + result = xp.to_cupy(host_values) * 2 -interpolate(x, y, z, basis, coeffs, out) # `out` is argument 5 +work_stream.synchronize() # on CuPy; the yielded value is None on NumPy ``` -Only the declared arguments are copied back; `basis` and `coeffs` make the trip to the host and no further. Containers and traversed objects may be declared too — every array nested inside them is copied back. An array that is *also* reachable from a declared output (e.g. passed as both an input and the output) is still copied back. +Do not call methods on the yielded value without checking the backend. Use +`xp.synchronize()` for code that should work on both backends. -Declare positional arguments by index (negatives count from the end) and keyword arguments by name, e.g. `outputs=("out",)` for `interpolate(..., out=out)`. The two are not interchangeable: pyccel-compiled kernels are builtins with no introspectable signature, so the wrapper cannot map a name onto a position. A declaration that matches no argument of the call raises `IndexError`/`KeyError` rather than silently copying nothing back. +## `PyccelKernel` -`outputs=()` declares that the kernel writes to none of its arguments. Leaving `outputs` unset keeps the always-correct default. Note that a *wrong* declaration is a silent-wrong-answer bug: an argument the kernel writes to but that you did not declare keeps its stale values on the GPU. +### Constructor + +```python +xp.PyccelKernel( + kernel, + use_cupy=None, + object_modules=(), + is_array=None, + outputs=None, +) +``` -#### Aliasing and cycles +Wraps a callable expecting host NumPy arrays so it can be used with CuPy +arrays. This can adapt a Pyccel-compiled kernel or an ordinary Python +callable. CuNumpy does not compile the callable or import Pyccel. + +When no conversion is needed, the original callable is invoked directly. If +conversion is needed, CuNumpy recursively replaces CuPy arrays in supported +arguments with NumPy host copies, calls the kernel, copies in-place updates +back to the corresponding CuPy arrays, and converts returned NumPy arrays to +CuPy arrays. + +### Parameters + +* `kernel`: callable that accepts the host-side arguments. +* `use_cupy`: `None` (default) chooses conversion for each call when the + global backend is CuPy or a CuPy array is present. `True` forces conversion; + `False` disables it. +* `object_modules`: module prefixes whose instances should be shallow-copied + and traversed by attributes. For example, + `object_modules=("my_project.",)`. +* `is_array`: predicate for host array values to convert back to CuPy. The + default is `isinstance(value, numpy.ndarray)`. +* `outputs`: sequence of arguments the kernel may write to. Entries are + positional indices or keyword names. If omitted, every converted array is + copied back. + +### Example and output declarations -Conversion is identity-aware: an array reachable by several paths — passed as two arguments, or both directly and as an attribute of a traversed object — becomes a single array on the host, so the kernel sees the aliasing the caller intended and no in-place update is lost on the way back. Reference cycles are handled rather than recursed into. +```python +def scale_and_shift(scale, values, out): + out[:] = scale * values + 1 + return out + +kernel = xp.PyccelKernel(scale_and_shift, outputs=(2,)) + +with xp.use_backend("cupy"): + values = xp.arange(8, dtype=xp.float64) + out = xp.empty_like(values) + result = kernel(2.0, values, out) +``` -Set `use_cupy` explicitly to force conversion on or off. By default it is decided per call: conversion happens when the active backend is CuPy, or when a CuPy array is passed in. +`outputs=(2,)` marks only `out` for copy-back. Read-only `values` is copied +to the host for the call but not transferred back. Declare a positional +argument by index (negative indices count from the end), and a keyword +argument by its name, such as `outputs=("out",)` for `kernel(..., out=out)`. +The forms are not interchangeable because compiled builtins may not expose a +Python signature. Invalid declarations raise `IndexError` or `KeyError`. + +An incorrect declaration is a correctness bug: if the kernel writes an +argument that was not declared, the device array will not receive that +update. Use `outputs=()` only when no converted input is mutated. Containers +and selected objects can be declared as outputs; every nested supported +array is then copied back. If an array is reachable through multiple paths, +declaring any path that includes it is sufficient. + +### Supported containers, aliasing, and returns + +Tuples, lists, and dictionaries are traversed recursively. Objects are +traversed only when their class module starts with one of the configured +`object_modules` prefixes; those objects are shallow-copied, and their +attributes are converted on the copy. Other objects are passed to the kernel +unchanged. + +The wrapper memoizes conversions within a call. If the same device array is +passed more than once or appears inside a supported container, the host kernel +sees the same NumPy array object, preserving aliasing. Supported reference +cycles terminate safely. Returned NumPy arrays (and arrays inside tuples or +lists) are converted back using `is_array`; dictionaries in return values are +not recursively converted. On the NumPy path, the original return value and +normal Python mutation and exception behavior are preserved. + +## Version + +`xp.__version__` is the installed package version. When package metadata is +not available (for example, some source-tree imports), it is +`"0.0.0+unknown"`. diff --git a/docs/source/array-api-compat.md b/docs/source/array-api-compat.md new file mode 100644 index 0000000..b79e6c5 --- /dev/null +++ b/docs/source/array-api-compat.md @@ -0,0 +1,103 @@ +# Why CuNumpy uses `array-api-compat` + +You can use CuNumpy without importing `array-api-compat` yourself. It is a +dependency that sits between CuNumpy and the array libraries it selects. This +page explains what that layer does and when you might notice it. + +## Start with the three pieces + +**NumPy** stores arrays in regular computer memory and performs operations on +the CPU. **CuPy** offers many similar operations on arrays stored on an NVIDIA +GPU. Their APIs overlap substantially, but the same function name does not +always accept the same arguments or follow the same rules. + +The **Python Array API standard** describes a shared set of array operations +and their expected behavior. It is a specification, not an array library that +stores data or runs calculations. [`array-api-compat`](https://data-apis.org/array-api-compat/) +provides small compatibility modules for existing libraries, including NumPy +and CuPy. Those modules expose the underlying libraries' functions while +adjusting operations covered by the standard to behave more consistently. + +CuNumpy uses those modules as its two backends: + +| CuNumpy backend | Module CuNumpy calls | Where the array lives | +| --- | --- | --- | +| `"numpy"` | `array_api_compat.numpy` | CPU memory | +| `"cupy"` | `array_api_compat.cupy` | GPU memory | + +For example, after `xp.set_backend("numpy")`, `xp.asarray(...)` is resolved +through the NumPy compatibility module. After `xp.set_backend("cupy")`, it is +resolved through the CuPy compatibility module. You still write `import +cunumpy as xp` and call `xp.asarray(...)`; CuNumpy chooses the module. + +## Why add this layer? + +Without a compatibility layer, a program that switches between NumPy and +CuPy must account for differences in shared operations. CuNumpy uses +`array-api-compat` so those operations have a more consistent interface. For +example, the compatibility namespace supports the Array API's `device` +argument on `asarray` for both backends: + +```python +import cunumpy as xp + +with xp.use_backend("numpy"): + values = xp.asarray([1, 2, 3], device=None) +``` + +`array-api-compat` also supplies array-type checks used by CuNumpy to tell +whether a particular value is a CuPy array. That is how +`xp.get_array_backend(values)` and helpers such as `xp.is_gpu(values)` can +inspect an array independently of the currently selected backend. + +This layer does **not** create a third kind of array. A NumPy-backed result is +still a NumPy array; a CuPy-backed result is still a CuPy array. It also does +not install CuPy or CUDA, move existing arrays when the global backend +changes, or make every NumPy operation available in CuPy. Functions outside +the shared standard may still differ between libraries. + +## When should you think about it? + +For code that creates arrays and works entirely on the active backend, you +usually do not need to think about the compatibility layer: + +```python +import cunumpy as xp + +xp.set_backend("numpy") +values = xp.arange(5) +print(xp.sum(values)) +``` + +It becomes useful when a function receives an array created elsewhere. The +global backend may be NumPy even though the argument is a CuPy array, or the +other way around. `xp.get_array_module(array)` returns the matching +compatibility module for that *array*, so operations inside the function use +the correct library: + +```python +def mean_center(values): + array_xp = xp.get_array_module(values) + return values - array_xp.mean(values) +``` + +`array_xp` is `array_api_compat.numpy` for a NumPy array and +`array_api_compat.cupy` for a CuPy array. The returned value stays on the +same backend as `values`. You do not need to import either compatibility +module directly for this pattern. + +If a function combines several inputs, first check that they live on the +same backend with `xp.assert_same_backend(a, b)`, or convert them explicitly +with `xp.to_numpy()`, `xp.to_cupy()`, or `xp.to_cunumpy()`. The compatibility +layer does not make mixed CPU/GPU arithmetic automatic. + +## What to remember + +* Use `xp.get_backend()` to ask which backend CuNumpy currently selects. +* Use `xp.get_array_backend(array)` to ask where a particular array lives. +* Use `xp.get_array_module(array)` when a function should follow its input + array rather than the global selection. +* Use explicit conversion helpers when data must move between CPU and GPU. + +The [`array-api-compat` documentation](https://data-apis.org/array-api-compat/) +explains the compatibility modules and the Array API standard in more depth. diff --git a/docs/source/conf.py b/docs/source/conf.py index 1689def..e6a5abb 100644 --- a/docs/source/conf.py +++ b/docs/source/conf.py @@ -3,27 +3,6 @@ # For the full list of built-in configuration values, see the documentation: # https://www.sphinx-doc.org/en/master/usage/configuration.html -import os -import shutil - - -def copy_tutorials(app): - src = os.path.abspath("../tutorials") - dst = os.path.abspath("source/tutorials") - - # Remove existing target directory if it exists - if os.path.exists(dst): - shutil.rmtree(dst) - - shutil.copytree(src, dst) - - -def setup(app): - app.connect("builder-inited", copy_tutorials) - # app.add_stylesheet("my-styles.css") - app.add_css_file("custom.css") - - # -- Project information ----------------------------------------------------- # https://www.sphinx-doc.org/en/master/usage/configuration.html#project-information diff --git a/docs/source/index.md b/docs/source/index.md index 2bcc674..29b1c3f 100644 --- a/docs/source/index.md +++ b/docs/source/index.md @@ -1,29 +1,28 @@ # CuNumpy -Simple wrapper for NumPy and CuPy. - -## Quick Start +CuNumpy provides a NumPy-like interface that can create and operate on NumPy +arrays on the CPU or CuPy arrays on an NVIDIA GPU. Select the backend once +for a program or temporarily for a section of code; use explicit conversion +helpers when data needs to cross between host and device. ```python import cunumpy as xp -# Use it exactly like NumPy -arr = xp.array([1, 2, 3]) -print(type(arr)) +xp.set_backend("cupy") +values = xp.arange(1_000) +print(xp.get_backend()) # 'cupy' when CuPy/CUDA is functional ``` +Install with `python -m pip install cunumpy`. GPU use also requires a CuPy +installation that matches the system's CUDA setup. CuNumpy falls back to +NumPy if CuPy cannot be used. + ```{toctree} :maxdepth: 2 -:caption: Contents: +:caption: Guides and reference: quickstart -pyodide -tutorials +array-api-compat api -``` - -## Installation - -```bash -pip install cunumpy +pyodide ``` diff --git a/docs/source/quickstart.md b/docs/source/quickstart.md index 99970f8..e8976c4 100644 --- a/docs/source/quickstart.md +++ b/docs/source/quickstart.md @@ -1,68 +1,284 @@ -# Quickstart +# User guide -## Installation +CuNumpy gives CPU and GPU programs a common NumPy-like entry point. This guide +starts with the default CPU workflow and then shows backend selection, +array-aware dispatch, data transfers, and GPU utilities. -Install `cunumpy` directly from PyPI: +## Install and import + +Install the package with pip: ```bash -pip install cunumpy +python -m pip install cunumpy ``` -## Basic Usage +CuNumpy depends on NumPy and `array-api-compat`. To use NVIDIA GPUs, install a +CuPy distribution that matches your CUDA environment separately. CUDA itself +is not installed by CuNumpy. + +`array-api-compat` is a small adapter that gives NumPy and CuPy a more +consistent interface for shared array operations. You do not need to import +it directly when using CuNumpy. See [Why CuNumpy uses +`array-api-compat`](array-api-compat.md) for a beginner-friendly explanation +and examples. -Replace `import numpy as np` with `import cunumpy as xp`. By default, it will use NumPy if CuPy is not installed or configured. +Import CuNumpy using the familiar alias `xp`: ```python import cunumpy as xp -# Create an array (automatically uses the active backend) -arr = xp.array([1, 2, 3]) - -print(f"Array type: {type(arr)}") -print(f"Active backend: {xp.get_backend(arr)}") +a = xp.array([1.0, 2.0, 3.0]) +b = xp.asarray([4.0, 5.0, 6.0]) +print(xp.dot(a, b)) ``` -## Backend Control +Most common array operations are available through the active NumPy-like +namespace, including array creation, indexing, arithmetic, reductions, +linear algebra, random operations via `xp.random`, and FFTs via `xp.fft`. +CuNumpy aims to keep the interface familiar; NumPy and CuPy are separate +libraries, so niche functions and edge-case behavior can differ. Consult the +upstream library documentation for operation-specific details. -You can control the backend using environment variables: +## Configure the active backend + +The default is NumPy. Select CuPy using an environment variable before +starting Python: ```bash -export ARRAY_BACKEND=cupy +ARRAY_BACKEND=cupy python my_program.py ``` -Or programmatically: +Or change the active backend within a running program: ```python -import cunumpy as xp +xp.set_backend("cupy") +print(xp.get_backend()) # active backend name + +data = xp.arange(1000) # created on the GPU while CuPy is active +``` + +Only `"numpy"` and `"cupy"` are valid names. If CuPy is requested but is +missing or fails its availability check, initialization falls back to NumPy. +Inspect `xp.get_backend()` after selection if the effective backend matters. +The boolean properties `xp.numpy_backend` and `xp.cupy_backend` are also +available for conditional code. + +Use a context manager when only a section should use a particular backend: -# Globally set backend +```python xp.set_backend("cupy") -# Scoped backend switching with xp.use_backend("numpy"): - arr_cpu = xp.zeros(10) - print(xp.is_cpu(arr_cpu)) # True + reference = xp.linspace(0, 1, 100) + print(xp.get_backend()) # 'numpy' -# Explicit conversion -arr_np = xp.to_numpy(arr) -arr_cp = xp.to_cupy(arr) +print(xp.get_backend()) # 'cupy' again ``` -## Synchronization +The old backend is restored even if the block raises an exception. Backend +selection is a process-wide setting and is not thread-safe; concurrent code +must not switch it independently from different threads or async tasks. -When using the GPU, it's important to synchronize for accurate timing: +## The active backend and an array's backend -```python -import time -import cunumpy as xp +These are deliberately separate concepts: + +* `xp.get_backend()` returns the globally selected backend for new CuNumpy + operations. +* `xp.get_array_backend(array)` returns `"numpy"` or `"cupy"` according to + where a particular array is stored. +* `xp.get_array_module(array)` returns the matching `array_api_compat` module. + +Changing the global selection does not migrate arrays already created. For +example, the active backend can be NumPy while a CuPy array is still alive: +```python xp.set_backend("cupy") -a = xp.random.rand(1000, 1000) +gpu_values = xp.arange(4) + +xp.set_backend("numpy") +print(xp.get_backend()) # 'numpy' +print(xp.get_array_backend(gpu_values)) # 'cupy' +``` -start = time.time() -b = xp.dot(a, a) -xp.synchronize() # Wait for GPU -end = time.time() +For a function that should follow its input array rather than global state, +use `get_array_module()`: -print(f"Elapsed: {end - start:.4f}s") +```python +def normalize(values): + array_xp = xp.get_array_module(values) + length = array_xp.sqrt(array_xp.sum(values * values)) + return values / length ``` + +This is often the right approach for reusable functions called with arrays +created by other libraries. `is_cpu(values)` and `is_gpu(values)` are concise +boolean checks. `same_backend(a, b)` checks whether all given arrays share a +backend, and `assert_same_backend(a, b)` raises a `TypeError` with the detected +backends if they do not. + +## Convert arrays explicitly + +Use explicit conversion at CPU/GPU boundaries: + +```python +host_array = xp.to_numpy(gpu_array) +device_array = xp.to_cupy(host_array) +active_array = xp.to_cunumpy(host_array) +``` + +`to_numpy()` always returns a host-side NumPy array. For a NumPy input it +uses `numpy.asarray`, so an existing array or view can be returned without a +copy. For a CuPy input it transfers data from device to host. `to_cupy()` +converts an array-like object to a CuPy array, and requires working CuPy/CUDA. +`to_cunumpy()` converts to whichever backend is currently active. + +Conversions do not mutate the source array. To avoid repeated transfer costs, +keep data on one device for a whole computational phase and transfer results +once at a boundary: + +```python +with xp.use_backend("cupy"): + signal_gpu = xp.asarray(signal_host) + spectrum_gpu = xp.fft.rfft(signal_gpu) + spectrum_host = xp.to_numpy(spectrum_gpu) +``` + +Mixing NumPy and CuPy arrays in the same operation is usually an error. Use +`assert_same_backend` to fail early in APIs that require matched arrays, or +convert the inputs to a common backend first. + +## Random generators and floating-point dtypes + +`xp.get_rng(seed)` chooses a random generator for the active backend: + +```python +rng = xp.get_rng(seed=1234) +noise = rng.normal(loc=0.0, scale=1.0, size=10_000) +``` + +The interface is similar across NumPy and CuPy, but generated values are not +expected to be bitwise identical across backends. When reproducibility +matters, keep the backend and library versions fixed as well as the seed. + +`xp.default_float_dtype()` returns the active module's `float64` dtype object. +It can be used to request an explicit portable precision: + +```python +weights = xp.asarray([0.25, 0.75], dtype=xp.default_float_dtype()) +``` + +Explicit dtypes are especially useful when code must not rely on backend- or +platform-specific inference for Python integers and floats. + +## Work with CUDA devices + +`xp.device_count()` reports the number of visible CUDA devices, or zero if +CuPy/CUDA is unavailable. It checks hardware visibility regardless of the +active array backend. + +```python +count = xp.device_count() +if count: + xp.set_backend("cupy") + xp.set_device(0) + values = xp.arange(100) +``` + +`set_device(device_id)` selects a CUDA device when the active backend is +CuPy and does nothing on NumPy. For a common one-rank-per-GPU MPI layout, +`set_device_for_rank(rank)` selects `rank % device_count()` and returns the +selected ID. Its default assumes ranks are arranged in contiguous device +blocks on each node; supply `devices_per_node` or call `set_device()` yourself +if your process mapping differs. + +`memory_info()` returns `(free_bytes, total_bytes)` for the active CUDA device, +or `None` on NumPy. This queries CUDA's runtime memory accounting, not only +CuPy's allocator. CuPy keeps freed blocks in pools for reuse, so cached memory +may remain visible as allocated. `free_memory()` releases free cached blocks +from CuPy's device and pinned-host pools; live arrays remain allocated. + +## Streams and synchronization + +CUDA operations are generally asynchronous with respect to the host. A stream +orders queued work; synchronization waits until that work has completed. +`xp.stream()` creates a non-blocking CuPy stream and yields it. On NumPy it is +a no-op context manager that yields `None`: + +```python +with xp.stream() as stream: + device_values = xp.to_cupy(host_values) + result = xp.exp(device_values) + +xp.synchronize() # wait before host code consumes the result +host_result = xp.to_numpy(result) +``` + +For finer control, synchronize the yielded stream with `stream.synchronize()`. +Use `xp.synchronize()` when you need to wait for all work on the current +device. Synchronization is a no-op on NumPy. + +`pin_memory(host_array)` creates a pinned (page-locked) host copy. Pinned +memory can improve host/device transfer throughput, especially for workloads +that overlap transfers and computation, but should be used when transfer +profiling indicates it is useful: + +```python +pinned = xp.pin_memory(host_array) +``` + +## Adapt NumPy-only or Pyccel kernels + +`xp.PyccelKernel` wraps a callable that expects NumPy arrays. It is useful for +compiled Pyccel functions and for other host-only kernels. It is an adapter, +not a compiler: compilation and imports remain the application's +responsibility. + +With the NumPy backend and NumPy arguments, the wrapped callable runs +directly. When conversion is needed (active CuPy backend or CuPy arrays among +the arguments), CuNumpy makes host copies of device arrays, invokes the +callable, copies declared in-place outputs back to the device, and converts +returned NumPy arrays to CuPy. + +```python +def axpy(alpha, x, y, out): + out[:] = alpha * x + y + return out + +axpy_cpu = xp.PyccelKernel(axpy, outputs=(3,)) + +with xp.use_backend("cupy"): + x = xp.arange(8, dtype=xp.float64) + y = xp.ones_like(x) + out = xp.empty_like(x) + result = axpy_cpu(2.0, x, y, out) +``` + +By default (`outputs=None`), all converted arrays are copied back after a +converted call. Specify only arguments the kernel may mutate to avoid copying +read-only inputs back. Positional arguments are declared by index; keyword +arguments by name. An argument passed by keyword must be declared by name, +because compiled builtins do not expose a signature for mapping it to a +position. A wrong output declaration can silently discard an in-place update +on the host copy, so include every argument the kernel writes. + +Lists, tuples, and dictionaries are traversed recursively. For instances from +your own modules, pass their module prefixes in `object_modules`, for example +`object_modules=("my_simulation.",)`. The wrapper shallow-copies those objects +and traverses their attributes, preserving the caller's original references. +`is_array` customizes which host result types are converted back to CuPy; its +default recognizes `numpy.ndarray`. + +Aliased device arrays are converted only once per call, so if the same array +appears in multiple arguments the kernel sees the same host array. Reference +cycles in supported containers are handled. `use_cupy=True` forces conversion +on and `use_cupy=False` forces it off; leave it as `None` for per-call +automatic selection. + +## Pyodide and WebAssembly + +CuNumpy's NumPy backend works in Pyodide. CuPy/CUDA and native Pyccel +compilation are not supported in that environment. Python-source functions +can still be wrapped with `PyccelKernel`; keep `use_cupy` unset or false. +Follow the [Pyodide guide](pyodide.md) for an install-and-run example and +runtime constraints. diff --git a/docs/source/tutorials.md b/docs/source/tutorials.md deleted file mode 100644 index 84e10de..0000000 --- a/docs/source/tutorials.md +++ /dev/null @@ -1,9 +0,0 @@ -# Tutorials - -Here are a few tutorials to help you get started. - -```{toctree} -:maxdepth: 1 -:glob: - -tutorials/* \ No newline at end of file diff --git a/pyproject.toml b/pyproject.toml index 7cf371f..794f7b3 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,7 +5,7 @@ requires = [ "setuptools", "wheel" ] [project] name = "cunumpy" -version = "0.1.5" +version = "0.2.0" description = "Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`." readme = "README.md" keywords = [ "python" ] diff --git a/src/cunumpy/__init__.py b/src/cunumpy/__init__.py index a35a539..4e9e194 100644 --- a/src/cunumpy/__init__.py +++ b/src/cunumpy/__init__.py @@ -6,12 +6,22 @@ from .xp import ( assert_same_backend, cupy_available, + default_float_dtype, + device_count, + free_memory, + get_array_backend, + get_array_module, get_backend, + get_rng, is_cpu, is_gpu, + memory_info, + pin_memory, same_backend, set_backend, set_device, + set_device_for_rank, + stream, synchronize, to_cunumpy, to_cupy, @@ -30,13 +40,23 @@ "assert_same_backend", "cupy_available", "cupy_backend", + "default_float_dtype", + "device_count", + "free_memory", + "get_array_backend", + "get_array_module", "get_backend", + "get_rng", "is_cpu", "is_gpu", + "memory_info", "numpy_backend", + "pin_memory", "same_backend", "set_backend", "set_device", + "set_device_for_rank", + "stream", "synchronize", "to_cunumpy", "to_cupy", diff --git a/src/cunumpy/__init__.pyi b/src/cunumpy/__init__.pyi index dc26342..d71893d 100644 --- a/src/cunumpy/__init__.pyi +++ b/src/cunumpy/__init__.pyi @@ -14,7 +14,9 @@ def to_numpy(array: Any) -> np.ndarray: ... def to_cupy(array: Any) -> Any: ... def to_cunumpy(array: Any) -> Any: ... def cupy_available() -> bool: ... -def get_backend(array: Any) -> str: ... +def get_array_module(array: Any) -> Any: ... +def get_backend() -> str: ... +def get_array_backend(array: Any) -> str: ... def is_gpu(array: Any) -> bool: ... def is_cpu(array: Any) -> bool: ... def same_backend(*arrays: Any) -> bool: ... @@ -23,6 +25,15 @@ def assert_same_backend(*arrays: Any) -> None: ... def use_backend(backend: str) -> Generator[None]: ... def set_backend(backend: str) -> None: ... def set_device(device_id: int) -> None: ... +def set_device_for_rank(rank: int, devices_per_node: int | None = ...) -> int: ... +def device_count() -> int: ... +def memory_info() -> tuple[int, int] | None: ... +def free_memory() -> None: ... +def pin_memory(array: Any) -> Any: ... +@contextmanager +def stream() -> Generator[Any]: ... +def get_rng(seed: int | None = ...) -> Any: ... +def default_float_dtype() -> Any: ... def synchronize() -> None: ... numpy_backend: bool diff --git a/src/cunumpy/kernel.py b/src/cunumpy/kernel.py index eb7ba69..5258633 100644 --- a/src/cunumpy/kernel.py +++ b/src/cunumpy/kernel.py @@ -41,6 +41,12 @@ class PyccelKernel: Module prefixes (e.g. ``("struphy.", "feectools.")``) whose instances should be traversed attribute-by-attribute when looking for arrays to convert. Objects from other modules are passed through untouched. + is_array : callable, optional + Predicate deciding whether a host-side value returned by the kernel + (or reachable from a declared output) counts as an array to move + back to the device. Defaults to ``isinstance(value, np.ndarray)``. + Override this if the kernel returns/mutates a NumPy subclass or a + custom host array type that should also be converted back to CuPy. outputs : sequence of int or str, optional Which arguments the kernel writes to. Only those are copied back to the device after the call, which avoids pointless device transfers for the @@ -68,11 +74,13 @@ def __init__( kernel: Callable[..., Any], use_cupy: bool | None = None, object_modules: Sequence[str] = (), + is_array: Callable[[Any], bool] | None = None, outputs: Sequence[int | str] | None = None, ) -> None: self._kernel = kernel self._use_cupy = use_cupy self._object_modules = tuple(object_modules) + self._is_array = is_array or (lambda value: isinstance(value, np.ndarray)) if outputs is None: self._outputs: tuple[int | str, ...] | None = None @@ -165,15 +173,14 @@ def _convert_to_numpy( return value - @staticmethod - def _convert_from_numpy(value: Any) -> Any: - """Move NumPy arrays returned by the kernel back to the device.""" - if isinstance(value, np.ndarray): + def _convert_from_numpy(self, value: Any) -> Any: + """Move host arrays returned by the kernel back to the device.""" + if self._is_array(value): return to_cupy(value) if isinstance(value, tuple): - return tuple(PyccelKernel._convert_from_numpy(item) for item in value) + return tuple(self._convert_from_numpy(item) for item in value) if isinstance(value, list): - return [PyccelKernel._convert_from_numpy(item) for item in value] + return [self._convert_from_numpy(item) for item in value] return value def _collect_host_arrays(self, value: Any, found: set[int], seen: set[int]) -> None: @@ -183,7 +190,7 @@ def _collect_host_arrays(self, value: Any, found: set[int], seen: set[int]) -> N :meth:`_convert_to_numpy`, so that an output declared as a container or an object contributes the arrays nested inside it. """ - if isinstance(value, np.ndarray): + if self._is_array(value): found.add(id(value)) return @@ -341,6 +348,11 @@ def object_modules(self) -> tuple[str, ...]: """Module prefixes whose instances are traversed for arrays.""" return self._object_modules + @property + def is_array(self) -> Callable[[Any], bool]: + """Predicate identifying host values to convert back to the device.""" + return self._is_array + @property def outputs(self) -> tuple[int | str, ...] | None: """Declared output arguments, or ``None`` if every array is copied back.""" diff --git a/src/cunumpy/xp.py b/src/cunumpy/xp.py index 46daea4..839c467 100644 --- a/src/cunumpy/xp.py +++ b/src/cunumpy/xp.py @@ -1,3 +1,5 @@ +from __future__ import annotations + import os import warnings from contextlib import contextmanager @@ -84,6 +86,8 @@ def xp(self) -> ModuleType: @contextmanager def use_backend(self, backend: BackendType) -> Generator[None, None, None]: """Temporarily change the backend.""" + if backend not in ("numpy", "cupy"): + raise ValueError("Array backend must be either 'numpy' or 'cupy'.") old_backend = self._backend old_xp = self._xp @@ -112,10 +116,17 @@ def use_backend(backend: BackendType) -> Generator[None, None, None]: def set_backend(backend: BackendType) -> None: """Set the backend globally.""" + if backend not in ("numpy", "cupy"): + raise ValueError("Array backend must be either 'numpy' or 'cupy'.") array_backend._backend = backend array_backend._xp = array_backend._load_backend(backend) +def get_backend() -> BackendType: + """Return the currently active global backend name.""" + return array_backend.backend + + def _cupy_backend() -> bool: """Check if the active global backend is CuPy.""" return array_backend.backend == "cupy" @@ -134,6 +145,141 @@ def set_device(device_id: int) -> None: cp.cuda.Device(device_id).use() +def device_count() -> int: + """Number of visible CUDA devices. + + Returns 0 on the NumPy backend, or if CuPy/CUDA is not available. + Independent of the currently active backend -- this reports what + hardware is visible, not what `xp.xp` currently dispatches to. + """ + if not cupy_available(): + return 0 + + import cupy as cp + + try: + return cp.cuda.runtime.getDeviceCount() + except Exception: # noqa: BLE001 - tolerate any driver/runtime failure + return 0 + + +def set_device_for_rank(rank: int, devices_per_node: int | None = None) -> int: + """Select a CUDA device for an MPI rank, round-robin across the node. + + Convenience for one-rank-per-GPU codes: computes + ``device_id = rank % devices_per_node`` and calls `set_device()` with + it. `devices_per_node` defaults to `device_count()`. Returns the + selected device id, or 0 as a no-op if there are no visible devices. + + This assumes ranks map to devices in contiguous blocks per node (i.e. + local rank == ``rank % devices_per_node``); codes with a different + rank-to-device layout should call `set_device()` directly instead. + """ + n = devices_per_node if devices_per_node is not None else device_count() + if n == 0: + return 0 + + device_id = rank % n + set_device(device_id) + return device_id + + +def memory_info() -> tuple[int, int] | None: + """Return `(free, total)` bytes of memory on the active CUDA device. + + Returns `None` on the NumPy backend. Queries the CUDA runtime directly, + so it reflects the whole device rather than just CuPy's memory pool. + """ + if array_backend.backend != "cupy": + return None + + import cupy as cp + + return cp.cuda.runtime.memGetInfo() + + +def free_memory() -> None: + """Release all free blocks held by CuPy's memory pools (no-op on NumPy). + + CuPy caches freed device (and pinned host) memory in pools rather than + returning it to the driver/OS immediately, which can look like a leak + in long-running processes. Call this to give it back. + """ + if array_backend.backend == "cupy": + import cupy as cp + + cp.get_default_memory_pool().free_all_blocks() + cp.get_default_pinned_memory_pool().free_all_blocks() + + +def pin_memory(array: Any) -> Any: + """Copy a host array into pinned (page-locked) CUDA host memory. + + Pinned memory transfers to/from the GPU faster than regular pageable + memory, since the driver can DMA it directly. `array` must already be + on the host (use `to_numpy()` first if it may be on the GPU). Raises + `ImportError` if CuPy is not available. + """ + if not cupy_available(): + raise ImportError("CuPy is not available or not functional.") + + import cupy as cp + + array = np.asarray(array) + mem = cp.cuda.alloc_pinned_memory(array.nbytes) + pinned = np.frombuffer(mem, array.dtype, array.size).reshape(array.shape) + pinned[...] = array + return pinned + + +@contextmanager +def stream() -> Generator[Any, None, None]: + """Context manager for a CUDA stream, to overlap transfers and compute. + + On the CuPy backend, operations issued inside the block are enqueued on + a new, non-blocking stream rather than the default one. Call + `xp.synchronize()` (or the yielded stream's own `.synchronize()`) before + reading results computed inside the block. No-op on the NumPy backend, + where it yields `None`. + """ + if array_backend.backend == "cupy": + import cupy as cp + + with cp.cuda.Stream(non_blocking=True) as s: + yield s + else: + yield None + + +def get_rng(seed: int | None = None) -> Any: + """Return a random Generator matching the active backend. + + NumPy and CuPy both provide `default_rng(seed)`, returning a + `Generator` with a largely-compatible distribution API, but picking the + right one requires branching on the backend -- this does that for you. + """ + if array_backend.backend == "cupy": + import cupy as cp + + return cp.random.default_rng(seed) + + import numpy as numpy_raw + + return numpy_raw.random.default_rng(seed) + + +def default_float_dtype() -> Any: + """Return the active backend's `float64` dtype object. + + NumPy and CuPy resolve Python `int`/`float` literals and the bare + `dtype=float` spelling to a platform- or backend-dependent default + (e.g. NumPy's default integer width differs between Windows and + Linux/macOS). Use ``dtype=xp.default_float_dtype()`` instead of + ``dtype=float`` when a specific, portable precision matters. + """ + return array_backend.xp.float64 + + def synchronize() -> None: """Wait for all kernels in all streams on current device to complete.""" if array_backend.backend == "cupy": @@ -154,7 +300,7 @@ def synchronize() -> None: def to_numpy(array: Any) -> np.ndarray: """Convert an array to a NumPy array.""" - if get_backend(array) == "cupy": + if get_array_backend(array) == "cupy": return array.get() return np.asarray(array) @@ -177,19 +323,36 @@ def to_cunumpy(array: Any) -> Any: return to_numpy(array) -def get_backend(array: Any) -> BackendType: - """Return 'cupy' or 'numpy' depending on the array type.""" +def get_array_backend(array: Any) -> BackendType: + """Return 'cupy' or 'numpy' depending on the array's type.""" return "cupy" if array_api_compat.is_cupy_array(array) else "numpy" +def get_array_module(array: Any) -> ModuleType: + """Return the array-api-compat module matching `array`'s own backend. + + Unlike `xp.xp`, which reflects the process-wide active backend, this + dispatches on the array itself -- useful for writing functions that + operate correctly regardless of what `set_backend`/`use_backend` last + selected. Mirrors `cupy.get_array_module`, but also works in Pyodide + (where `cupy` cannot be imported) and returns array-api-compat modules + for standard-conformant behavior, consistent with `xp.xp`. + """ + if get_array_backend(array) == "cupy": + import array_api_compat.cupy as cp + + return cp + return np + + def is_gpu(array: Any) -> bool: """Check if the array is stored on a GPU (CuPy).""" - return get_backend(array) == "cupy" + return get_array_backend(array) == "cupy" def is_cpu(array: Any) -> bool: """Check if the array is stored on a CPU (NumPy).""" - return get_backend(array) == "numpy" + return get_array_backend(array) == "numpy" def same_backend(*arrays: Any) -> bool: @@ -199,7 +362,7 @@ def same_backend(*arrays: Any) -> bool: """ if len(arrays) <= 1: return True - backends = {get_backend(array) for array in arrays} + backends = {get_array_backend(array) for array in arrays} return len(backends) == 1 @@ -213,7 +376,7 @@ def assert_same_backend(*arrays: Any) -> None: to align mismatched arrays onto one backend first. """ if not same_backend(*arrays): - backends = [get_backend(array) for array in arrays] + backends = [get_array_backend(array) for array in arrays] raise TypeError( f"Arrays are on mismatched backends: {backends}. Use " "xp.to_cunumpy()/xp.to_numpy()/xp.to_cupy() to align them first." diff --git a/tests/portable/test_numpy_runtime.py b/tests/portable/test_numpy_runtime.py index 6168af0..df174c4 100644 --- a/tests/portable/test_numpy_runtime.py +++ b/tests/portable/test_numpy_runtime.py @@ -18,7 +18,7 @@ def test_array_operations(): np.testing.assert_array_equal(a @ a.T, [[5, 14], [14, 50]]) np.testing.assert_allclose(xp.linalg.solve(xp.eye(2), xp.ones(2)), [1, 1]) assert xp.numpy_backend and not xp.cupy_backend - assert xp.get_backend(a) == "numpy" + assert xp.get_array_backend(a) == "numpy" assert xp.is_cpu(a) and not xp.is_gpu(a) assert xp.same_backend(a, xp.ones(2)) xp.assert_same_backend(a, xp.ones(2)) @@ -44,11 +44,10 @@ def test_backend_context_restoration(raises): original = backend.array_backend.xp try: - with xp.use_backend("numpy"): - with xp.use_backend("numpy"): - xp.set_backend("numpy") - if raises: - raise RuntimeError("kernel failed") + with xp.use_backend("numpy"), xp.use_backend("numpy"): + xp.set_backend("numpy") + if raises: + raise RuntimeError("kernel failed") except RuntimeError: assert raises assert backend.array_backend.xp is original diff --git a/tests/unit/test_array_api_compat_backend.py b/tests/unit/test_array_api_compat_backend.py index 6d497cd..175be61 100644 --- a/tests/unit/test_array_api_compat_backend.py +++ b/tests/unit/test_array_api_compat_backend.py @@ -69,7 +69,7 @@ def test_round_trip_preserves_values_and_dtype(dtype): def test_round_trip_preserves_shape(shape): _skip_without_cupy() - original = np.random.rand(*shape).astype(np.float64) + original = np.asarray(np.random.rand(*shape), dtype=np.float64) gpu = xp.to_cupy(original) back = xp.to_numpy(gpu) @@ -77,23 +77,23 @@ def test_round_trip_preserves_shape(shape): assert np.allclose(back, original) -def test_get_backend_is_gpu_is_cpu_consistency(): +def test_get_array_backend_is_gpu_is_cpu_consistency(): _skip_without_cupy() arr_cpu = np.array([1, 2, 3]) arr_gpu = xp.to_cupy(arr_cpu) - assert xp.get_backend(arr_cpu) == "numpy" + assert xp.get_array_backend(arr_cpu) == "numpy" assert xp.is_cpu(arr_cpu) is True assert xp.is_gpu(arr_cpu) is False - assert xp.get_backend(arr_gpu) == "cupy" + assert xp.get_array_backend(arr_gpu) == "cupy" assert xp.is_gpu(arr_gpu) is True assert xp.is_cpu(arr_gpu) is False def test_array_api_compat_agrees_with_cunumpy_on_real_gpu_array(): - """Sanity-check our get_backend() against array_api_compat's own + """Sanity-check our get_array_backend() against array_api_compat's own detector directly, on a real (non-mocked) GPU array.""" _skip_without_cupy() diff --git a/tests/unit/test_cunumpy.py b/tests/unit/test_cunumpy.py index 38bf4d4..938c9bd 100644 --- a/tests/unit/test_cunumpy.py +++ b/tests/unit/test_cunumpy.py @@ -24,11 +24,60 @@ def test_to_cunumpy(): def test_get_backend_and_is_gpu_cpu(): arr = np.array([1, 2, 3]) - assert xp.get_backend(arr) == "numpy" + assert xp.get_array_backend(arr) == "numpy" assert xp.is_gpu(arr) is False assert xp.is_cpu(arr) is True +def test_get_backend_reports_active_selection_independently_of_array(): + arr = np.array([1, 2, 3]) + with xp.use_backend("numpy"): + assert xp.get_backend() == "numpy" + assert xp.get_array_backend(arr) == "numpy" + + with xp.use_backend("cupy"): + expected = "cupy" if xp.cupy_available() else "numpy" + assert xp.get_backend() == expected + assert xp.get_array_backend(arr) == "numpy" + + assert xp.get_backend() == "numpy" + + +def test_invalid_backend_selection_preserves_active_backend(): + with xp.use_backend("numpy"): + with pytest.raises(ValueError, match="Array backend"): + xp.set_backend("invalid") + assert xp.get_backend() == "numpy" + + with pytest.raises(ValueError, match="Array backend"): + with xp.use_backend("invalid"): + pass + assert xp.get_backend() == "numpy" + + +def test_get_array_module_numpy(): + arr = np.array([1, 2, 3]) + mod = xp.get_array_module(arr) + assert "numpy" in mod.__name__ + assert mod.asarray(arr) is not None + + +def test_get_array_module_matches_array_not_global_backend(): + if not xp.cupy_available(): + pytest.skip("CuPy not installed or not functional") + + a_cpu = np.array([1, 2, 3]) + a_gpu = xp.to_cupy(a_cpu) + + with xp.use_backend("cupy"): + # Global backend is cupy, but the array itself is on the host. + assert "numpy" in xp.get_array_module(a_cpu).__name__ + + with xp.use_backend("numpy"): + # Global backend is numpy, but the array itself is on the device. + assert "cupy" in xp.get_array_module(a_gpu).__name__ + + def test_same_backend_trivially_true_for_zero_or_one_array(): assert xp.same_backend() is True assert xp.same_backend(np.array([1, 2, 3])) is True @@ -93,6 +142,135 @@ def test_set_backend(): assert arr2 is not None +def test_device_count_is_zero_without_cupy(): + if xp.cupy_available(): + pytest.skip("CuPy is installed/functional; device_count() may be > 0") + assert xp.device_count() == 0 + + +def test_device_count_matches_cupy_when_available(): + if not xp.cupy_available(): + pytest.skip("CuPy not installed or not functional") + import cupy as cp + + assert xp.device_count() == cp.cuda.runtime.getDeviceCount() + + +def test_set_device_for_rank_is_noop_without_gpus(): + if xp.cupy_available(): + pytest.skip("CuPy is installed/functional") + assert xp.set_device_for_rank(3) == 0 + + +def test_set_device_for_rank_wraps_around_devices_per_node(): + if not xp.cupy_available(): + pytest.skip("CuPy not installed or not functional") + n = xp.device_count() + assert xp.set_device_for_rank(n, devices_per_node=n) == 0 + assert xp.set_device_for_rank(n + 1, devices_per_node=n) == 1 % n + + +def test_memory_info_is_none_on_numpy_backend(): + with xp.use_backend("numpy"): + assert xp.memory_info() is None + + +def test_memory_info_returns_free_and_total_on_cupy(): + if not xp.cupy_available(): + pytest.skip("CuPy not installed or not functional") + with xp.use_backend("cupy"): + free, total = xp.memory_info() + assert 0 <= free <= total + + +def test_free_memory_is_noop_on_numpy_backend(): + with xp.use_backend("numpy"): + xp.free_memory() # must not raise + + +def test_free_memory_does_not_increase_cupy_pool_cache(): + if not xp.cupy_available(): + pytest.skip("CuPy not installed or not functional") + import cupy as cp + + with xp.use_backend("cupy"): + array = xp.to_cupy(np.ones(1_000)) + del array + pool = cp.get_default_memory_pool() + cached_before = pool.free_bytes() + xp.free_memory() # must not raise + # CuPy can retain split blocks even after free_all_blocks(). + assert pool.free_bytes() <= cached_before + + +def test_pin_memory_requires_cupy(): + if xp.cupy_available(): + pytest.skip("CuPy is installed/functional") + with pytest.raises(ImportError): + xp.pin_memory(np.ones(3)) + + +def test_pin_memory_round_trips_values(): + if not xp.cupy_available(): + pytest.skip("CuPy not installed or not functional") + arr = np.array([1.0, 2.0, 3.0]) + pinned = xp.pin_memory(arr) + assert np.array_equal(pinned, arr) + + +def test_stream_is_noop_on_numpy_backend(): + with xp.use_backend("numpy"), xp.stream() as s: + assert s is None + + +def test_stream_yields_a_cupy_stream_on_cupy_backend(): + if not xp.cupy_available(): + pytest.skip("CuPy not installed or not functional") + import cupy as cp + + with xp.use_backend("cupy"): + with xp.stream() as s: + assert isinstance(s, cp.cuda.Stream) + arr = xp.zeros(10) + assert xp.is_gpu(arr) + xp.synchronize() + + +def test_get_rng_returns_numpy_generator_on_numpy_backend(): + with xp.use_backend("numpy"): + rng = xp.get_rng(42) + assert isinstance(rng, np.random.Generator) + assert rng.random(3).shape == (3,) + + +def test_get_rng_returns_cupy_generator_on_cupy_backend(): + if not xp.cupy_available(): + pytest.skip("CuPy not installed or not functional") + import cupy as cp + + with xp.use_backend("cupy"): + rng = xp.get_rng(42) + assert isinstance(rng, cp.random.Generator) + + +def test_get_rng_is_reproducible_given_a_seed(): + with xp.use_backend("numpy"): + a = xp.get_rng(123).random(5) + b = xp.get_rng(123).random(5) + assert np.array_equal(a, b) + + +def test_default_float_dtype_matches_active_backend(): + with xp.use_backend("numpy"): + assert xp.default_float_dtype() == np.float64 + + if xp.cupy_available(): + import cupy as cp + + with xp.use_backend("cupy"): + assert xp.default_float_dtype() == cp.float64 + + def test_backend_bools(): with xp.use_backend("numpy"): assert xp.numpy_backend is True diff --git a/tests/unit/test_numpy.py b/tests/unit/test_numpy.py index 2247f10..b9e9f48 100644 --- a/tests/unit/test_numpy.py +++ b/tests/unit/test_numpy.py @@ -24,6 +24,7 @@ def test_numpy_symbols_accessible(): "to_cupy", "to_cunumpy", "get_backend", + "get_array_backend", "is_gpu", "is_cpu", "use_backend", diff --git a/tests/unit/test_pyccel_kernel.py b/tests/unit/test_pyccel_kernel.py index 1288b6c..e25abc5 100644 --- a/tests/unit/test_pyccel_kernel.py +++ b/tests/unit/test_pyccel_kernel.py @@ -282,6 +282,48 @@ def kernel(obj): PyccelKernel(kernel)(holder) +# --------------------------------------------------------------------------- +# Custom is_array predicate +# --------------------------------------------------------------------------- + + +class _HostArrayLike: + """Stand-in for a custom host array type that isn't a `np.ndarray`.""" + + def __init__(self, data: np.ndarray) -> None: + self.data = data + + def __array__(self, dtype=None): + return np.asarray(self.data, dtype=dtype) + + +def test_default_is_array_ignores_custom_array_like_return_value(): + _skip_without_cupy() + + def kernel(): + return _HostArrayLike(np.ones(3)) + + result = PyccelKernel(kernel, use_cupy=True)() + assert isinstance(result, _HostArrayLike) + assert xp.is_cpu(result.data) # not moved back: not a np.ndarray + + +def test_custom_is_array_moves_custom_return_value_back_to_device(): + _skip_without_cupy() + + def kernel(): + return _HostArrayLike(np.ones(3)) + + wrapped = PyccelKernel( + kernel, + use_cupy=True, + is_array=lambda v: isinstance(v, _HostArrayLike), + ) + result = wrapped() + assert xp.is_gpu(result) + assert wrapped.is_array is wrapped._is_array + + # --------------------------------------------------------------------------- # Declared outputs # --------------------------------------------------------------------------- @@ -431,7 +473,7 @@ def test_compiled_scale_inplace(kernels, backend): PyccelKernel(kernels.scale_inplace)(x, 3.0) - assert xp.get_backend(x) == backend + assert xp.get_array_backend(x) == backend assert np.allclose(xp.to_numpy(x), 3.0 * np.arange(4)) diff --git a/tutorials/example_tutorial.ipynb b/tutorials/example_tutorial.ipynb deleted file mode 100644 index fc03a2d..0000000 --- a/tutorials/example_tutorial.ipynb +++ /dev/null @@ -1,56 +0,0 @@ -{ - "cells": [ - { - "cell_type": "markdown", - "id": "efb79425-958b-412c-afa9-055e27a69baa", - "metadata": {}, - "source": [ - "# Tutorial 1" - ] - }, - { - "cell_type": "code", - "execution_count": 1, - "id": "68ad1562-4953-444c-96c7-9026bcc54cc7", - "metadata": {}, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "arr = array([2, 4]) type(arr) = \n" - ] - } - ], - "source": [ - "import cunumpy as xp\n", - "\n", - "arr = xp.array([1, 2])\n", - "arr *= 2\n", - "\n", - "print(f\"{arr = } {type(arr) = }\")" - ] - } - ], - "metadata": { - "kernelspec": { - "display_name": "Python 3 (ipykernel)", - "language": "python", - "name": "python3" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.12.3" - } - }, - "nbformat": 4, - "nbformat_minor": 5 -}