cunumpy 0.3.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cunumpy-0.5.0/PKG-INFO +587 -0
- cunumpy-0.5.0/README.md +548 -0
- {cunumpy-0.3.0 → cunumpy-0.5.0}/pyproject.toml +13 -7
- cunumpy-0.5.0/src/cunumpy/LLM_GUIDE.md +541 -0
- cunumpy-0.5.0/src/cunumpy/__init__.py +207 -0
- cunumpy-0.5.0/src/cunumpy/__init__.pyi +46 -0
- cunumpy-0.5.0/src/cunumpy/_algorithms.py +253 -0
- cunumpy-0.5.0/src/cunumpy/_cuda_kernel.py +2719 -0
- cunumpy-0.5.0/src/cunumpy/_deprecated.py +19 -0
- cunumpy-0.5.0/src/cunumpy/_device.py +260 -0
- cunumpy-0.5.0/src/cunumpy/_dispatch.py +910 -0
- cunumpy-0.5.0/src/cunumpy/_emulation.py +439 -0
- cunumpy-0.5.0/src/cunumpy/_fake_cupy.py +97 -0
- cunumpy-0.5.0/src/cunumpy/_fake_cupy_impl.py +558 -0
- cunumpy-0.5.0/src/cunumpy/_fusion.py +115 -0
- cunumpy-0.5.0/src/cunumpy/_kernel.py +916 -0
- cunumpy-0.5.0/src/cunumpy/_mirror.py +252 -0
- cunumpy-0.5.0/src/cunumpy/_morton.py +189 -0
- cunumpy-0.5.0/src/cunumpy/_mpi.py +394 -0
- cunumpy-0.5.0/src/cunumpy/_mpi_serial.py +657 -0
- cunumpy-0.5.0/src/cunumpy/_philox.py +149 -0
- cunumpy-0.5.0/src/cunumpy/_profiling.py +148 -0
- cunumpy-0.5.0/src/cunumpy/_random_streams.py +214 -0
- cunumpy-0.5.0/src/cunumpy/_scipy_backend.py +157 -0
- cunumpy-0.5.0/src/cunumpy/_staging.py +292 -0
- cunumpy-0.5.0/src/cunumpy/_streams.py +97 -0
- cunumpy-0.5.0/src/cunumpy/_transfers.py +299 -0
- cunumpy-0.5.0/src/cunumpy/algorithms.py +39 -0
- cunumpy-0.5.0/src/cunumpy/cuda/__init__.py +103 -0
- cunumpy-0.5.0/src/cunumpy/cuda/include/cunumpy/array_view.cuh +128 -0
- cunumpy-0.5.0/src/cunumpy/cuda/include/cunumpy/atomic.cuh +110 -0
- cunumpy-0.5.0/src/cunumpy/cuda/include/cunumpy/index.cuh +52 -0
- cunumpy-0.5.0/src/cunumpy/cuda/include/cunumpy/morton.cuh +128 -0
- cunumpy-0.5.0/src/cunumpy/cuda/include/cunumpy/random.cuh +127 -0
- cunumpy-0.5.0/src/cunumpy/cuda/include/cunumpy/reduce.cuh +192 -0
- cunumpy-0.5.0/src/cunumpy/cuda/include/cunumpy/scan.cuh +84 -0
- cunumpy-0.5.0/src/cunumpy/cuda_kernel.py +8 -0
- cunumpy-0.5.0/src/cunumpy/dispatch.py +8 -0
- cunumpy-0.5.0/src/cunumpy/kernel.py +8 -0
- cunumpy-0.5.0/src/cunumpy/kernel_testing.py +637 -0
- cunumpy-0.5.0/src/cunumpy/kernels.py +65 -0
- cunumpy-0.5.0/src/cunumpy/memory.py +22 -0
- cunumpy-0.5.0/src/cunumpy/mpi.py +60 -0
- cunumpy-0.5.0/src/cunumpy/petsc.py +122 -0
- cunumpy-0.5.0/src/cunumpy/profiling.py +32 -0
- cunumpy-0.5.0/src/cunumpy/rng.py +43 -0
- cunumpy-0.5.0/src/cunumpy/testing.py +9 -0
- cunumpy-0.5.0/src/cunumpy/xp.py +494 -0
- cunumpy-0.5.0/src/cunumpy.egg-info/PKG-INFO +587 -0
- cunumpy-0.5.0/src/cunumpy.egg-info/SOURCES.txt +54 -0
- {cunumpy-0.3.0 → cunumpy-0.5.0}/src/cunumpy.egg-info/requires.txt +2 -2
- cunumpy-0.3.0/PKG-INFO +0 -356
- cunumpy-0.3.0/README.md +0 -317
- cunumpy-0.3.0/src/cunumpy/__init__.py +0 -102
- cunumpy-0.3.0/src/cunumpy/__init__.pyi +0 -55
- cunumpy-0.3.0/src/cunumpy/cuda_kernel.py +0 -1003
- cunumpy-0.3.0/src/cunumpy/dispatch.py +0 -352
- cunumpy-0.3.0/src/cunumpy/kernel.py +0 -360
- cunumpy-0.3.0/src/cunumpy/xp.py +0 -486
- cunumpy-0.3.0/src/cunumpy.egg-info/PKG-INFO +0 -356
- cunumpy-0.3.0/src/cunumpy.egg-info/SOURCES.txt +0 -15
- {cunumpy-0.3.0 → cunumpy-0.5.0}/setup.cfg +0 -0
- {cunumpy-0.3.0 → cunumpy-0.5.0}/src/cunumpy/main.py +0 -0
- {cunumpy-0.3.0 → cunumpy-0.5.0}/src/cunumpy/py.typed +0 -0
- {cunumpy-0.3.0 → cunumpy-0.5.0}/src/cunumpy.egg-info/dependency_links.txt +0 -0
- {cunumpy-0.3.0 → cunumpy-0.5.0}/src/cunumpy.egg-info/top_level.txt +0 -0
cunumpy-0.5.0/PKG-INFO
ADDED
|
@@ -0,0 +1,587 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: cunumpy
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`.
|
|
5
|
+
Author: Max
|
|
6
|
+
Project-URL: Source, https://github.com/max-models/cunumpy
|
|
7
|
+
Keywords: python
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
Requires-Dist: array-api-compat
|
|
18
|
+
Requires-Dist: numpy
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: ruff; extra == "dev"
|
|
21
|
+
Requires-Dist: cunumpy[docs,test-compiled]; extra == "dev"
|
|
22
|
+
Provides-Extra: docs
|
|
23
|
+
Requires-Dist: ipykernel; extra == "docs"
|
|
24
|
+
Requires-Dist: myst-parser; extra == "docs"
|
|
25
|
+
Requires-Dist: nbconvert; extra == "docs"
|
|
26
|
+
Requires-Dist: nbsphinx; extra == "docs"
|
|
27
|
+
Requires-Dist: jupyterlab; extra == "docs"
|
|
28
|
+
Requires-Dist: pre-commit; extra == "docs"
|
|
29
|
+
Requires-Dist: pyproject-fmt; extra == "docs"
|
|
30
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
31
|
+
Requires-Dist: sphinx-book-theme; extra == "docs"
|
|
32
|
+
Provides-Extra: test
|
|
33
|
+
Requires-Dist: coverage; extra == "test"
|
|
34
|
+
Requires-Dist: pytest; extra == "test"
|
|
35
|
+
Requires-Dist: scipy; extra == "test"
|
|
36
|
+
Provides-Extra: test-compiled
|
|
37
|
+
Requires-Dist: cunumpy[test]; extra == "test-compiled"
|
|
38
|
+
Requires-Dist: pyccel; extra == "test-compiled"
|
|
39
|
+
|
|
40
|
+
# CuNumpy
|
|
41
|
+
|
|
42
|
+
CuNumpy lets a Python program use a NumPy-like API while choosing NumPy arrays
|
|
43
|
+
on the CPU or CuPy arrays on an NVIDIA GPU. In the simplest case, replace
|
|
44
|
+
`import numpy as np` with `import cunumpy as xp`; the array operations you
|
|
45
|
+
already know then run on the selected backend.
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
import cunumpy as xp
|
|
49
|
+
|
|
50
|
+
values = xp.arange(5, dtype=xp.float64)
|
|
51
|
+
print(values * 2)
|
|
52
|
+
print(xp.get_backend()) # 'numpy' by default
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
CuNumpy selects an array library for newly requested operations. It does not
|
|
56
|
+
move existing arrays just because the selected backend changes. This guide
|
|
57
|
+
covers backend selection, array movement, mixed CPU/GPU workflows, and the
|
|
58
|
+
helper APIs CuNumpy provides around NumPy and CuPy.
|
|
59
|
+
|
|
60
|
+
The top level of `cunumpy` is the NumPy (or CuPy) namespace plus backend
|
|
61
|
+
selection and array conversion. The helpers are in submodules, so that they
|
|
62
|
+
never hide a NumPy name:
|
|
63
|
+
|
|
64
|
+
| Submodule | Contents |
|
|
65
|
+
|---|---|
|
|
66
|
+
| `xp.cuda` | CUDA only: `CudaKernel`, `CudaStruct`, CUDA headers, devices, streams |
|
|
67
|
+
| `xp.kernels` | `Kernel`, `KernelCatalog`, `PyccelKernel`, host implementations, `fuse` |
|
|
68
|
+
| `xp.rng` | `random_streams`, `get_rng`, `philox_*` |
|
|
69
|
+
| `xp.algorithms` | `morton_*`, `sort_by_key`, `cell_offsets`, `segment_boundaries`, `segment_sum`, `SegmentPlan` |
|
|
70
|
+
| `xp.mpi` | `mpi_buffer`, reusable `MPIStaging`, CUDA-aware MPI |
|
|
71
|
+
| `xp.profiling` | `timed_region`, `nvtx_range`, `count_transfers` |
|
|
72
|
+
| `xp.memory` | `HostStaging`, `DeviceMirror` |
|
|
73
|
+
| `xp.petsc` | `petsc_vec` |
|
|
74
|
+
| `cunumpy.kernel_testing` | pytest helpers for host/CUDA kernel pairs |
|
|
75
|
+
|
|
76
|
+
Everything except `xp.cuda` works on both backends.
|
|
77
|
+
|
|
78
|
+
## Install
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
python -m pip install cunumpy
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
NumPy and `array-api-compat` are installed as dependencies. To use a GPU,
|
|
85
|
+
install a CuPy package compatible with your CUDA environment as well. CuPy
|
|
86
|
+
installation depends on the CUDA version and platform; follow the CuPy
|
|
87
|
+
installation instructions for your system. CuNumpy does not install CUDA.
|
|
88
|
+
|
|
89
|
+
`array-api-compat` supplies NumPy and CuPy compatibility modules with more
|
|
90
|
+
consistent behavior for shared array operations. CuNumpy uses them internally;
|
|
91
|
+
your arrays remain ordinary NumPy or CuPy arrays. See [why CuNumpy uses
|
|
92
|
+
`array-api-compat`](docs/source/array-api-compat.md) for a plain-language
|
|
93
|
+
explanation and examples.
|
|
94
|
+
|
|
95
|
+
## Choose a backend
|
|
96
|
+
|
|
97
|
+
CuNumpy starts with NumPy unless `CUNUMPY_BACKEND=cupy` is set before import.
|
|
98
|
+
You can also choose at runtime:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
import cunumpy as xp
|
|
102
|
+
|
|
103
|
+
xp.set_backend("cupy")
|
|
104
|
+
print(xp.get_backend()) # 'cupy' if CuPy and CUDA are functional
|
|
105
|
+
|
|
106
|
+
values = xp.arange(5) # created by the active backend
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
The accepted backend names are `"numpy"` and `"cupy"`. If CuPy is requested
|
|
110
|
+
but unavailable or not functional, CuNumpy falls back to NumPy. Always check
|
|
111
|
+
`get_backend()` when the effective backend matters, such as when reporting
|
|
112
|
+
configuration or deciding whether GPU-specific work will happen.
|
|
113
|
+
|
|
114
|
+
Use `xp.set_backend("cupy", strict=True)` to raise when CUDA is unavailable,
|
|
115
|
+
preserving the previous backend. `xp.backend_info()` returns structured backend,
|
|
116
|
+
dependency, and CUDA diagnostics. Reusable streams/events, MPI staging, cell
|
|
117
|
+
ranges, and prepared reductions are described in the
|
|
118
|
+
[execution helpers guide](docs/source/guides/execution-helpers.md).
|
|
119
|
+
|
|
120
|
+
Use `use_backend()` for a temporary selection. It restores the previous
|
|
121
|
+
selection when the block exits, including when an exception is raised:
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
with xp.use_backend("numpy"):
|
|
125
|
+
cpu_values = xp.linspace(0, 1, 100)
|
|
126
|
+
assert xp.get_backend() == "numpy"
|
|
127
|
+
|
|
128
|
+
# The previous global backend is active again here.
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
The backend selection is process-wide shared state. Do not switch it
|
|
132
|
+
independently from multiple threads or async tasks; those changes can
|
|
133
|
+
interfere. A context manager is useful for sequential code, tests, and
|
|
134
|
+
notebooks.
|
|
135
|
+
|
|
136
|
+
## Understand the two backend questions
|
|
137
|
+
|
|
138
|
+
The active backend controls which library CuNumpy exposes through its NumPy
|
|
139
|
+
like operations. The array backend reports where one particular array lives.
|
|
140
|
+
These can differ: changing the active backend does not convert arrays that
|
|
141
|
+
already exist.
|
|
142
|
+
|
|
143
|
+
```python
|
|
144
|
+
xp.set_backend("numpy")
|
|
145
|
+
cpu_values = xp.arange(3)
|
|
146
|
+
|
|
147
|
+
gpu_values = xp.to_cupy(cpu_values) # explicit transfer
|
|
148
|
+
print(xp.get_backend()) # 'numpy'
|
|
149
|
+
print(xp.get_array_backend(gpu_values)) # 'cupy'
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Use `is_cpu(array)`, `is_gpu(array)`, or `get_array_backend(array)` when
|
|
153
|
+
dispatch should follow the array passed to a function. `get_array_module()`
|
|
154
|
+
returns the matching `array_api_compat` module, which is useful when writing
|
|
155
|
+
backend-generic functions:
|
|
156
|
+
|
|
157
|
+
```python
|
|
158
|
+
def vector_norm(values):
|
|
159
|
+
array_xp = xp.get_array_module(values)
|
|
160
|
+
return array_xp.sqrt(array_xp.sum(values * values))
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
## Move data between CPU and GPU
|
|
164
|
+
|
|
165
|
+
Transfers are explicit so it is clear when data crosses the CPU/GPU boundary:
|
|
166
|
+
|
|
167
|
+
```python
|
|
168
|
+
host = xp.to_numpy(gpu_values) # CuPy -> NumPy (host)
|
|
169
|
+
device = xp.to_cupy(host) # NumPy/array-like -> CuPy (device)
|
|
170
|
+
active = xp.to_cunumpy(host) # convert to the currently selected backend
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
`to_numpy()` also accepts ordinary array-like values. `to_cupy()` raises
|
|
174
|
+
`ImportError` when CuPy or a functional CUDA runtime is unavailable.
|
|
175
|
+
`to_cunumpy()` is useful at API boundaries where the consumer expects the
|
|
176
|
+
currently selected backend. It does not change the original array.
|
|
177
|
+
|
|
178
|
+
Avoid transferring data inside a tight loop. Keep intermediate arrays on one
|
|
179
|
+
backend and move only at boundaries such as file I/O, plotting, or a
|
|
180
|
+
CPU-only library call. For example:
|
|
181
|
+
|
|
182
|
+
```python
|
|
183
|
+
with xp.use_backend("cupy"):
|
|
184
|
+
signal = xp.asarray(host_signal)
|
|
185
|
+
filtered = xp.fft.rfft(signal)
|
|
186
|
+
result = xp.to_numpy(filtered) # one transfer for a CPU-only consumer
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
To verify that a block, such as a time step, makes no transfer at all, count
|
|
190
|
+
them: `count_transfers()` records every `to_numpy()`, `to_cupy()` and
|
|
191
|
+
`to_cunumpy()` call that actually copies, mirror/staging refreshes, argument
|
|
192
|
+
conversion and kernel output copy-back, with call sites and payload byte counts.
|
|
193
|
+
Host kernel conversions and fallbacks have separate explanatory markers.
|
|
194
|
+
`assert_no_transfers()` rejects host/device movement and permits device-only
|
|
195
|
+
conversions. Only CuNumpy execution/conversion helpers are counted; forwarded
|
|
196
|
+
backend calls such as `xp.asarray()` and raw
|
|
197
|
+
`cupy.ndarray.get()` or `cupy.asarray()` calls need a profiler such as `nsys`.
|
|
198
|
+
|
|
199
|
+
```python
|
|
200
|
+
with xp.profiling.count_transfers() as counter:
|
|
201
|
+
propagator(dt)
|
|
202
|
+
|
|
203
|
+
assert counter.to_host == counter.to_device == 0, counter.report()
|
|
204
|
+
print(counter.bytes_to_host, counter.bytes_to_device)
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
## Random numbers and dtypes
|
|
208
|
+
|
|
209
|
+
`get_rng(seed)` returns a random generator for the active backend. NumPy and
|
|
210
|
+
CuPy have similar generator APIs, though exact bit-for-bit sequences are not
|
|
211
|
+
guaranteed to match between libraries:
|
|
212
|
+
|
|
213
|
+
```python
|
|
214
|
+
rng = xp.rng.get_rng(seed=42)
|
|
215
|
+
samples = rng.normal(size=1000)
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
Use `default_float_dtype()` when code needs to explicitly request the active
|
|
219
|
+
backend's `float64` dtype rather than rely on Python scalar inference:
|
|
220
|
+
|
|
221
|
+
```python
|
|
222
|
+
x = xp.asarray([1.0, 2.0], dtype=xp.default_float_dtype())
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
## GPU selection and memory helpers
|
|
226
|
+
|
|
227
|
+
These helpers are useful for multi-GPU programs and for understanding CuPy's
|
|
228
|
+
memory behavior:
|
|
229
|
+
|
|
230
|
+
```python
|
|
231
|
+
print("visible GPUs:", xp.cuda.device_count())
|
|
232
|
+
xp.cuda.set_device(0) # selects CUDA device 0 when CuPy is active
|
|
233
|
+
print("memory (free, total):", xp.cuda.memory_info())
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
`set_device()` is a no-op on NumPy. `device_count()` checks visible CUDA
|
|
237
|
+
hardware even if the active backend is NumPy; it returns zero when CuPy/CUDA
|
|
238
|
+
cannot be used. `memory_info()` returns `(free_bytes, total_bytes)` on the
|
|
239
|
+
active CuPy device and `None` on NumPy. `set_device_for_rank(rank)` is a
|
|
240
|
+
round-robin convenience for MPI layouts where local ranks map contiguously to
|
|
241
|
+
GPUs. If your scheduler uses a different mapping, select the device directly.
|
|
242
|
+
|
|
243
|
+
For MPI programs with one rank per GPU, the startup sequence is:
|
|
244
|
+
|
|
245
|
+
1. `bind_local_device()` selects the GPU from the node-local rank that the MPI
|
|
246
|
+
launcher exports (`local_rank()`) and creates its CUDA context. It runs
|
|
247
|
+
before MPI is initialized because a CUDA-aware MPI binds to the device that
|
|
248
|
+
is current at `MPI_Init`; without it, every rank of a node would use
|
|
249
|
+
device 0.
|
|
250
|
+
2. `from mpi4py import MPI` initializes MPI.
|
|
251
|
+
3. `require_cuda_aware_mpi()` (or `mpi_is_cuda_aware(comm)`) checks, with one
|
|
252
|
+
tiny device `Sendrecv` on every rank, that the MPI library can pass device
|
|
253
|
+
buffers at all. Passing CuPy arrays to a plain MPI build segfaults or
|
|
254
|
+
silently sends garbage; the check turns that into a clear error at
|
|
255
|
+
startup. It is a no-op on the NumPy backend.
|
|
256
|
+
4. `synchronize_for_mpi(*buffers)` before every MPI call with device buffers:
|
|
257
|
+
kernels run asynchronously, and MPI would otherwise send a buffer a kernel
|
|
258
|
+
is still writing, without an error.
|
|
259
|
+
|
|
260
|
+
```python
|
|
261
|
+
xp.set_backend("cupy")
|
|
262
|
+
xp.cuda.bind_local_device() # before MPI_Init
|
|
263
|
+
from mpi4py import MPI # MPI_Init
|
|
264
|
+
|
|
265
|
+
xp.mpi.require_cuda_aware_mpi() # once, on all ranks
|
|
266
|
+
|
|
267
|
+
xp.mpi.synchronize_for_mpi(send, recv)
|
|
268
|
+
MPI.COMM_WORLD.Sendrecv(send, dest, recvbuf=recv, source=source)
|
|
269
|
+
```
|
|
270
|
+
|
|
271
|
+
CuPy caches released allocations in memory pools. This can make process-level
|
|
272
|
+
GPU memory appear occupied after arrays go out of scope. `free_memory()` asks
|
|
273
|
+
CuPy to release currently free cached blocks; it does not free memory still
|
|
274
|
+
referenced by live arrays.
|
|
275
|
+
|
|
276
|
+
`pin_memory(host_array)` makes a pinned host copy, which can improve transfer
|
|
277
|
+
throughput for workloads that explicitly manage asynchronous transfers.
|
|
278
|
+
`stream()` creates a non-blocking CuPy stream and yields it; it yields `None`
|
|
279
|
+
on NumPy. GPU work is asynchronous, so synchronize before reading results on
|
|
280
|
+
the host:
|
|
281
|
+
|
|
282
|
+
```python
|
|
283
|
+
with xp.cuda.stream():
|
|
284
|
+
device = xp.to_cupy(host)
|
|
285
|
+
transformed = xp.fft.fft(device)
|
|
286
|
+
|
|
287
|
+
xp.synchronize()
|
|
288
|
+
result = xp.to_numpy(transformed)
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
Because GPU work is asynchronous, a wall-clock timer around a kernel launch
|
|
292
|
+
measures the launch, not the kernel. `timed_region(name)` synchronizes the
|
|
293
|
+
device before reading the clock (on NumPy it is a plain timer), and
|
|
294
|
+
`nvtx_range(name)` marks a region so it shows up in `nsys`/Nsight; both are
|
|
295
|
+
no-ops or plain timers on NumPy, and `nvtx_range` also works as a decorator:
|
|
296
|
+
|
|
297
|
+
```python
|
|
298
|
+
with xp.profiling.timed_region("fft") as timing:
|
|
299
|
+
transformed = xp.fft.fft(device)
|
|
300
|
+
print(timing.elapsed, timing.synced)
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
@xp.profiling.nvtx_range("step")
|
|
304
|
+
def step(dt): ...
|
|
305
|
+
```
|
|
306
|
+
|
|
307
|
+
## Use NumPy-only kernels with CuPy arrays
|
|
308
|
+
|
|
309
|
+
`PyccelKernel` adapts a callable that expects NumPy arrays. When conversion is
|
|
310
|
+
needed, CuNumpy copies CuPy inputs to the host, calls the wrapped function,
|
|
311
|
+
copies in-place output changes back to the device, and moves returned NumPy
|
|
312
|
+
arrays to CuPy. With NumPy inputs, the wrapper calls the function directly.
|
|
313
|
+
CuNumpy does not compile functions or import Pyccel for you.
|
|
314
|
+
|
|
315
|
+
```python
|
|
316
|
+
import cunumpy as xp
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def scale_in_place(values, factor):
|
|
320
|
+
values[:] *= factor
|
|
321
|
+
return values
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
scale = xp.kernels.PyccelKernel(scale_in_place, outputs=(0,))
|
|
325
|
+
|
|
326
|
+
with xp.use_backend("cupy"):
|
|
327
|
+
values = xp.arange(5, dtype=xp.float64)
|
|
328
|
+
returned = scale(values, 3.0)
|
|
329
|
+
xp.synchronize()
|
|
330
|
+
```
|
|
331
|
+
|
|
332
|
+
By default every converted argument is copied back, because the wrapper cannot
|
|
333
|
+
know which arguments the kernel changed. `outputs=(0,)` declares that
|
|
334
|
+
positional argument 0 is written, avoiding unnecessary copy-back for
|
|
335
|
+
read-only inputs. For a keyword call, declare the keyword name, such as
|
|
336
|
+
`outputs=("out",)`. A wrong declaration can leave GPU output values stale.
|
|
337
|
+
The wrapper can also traverse arrays nested in lists, tuples, dictionaries,
|
|
338
|
+
and selected application objects; see the full [API reference](docs/source/api.md)
|
|
339
|
+
for `object_modules`, `is_array`, aliasing, and output declarations.
|
|
340
|
+
|
|
341
|
+
## Write CUDA kernels next to host kernels
|
|
342
|
+
|
|
343
|
+
`CudaKernel` wraps a CUDA C kernel (compiled with NVRTC through
|
|
344
|
+
`cupy.RawKernel`) so that it is called with the same arguments as the host
|
|
345
|
+
kernel it mirrors. Thread counts default to the first array's leading shape
|
|
346
|
+
axes: one thread per row for 1D blocks, matching axes for 2D/3D blocks. Explicit
|
|
347
|
+
`n_threads`, `grid`, or a custom `n_threads_from` controls the launch when needed.
|
|
348
|
+
Arrays are never copied: they
|
|
349
|
+
must be C-contiguous CuPy arrays. The `extern "C" __global__` signature is
|
|
350
|
+
parsed once and every call is checked against it: Python scalars are cast to
|
|
351
|
+
the declared C types, and a wrong argument count, an array of the wrong dtype
|
|
352
|
+
or a non-contiguous view, or a scalar that does not fit its type raises instead
|
|
353
|
+
of silently producing wrong values.
|
|
354
|
+
|
|
355
|
+
`Kernel` pairs a host kernel with its CUDA kernel and calls the one matching
|
|
356
|
+
the active backend, so kernels can be ported to CUDA one at a time:
|
|
357
|
+
|
|
358
|
+
```python
|
|
359
|
+
import cunumpy as xp
|
|
360
|
+
|
|
361
|
+
AXPY = r"""
|
|
362
|
+
extern "C" __global__
|
|
363
|
+
void axpy(double a, const double* x, double* y, int n) {
|
|
364
|
+
int i = blockDim.x * blockIdx.x + threadIdx.x;
|
|
365
|
+
if (i < n) y[i] += a * x[i];
|
|
366
|
+
}
|
|
367
|
+
"""
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def axpy(a, x, y, n): # host version, e.g. compiled with Pyccel
|
|
371
|
+
for i in range(n):
|
|
372
|
+
y[i] += a * x[i]
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
kernel = xp.kernels.Kernel(axpy, xp.cuda.CudaKernel(AXPY, "axpy"))
|
|
376
|
+
|
|
377
|
+
with xp.use_backend("cupy"):
|
|
378
|
+
x = xp.arange(1000, dtype=xp.float64)
|
|
379
|
+
y = xp.zeros(1000)
|
|
380
|
+
kernel(2.0, x, y, 1000) # infer n_threads = x.shape[0], run the CUDA kernel
|
|
381
|
+
```
|
|
382
|
+
|
|
383
|
+
On the CuPy backend, a `Kernel` without CUDA kernel raises
|
|
384
|
+
`NotImplementedError` (or, with `missing_cuda="fallback"`, runs the host kernel
|
|
385
|
+
through `PyccelKernel`, with host copies; `host_options` configure that
|
|
386
|
+
`PyccelKernel`). `KernelCatalog.from_package()` collects kernel pairs from a
|
|
387
|
+
package with one folder per kernel (`name/name_kernels.py` and
|
|
388
|
+
`name/name_cuda.cu`), and `catalog.compile_all()` compiles all CUDA kernels at
|
|
389
|
+
setup.
|
|
390
|
+
|
|
391
|
+
Groups of arguments can be passed as one: objects implementing
|
|
392
|
+
`__cuda_args__()` (see `CudaArguments`) are flattened into several kernel
|
|
393
|
+
arguments, and `CudaStruct` defines a C struct once (its C `declaration` and
|
|
394
|
+
the matching memory layout) and packs values into it, which the kernel takes
|
|
395
|
+
as one parameter:
|
|
396
|
+
|
|
397
|
+
```python
|
|
398
|
+
Vec = xp.cuda.CudaStruct("Vec", [("data", "double*"), ("n", "int")])
|
|
399
|
+
scale = xp.cuda.CudaKernel(
|
|
400
|
+
Vec.declaration
|
|
401
|
+
+ r"""
|
|
402
|
+
extern "C" __global__ void scale(Vec v, double a) {
|
|
403
|
+
int i = blockDim.x * blockIdx.x + threadIdx.x;
|
|
404
|
+
if (i < v.n) v.data[i] *= a;
|
|
405
|
+
}""",
|
|
406
|
+
"scale",
|
|
407
|
+
structs=[Vec],
|
|
408
|
+
)
|
|
409
|
+
scale(Vec(data=y, n=y.size), 0.5, n_threads=y.size)
|
|
410
|
+
```
|
|
411
|
+
|
|
412
|
+
When building such argument objects, `xp.as_device_array(value, dtype,
|
|
413
|
+
ndim=None)` applies the "reference or copy once" rule: a CuPy array that
|
|
414
|
+
already has the dtype and is C-contiguous is returned as it is, anything else
|
|
415
|
+
(a tuple such as `degree = (3, 3, 3)`, a host array, another dtype, a
|
|
416
|
+
non-contiguous view) is converted; dtype and layout changes can require separate
|
|
417
|
+
device copies. Call it once when the object is
|
|
418
|
+
built, not per kernel call; on the NumPy backend it raises, so host data is
|
|
419
|
+
never copied to the device implicitly.
|
|
420
|
+
When the host kernel takes such a group as one object too (e.g. a Pyccel class
|
|
421
|
+
holding NumPy arrays), give the group both forms with `KernelArguments`:
|
|
422
|
+
`__host_args__()` returns the object for the host kernel, `__cuda_args__()`
|
|
423
|
+
the flattened device arguments. `Kernel` and `PyccelKernel` resolve
|
|
424
|
+
`__host_args__()` on the host path and `CudaKernel` flattens `__cuda_args__()`
|
|
425
|
+
on the CUDA path, so the call site is the same on both backends and each form
|
|
426
|
+
can be built lazily on first access (a CPU run never builds device arguments):
|
|
427
|
+
|
|
428
|
+
```python
|
|
429
|
+
class ParticleArguments(xp.kernels.KernelArguments):
|
|
430
|
+
def __init__(self, markers):
|
|
431
|
+
self.markers = markers
|
|
432
|
+
self._host = None
|
|
433
|
+
|
|
434
|
+
def __host_args__(self):
|
|
435
|
+
if self._host is None:
|
|
436
|
+
self._host = MarkerArguments(self.markers) # Pyccel class
|
|
437
|
+
return self._host
|
|
438
|
+
|
|
439
|
+
def __cuda_args__(self):
|
|
440
|
+
return (self.markers, self.markers.shape[0])
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
kernel(particles.kernel_args, dt, n_threads=n) # host or CUDA kernel
|
|
444
|
+
```
|
|
445
|
+
|
|
446
|
+
Kernels ported from pyccel index arrays like `markers[ip, j]`, which needs
|
|
447
|
+
shapes and strides rather than bare pointers. The shipped header
|
|
448
|
+
`cunumpy/array_view.cuh` (found by every `CudaKernel`) provides the strided
|
|
449
|
+
views `Array1D<T>` to `Array4D<T>`; a parameter or struct field of that type
|
|
450
|
+
takes a CuPy array, contiguous or not, and indexes `a(i, j)`. The struct can be
|
|
451
|
+
generated from the annotations of the pyccel argument class, so the Python
|
|
452
|
+
class is the one definition, and written to a header that a test keeps in sync:
|
|
453
|
+
|
|
454
|
+
```python
|
|
455
|
+
class MarkerArguments:
|
|
456
|
+
def __init__(self, markers: "float[:, :]", n_markers: int, valid: "bool[:]"): ...
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
MarkerArgs = xp.cuda.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs")
|
|
460
|
+
MarkerArgs.to_header(
|
|
461
|
+
"marker_args.cuh"
|
|
462
|
+
) # Array2D<double> markers; long long n_markers; ...
|
|
463
|
+
push = xp.cuda.CudaKernel(
|
|
464
|
+
r"""
|
|
465
|
+
#include "marker_args.cuh"
|
|
466
|
+
#include <cunumpy/index.cuh>
|
|
467
|
+
extern "C" __global__ void push(MarkerArgs m, double dt) {
|
|
468
|
+
CUNUMPY_THREAD_1D(ip, m.n_markers);
|
|
469
|
+
if (m.valid(ip)) m.markers(ip, 0) += dt * m.markers(ip, 3);
|
|
470
|
+
}""",
|
|
471
|
+
"push",
|
|
472
|
+
structs=[MarkerArgs],
|
|
473
|
+
include_dirs=["."],
|
|
474
|
+
)
|
|
475
|
+
push(
|
|
476
|
+
MarkerArgs(markers=markers, n_markers=markers.shape[0], valid=valid),
|
|
477
|
+
0.1,
|
|
478
|
+
n_threads=markers.shape[0],
|
|
479
|
+
)
|
|
480
|
+
```
|
|
481
|
+
|
|
482
|
+
Launches can be 1D to 3D (`n_threads=(nx, ny)`, `block_size=(16, 16)`) or use
|
|
483
|
+
an explicit `grid`, with dynamic shared memory (`shared_mem`) and a `stream`.
|
|
484
|
+
C++ function templates are instantiated with `template_args`, and
|
|
485
|
+
`CudaKernelVariants` caches kernels whose source is generated per variant
|
|
486
|
+
(e.g. per dimension and dtype). See the [API reference](docs/source/api.md) for
|
|
487
|
+
details.
|
|
488
|
+
|
|
489
|
+
Kernels run asynchronously, so a CUDA error (an illegal memory access, say)
|
|
490
|
+
normally surfaces at a later `.get()` or MPI call, far from the kernel that
|
|
491
|
+
caused it. In debug mode, enabled with `xp.cuda.set_cuda_debug(True)`, the
|
|
492
|
+
context manager `xp.cuda.cuda_debug()`, `CudaKernel(..., debug=True)` or the
|
|
493
|
+
environment variable `CUNUMPY_CUDA_DEBUG=1`, kernels are compiled with
|
|
494
|
+
`-lineinfo` and `-DCUNUMPY_BOUNDS_CHECK` and every launch is synchronized, so
|
|
495
|
+
the error is raised as a `RuntimeError` naming the kernel and its launch shape.
|
|
496
|
+
To find the faulting line and out-of-bounds accesses that do not crash, the
|
|
497
|
+
next step is NVIDIA's memory checker:
|
|
498
|
+
`CUNUMPY_CUDA_DEBUG=1 compute-sanitizer python -m pytest ...`.
|
|
499
|
+
|
|
500
|
+
## Test kernel pairs
|
|
501
|
+
|
|
502
|
+
`cunumpy.kernel_testing` helps to test the ports with pytest. `assert_kernels_agree`
|
|
503
|
+
builds the arguments on both backends, runs the host and the CUDA kernel and
|
|
504
|
+
compares the arrays they wrote; with `catalog.parity_cases()`, one
|
|
505
|
+
parametrised test covers every ported kernel of a catalog. `BACKENDS` and
|
|
506
|
+
`requires_cupy` parametrize tests over the backends, skipping CuPy without a
|
|
507
|
+
GPU, and `device_function_kernel` wraps a `__device__` helper in an elementwise
|
|
508
|
+
kernel so it can be checked against its host version without writing a test
|
|
509
|
+
kernel:
|
|
510
|
+
|
|
511
|
+
```python
|
|
512
|
+
import pytest
|
|
513
|
+
from cunumpy.kernel_testing import assert_kernels_agree
|
|
514
|
+
|
|
515
|
+
|
|
516
|
+
def make_args(backend, seed):
|
|
517
|
+
x = xp.to_cunumpy(np.random.default_rng(seed).random(1000))
|
|
518
|
+
return (x, 2.0, x.size)
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
@pytest.mark.parametrize("name, kernel", catalog.parity_cases())
|
|
522
|
+
def test_parity(name, kernel):
|
|
523
|
+
assert_kernels_agree(kernel, make_args, n_threads=1000)
|
|
524
|
+
```
|
|
525
|
+
|
|
526
|
+
Accumulation kernels often write into a buffer that another library owns on
|
|
527
|
+
the host (a stencil vector's `_data`, exchanged over MPI). `DeviceMirror`
|
|
528
|
+
pairs that NumPy array with a device copy: `mirror.device` is the CuPy array
|
|
529
|
+
on the GPU and the host array itself on the CPU, `to_host()` copies back in
|
|
530
|
+
place (the host array keeps its identity) and `zero()` clears the buffer, so
|
|
531
|
+
the one transfer per accumulation is explicit. The shipped header
|
|
532
|
+
`cunumpy/atomic.cuh` (found automatically, see `cuda_include_dir()`) provides
|
|
533
|
+
`cunumpy_atomic_add()` and 2D/3D indexed variants for the many-threads-to-one-cell
|
|
534
|
+
writes:
|
|
535
|
+
|
|
536
|
+
```python
|
|
537
|
+
mirror = xp.memory.DeviceMirror(vector._data)
|
|
538
|
+
mirror.zero()
|
|
539
|
+
accumulate(markers, mirror.device, n_threads=n_markers)
|
|
540
|
+
mirror.to_host() # vector._data holds the result on both backends
|
|
541
|
+
```
|
|
542
|
+
|
|
543
|
+
## Pyodide
|
|
544
|
+
|
|
545
|
+
CuNumpy supports the NumPy backend in Pyodide. It does not provide CuPy/CUDA
|
|
546
|
+
there. Ordinary Python callables can be wrapped with `PyccelKernel` without
|
|
547
|
+
compilation. See the [Pyodide guide](docs/source/pyodide.md) for a complete
|
|
548
|
+
installation example and compatibility notes.
|
|
549
|
+
|
|
550
|
+
## Documentation
|
|
551
|
+
|
|
552
|
+
The full documentation lives in [`docs/source`](docs/source/index.md) and is
|
|
553
|
+
published at <https://max-models.github.io/cunumpy/>:
|
|
554
|
+
|
|
555
|
+
* Getting started: [installation](docs/source/installation.md) and a
|
|
556
|
+
[quickstart](docs/source/quickstart.md) with a map of which guide covers what.
|
|
557
|
+
* User guide: [choosing a backend](docs/source/guides/backends.md),
|
|
558
|
+
[backend-agnostic code](docs/source/guides/portable-code.md),
|
|
559
|
+
[data movement](docs/source/guides/data-movement.md),
|
|
560
|
+
[devices, memory and streams](docs/source/guides/gpu-devices.md),
|
|
561
|
+
[MPI with one rank per GPU](docs/source/guides/mpi.md),
|
|
562
|
+
[timing and profiling](docs/source/guides/profiling.md).
|
|
563
|
+
* Porting kernels: [overview](docs/source/kernels/overview.md),
|
|
564
|
+
[`PyccelKernel`](docs/source/kernels/pyccel-kernel.md),
|
|
565
|
+
[`CudaKernel`](docs/source/kernels/cuda-kernel.md),
|
|
566
|
+
[`Kernel` and `KernelCatalog`](docs/source/kernels/dispatch.md),
|
|
567
|
+
[argument objects and structs](docs/source/kernels/arguments.md),
|
|
568
|
+
[accumulation kernels](docs/source/kernels/accumulation.md),
|
|
569
|
+
[debugging](docs/source/kernels/debugging.md),
|
|
570
|
+
[testing](docs/source/kernels/testing.md).
|
|
571
|
+
* [Worked examples](docs/source/examples/index.md),
|
|
572
|
+
[best practices](docs/source/best-practices.md),
|
|
573
|
+
[troubleshooting](docs/source/troubleshooting.md),
|
|
574
|
+
[Pyodide](docs/source/pyodide.md) and the
|
|
575
|
+
[API reference](docs/source/api.md).
|
|
576
|
+
|
|
577
|
+
### For AI coding assistants
|
|
578
|
+
|
|
579
|
+
[`src/cunumpy/LLM_GUIDE.md`](src/cunumpy/LLM_GUIDE.md) is a compact,
|
|
580
|
+
self-contained guide to the API and its rules for LLM-based coding assistants.
|
|
581
|
+
It ships inside the installed package, so an assistant working in a project that
|
|
582
|
+
depends on CuNumpy can read it from `site-packages/cunumpy/LLM_GUIDE.md`, or
|
|
583
|
+
locate it with:
|
|
584
|
+
|
|
585
|
+
```bash
|
|
586
|
+
python -c "import cunumpy, pathlib; print(pathlib.Path(cunumpy.__file__).parent / 'LLM_GUIDE.md')"
|
|
587
|
+
```
|