cunumpy 0.1.4__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cunumpy-0.2.0/PKG-INFO +267 -0
- cunumpy-0.2.0/README.md +227 -0
- {cunumpy-0.1.4 → cunumpy-0.2.0}/pyproject.toml +4 -3
- {cunumpy-0.1.4 → cunumpy-0.2.0}/src/cunumpy/__init__.py +24 -0
- {cunumpy-0.1.4 → cunumpy-0.2.0}/src/cunumpy/__init__.pyi +14 -1
- {cunumpy-0.1.4 → cunumpy-0.2.0}/src/cunumpy/kernel.py +23 -8
- cunumpy-0.2.0/src/cunumpy/xp.py +399 -0
- cunumpy-0.2.0/src/cunumpy.egg-info/PKG-INFO +267 -0
- {cunumpy-0.1.4 → cunumpy-0.2.0}/src/cunumpy.egg-info/requires.txt +5 -2
- cunumpy-0.1.4/PKG-INFO +0 -93
- cunumpy-0.1.4/README.md +0 -55
- cunumpy-0.1.4/src/cunumpy/xp.py +0 -208
- cunumpy-0.1.4/src/cunumpy.egg-info/PKG-INFO +0 -93
- {cunumpy-0.1.4 → cunumpy-0.2.0}/setup.cfg +0 -0
- {cunumpy-0.1.4 → cunumpy-0.2.0}/src/cunumpy/main.py +0 -0
- {cunumpy-0.1.4 → cunumpy-0.2.0}/src/cunumpy/py.typed +0 -0
- {cunumpy-0.1.4 → cunumpy-0.2.0}/src/cunumpy.egg-info/SOURCES.txt +0 -0
- {cunumpy-0.1.4 → cunumpy-0.2.0}/src/cunumpy.egg-info/dependency_links.txt +0 -0
- {cunumpy-0.1.4 → cunumpy-0.2.0}/src/cunumpy.egg-info/top_level.txt +0 -0
cunumpy-0.2.0/PKG-INFO
ADDED
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: cunumpy
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`.
|
|
5
|
+
Author: Max
|
|
6
|
+
Project-URL: Source, https://github.com/max-models/cunumpy
|
|
7
|
+
Keywords: python
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Requires-Python: >=3.8
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
Requires-Dist: array-api-compat
|
|
19
|
+
Requires-Dist: numpy
|
|
20
|
+
Provides-Extra: dev
|
|
21
|
+
Requires-Dist: black[jupyter]; extra == "dev"
|
|
22
|
+
Requires-Dist: isort; extra == "dev"
|
|
23
|
+
Requires-Dist: cunumpy[docs,test-compiled]; extra == "dev"
|
|
24
|
+
Provides-Extra: docs
|
|
25
|
+
Requires-Dist: ipykernel; extra == "docs"
|
|
26
|
+
Requires-Dist: myst-parser; extra == "docs"
|
|
27
|
+
Requires-Dist: nbconvert; extra == "docs"
|
|
28
|
+
Requires-Dist: nbsphinx; extra == "docs"
|
|
29
|
+
Requires-Dist: jupyterlab; extra == "docs"
|
|
30
|
+
Requires-Dist: pre-commit; extra == "docs"
|
|
31
|
+
Requires-Dist: pyproject-fmt; extra == "docs"
|
|
32
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
33
|
+
Requires-Dist: sphinx-book-theme; extra == "docs"
|
|
34
|
+
Provides-Extra: test
|
|
35
|
+
Requires-Dist: coverage; extra == "test"
|
|
36
|
+
Requires-Dist: pytest; extra == "test"
|
|
37
|
+
Provides-Extra: test-compiled
|
|
38
|
+
Requires-Dist: cunumpy[test]; extra == "test-compiled"
|
|
39
|
+
Requires-Dist: pyccel; extra == "test-compiled"
|
|
40
|
+
|
|
41
|
+
# CuNumpy
|
|
42
|
+
|
|
43
|
+
CuNumpy lets a Python program use a NumPy-like API while choosing NumPy arrays
|
|
44
|
+
on the CPU or CuPy arrays on an NVIDIA GPU. In the simplest case, replace
|
|
45
|
+
`import numpy as np` with `import cunumpy as xp`; the array operations you
|
|
46
|
+
already know then run on the selected backend.
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
import cunumpy as xp
|
|
50
|
+
|
|
51
|
+
values = xp.arange(5, dtype=xp.float64)
|
|
52
|
+
print(values * 2)
|
|
53
|
+
print(xp.get_backend()) # 'numpy' by default
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
CuNumpy selects an array library for newly requested operations. It does not
|
|
57
|
+
move existing arrays just because the selected backend changes. This guide
|
|
58
|
+
covers backend selection, array movement, mixed CPU/GPU workflows, and the
|
|
59
|
+
helper APIs CuNumpy provides around NumPy and CuPy.
|
|
60
|
+
|
|
61
|
+
## Install
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
python -m pip install cunumpy
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
NumPy and `array-api-compat` are installed as dependencies. To use a GPU,
|
|
68
|
+
install a CuPy package compatible with your CUDA environment as well. CuPy
|
|
69
|
+
installation depends on the CUDA version and platform; follow the CuPy
|
|
70
|
+
installation instructions for your system. CuNumpy does not install CUDA.
|
|
71
|
+
|
|
72
|
+
`array-api-compat` supplies NumPy and CuPy compatibility modules with more
|
|
73
|
+
consistent behavior for shared array operations. CuNumpy uses them internally;
|
|
74
|
+
your arrays remain ordinary NumPy or CuPy arrays. See [why CuNumpy uses
|
|
75
|
+
`array-api-compat`](docs/source/array-api-compat.md) for a plain-language
|
|
76
|
+
explanation and examples.
|
|
77
|
+
|
|
78
|
+
## Choose a backend
|
|
79
|
+
|
|
80
|
+
CuNumpy starts with NumPy unless `ARRAY_BACKEND=cupy` is set before import.
|
|
81
|
+
You can also choose at runtime:
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
import cunumpy as xp
|
|
85
|
+
|
|
86
|
+
xp.set_backend("cupy")
|
|
87
|
+
print(xp.get_backend()) # 'cupy' if CuPy and CUDA are functional
|
|
88
|
+
|
|
89
|
+
values = xp.arange(5) # created by the active backend
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
The accepted backend names are `"numpy"` and `"cupy"`. If CuPy is requested
|
|
93
|
+
but unavailable or not functional, CuNumpy falls back to NumPy. Always check
|
|
94
|
+
`get_backend()` when the effective backend matters, such as when reporting
|
|
95
|
+
configuration or deciding whether GPU-specific work will happen.
|
|
96
|
+
|
|
97
|
+
Use `use_backend()` for a temporary selection. It restores the previous
|
|
98
|
+
selection when the block exits, including when an exception is raised:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
with xp.use_backend("numpy"):
|
|
102
|
+
cpu_values = xp.linspace(0, 1, 100)
|
|
103
|
+
assert xp.get_backend() == "numpy"
|
|
104
|
+
|
|
105
|
+
# The previous global backend is active again here.
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
The backend selection is process-wide shared state. Do not switch it
|
|
109
|
+
independently from multiple threads or async tasks; those changes can
|
|
110
|
+
interfere. A context manager is useful for sequential code, tests, and
|
|
111
|
+
notebooks.
|
|
112
|
+
|
|
113
|
+
## Understand the two backend questions
|
|
114
|
+
|
|
115
|
+
The active backend controls which library CuNumpy exposes through its NumPy
|
|
116
|
+
like operations. The array backend reports where one particular array lives.
|
|
117
|
+
These can differ: changing the active backend does not convert arrays that
|
|
118
|
+
already exist.
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
xp.set_backend("numpy")
|
|
122
|
+
cpu_values = xp.arange(3)
|
|
123
|
+
|
|
124
|
+
gpu_values = xp.to_cupy(cpu_values) # explicit transfer
|
|
125
|
+
print(xp.get_backend()) # 'numpy'
|
|
126
|
+
print(xp.get_array_backend(gpu_values)) # 'cupy'
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Use `is_cpu(array)`, `is_gpu(array)`, or `get_array_backend(array)` when
|
|
130
|
+
dispatch should follow the array passed to a function. `get_array_module()`
|
|
131
|
+
returns the matching `array_api_compat` module, which is useful when writing
|
|
132
|
+
backend-generic functions:
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
def vector_norm(values):
|
|
136
|
+
array_xp = xp.get_array_module(values)
|
|
137
|
+
return array_xp.sqrt(array_xp.sum(values * values))
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
## Move data between CPU and GPU
|
|
141
|
+
|
|
142
|
+
Transfers are explicit so it is clear when data crosses the CPU/GPU boundary:
|
|
143
|
+
|
|
144
|
+
```python
|
|
145
|
+
host = xp.to_numpy(gpu_values) # CuPy -> NumPy (host)
|
|
146
|
+
device = xp.to_cupy(host) # NumPy/array-like -> CuPy (device)
|
|
147
|
+
active = xp.to_cunumpy(host) # convert to the currently selected backend
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
`to_numpy()` also accepts ordinary array-like values. `to_cupy()` raises
|
|
151
|
+
`ImportError` when CuPy or a functional CUDA runtime is unavailable.
|
|
152
|
+
`to_cunumpy()` is useful at API boundaries where the consumer expects the
|
|
153
|
+
currently selected backend. It does not change the original array.
|
|
154
|
+
|
|
155
|
+
Avoid transferring data inside a tight loop. Keep intermediate arrays on one
|
|
156
|
+
backend and move only at boundaries such as file I/O, plotting, or a
|
|
157
|
+
CPU-only library call. For example:
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
with xp.use_backend("cupy"):
|
|
161
|
+
signal = xp.asarray(host_signal)
|
|
162
|
+
filtered = xp.fft.rfft(signal)
|
|
163
|
+
result = xp.to_numpy(filtered) # one transfer for a CPU-only consumer
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## Random numbers and dtypes
|
|
167
|
+
|
|
168
|
+
`get_rng(seed)` returns a random generator for the active backend. NumPy and
|
|
169
|
+
CuPy have similar generator APIs, though exact bit-for-bit sequences are not
|
|
170
|
+
guaranteed to match between libraries:
|
|
171
|
+
|
|
172
|
+
```python
|
|
173
|
+
rng = xp.get_rng(seed=42)
|
|
174
|
+
samples = rng.normal(size=1000)
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
Use `default_float_dtype()` when code needs to explicitly request the active
|
|
178
|
+
backend's `float64` dtype rather than rely on Python scalar inference:
|
|
179
|
+
|
|
180
|
+
```python
|
|
181
|
+
x = xp.asarray([1.0, 2.0], dtype=xp.default_float_dtype())
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
## GPU selection and memory helpers
|
|
185
|
+
|
|
186
|
+
These helpers are useful for multi-GPU programs and for understanding CuPy's
|
|
187
|
+
memory behavior:
|
|
188
|
+
|
|
189
|
+
```python
|
|
190
|
+
print("visible GPUs:", xp.device_count())
|
|
191
|
+
xp.set_device(0) # selects CUDA device 0 when CuPy is active
|
|
192
|
+
print("memory (free, total):", xp.memory_info())
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
`set_device()` is a no-op on NumPy. `device_count()` checks visible CUDA
|
|
196
|
+
hardware even if the active backend is NumPy; it returns zero when CuPy/CUDA
|
|
197
|
+
cannot be used. `memory_info()` returns `(free_bytes, total_bytes)` on the
|
|
198
|
+
active CuPy device and `None` on NumPy. `set_device_for_rank(rank)` is a
|
|
199
|
+
round-robin convenience for MPI layouts where local ranks map contiguously to
|
|
200
|
+
GPUs. If your scheduler uses a different mapping, select the device directly.
|
|
201
|
+
|
|
202
|
+
CuPy caches released allocations in memory pools. This can make process-level
|
|
203
|
+
GPU memory appear occupied after arrays go out of scope. `free_memory()` asks
|
|
204
|
+
CuPy to release currently free cached blocks; it does not free memory still
|
|
205
|
+
referenced by live arrays.
|
|
206
|
+
|
|
207
|
+
`pin_memory(host_array)` makes a pinned host copy, which can improve transfer
|
|
208
|
+
throughput for workloads that explicitly manage asynchronous transfers.
|
|
209
|
+
`stream()` creates a non-blocking CuPy stream and yields it; it yields `None`
|
|
210
|
+
on NumPy. GPU work is asynchronous, so synchronize before reading results on
|
|
211
|
+
the host:
|
|
212
|
+
|
|
213
|
+
```python
|
|
214
|
+
with xp.stream():
|
|
215
|
+
device = xp.to_cupy(host)
|
|
216
|
+
transformed = xp.fft.fft(device)
|
|
217
|
+
|
|
218
|
+
xp.synchronize()
|
|
219
|
+
result = xp.to_numpy(transformed)
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
## Use NumPy-only kernels with CuPy arrays
|
|
223
|
+
|
|
224
|
+
`PyccelKernel` adapts a callable that expects NumPy arrays. When conversion is
|
|
225
|
+
needed, CuNumpy copies CuPy inputs to the host, calls the wrapped function,
|
|
226
|
+
copies in-place output changes back to the device, and moves returned NumPy
|
|
227
|
+
arrays to CuPy. With NumPy inputs, the wrapper calls the function directly.
|
|
228
|
+
CuNumpy does not compile functions or import Pyccel for you.
|
|
229
|
+
|
|
230
|
+
```python
|
|
231
|
+
import cunumpy as xp
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def scale_in_place(values, factor):
|
|
235
|
+
values[:] *= factor
|
|
236
|
+
return values
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
scale = xp.PyccelKernel(scale_in_place, outputs=(0,))
|
|
240
|
+
|
|
241
|
+
with xp.use_backend("cupy"):
|
|
242
|
+
values = xp.arange(5, dtype=xp.float64)
|
|
243
|
+
returned = scale(values, 3.0)
|
|
244
|
+
xp.synchronize()
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
By default every converted argument is copied back, because the wrapper cannot
|
|
248
|
+
know which arguments the kernel changed. `outputs=(0,)` declares that
|
|
249
|
+
positional argument 0 is written, avoiding unnecessary copy-back for
|
|
250
|
+
read-only inputs. For a keyword call, declare the keyword name, such as
|
|
251
|
+
`outputs=("out",)`. A wrong declaration can leave GPU output values stale.
|
|
252
|
+
The wrapper can also traverse arrays nested in lists, tuples, dictionaries,
|
|
253
|
+
and selected application objects; see the full [API reference](docs/source/api.md)
|
|
254
|
+
for `object_modules`, `is_array`, aliasing, and output declarations.
|
|
255
|
+
|
|
256
|
+
## Pyodide
|
|
257
|
+
|
|
258
|
+
CuNumpy supports the NumPy backend in Pyodide. It does not provide CuPy/CUDA
|
|
259
|
+
there. Ordinary Python callables can be wrapped with `PyccelKernel` without
|
|
260
|
+
compilation. See the [Pyodide guide](docs/source/pyodide.md) for a complete
|
|
261
|
+
installation example and compatibility notes.
|
|
262
|
+
|
|
263
|
+
## Documentation
|
|
264
|
+
|
|
265
|
+
The [user guide](docs/source/quickstart.md) explains common workflows. The
|
|
266
|
+
[API reference](docs/source/api.md) documents each helper and its behavior.
|
|
267
|
+
The [Pyodide guide](docs/source/pyodide.md) covers WebAssembly usage.
|
cunumpy-0.2.0/README.md
ADDED
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
# CuNumpy
|
|
2
|
+
|
|
3
|
+
CuNumpy lets a Python program use a NumPy-like API while choosing NumPy arrays
|
|
4
|
+
on the CPU or CuPy arrays on an NVIDIA GPU. In the simplest case, replace
|
|
5
|
+
`import numpy as np` with `import cunumpy as xp`; the array operations you
|
|
6
|
+
already know then run on the selected backend.
|
|
7
|
+
|
|
8
|
+
```python
|
|
9
|
+
import cunumpy as xp
|
|
10
|
+
|
|
11
|
+
values = xp.arange(5, dtype=xp.float64)
|
|
12
|
+
print(values * 2)
|
|
13
|
+
print(xp.get_backend()) # 'numpy' by default
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
CuNumpy selects an array library for newly requested operations. It does not
|
|
17
|
+
move existing arrays just because the selected backend changes. This guide
|
|
18
|
+
covers backend selection, array movement, mixed CPU/GPU workflows, and the
|
|
19
|
+
helper APIs CuNumpy provides around NumPy and CuPy.
|
|
20
|
+
|
|
21
|
+
## Install
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
python -m pip install cunumpy
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
NumPy and `array-api-compat` are installed as dependencies. To use a GPU,
|
|
28
|
+
install a CuPy package compatible with your CUDA environment as well. CuPy
|
|
29
|
+
installation depends on the CUDA version and platform; follow the CuPy
|
|
30
|
+
installation instructions for your system. CuNumpy does not install CUDA.
|
|
31
|
+
|
|
32
|
+
`array-api-compat` supplies NumPy and CuPy compatibility modules with more
|
|
33
|
+
consistent behavior for shared array operations. CuNumpy uses them internally;
|
|
34
|
+
your arrays remain ordinary NumPy or CuPy arrays. See [why CuNumpy uses
|
|
35
|
+
`array-api-compat`](docs/source/array-api-compat.md) for a plain-language
|
|
36
|
+
explanation and examples.
|
|
37
|
+
|
|
38
|
+
## Choose a backend
|
|
39
|
+
|
|
40
|
+
CuNumpy starts with NumPy unless `ARRAY_BACKEND=cupy` is set before import.
|
|
41
|
+
You can also choose at runtime:
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
import cunumpy as xp
|
|
45
|
+
|
|
46
|
+
xp.set_backend("cupy")
|
|
47
|
+
print(xp.get_backend()) # 'cupy' if CuPy and CUDA are functional
|
|
48
|
+
|
|
49
|
+
values = xp.arange(5) # created by the active backend
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
The accepted backend names are `"numpy"` and `"cupy"`. If CuPy is requested
|
|
53
|
+
but unavailable or not functional, CuNumpy falls back to NumPy. Always check
|
|
54
|
+
`get_backend()` when the effective backend matters, such as when reporting
|
|
55
|
+
configuration or deciding whether GPU-specific work will happen.
|
|
56
|
+
|
|
57
|
+
Use `use_backend()` for a temporary selection. It restores the previous
|
|
58
|
+
selection when the block exits, including when an exception is raised:
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
with xp.use_backend("numpy"):
|
|
62
|
+
cpu_values = xp.linspace(0, 1, 100)
|
|
63
|
+
assert xp.get_backend() == "numpy"
|
|
64
|
+
|
|
65
|
+
# The previous global backend is active again here.
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
The backend selection is process-wide shared state. Do not switch it
|
|
69
|
+
independently from multiple threads or async tasks; those changes can
|
|
70
|
+
interfere. A context manager is useful for sequential code, tests, and
|
|
71
|
+
notebooks.
|
|
72
|
+
|
|
73
|
+
## Understand the two backend questions
|
|
74
|
+
|
|
75
|
+
The active backend controls which library CuNumpy exposes through its NumPy
|
|
76
|
+
like operations. The array backend reports where one particular array lives.
|
|
77
|
+
These can differ: changing the active backend does not convert arrays that
|
|
78
|
+
already exist.
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
xp.set_backend("numpy")
|
|
82
|
+
cpu_values = xp.arange(3)
|
|
83
|
+
|
|
84
|
+
gpu_values = xp.to_cupy(cpu_values) # explicit transfer
|
|
85
|
+
print(xp.get_backend()) # 'numpy'
|
|
86
|
+
print(xp.get_array_backend(gpu_values)) # 'cupy'
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Use `is_cpu(array)`, `is_gpu(array)`, or `get_array_backend(array)` when
|
|
90
|
+
dispatch should follow the array passed to a function. `get_array_module()`
|
|
91
|
+
returns the matching `array_api_compat` module, which is useful when writing
|
|
92
|
+
backend-generic functions:
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
def vector_norm(values):
|
|
96
|
+
array_xp = xp.get_array_module(values)
|
|
97
|
+
return array_xp.sqrt(array_xp.sum(values * values))
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
## Move data between CPU and GPU
|
|
101
|
+
|
|
102
|
+
Transfers are explicit so it is clear when data crosses the CPU/GPU boundary:
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
host = xp.to_numpy(gpu_values) # CuPy -> NumPy (host)
|
|
106
|
+
device = xp.to_cupy(host) # NumPy/array-like -> CuPy (device)
|
|
107
|
+
active = xp.to_cunumpy(host) # convert to the currently selected backend
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
`to_numpy()` also accepts ordinary array-like values. `to_cupy()` raises
|
|
111
|
+
`ImportError` when CuPy or a functional CUDA runtime is unavailable.
|
|
112
|
+
`to_cunumpy()` is useful at API boundaries where the consumer expects the
|
|
113
|
+
currently selected backend. It does not change the original array.
|
|
114
|
+
|
|
115
|
+
Avoid transferring data inside a tight loop. Keep intermediate arrays on one
|
|
116
|
+
backend and move only at boundaries such as file I/O, plotting, or a
|
|
117
|
+
CPU-only library call. For example:
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
with xp.use_backend("cupy"):
|
|
121
|
+
signal = xp.asarray(host_signal)
|
|
122
|
+
filtered = xp.fft.rfft(signal)
|
|
123
|
+
result = xp.to_numpy(filtered) # one transfer for a CPU-only consumer
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
## Random numbers and dtypes
|
|
127
|
+
|
|
128
|
+
`get_rng(seed)` returns a random generator for the active backend. NumPy and
|
|
129
|
+
CuPy have similar generator APIs, though exact bit-for-bit sequences are not
|
|
130
|
+
guaranteed to match between libraries:
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
rng = xp.get_rng(seed=42)
|
|
134
|
+
samples = rng.normal(size=1000)
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
Use `default_float_dtype()` when code needs to explicitly request the active
|
|
138
|
+
backend's `float64` dtype rather than rely on Python scalar inference:
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
x = xp.asarray([1.0, 2.0], dtype=xp.default_float_dtype())
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
## GPU selection and memory helpers
|
|
145
|
+
|
|
146
|
+
These helpers are useful for multi-GPU programs and for understanding CuPy's
|
|
147
|
+
memory behavior:
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
print("visible GPUs:", xp.device_count())
|
|
151
|
+
xp.set_device(0) # selects CUDA device 0 when CuPy is active
|
|
152
|
+
print("memory (free, total):", xp.memory_info())
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
`set_device()` is a no-op on NumPy. `device_count()` checks visible CUDA
|
|
156
|
+
hardware even if the active backend is NumPy; it returns zero when CuPy/CUDA
|
|
157
|
+
cannot be used. `memory_info()` returns `(free_bytes, total_bytes)` on the
|
|
158
|
+
active CuPy device and `None` on NumPy. `set_device_for_rank(rank)` is a
|
|
159
|
+
round-robin convenience for MPI layouts where local ranks map contiguously to
|
|
160
|
+
GPUs. If your scheduler uses a different mapping, select the device directly.
|
|
161
|
+
|
|
162
|
+
CuPy caches released allocations in memory pools. This can make process-level
|
|
163
|
+
GPU memory appear occupied after arrays go out of scope. `free_memory()` asks
|
|
164
|
+
CuPy to release currently free cached blocks; it does not free memory still
|
|
165
|
+
referenced by live arrays.
|
|
166
|
+
|
|
167
|
+
`pin_memory(host_array)` makes a pinned host copy, which can improve transfer
|
|
168
|
+
throughput for workloads that explicitly manage asynchronous transfers.
|
|
169
|
+
`stream()` creates a non-blocking CuPy stream and yields it; it yields `None`
|
|
170
|
+
on NumPy. GPU work is asynchronous, so synchronize before reading results on
|
|
171
|
+
the host:
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
with xp.stream():
|
|
175
|
+
device = xp.to_cupy(host)
|
|
176
|
+
transformed = xp.fft.fft(device)
|
|
177
|
+
|
|
178
|
+
xp.synchronize()
|
|
179
|
+
result = xp.to_numpy(transformed)
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
## Use NumPy-only kernels with CuPy arrays
|
|
183
|
+
|
|
184
|
+
`PyccelKernel` adapts a callable that expects NumPy arrays. When conversion is
|
|
185
|
+
needed, CuNumpy copies CuPy inputs to the host, calls the wrapped function,
|
|
186
|
+
copies in-place output changes back to the device, and moves returned NumPy
|
|
187
|
+
arrays to CuPy. With NumPy inputs, the wrapper calls the function directly.
|
|
188
|
+
CuNumpy does not compile functions or import Pyccel for you.
|
|
189
|
+
|
|
190
|
+
```python
|
|
191
|
+
import cunumpy as xp
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def scale_in_place(values, factor):
|
|
195
|
+
values[:] *= factor
|
|
196
|
+
return values
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
scale = xp.PyccelKernel(scale_in_place, outputs=(0,))
|
|
200
|
+
|
|
201
|
+
with xp.use_backend("cupy"):
|
|
202
|
+
values = xp.arange(5, dtype=xp.float64)
|
|
203
|
+
returned = scale(values, 3.0)
|
|
204
|
+
xp.synchronize()
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
By default every converted argument is copied back, because the wrapper cannot
|
|
208
|
+
know which arguments the kernel changed. `outputs=(0,)` declares that
|
|
209
|
+
positional argument 0 is written, avoiding unnecessary copy-back for
|
|
210
|
+
read-only inputs. For a keyword call, declare the keyword name, such as
|
|
211
|
+
`outputs=("out",)`. A wrong declaration can leave GPU output values stale.
|
|
212
|
+
The wrapper can also traverse arrays nested in lists, tuples, dictionaries,
|
|
213
|
+
and selected application objects; see the full [API reference](docs/source/api.md)
|
|
214
|
+
for `object_modules`, `is_array`, aliasing, and output declarations.
|
|
215
|
+
|
|
216
|
+
## Pyodide
|
|
217
|
+
|
|
218
|
+
CuNumpy supports the NumPy backend in Pyodide. It does not provide CuPy/CUDA
|
|
219
|
+
there. Ordinary Python callables can be wrapped with `PyccelKernel` without
|
|
220
|
+
compilation. See the [Pyodide guide](docs/source/pyodide.md) for a complete
|
|
221
|
+
installation example and compatibility notes.
|
|
222
|
+
|
|
223
|
+
## Documentation
|
|
224
|
+
|
|
225
|
+
The [user guide](docs/source/quickstart.md) explains common workflows. The
|
|
226
|
+
[API reference](docs/source/api.md) documents each helper and its behavior.
|
|
227
|
+
The [Pyodide guide](docs/source/pyodide.md) covers WebAssembly usage.
|
|
@@ -5,7 +5,7 @@ requires = [ "setuptools", "wheel" ]
|
|
|
5
5
|
|
|
6
6
|
[project]
|
|
7
7
|
name = "cunumpy"
|
|
8
|
-
version = "0.
|
|
8
|
+
version = "0.2.0"
|
|
9
9
|
description = "Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`."
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
keywords = [ "python" ]
|
|
@@ -30,7 +30,7 @@ dependencies = [
|
|
|
30
30
|
optional-dependencies.dev = [
|
|
31
31
|
"black[jupyter]",
|
|
32
32
|
"isort",
|
|
33
|
-
"cunumpy[test,docs]",
|
|
33
|
+
"cunumpy[test-compiled,docs]",
|
|
34
34
|
]
|
|
35
35
|
# https://medium.com/@pratikdomadiya123/build-project-documentation-quickly-with-the-sphinx-python-2a9732b66594
|
|
36
36
|
optional-dependencies.docs = [
|
|
@@ -44,7 +44,8 @@ optional-dependencies.docs = [
|
|
|
44
44
|
"sphinx",
|
|
45
45
|
"sphinx-book-theme",
|
|
46
46
|
]
|
|
47
|
-
optional-dependencies.test = [ "coverage", "
|
|
47
|
+
optional-dependencies.test = [ "coverage", "pytest" ]
|
|
48
|
+
optional-dependencies.test-compiled = [ "cunumpy[test]", "pyccel" ]
|
|
48
49
|
urls."Source" = "https://github.com/max-models/cunumpy"
|
|
49
50
|
|
|
50
51
|
[tool.setuptools.packages.find]
|
|
@@ -4,12 +4,24 @@ from importlib.metadata import PackageNotFoundError, version
|
|
|
4
4
|
from . import xp
|
|
5
5
|
from .kernel import PyccelKernel
|
|
6
6
|
from .xp import (
|
|
7
|
+
assert_same_backend,
|
|
7
8
|
cupy_available,
|
|
9
|
+
default_float_dtype,
|
|
10
|
+
device_count,
|
|
11
|
+
free_memory,
|
|
12
|
+
get_array_backend,
|
|
13
|
+
get_array_module,
|
|
8
14
|
get_backend,
|
|
15
|
+
get_rng,
|
|
9
16
|
is_cpu,
|
|
10
17
|
is_gpu,
|
|
18
|
+
memory_info,
|
|
19
|
+
pin_memory,
|
|
20
|
+
same_backend,
|
|
11
21
|
set_backend,
|
|
12
22
|
set_device,
|
|
23
|
+
set_device_for_rank,
|
|
24
|
+
stream,
|
|
13
25
|
synchronize,
|
|
14
26
|
to_cunumpy,
|
|
15
27
|
to_cupy,
|
|
@@ -25,14 +37,26 @@ except PackageNotFoundError:
|
|
|
25
37
|
__all__ = [
|
|
26
38
|
"PyccelKernel",
|
|
27
39
|
"__version__",
|
|
40
|
+
"assert_same_backend",
|
|
28
41
|
"cupy_available",
|
|
29
42
|
"cupy_backend",
|
|
43
|
+
"default_float_dtype",
|
|
44
|
+
"device_count",
|
|
45
|
+
"free_memory",
|
|
46
|
+
"get_array_backend",
|
|
47
|
+
"get_array_module",
|
|
30
48
|
"get_backend",
|
|
49
|
+
"get_rng",
|
|
31
50
|
"is_cpu",
|
|
32
51
|
"is_gpu",
|
|
52
|
+
"memory_info",
|
|
33
53
|
"numpy_backend",
|
|
54
|
+
"pin_memory",
|
|
55
|
+
"same_backend",
|
|
34
56
|
"set_backend",
|
|
35
57
|
"set_device",
|
|
58
|
+
"set_device_for_rank",
|
|
59
|
+
"stream",
|
|
36
60
|
"synchronize",
|
|
37
61
|
"to_cunumpy",
|
|
38
62
|
"to_cupy",
|
|
@@ -14,13 +14,26 @@ def to_numpy(array: Any) -> np.ndarray: ...
|
|
|
14
14
|
def to_cupy(array: Any) -> Any: ...
|
|
15
15
|
def to_cunumpy(array: Any) -> Any: ...
|
|
16
16
|
def cupy_available() -> bool: ...
|
|
17
|
-
def
|
|
17
|
+
def get_array_module(array: Any) -> Any: ...
|
|
18
|
+
def get_backend() -> str: ...
|
|
19
|
+
def get_array_backend(array: Any) -> str: ...
|
|
18
20
|
def is_gpu(array: Any) -> bool: ...
|
|
19
21
|
def is_cpu(array: Any) -> bool: ...
|
|
22
|
+
def same_backend(*arrays: Any) -> bool: ...
|
|
23
|
+
def assert_same_backend(*arrays: Any) -> None: ...
|
|
20
24
|
@contextmanager
|
|
21
25
|
def use_backend(backend: str) -> Generator[None]: ...
|
|
22
26
|
def set_backend(backend: str) -> None: ...
|
|
23
27
|
def set_device(device_id: int) -> None: ...
|
|
28
|
+
def set_device_for_rank(rank: int, devices_per_node: int | None = ...) -> int: ...
|
|
29
|
+
def device_count() -> int: ...
|
|
30
|
+
def memory_info() -> tuple[int, int] | None: ...
|
|
31
|
+
def free_memory() -> None: ...
|
|
32
|
+
def pin_memory(array: Any) -> Any: ...
|
|
33
|
+
@contextmanager
|
|
34
|
+
def stream() -> Generator[Any]: ...
|
|
35
|
+
def get_rng(seed: int | None = ...) -> Any: ...
|
|
36
|
+
def default_float_dtype() -> Any: ...
|
|
24
37
|
def synchronize() -> None: ...
|
|
25
38
|
|
|
26
39
|
numpy_backend: bool
|
|
@@ -8,6 +8,9 @@ made by the kernel are copied back to the device afterwards, and any arrays
|
|
|
8
8
|
returned by the kernel are moved back to the device.
|
|
9
9
|
|
|
10
10
|
On the NumPy backend the wrapper is a no-op and the kernel is called directly.
|
|
11
|
+
Ordinary Python callables are supported, including in Pyodide. This module
|
|
12
|
+
neither imports Pyccel nor compiles kernels; compilation, if desired, is the
|
|
13
|
+
caller's responsibility.
|
|
11
14
|
"""
|
|
12
15
|
|
|
13
16
|
from __future__ import annotations
|
|
@@ -24,7 +27,7 @@ __all__ = ["PyccelKernel"]
|
|
|
24
27
|
|
|
25
28
|
|
|
26
29
|
class PyccelKernel:
|
|
27
|
-
"""Call a Pyccel-compiled kernel with NumPy or CuPy arrays.
|
|
30
|
+
"""Call a NumPy callable or Pyccel-compiled kernel with NumPy or CuPy arrays.
|
|
28
31
|
|
|
29
32
|
Parameters
|
|
30
33
|
----------
|
|
@@ -38,6 +41,12 @@ class PyccelKernel:
|
|
|
38
41
|
Module prefixes (e.g. ``("struphy.", "feectools.")``) whose instances
|
|
39
42
|
should be traversed attribute-by-attribute when looking for arrays to
|
|
40
43
|
convert. Objects from other modules are passed through untouched.
|
|
44
|
+
is_array : callable, optional
|
|
45
|
+
Predicate deciding whether a host-side value returned by the kernel
|
|
46
|
+
(or reachable from a declared output) counts as an array to move
|
|
47
|
+
back to the device. Defaults to ``isinstance(value, np.ndarray)``.
|
|
48
|
+
Override this if the kernel returns/mutates a NumPy subclass or a
|
|
49
|
+
custom host array type that should also be converted back to CuPy.
|
|
41
50
|
outputs : sequence of int or str, optional
|
|
42
51
|
Which arguments the kernel writes to. Only those are copied back to the
|
|
43
52
|
device after the call, which avoids pointless device transfers for the
|
|
@@ -65,11 +74,13 @@ class PyccelKernel:
|
|
|
65
74
|
kernel: Callable[..., Any],
|
|
66
75
|
use_cupy: bool | None = None,
|
|
67
76
|
object_modules: Sequence[str] = (),
|
|
77
|
+
is_array: Callable[[Any], bool] | None = None,
|
|
68
78
|
outputs: Sequence[int | str] | None = None,
|
|
69
79
|
) -> None:
|
|
70
80
|
self._kernel = kernel
|
|
71
81
|
self._use_cupy = use_cupy
|
|
72
82
|
self._object_modules = tuple(object_modules)
|
|
83
|
+
self._is_array = is_array or (lambda value: isinstance(value, np.ndarray))
|
|
73
84
|
|
|
74
85
|
if outputs is None:
|
|
75
86
|
self._outputs: tuple[int | str, ...] | None = None
|
|
@@ -162,15 +173,14 @@ class PyccelKernel:
|
|
|
162
173
|
|
|
163
174
|
return value
|
|
164
175
|
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
if isinstance(value, np.ndarray):
|
|
176
|
+
def _convert_from_numpy(self, value: Any) -> Any:
|
|
177
|
+
"""Move host arrays returned by the kernel back to the device."""
|
|
178
|
+
if self._is_array(value):
|
|
169
179
|
return to_cupy(value)
|
|
170
180
|
if isinstance(value, tuple):
|
|
171
|
-
return tuple(
|
|
181
|
+
return tuple(self._convert_from_numpy(item) for item in value)
|
|
172
182
|
if isinstance(value, list):
|
|
173
|
-
return [
|
|
183
|
+
return [self._convert_from_numpy(item) for item in value]
|
|
174
184
|
return value
|
|
175
185
|
|
|
176
186
|
def _collect_host_arrays(self, value: Any, found: set[int], seen: set[int]) -> None:
|
|
@@ -180,7 +190,7 @@ class PyccelKernel:
|
|
|
180
190
|
:meth:`_convert_to_numpy`, so that an output declared as a container or
|
|
181
191
|
an object contributes the arrays nested inside it.
|
|
182
192
|
"""
|
|
183
|
-
if
|
|
193
|
+
if self._is_array(value):
|
|
184
194
|
found.add(id(value))
|
|
185
195
|
return
|
|
186
196
|
|
|
@@ -338,6 +348,11 @@ class PyccelKernel:
|
|
|
338
348
|
"""Module prefixes whose instances are traversed for arrays."""
|
|
339
349
|
return self._object_modules
|
|
340
350
|
|
|
351
|
+
@property
|
|
352
|
+
def is_array(self) -> Callable[[Any], bool]:
|
|
353
|
+
"""Predicate identifying host values to convert back to the device."""
|
|
354
|
+
return self._is_array
|
|
355
|
+
|
|
341
356
|
@property
|
|
342
357
|
def outputs(self) -> tuple[int | str, ...] | None:
|
|
343
358
|
"""Declared output arguments, or ``None`` if every array is copied back."""
|