cunumpy 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cunumpy-0.2.0 → cunumpy-0.3.0}/PKG-INFO +93 -4
- {cunumpy-0.2.0 → cunumpy-0.3.0}/README.md +90 -0
- {cunumpy-0.2.0 → cunumpy-0.3.0}/pyproject.toml +3 -4
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy/__init__.py +27 -0
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy/__init__.pyi +15 -1
- cunumpy-0.3.0/src/cunumpy/cuda_kernel.py +1003 -0
- cunumpy-0.3.0/src/cunumpy/dispatch.py +352 -0
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy/kernel.py +2 -1
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy/xp.py +88 -1
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy.egg-info/PKG-INFO +93 -4
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy.egg-info/SOURCES.txt +2 -0
- {cunumpy-0.2.0 → cunumpy-0.3.0}/setup.cfg +0 -0
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy/main.py +0 -0
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy/py.typed +0 -0
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy.egg-info/dependency_links.txt +0 -0
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy.egg-info/requires.txt +0 -0
- {cunumpy-0.2.0 → cunumpy-0.3.0}/src/cunumpy.egg-info/top_level.txt +0 -0
|
@@ -1,19 +1,18 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cunumpy
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`.
|
|
5
5
|
Author: Max
|
|
6
6
|
Project-URL: Source, https://github.com/max-models/cunumpy
|
|
7
7
|
Keywords: python
|
|
8
8
|
Classifier: Development Status :: 3 - Alpha
|
|
9
9
|
Classifier: Programming Language :: Python :: 3 :: Only
|
|
10
|
-
Classifier: Programming Language :: Python :: 3.8
|
|
11
|
-
Classifier: Programming Language :: Python :: 3.9
|
|
12
10
|
Classifier: Programming Language :: Python :: 3.10
|
|
13
11
|
Classifier: Programming Language :: Python :: 3.11
|
|
14
12
|
Classifier: Programming Language :: Python :: 3.12
|
|
15
13
|
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
-
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
|
+
Requires-Python: >=3.10
|
|
17
16
|
Description-Content-Type: text/markdown
|
|
18
17
|
Requires-Dist: array-api-compat
|
|
19
18
|
Requires-Dist: numpy
|
|
@@ -199,6 +198,22 @@ active CuPy device and `None` on NumPy. `set_device_for_rank(rank)` is a
|
|
|
199
198
|
round-robin convenience for MPI layouts where local ranks map contiguously to
|
|
200
199
|
GPUs. If your scheduler uses a different mapping, select the device directly.
|
|
201
200
|
|
|
201
|
+
For MPI programs with one rank per GPU, `bind_local_device()` selects the GPU
|
|
202
|
+
from the node-local rank that the MPI launcher exports (`local_rank()`), so it
|
|
203
|
+
can run before MPI is initialized, as CUDA-aware MPI requires. Before passing
|
|
204
|
+
device buffers to MPI, call `synchronize_for_mpi(*buffers)`: kernels run
|
|
205
|
+
asynchronously, and MPI would otherwise send a buffer a kernel is still
|
|
206
|
+
writing, without an error.
|
|
207
|
+
|
|
208
|
+
```python
|
|
209
|
+
xp.set_backend("cupy")
|
|
210
|
+
xp.bind_local_device() # before MPI_Init
|
|
211
|
+
from mpi4py import MPI
|
|
212
|
+
|
|
213
|
+
xp.synchronize_for_mpi(send, recv)
|
|
214
|
+
MPI.COMM_WORLD.Sendrecv(send, dest, recvbuf=recv, source=source)
|
|
215
|
+
```
|
|
216
|
+
|
|
202
217
|
CuPy caches released allocations in memory pools. This can make process-level
|
|
203
218
|
GPU memory appear occupied after arrays go out of scope. `free_memory()` asks
|
|
204
219
|
CuPy to release currently free cached blocks; it does not free memory still
|
|
@@ -253,6 +268,80 @@ The wrapper can also traverse arrays nested in lists, tuples, dictionaries,
|
|
|
253
268
|
and selected application objects; see the full [API reference](docs/source/api.md)
|
|
254
269
|
for `object_modules`, `is_array`, aliasing, and output declarations.
|
|
255
270
|
|
|
271
|
+
## Write CUDA kernels next to host kernels
|
|
272
|
+
|
|
273
|
+
`CudaKernel` wraps a CUDA C kernel (compiled with NVRTC through
|
|
274
|
+
`cupy.RawKernel`) so that it is called with the same arguments as the host
|
|
275
|
+
kernel it mirrors, plus the number of threads. Arrays are never copied: they
|
|
276
|
+
must be CuPy arrays. The `extern "C" __global__` signature is parsed once and
|
|
277
|
+
every call is checked against it: Python scalars are cast to the declared C
|
|
278
|
+
types, and a wrong argument count, an array of the wrong dtype, or a scalar
|
|
279
|
+
that does not fit its type raises instead of silently producing wrong values.
|
|
280
|
+
|
|
281
|
+
`Kernel` pairs a host kernel with its CUDA kernel and calls the one matching
|
|
282
|
+
the active backend, so kernels can be ported to CUDA one at a time:
|
|
283
|
+
|
|
284
|
+
```python
|
|
285
|
+
import cunumpy as xp
|
|
286
|
+
|
|
287
|
+
AXPY = r"""
|
|
288
|
+
extern "C" __global__
|
|
289
|
+
void axpy(double a, const double* x, double* y, int n) {
|
|
290
|
+
int i = blockDim.x * blockIdx.x + threadIdx.x;
|
|
291
|
+
if (i < n) y[i] += a * x[i];
|
|
292
|
+
}
|
|
293
|
+
"""
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def axpy(a, x, y, n): # host version, e.g. compiled with Pyccel
|
|
297
|
+
for i in range(n):
|
|
298
|
+
y[i] += a * x[i]
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
kernel = xp.Kernel(axpy, xp.CudaKernel(AXPY, "axpy"))
|
|
302
|
+
|
|
303
|
+
with xp.use_backend("cupy"):
|
|
304
|
+
x = xp.arange(1000, dtype=xp.float64)
|
|
305
|
+
y = xp.zeros(1000)
|
|
306
|
+
kernel(2.0, x, y, 1000, n_threads=1000) # runs the CUDA kernel
|
|
307
|
+
```
|
|
308
|
+
|
|
309
|
+
On the CuPy backend, a `Kernel` without CUDA kernel raises
|
|
310
|
+
`NotImplementedError` (or, with `missing_cuda="fallback"`, runs the host kernel
|
|
311
|
+
through `PyccelKernel`, with host copies; `host_options` configure that
|
|
312
|
+
`PyccelKernel`). `KernelCatalog.from_package()` collects kernel pairs from a
|
|
313
|
+
package with one folder per kernel (`name/name_kernels.py` and
|
|
314
|
+
`name/name_cuda.cu`), and `catalog.compile_all()` compiles all CUDA kernels at
|
|
315
|
+
setup.
|
|
316
|
+
|
|
317
|
+
Groups of arguments can be passed as one: objects implementing
|
|
318
|
+
`__cuda_args__()` (see `CudaArguments`) are flattened into several kernel
|
|
319
|
+
arguments, and `CudaStruct` defines a C struct once (its C `declaration` and
|
|
320
|
+
the matching memory layout) and packs values into it, which the kernel takes
|
|
321
|
+
as one parameter:
|
|
322
|
+
|
|
323
|
+
```python
|
|
324
|
+
Vec = xp.CudaStruct("Vec", [("data", "double*"), ("n", "int")])
|
|
325
|
+
scale = xp.CudaKernel(
|
|
326
|
+
Vec.declaration
|
|
327
|
+
+ r"""
|
|
328
|
+
extern "C" __global__ void scale(Vec v, double a) {
|
|
329
|
+
int i = blockDim.x * blockIdx.x + threadIdx.x;
|
|
330
|
+
if (i < v.n) v.data[i] *= a;
|
|
331
|
+
}""",
|
|
332
|
+
"scale",
|
|
333
|
+
structs=[Vec],
|
|
334
|
+
)
|
|
335
|
+
scale(Vec(data=y, n=y.size), 0.5, n_threads=y.size)
|
|
336
|
+
```
|
|
337
|
+
|
|
338
|
+
Launches can be 1D to 3D (`n_threads=(nx, ny)`, `block_size=(16, 16)`) or use
|
|
339
|
+
an explicit `grid`, with dynamic shared memory (`shared_mem`) and a `stream`.
|
|
340
|
+
C++ function templates are instantiated with `template_args`, and
|
|
341
|
+
`CudaKernelVariants` caches kernels whose source is generated per variant
|
|
342
|
+
(e.g. per dimension and dtype). See the [API reference](docs/source/api.md) for
|
|
343
|
+
details.
|
|
344
|
+
|
|
256
345
|
## Pyodide
|
|
257
346
|
|
|
258
347
|
CuNumpy supports the NumPy backend in Pyodide. It does not provide CuPy/CUDA
|
|
@@ -159,6 +159,22 @@ active CuPy device and `None` on NumPy. `set_device_for_rank(rank)` is a
|
|
|
159
159
|
round-robin convenience for MPI layouts where local ranks map contiguously to
|
|
160
160
|
GPUs. If your scheduler uses a different mapping, select the device directly.
|
|
161
161
|
|
|
162
|
+
For MPI programs with one rank per GPU, `bind_local_device()` selects the GPU
|
|
163
|
+
from the node-local rank that the MPI launcher exports (`local_rank()`), so it
|
|
164
|
+
can run before MPI is initialized, as CUDA-aware MPI requires. Before passing
|
|
165
|
+
device buffers to MPI, call `synchronize_for_mpi(*buffers)`: kernels run
|
|
166
|
+
asynchronously, and MPI would otherwise send a buffer a kernel is still
|
|
167
|
+
writing, without an error.
|
|
168
|
+
|
|
169
|
+
```python
|
|
170
|
+
xp.set_backend("cupy")
|
|
171
|
+
xp.bind_local_device() # before MPI_Init
|
|
172
|
+
from mpi4py import MPI
|
|
173
|
+
|
|
174
|
+
xp.synchronize_for_mpi(send, recv)
|
|
175
|
+
MPI.COMM_WORLD.Sendrecv(send, dest, recvbuf=recv, source=source)
|
|
176
|
+
```
|
|
177
|
+
|
|
162
178
|
CuPy caches released allocations in memory pools. This can make process-level
|
|
163
179
|
GPU memory appear occupied after arrays go out of scope. `free_memory()` asks
|
|
164
180
|
CuPy to release currently free cached blocks; it does not free memory still
|
|
@@ -213,6 +229,80 @@ The wrapper can also traverse arrays nested in lists, tuples, dictionaries,
|
|
|
213
229
|
and selected application objects; see the full [API reference](docs/source/api.md)
|
|
214
230
|
for `object_modules`, `is_array`, aliasing, and output declarations.
|
|
215
231
|
|
|
232
|
+
## Write CUDA kernels next to host kernels
|
|
233
|
+
|
|
234
|
+
`CudaKernel` wraps a CUDA C kernel (compiled with NVRTC through
|
|
235
|
+
`cupy.RawKernel`) so that it is called with the same arguments as the host
|
|
236
|
+
kernel it mirrors, plus the number of threads. Arrays are never copied: they
|
|
237
|
+
must be CuPy arrays. The `extern "C" __global__` signature is parsed once and
|
|
238
|
+
every call is checked against it: Python scalars are cast to the declared C
|
|
239
|
+
types, and a wrong argument count, an array of the wrong dtype, or a scalar
|
|
240
|
+
that does not fit its type raises instead of silently producing wrong values.
|
|
241
|
+
|
|
242
|
+
`Kernel` pairs a host kernel with its CUDA kernel and calls the one matching
|
|
243
|
+
the active backend, so kernels can be ported to CUDA one at a time:
|
|
244
|
+
|
|
245
|
+
```python
|
|
246
|
+
import cunumpy as xp
|
|
247
|
+
|
|
248
|
+
AXPY = r"""
|
|
249
|
+
extern "C" __global__
|
|
250
|
+
void axpy(double a, const double* x, double* y, int n) {
|
|
251
|
+
int i = blockDim.x * blockIdx.x + threadIdx.x;
|
|
252
|
+
if (i < n) y[i] += a * x[i];
|
|
253
|
+
}
|
|
254
|
+
"""
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def axpy(a, x, y, n): # host version, e.g. compiled with Pyccel
|
|
258
|
+
for i in range(n):
|
|
259
|
+
y[i] += a * x[i]
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
kernel = xp.Kernel(axpy, xp.CudaKernel(AXPY, "axpy"))
|
|
263
|
+
|
|
264
|
+
with xp.use_backend("cupy"):
|
|
265
|
+
x = xp.arange(1000, dtype=xp.float64)
|
|
266
|
+
y = xp.zeros(1000)
|
|
267
|
+
kernel(2.0, x, y, 1000, n_threads=1000) # runs the CUDA kernel
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
On the CuPy backend, a `Kernel` without CUDA kernel raises
|
|
271
|
+
`NotImplementedError` (or, with `missing_cuda="fallback"`, runs the host kernel
|
|
272
|
+
through `PyccelKernel`, with host copies; `host_options` configure that
|
|
273
|
+
`PyccelKernel`). `KernelCatalog.from_package()` collects kernel pairs from a
|
|
274
|
+
package with one folder per kernel (`name/name_kernels.py` and
|
|
275
|
+
`name/name_cuda.cu`), and `catalog.compile_all()` compiles all CUDA kernels at
|
|
276
|
+
setup.
|
|
277
|
+
|
|
278
|
+
Groups of arguments can be passed as one: objects implementing
|
|
279
|
+
`__cuda_args__()` (see `CudaArguments`) are flattened into several kernel
|
|
280
|
+
arguments, and `CudaStruct` defines a C struct once (its C `declaration` and
|
|
281
|
+
the matching memory layout) and packs values into it, which the kernel takes
|
|
282
|
+
as one parameter:
|
|
283
|
+
|
|
284
|
+
```python
|
|
285
|
+
Vec = xp.CudaStruct("Vec", [("data", "double*"), ("n", "int")])
|
|
286
|
+
scale = xp.CudaKernel(
|
|
287
|
+
Vec.declaration
|
|
288
|
+
+ r"""
|
|
289
|
+
extern "C" __global__ void scale(Vec v, double a) {
|
|
290
|
+
int i = blockDim.x * blockIdx.x + threadIdx.x;
|
|
291
|
+
if (i < v.n) v.data[i] *= a;
|
|
292
|
+
}""",
|
|
293
|
+
"scale",
|
|
294
|
+
structs=[Vec],
|
|
295
|
+
)
|
|
296
|
+
scale(Vec(data=y, n=y.size), 0.5, n_threads=y.size)
|
|
297
|
+
```
|
|
298
|
+
|
|
299
|
+
Launches can be 1D to 3D (`n_threads=(nx, ny)`, `block_size=(16, 16)`) or use
|
|
300
|
+
an explicit `grid`, with dynamic shared memory (`shared_mem`) and a `stream`.
|
|
301
|
+
C++ function templates are instantiated with `template_args`, and
|
|
302
|
+
`CudaKernelVariants` caches kernels whose source is generated per variant
|
|
303
|
+
(e.g. per dimension and dtype). See the [API reference](docs/source/api.md) for
|
|
304
|
+
details.
|
|
305
|
+
|
|
216
306
|
## Pyodide
|
|
217
307
|
|
|
218
308
|
CuNumpy supports the NumPy backend in Pyodide. It does not provide CuPy/CUDA
|
|
@@ -5,22 +5,21 @@ requires = [ "setuptools", "wheel" ]
|
|
|
5
5
|
|
|
6
6
|
[project]
|
|
7
7
|
name = "cunumpy"
|
|
8
|
-
version = "0.
|
|
8
|
+
version = "0.3.0"
|
|
9
9
|
description = "Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`."
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
keywords = [ "python" ]
|
|
12
12
|
license = { file = "LICENSE.txt" }
|
|
13
13
|
authors = [ { name = "Max" } ]
|
|
14
|
-
requires-python = ">=3.
|
|
14
|
+
requires-python = ">=3.10"
|
|
15
15
|
classifiers = [
|
|
16
16
|
"Development Status :: 3 - Alpha",
|
|
17
17
|
"Programming Language :: Python :: 3 :: Only",
|
|
18
|
-
"Programming Language :: Python :: 3.8",
|
|
19
|
-
"Programming Language :: Python :: 3.9",
|
|
20
18
|
"Programming Language :: Python :: 3.10",
|
|
21
19
|
"Programming Language :: Python :: 3.11",
|
|
22
20
|
"Programming Language :: Python :: 3.12",
|
|
23
21
|
"Programming Language :: Python :: 3.13",
|
|
22
|
+
"Programming Language :: Python :: 3.14",
|
|
24
23
|
]
|
|
25
24
|
dependencies = [
|
|
26
25
|
"array-api-compat",
|
|
@@ -2,9 +2,21 @@
|
|
|
2
2
|
from importlib.metadata import PackageNotFoundError, version
|
|
3
3
|
|
|
4
4
|
from . import xp
|
|
5
|
+
from .cuda_kernel import (
|
|
6
|
+
CudaArguments,
|
|
7
|
+
CudaKernel,
|
|
8
|
+
CudaKernelVariants,
|
|
9
|
+
CudaParameter,
|
|
10
|
+
CudaStruct,
|
|
11
|
+
CudaStructValue,
|
|
12
|
+
ctype_of,
|
|
13
|
+
parse_cuda_signature,
|
|
14
|
+
)
|
|
15
|
+
from .dispatch import Kernel, KernelCatalog
|
|
5
16
|
from .kernel import PyccelKernel
|
|
6
17
|
from .xp import (
|
|
7
18
|
assert_same_backend,
|
|
19
|
+
bind_local_device,
|
|
8
20
|
cupy_available,
|
|
9
21
|
default_float_dtype,
|
|
10
22
|
device_count,
|
|
@@ -15,6 +27,7 @@ from .xp import (
|
|
|
15
27
|
get_rng,
|
|
16
28
|
is_cpu,
|
|
17
29
|
is_gpu,
|
|
30
|
+
local_rank,
|
|
18
31
|
memory_info,
|
|
19
32
|
pin_memory,
|
|
20
33
|
same_backend,
|
|
@@ -23,6 +36,7 @@ from .xp import (
|
|
|
23
36
|
set_device_for_rank,
|
|
24
37
|
stream,
|
|
25
38
|
synchronize,
|
|
39
|
+
synchronize_for_mpi,
|
|
26
40
|
to_cunumpy,
|
|
27
41
|
to_cupy,
|
|
28
42
|
to_numpy,
|
|
@@ -35,9 +49,19 @@ except PackageNotFoundError:
|
|
|
35
49
|
__version__ = "0.0.0+unknown"
|
|
36
50
|
|
|
37
51
|
__all__ = [
|
|
52
|
+
"CudaArguments",
|
|
53
|
+
"CudaKernel",
|
|
54
|
+
"CudaKernelVariants",
|
|
55
|
+
"CudaParameter",
|
|
56
|
+
"CudaStruct",
|
|
57
|
+
"CudaStructValue",
|
|
58
|
+
"Kernel",
|
|
59
|
+
"KernelCatalog",
|
|
38
60
|
"PyccelKernel",
|
|
39
61
|
"__version__",
|
|
40
62
|
"assert_same_backend",
|
|
63
|
+
"bind_local_device",
|
|
64
|
+
"ctype_of",
|
|
41
65
|
"cupy_available",
|
|
42
66
|
"cupy_backend",
|
|
43
67
|
"default_float_dtype",
|
|
@@ -49,8 +73,10 @@ __all__ = [
|
|
|
49
73
|
"get_rng",
|
|
50
74
|
"is_cpu",
|
|
51
75
|
"is_gpu",
|
|
76
|
+
"local_rank",
|
|
52
77
|
"memory_info",
|
|
53
78
|
"numpy_backend",
|
|
79
|
+
"parse_cuda_signature",
|
|
54
80
|
"pin_memory",
|
|
55
81
|
"same_backend",
|
|
56
82
|
"set_backend",
|
|
@@ -58,6 +84,7 @@ __all__ = [
|
|
|
58
84
|
"set_device_for_rank",
|
|
59
85
|
"stream",
|
|
60
86
|
"synchronize",
|
|
87
|
+
"synchronize_for_mpi",
|
|
61
88
|
"to_cunumpy",
|
|
62
89
|
"to_cupy",
|
|
63
90
|
"to_numpy",
|
|
@@ -1,13 +1,24 @@
|
|
|
1
1
|
# Stub file for Pylance/mypy: exposes all numpy symbols so that
|
|
2
2
|
# `import cunumpy as xp` followed by `xp.<Tab>` shows numpy completions.
|
|
3
3
|
# At runtime the real __init__.py dispatches to numpy or cupy via __getattr__.
|
|
4
|
+
from collections.abc import Generator
|
|
4
5
|
from contextlib import contextmanager
|
|
5
|
-
from typing import Any
|
|
6
|
+
from typing import Any
|
|
6
7
|
|
|
7
8
|
import numpy as np
|
|
8
9
|
from numpy import *
|
|
9
10
|
|
|
10
11
|
from . import xp as xp
|
|
12
|
+
from .cuda_kernel import CudaArguments as CudaArguments
|
|
13
|
+
from .cuda_kernel import CudaKernel as CudaKernel
|
|
14
|
+
from .cuda_kernel import CudaKernelVariants as CudaKernelVariants
|
|
15
|
+
from .cuda_kernel import CudaParameter as CudaParameter
|
|
16
|
+
from .cuda_kernel import CudaStruct as CudaStruct
|
|
17
|
+
from .cuda_kernel import CudaStructValue as CudaStructValue
|
|
18
|
+
from .cuda_kernel import ctype_of as ctype_of
|
|
19
|
+
from .cuda_kernel import parse_cuda_signature as parse_cuda_signature
|
|
20
|
+
from .dispatch import Kernel as Kernel
|
|
21
|
+
from .dispatch import KernelCatalog as KernelCatalog
|
|
11
22
|
from .kernel import PyccelKernel as PyccelKernel
|
|
12
23
|
|
|
13
24
|
def to_numpy(array: Any) -> np.ndarray: ...
|
|
@@ -26,6 +37,9 @@ def use_backend(backend: str) -> Generator[None]: ...
|
|
|
26
37
|
def set_backend(backend: str) -> None: ...
|
|
27
38
|
def set_device(device_id: int) -> None: ...
|
|
28
39
|
def set_device_for_rank(rank: int, devices_per_node: int | None = ...) -> int: ...
|
|
40
|
+
def local_rank() -> int: ...
|
|
41
|
+
def bind_local_device() -> int | None: ...
|
|
42
|
+
def synchronize_for_mpi(*arrays: Any) -> None: ...
|
|
29
43
|
def device_count() -> int: ...
|
|
30
44
|
def memory_info() -> tuple[int, int] | None: ...
|
|
31
45
|
def free_memory() -> None: ...
|