cunumpy 0.5.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cunumpy-0.5.0 → cunumpy-0.6.0}/PKG-INFO +18 -30
- {cunumpy-0.5.0 → cunumpy-0.6.0}/README.md +16 -29
- {cunumpy-0.5.0 → cunumpy-0.6.0}/pyproject.toml +2 -1
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/LLM_GUIDE.md +28 -37
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/__init__.py +15 -53
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/__init__.pyi +1 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_cuda_kernel.py +83 -171
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_device.py +1 -1
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_dispatch.py +13 -24
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_emulation.py +11 -4
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_fake_cupy.py +2 -2
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_kernel.py +1 -132
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_mpi.py +6 -17
- cunumpy-0.6.0/src/cunumpy/arguments.py +42 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/__init__.py +10 -23
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/array_view.cuh +114 -2
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/kernel_testing.py +4 -4
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/kernels.py +8 -9
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/mpi.py +32 -11
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/PKG-INFO +18 -30
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/SOURCES.txt +1 -7
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/requires.txt +1 -0
- cunumpy-0.5.0/src/cunumpy/_deprecated.py +0 -19
- cunumpy-0.5.0/src/cunumpy/_mpi_serial.py +0 -657
- cunumpy-0.5.0/src/cunumpy/cuda_kernel.py +0 -8
- cunumpy-0.5.0/src/cunumpy/dispatch.py +0 -8
- cunumpy-0.5.0/src/cunumpy/kernel.py +0 -8
- cunumpy-0.5.0/src/cunumpy/main.py +0 -8
- cunumpy-0.5.0/src/cunumpy/testing.py +0 -9
- {cunumpy-0.5.0 → cunumpy-0.6.0}/setup.cfg +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_algorithms.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_fake_cupy_impl.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_fusion.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_mirror.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_morton.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_philox.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_profiling.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_random_streams.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_scipy_backend.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_staging.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_streams.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_transfers.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/algorithms.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/atomic.cuh +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/index.cuh +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/morton.cuh +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/random.cuh +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/reduce.cuh +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/scan.cuh +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/memory.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/petsc.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/profiling.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/py.typed +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/rng.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/xp.py +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/dependency_links.txt +0 -0
- {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cunumpy
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`.
|
|
5
5
|
Author: Max
|
|
6
6
|
Project-URL: Source, https://github.com/max-models/cunumpy
|
|
@@ -15,6 +15,7 @@ Classifier: Programming Language :: Python :: 3.14
|
|
|
15
15
|
Requires-Python: >=3.10
|
|
16
16
|
Description-Content-Type: text/markdown
|
|
17
17
|
Requires-Dist: array-api-compat
|
|
18
|
+
Requires-Dist: maybempi>=0.1.2
|
|
18
19
|
Requires-Dist: numpy
|
|
19
20
|
Provides-Extra: dev
|
|
20
21
|
Requires-Dist: ruff; extra == "dev"
|
|
@@ -63,8 +64,9 @@ never hide a NumPy name:
|
|
|
63
64
|
|
|
64
65
|
| Submodule | Contents |
|
|
65
66
|
|---|---|
|
|
66
|
-
| `xp.
|
|
67
|
-
| `xp.
|
|
67
|
+
| `xp.kernels` | `Kernel`, `KernelCatalog`, `PyccelKernel`, `CudaKernel`, host implementations, `fuse` |
|
|
68
|
+
| `xp.arguments` | CUDA only: `CudaStruct`, `CudaStructArguments`, `CudaArguments` |
|
|
69
|
+
| `xp.cuda` | CUDA only: devices, streams, debug mode, CUDA headers |
|
|
68
70
|
| `xp.rng` | `random_streams`, `get_rng`, `philox_*` |
|
|
69
71
|
| `xp.algorithms` | `morton_*`, `sort_by_key`, `cell_offsets`, `segment_boundaries`, `segment_sum`, `SegmentPlan` |
|
|
70
72
|
| `xp.mpi` | `mpi_buffer`, reusable `MPIStaging`, CUDA-aware MPI |
|
|
@@ -73,7 +75,7 @@ never hide a NumPy name:
|
|
|
73
75
|
| `xp.petsc` | `petsc_vec` |
|
|
74
76
|
| `cunumpy.kernel_testing` | pytest helpers for host/CUDA kernel pairs |
|
|
75
77
|
|
|
76
|
-
Everything except `xp.cuda` works on both backends.
|
|
78
|
+
Everything except `xp.cuda`, `xp.arguments` and `CudaKernel` works on both backends.
|
|
77
79
|
|
|
78
80
|
## Install
|
|
79
81
|
|
|
@@ -372,7 +374,7 @@ def axpy(a, x, y, n): # host version, e.g. compiled with Pyccel
|
|
|
372
374
|
y[i] += a * x[i]
|
|
373
375
|
|
|
374
376
|
|
|
375
|
-
kernel = xp.kernels.Kernel(axpy, xp.
|
|
377
|
+
kernel = xp.kernels.Kernel(axpy, xp.kernels.CudaKernel(AXPY, "axpy"))
|
|
376
378
|
|
|
377
379
|
with xp.use_backend("cupy"):
|
|
378
380
|
x = xp.arange(1000, dtype=xp.float64)
|
|
@@ -395,8 +397,8 @@ the matching memory layout) and packs values into it, which the kernel takes
|
|
|
395
397
|
as one parameter:
|
|
396
398
|
|
|
397
399
|
```python
|
|
398
|
-
Vec = xp.
|
|
399
|
-
scale = xp.
|
|
400
|
+
Vec = xp.arguments.CudaStruct("Vec", [("data", "double*"), ("n", "int")])
|
|
401
|
+
scale = xp.kernels.CudaKernel(
|
|
400
402
|
Vec.declaration
|
|
401
403
|
+ r"""
|
|
402
404
|
extern "C" __global__ void scale(Vec v, double a) {
|
|
@@ -418,29 +420,15 @@ device copies. Call it once when the object is
|
|
|
418
420
|
built, not per kernel call; on the NumPy backend it raises, so host data is
|
|
419
421
|
never copied to the device implicitly.
|
|
420
422
|
When the host kernel takes such a group as one object too (e.g. a Pyccel class
|
|
421
|
-
holding NumPy arrays),
|
|
422
|
-
|
|
423
|
-
the
|
|
424
|
-
|
|
425
|
-
on the CUDA path, so the call site is the same on both backends and each form
|
|
426
|
-
can be built lazily on first access (a CPU run never builds device arguments):
|
|
423
|
+
holding NumPy arrays), write a CUDA class with the same constructor and
|
|
424
|
+
attributes (a `CudaStructArguments`, see below) and let the owner of the arrays
|
|
425
|
+
build the one for the active backend. CuNumpy passes argument objects through
|
|
426
|
+
as they are and never converts one form into the other:
|
|
427
427
|
|
|
428
428
|
```python
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
self._host = None
|
|
433
|
-
|
|
434
|
-
def __host_args__(self):
|
|
435
|
-
if self._host is None:
|
|
436
|
-
self._host = MarkerArguments(self.markers) # Pyccel class
|
|
437
|
-
return self._host
|
|
438
|
-
|
|
439
|
-
def __cuda_args__(self):
|
|
440
|
-
return (self.markers, self.markers.shape[0])
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
kernel(particles.kernel_args, dt, n_threads=n) # host or CUDA kernel
|
|
429
|
+
args_class = CudaMarkerArguments if xp.is_gpu(markers) else MarkerArguments
|
|
430
|
+
particles.args_markers = args_class(markers, markers.shape[0])
|
|
431
|
+
kernel(particles.args_markers, dt) # host or CUDA kernel
|
|
444
432
|
```
|
|
445
433
|
|
|
446
434
|
Kernels ported from pyccel index arrays like `markers[ip, j]`, which needs
|
|
@@ -456,11 +444,11 @@ class MarkerArguments:
|
|
|
456
444
|
def __init__(self, markers: "float[:, :]", n_markers: int, valid: "bool[:]"): ...
|
|
457
445
|
|
|
458
446
|
|
|
459
|
-
MarkerArgs = xp.
|
|
447
|
+
MarkerArgs = xp.arguments.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs")
|
|
460
448
|
MarkerArgs.to_header(
|
|
461
449
|
"marker_args.cuh"
|
|
462
450
|
) # Array2D<double> markers; long long n_markers; ...
|
|
463
|
-
push = xp.
|
|
451
|
+
push = xp.kernels.CudaKernel(
|
|
464
452
|
r"""
|
|
465
453
|
#include "marker_args.cuh"
|
|
466
454
|
#include <cunumpy/index.cuh>
|
|
@@ -24,8 +24,9 @@ never hide a NumPy name:
|
|
|
24
24
|
|
|
25
25
|
| Submodule | Contents |
|
|
26
26
|
|---|---|
|
|
27
|
-
| `xp.
|
|
28
|
-
| `xp.
|
|
27
|
+
| `xp.kernels` | `Kernel`, `KernelCatalog`, `PyccelKernel`, `CudaKernel`, host implementations, `fuse` |
|
|
28
|
+
| `xp.arguments` | CUDA only: `CudaStruct`, `CudaStructArguments`, `CudaArguments` |
|
|
29
|
+
| `xp.cuda` | CUDA only: devices, streams, debug mode, CUDA headers |
|
|
29
30
|
| `xp.rng` | `random_streams`, `get_rng`, `philox_*` |
|
|
30
31
|
| `xp.algorithms` | `morton_*`, `sort_by_key`, `cell_offsets`, `segment_boundaries`, `segment_sum`, `SegmentPlan` |
|
|
31
32
|
| `xp.mpi` | `mpi_buffer`, reusable `MPIStaging`, CUDA-aware MPI |
|
|
@@ -34,7 +35,7 @@ never hide a NumPy name:
|
|
|
34
35
|
| `xp.petsc` | `petsc_vec` |
|
|
35
36
|
| `cunumpy.kernel_testing` | pytest helpers for host/CUDA kernel pairs |
|
|
36
37
|
|
|
37
|
-
Everything except `xp.cuda` works on both backends.
|
|
38
|
+
Everything except `xp.cuda`, `xp.arguments` and `CudaKernel` works on both backends.
|
|
38
39
|
|
|
39
40
|
## Install
|
|
40
41
|
|
|
@@ -333,7 +334,7 @@ def axpy(a, x, y, n): # host version, e.g. compiled with Pyccel
|
|
|
333
334
|
y[i] += a * x[i]
|
|
334
335
|
|
|
335
336
|
|
|
336
|
-
kernel = xp.kernels.Kernel(axpy, xp.
|
|
337
|
+
kernel = xp.kernels.Kernel(axpy, xp.kernels.CudaKernel(AXPY, "axpy"))
|
|
337
338
|
|
|
338
339
|
with xp.use_backend("cupy"):
|
|
339
340
|
x = xp.arange(1000, dtype=xp.float64)
|
|
@@ -356,8 +357,8 @@ the matching memory layout) and packs values into it, which the kernel takes
|
|
|
356
357
|
as one parameter:
|
|
357
358
|
|
|
358
359
|
```python
|
|
359
|
-
Vec = xp.
|
|
360
|
-
scale = xp.
|
|
360
|
+
Vec = xp.arguments.CudaStruct("Vec", [("data", "double*"), ("n", "int")])
|
|
361
|
+
scale = xp.kernels.CudaKernel(
|
|
361
362
|
Vec.declaration
|
|
362
363
|
+ r"""
|
|
363
364
|
extern "C" __global__ void scale(Vec v, double a) {
|
|
@@ -379,29 +380,15 @@ device copies. Call it once when the object is
|
|
|
379
380
|
built, not per kernel call; on the NumPy backend it raises, so host data is
|
|
380
381
|
never copied to the device implicitly.
|
|
381
382
|
When the host kernel takes such a group as one object too (e.g. a Pyccel class
|
|
382
|
-
holding NumPy arrays),
|
|
383
|
-
|
|
384
|
-
the
|
|
385
|
-
|
|
386
|
-
on the CUDA path, so the call site is the same on both backends and each form
|
|
387
|
-
can be built lazily on first access (a CPU run never builds device arguments):
|
|
383
|
+
holding NumPy arrays), write a CUDA class with the same constructor and
|
|
384
|
+
attributes (a `CudaStructArguments`, see below) and let the owner of the arrays
|
|
385
|
+
build the one for the active backend. CuNumpy passes argument objects through
|
|
386
|
+
as they are and never converts one form into the other:
|
|
388
387
|
|
|
389
388
|
```python
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
self._host = None
|
|
394
|
-
|
|
395
|
-
def __host_args__(self):
|
|
396
|
-
if self._host is None:
|
|
397
|
-
self._host = MarkerArguments(self.markers) # Pyccel class
|
|
398
|
-
return self._host
|
|
399
|
-
|
|
400
|
-
def __cuda_args__(self):
|
|
401
|
-
return (self.markers, self.markers.shape[0])
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
kernel(particles.kernel_args, dt, n_threads=n) # host or CUDA kernel
|
|
389
|
+
args_class = CudaMarkerArguments if xp.is_gpu(markers) else MarkerArguments
|
|
390
|
+
particles.args_markers = args_class(markers, markers.shape[0])
|
|
391
|
+
kernel(particles.args_markers, dt) # host or CUDA kernel
|
|
405
392
|
```
|
|
406
393
|
|
|
407
394
|
Kernels ported from pyccel index arrays like `markers[ip, j]`, which needs
|
|
@@ -417,11 +404,11 @@ class MarkerArguments:
|
|
|
417
404
|
def __init__(self, markers: "float[:, :]", n_markers: int, valid: "bool[:]"): ...
|
|
418
405
|
|
|
419
406
|
|
|
420
|
-
MarkerArgs = xp.
|
|
407
|
+
MarkerArgs = xp.arguments.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs")
|
|
421
408
|
MarkerArgs.to_header(
|
|
422
409
|
"marker_args.cuh"
|
|
423
410
|
) # Array2D<double> markers; long long n_markers; ...
|
|
424
|
-
push = xp.
|
|
411
|
+
push = xp.kernels.CudaKernel(
|
|
425
412
|
r"""
|
|
426
413
|
#include "marker_args.cuh"
|
|
427
414
|
#include <cunumpy/index.cuh>
|
|
@@ -5,7 +5,7 @@ requires = [ "setuptools", "wheel" ]
|
|
|
5
5
|
|
|
6
6
|
[project]
|
|
7
7
|
name = "cunumpy"
|
|
8
|
-
version = "0.
|
|
8
|
+
version = "0.6.0"
|
|
9
9
|
description = "Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`."
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
keywords = [ "python" ]
|
|
@@ -23,6 +23,7 @@ classifiers = [
|
|
|
23
23
|
]
|
|
24
24
|
dependencies = [
|
|
25
25
|
"array-api-compat",
|
|
26
|
+
"maybempi>=0.1.2",
|
|
26
27
|
"numpy",
|
|
27
28
|
]
|
|
28
29
|
|
|
@@ -18,7 +18,7 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository.
|
|
|
18
18
|
with one rank per GPU, profiling.
|
|
19
19
|
* A kernel layer for porting compiled CPU kernels (Pyccel, Numba, Python loops)
|
|
20
20
|
to CUDA one at a time: `PyccelKernel`, `CudaKernel`, `Kernel`,
|
|
21
|
-
`KernelCatalog`, `
|
|
21
|
+
`KernelCatalog`, `CudaStruct`, `CudaStructArguments`, `DeviceMirror`, and test
|
|
22
22
|
helpers in `cunumpy.kernel_testing`.
|
|
23
23
|
|
|
24
24
|
## Hard rules
|
|
@@ -54,12 +54,12 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository.
|
|
|
54
54
|
An argument the kernel writes but that is not declared leaves stale device
|
|
55
55
|
data, silently. If unsure, leave `outputs=None` (copies everything back).
|
|
56
56
|
11. **Helpers are in submodules; the top level is NumPy plus backend control.**
|
|
57
|
-
`xp.
|
|
57
|
+
`xp.kernels.CudaKernel`, `xp.kernels.Kernel`, `xp.arguments.CudaStructArguments`,
|
|
58
|
+
`xp.rng.random_streams`,
|
|
58
59
|
`xp.algorithms.morton_keys`, `xp.mpi.mpi_buffer`, `xp.profiling.timed_region`,
|
|
59
60
|
`xp.memory.HostStaging`, `xp.petsc.petsc_vec`; kernel test helpers in
|
|
60
|
-
`cunumpy.kernel_testing`.
|
|
61
|
-
`cunumpy.testing
|
|
62
|
-
with them. Modules starting with `_` (`cunumpy._cuda_kernel`, ...) are
|
|
61
|
+
`cunumpy.kernel_testing`. There is no `xp.CudaKernel`, `xp.cuda.CudaKernel`
|
|
62
|
+
or `cunumpy.testing`; use the submodule names above. Modules starting with `_` (`cunumpy._cuda_kernel`, ...) are
|
|
63
63
|
private; never import from them.
|
|
64
64
|
|
|
65
65
|
## Decision guide
|
|
@@ -71,7 +71,7 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository.
|
|
|
71
71
|
| normalize inputs at an API boundary | `xp.to_cunumpy(a)` once |
|
|
72
72
|
| hand data to SciPy/matplotlib/h5py | `xp.to_numpy(a)` |
|
|
73
73
|
| call an existing NumPy-only kernel with GPU arrays (slow, correct) | `xp.kernels.PyccelKernel(fn, outputs=(...))` |
|
|
74
|
-
| launch a hand-written CUDA C kernel | `xp.
|
|
74
|
+
| launch a hand-written CUDA C kernel | `xp.kernels.CudaKernel(source, "name")` / `CudaKernel.from_file(path)` |
|
|
75
75
|
| host kernel + CUDA port, chosen by backend | `xp.kernels.Kernel(host_fn, cuda_kernel_or_None)` |
|
|
76
76
|
| many kernels in a package, ported incrementally | `xp.kernels.KernelCatalog.from_package(__name__, missing_cuda="fallback")` |
|
|
77
77
|
| host kernels compiled at first call (your compile function), NumPy fallback | `from_package(..., host_suffix="_pyccel", compile_host=my_compile, host_fallback={...})` -> `xp.kernels.CompiledHostKernel` |
|
|
@@ -89,8 +89,8 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository.
|
|
|
89
89
|
| copy device arrays to the host for output without stalling | `xp.memory.HostStaging(shape, dtype)`: `c = staging.copy(a)` ... `c.result()` |
|
|
90
90
|
| PIC recipes (compaction, sort by cell, MPI exchange, graphs) | docs guide "Particle codes" |
|
|
91
91
|
| reproducible random numbers per MPI rank | `xp.rng.random_streams.seed(seed, rank=rank)`, then `xp.rng.random_streams.normal(...)` / `.generator()` |
|
|
92
|
-
| group arrays/scalars into one kernel argument | `xp.
|
|
93
|
-
| CUDA struct from a Pyccel argument class | `xp.
|
|
92
|
+
| group arrays/scalars into one kernel argument | `xp.arguments.CudaArguments` (flattened), `xp.arguments.CudaStruct` (C struct), `xp.arguments.CudaStructArguments` (C struct as a class); host kernels take their own argument objects, the caller picks one per backend |
|
|
93
|
+
| CUDA struct from a Pyccel argument class | `xp.arguments.CudaStruct.from_signature(Cls.__init__, "Name")`, `xp.arguments.write_cuda_header(...)` |
|
|
94
94
|
| SciPy (sparse, sparse.linalg, fft, special, ndimage, ...) on either backend | `xp.scipy.<subpackage>.<name>` (SciPy or `cupyx.scipy`); `xp.scipy.special.available(name)` |
|
|
95
95
|
| chain of elementwise operations as one GPU kernel | `@xp.kernels.fuse` (`cupy.fuse` for CuPy arrays, plain call otherwise) |
|
|
96
96
|
| PETSc solve on device arrays without copies | `xp.petsc.petsc_vec(array)` (CUDA/HIP petsc4py for CuPy arrays); `xp.synchronize()` around PETSc calls |
|
|
@@ -153,6 +153,7 @@ MPI = (
|
|
|
153
153
|
xp.mpi.get_mpi()
|
|
154
154
|
) # mpi4py.MPI under mpirun/srun, else a serial stand-in (no MPI_Init)
|
|
155
155
|
xp.mpi.launched_under_mpi() # from the launcher env, without importing mpi4py
|
|
156
|
+
xp.mpi.is_serial(MPI) # True for the stand-in (re-exported from maybempi; MAYBEMPI=0/1)
|
|
156
157
|
xp.mpi.mpi_is_cuda_aware(comm) # collective, once at startup; remembered
|
|
157
158
|
with xp.mpi.mpi_buffer(a) as buf:
|
|
158
159
|
comm.Send(buf, ...) # host array, CUDA-aware device
|
|
@@ -263,17 +264,17 @@ CuPy. Does not compile anything.
|
|
|
263
264
|
`CudaKernel`:
|
|
264
265
|
|
|
265
266
|
```python
|
|
266
|
-
k = xp.
|
|
267
|
+
k = xp.kernels.CudaKernel(source, name, *, block_size=128, options=(), include_dirs=(),
|
|
267
268
|
source_dir=None, structs=(), template_args=None,
|
|
268
269
|
check_signature=True, debug=None)
|
|
269
|
-
k = xp.
|
|
270
|
-
ks = xp.
|
|
270
|
+
k = xp.kernels.CudaKernel.from_file("push/push_cuda.cu") # name "push"
|
|
271
|
+
ks = xp.kernels.CudaKernel.all_from_file("ops.cu") # dict name -> kernel
|
|
271
272
|
k(*args, n_threads=None, grid=None, block=None, shared_mem=0, stream=None)
|
|
272
273
|
k.compile(log_stream=None); k.recompile(log_stream=None)
|
|
273
274
|
k.is_compiled # successful compilation on the current CUDA device
|
|
274
275
|
k.launch_shape(n_threads=None, *, grid=None, block=None, args=None) -> (grid, block)
|
|
275
276
|
k.included_headers; k.compile_options(); k.debug_active()
|
|
276
|
-
xp.
|
|
277
|
+
xp.kernels.CudaKernelVariants(factory).get(*key); .compile_all(keys, jobs=1)
|
|
277
278
|
xp.cuda.ctype_of(np.float64) == "double"
|
|
278
279
|
xp.cuda.cuda_kernel_names(source); xp.cuda.parse_cuda_signature(source, name)
|
|
279
280
|
xp.cuda.cuda_include_dir()
|
|
@@ -339,20 +340,12 @@ function `<name>` (host); optional `pkg/<name>/<name>_cuda.cu` defines
|
|
|
339
340
|
Argument objects:
|
|
340
341
|
|
|
341
342
|
```python
|
|
342
|
-
class Dev(xp.
|
|
343
|
+
class Dev(xp.arguments.CudaArguments): # flattened into several CUDA params
|
|
343
344
|
def __init__(self, x, n):
|
|
344
345
|
super().__init__(x, n)
|
|
345
346
|
|
|
346
347
|
|
|
347
|
-
|
|
348
|
-
def __host_args__(self):
|
|
349
|
-
return host_object # host kernel gets this
|
|
350
|
-
|
|
351
|
-
def __cuda_args__(self):
|
|
352
|
-
return (arr, n, ...) # CUDA kernel gets these, flattened
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
S = xp.cuda.CudaStruct(
|
|
348
|
+
S = xp.arguments.CudaStruct(
|
|
356
349
|
"S", [("x", "double*"), ("n", "long long"), ("a", "Array2D<double>")]
|
|
357
350
|
)
|
|
358
351
|
S.declaration
|
|
@@ -360,12 +353,12 @@ S.dtype
|
|
|
360
353
|
S.to_header(path)
|
|
361
354
|
value = S(x=..., n=..., a=...)
|
|
362
355
|
S.verify_layout() # GPU test: compiler layout == S.dtype (also verify_layout("hdr.cuh"))
|
|
363
|
-
S = xp.
|
|
364
|
-
xp.
|
|
356
|
+
S = xp.arguments.CudaStruct.from_signature(Cls.__init__, "S", int_type="long long")
|
|
357
|
+
xp.arguments.write_cuda_header("args.cuh", [S1, S2])
|
|
365
358
|
|
|
366
359
|
|
|
367
360
|
class A(
|
|
368
|
-
xp.
|
|
361
|
+
xp.arguments.CudaStructArguments
|
|
369
362
|
): # the struct as a class; A.struct is the CudaStruct
|
|
370
363
|
struct_name = "A"
|
|
371
364
|
fields = (("x", "double*"), ("n", "int"))
|
|
@@ -375,12 +368,14 @@ class A(
|
|
|
375
368
|
self.pack() # repacks itself when a field changes; copies repack
|
|
376
369
|
|
|
377
370
|
|
|
378
|
-
xp.
|
|
379
|
-
|
|
371
|
+
xp.kernels.CudaKernel(S.declaration + src, "k", structs=[S])
|
|
372
|
+
|
|
373
|
+
# host kernels: a pyccel class and a CudaStructArguments with the same
|
|
374
|
+
# constructor; the owner builds one per backend, cunumpy never converts them
|
|
375
|
+
args = (CudaMarkerArguments if xp.is_gpu(markers) else MarkerArguments)(markers, n)
|
|
380
376
|
```
|
|
381
377
|
|
|
382
|
-
|
|
383
|
-
them when the underlying arrays are replaced. A packed struct holds device
|
|
378
|
+
A packed struct holds device
|
|
384
379
|
addresses: re-pack after replacing an array (a `CudaStructArguments` does this
|
|
385
380
|
itself at the next launch; make its fields properties to follow an owner's arrays).
|
|
386
381
|
|
|
@@ -409,7 +404,7 @@ Debugging:
|
|
|
409
404
|
|
|
410
405
|
```python
|
|
411
406
|
xp.cuda.set_cuda_debug(True); xp.cuda.get_cuda_debug(); with xp.cuda.cuda_debug(): ...
|
|
412
|
-
xp.
|
|
407
|
+
xp.kernels.CudaKernel(..., debug=True) # env: CUNUMPY_CUDA_DEBUG=1
|
|
413
408
|
```
|
|
414
409
|
|
|
415
410
|
Debug mode adds `-lineinfo -DCUNUMPY_BOUNDS_CHECK` at compile time and
|
|
@@ -443,12 +438,8 @@ def test_parity(kernel): check_parity(kernel)
|
|
|
443
438
|
# without a GPU: CUNUMPY_FAKE_CUPY=1 CUNUMPY_BACKEND=cupy pytest (fake CuPy: strict host
|
|
444
439
|
# stand-in, no kernel launches; fake_cupy_active(); requires_cupy skips)
|
|
445
440
|
|
|
446
|
-
#
|
|
447
|
-
|
|
448
|
-
struct_name = "MarkerArgs"; fields = (("markers", "Array2D<double>"), ("Np", "long long"))
|
|
449
|
-
host_class = pusher_args_kernels.MarkerArguments # pyccel class; cannot inherit
|
|
450
|
-
host_fields = ("markers", "Np") # its constructor args, in order
|
|
451
|
-
MarkerArgs = xp.cuda.CudaStruct.from_pyccel_class("pusher_args_kernels.py", "MarkerArguments", "MarkerArgs")
|
|
441
|
+
# struct fields from a pyccel argument class (contiguous=True or names -> CArray2D<T>)
|
|
442
|
+
MarkerArgs = xp.arguments.CudaStruct.from_pyccel_class("pusher_args_kernels.py", "MarkerArguments", "MarkerArgs")
|
|
452
443
|
kernel.n_threads_from = lambda args: args[0].n_markers # launch size from an argument
|
|
453
444
|
kernel.check_finite = True # NaN/inf after each launch (debug)
|
|
454
445
|
```
|
|
@@ -497,7 +488,7 @@ def scale_host(x: "float[:]", a: float, n: int):
|
|
|
497
488
|
|
|
498
489
|
|
|
499
490
|
scale = xp.kernels.Kernel(
|
|
500
|
-
scale_host, xp.
|
|
491
|
+
scale_host, xp.kernels.CudaKernel(SRC, "scale"), host_options={"outputs": (0,)}
|
|
501
492
|
)
|
|
502
493
|
scale(x, 2.0, x.size, n_threads=x.size)
|
|
503
494
|
```
|
|
@@ -1,9 +1,19 @@
|
|
|
1
1
|
# cunumpy/__init__.py
|
|
2
2
|
import re as _re
|
|
3
|
-
import warnings as _warnings
|
|
4
3
|
from importlib.metadata import PackageNotFoundError, version
|
|
5
4
|
|
|
6
|
-
from cunumpy import
|
|
5
|
+
from cunumpy import (
|
|
6
|
+
algorithms,
|
|
7
|
+
arguments,
|
|
8
|
+
cuda,
|
|
9
|
+
kernels,
|
|
10
|
+
memory,
|
|
11
|
+
mpi,
|
|
12
|
+
petsc,
|
|
13
|
+
profiling,
|
|
14
|
+
rng,
|
|
15
|
+
xp,
|
|
16
|
+
)
|
|
7
17
|
from cunumpy._scipy_backend import scipy
|
|
8
18
|
from cunumpy.xp import (
|
|
9
19
|
as_device_array,
|
|
@@ -25,42 +35,6 @@ from cunumpy.xp import (
|
|
|
25
35
|
use_backend,
|
|
26
36
|
)
|
|
27
37
|
|
|
28
|
-
# Names that were at the top level before cunumpy 0.5, and the submodule each
|
|
29
|
-
# moved to. They still resolve (with a DeprecationWarning) until cunumpy 0.6.
|
|
30
|
-
_MOVED = {
|
|
31
|
-
**dict.fromkeys(cuda.__all__, "cuda"),
|
|
32
|
-
**dict.fromkeys(
|
|
33
|
-
(
|
|
34
|
-
name
|
|
35
|
-
for name in kernels.__all__
|
|
36
|
-
if not name.endswith(
|
|
37
|
-
("_host_kernel_implementation", "_device_kernel_implementation")
|
|
38
|
-
)
|
|
39
|
-
and name != "DEVICE_IMPLEMENTATIONS"
|
|
40
|
-
),
|
|
41
|
-
"kernels",
|
|
42
|
-
),
|
|
43
|
-
**dict.fromkeys(rng.__all__, "rng"),
|
|
44
|
-
**dict.fromkeys(algorithms.__all__, "algorithms"),
|
|
45
|
-
# the names of cunumpy.mpi that were at the top level (not the later ones)
|
|
46
|
-
**dict.fromkeys(
|
|
47
|
-
(
|
|
48
|
-
"get_mpi_cuda_aware",
|
|
49
|
-
"local_rank",
|
|
50
|
-
"mpi_buffer",
|
|
51
|
-
"mpi_is_cuda_aware",
|
|
52
|
-
"require_cuda_aware_mpi",
|
|
53
|
-
"set_mpi_cuda_aware",
|
|
54
|
-
"synchronize_for_mpi",
|
|
55
|
-
),
|
|
56
|
-
"mpi",
|
|
57
|
-
),
|
|
58
|
-
**dict.fromkeys(profiling.__all__, "profiling"),
|
|
59
|
-
**dict.fromkeys(memory.__all__, "memory"),
|
|
60
|
-
"petsc_vec": "petsc",
|
|
61
|
-
}
|
|
62
|
-
_MOVED.pop("BIT_GENERATORS") # never was at the top level
|
|
63
|
-
|
|
64
38
|
try:
|
|
65
39
|
__version__ = version("cunumpy")
|
|
66
40
|
except PackageNotFoundError:
|
|
@@ -97,6 +71,7 @@ def require_version(minimum: str) -> None:
|
|
|
97
71
|
__all__ = [
|
|
98
72
|
"__version__",
|
|
99
73
|
"algorithms",
|
|
74
|
+
"arguments",
|
|
100
75
|
"as_device_array",
|
|
101
76
|
"assert_same_backend",
|
|
102
77
|
"backend_info",
|
|
@@ -134,31 +109,19 @@ def __getattr__(name: str):
|
|
|
134
109
|
|
|
135
110
|
The public names of the active backend are copied into this namespace (see
|
|
136
111
|
`_sync_backend_namespace`), so this only runs for names missing from the
|
|
137
|
-
backend's ``__all__
|
|
138
|
-
names moved to a submodule in cunumpy 0.5 (see ``_MOVED``), which still
|
|
139
|
-
resolve with a ``DeprecationWarning``.
|
|
112
|
+
backend's ``__all__`` and for ``numpy_backend``/``cupy_backend``.
|
|
140
113
|
"""
|
|
141
114
|
if name == "numpy_backend":
|
|
142
115
|
return xp.numpy_backend
|
|
143
116
|
if name == "cupy_backend":
|
|
144
117
|
return xp.cupy_backend
|
|
145
|
-
submodule = _MOVED.get(name)
|
|
146
|
-
if submodule is not None:
|
|
147
|
-
_warnings.warn(
|
|
148
|
-
f"cunumpy.{name} moved to cunumpy.{submodule}.{name}; the top-level "
|
|
149
|
-
"name is deprecated and will be removed in cunumpy 0.6",
|
|
150
|
-
DeprecationWarning,
|
|
151
|
-
stacklevel=2,
|
|
152
|
-
)
|
|
153
|
-
return getattr(globals()[submodule], name)
|
|
154
118
|
return getattr(xp.xp, name)
|
|
155
119
|
|
|
156
120
|
|
|
157
121
|
# `xp.zeros` must be as fast as `numpy.zeros`. A module-level __getattr__ runs
|
|
158
122
|
# only after the normal lookup failed, which costs about 3 us per access, so the
|
|
159
123
|
# public names of the active backend module are copied into this namespace, and
|
|
160
|
-
# replaced whenever the backend changes. cunumpy's own names
|
|
161
|
-
# names of _MOVED (e.g. `fuse`, which CuPy also has) are never overwritten.
|
|
124
|
+
# replaced whenever the backend changes. cunumpy's own names are never overwritten.
|
|
162
125
|
_OWN_NAMES = frozenset(globals())
|
|
163
126
|
_backend_names: dict[int, dict[str, object]] = {} # id(module) -> names to copy
|
|
164
127
|
_switches: dict[tuple[int, int], tuple[tuple[str, ...], dict[str, object]]] = {}
|
|
@@ -173,7 +136,6 @@ def _names_of(module) -> dict[str, object]:
|
|
|
173
136
|
for name in getattr(module, "__all__", ())
|
|
174
137
|
if not name.startswith("_")
|
|
175
138
|
and name not in _OWN_NAMES
|
|
176
|
-
and name not in _MOVED
|
|
177
139
|
and hasattr(module, name)
|
|
178
140
|
}
|
|
179
141
|
_backend_names[id(module)] = names
|