cunumpy 0.5.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. {cunumpy-0.5.0 → cunumpy-0.6.0}/PKG-INFO +18 -30
  2. {cunumpy-0.5.0 → cunumpy-0.6.0}/README.md +16 -29
  3. {cunumpy-0.5.0 → cunumpy-0.6.0}/pyproject.toml +2 -1
  4. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/LLM_GUIDE.md +28 -37
  5. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/__init__.py +15 -53
  6. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/__init__.pyi +1 -0
  7. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_cuda_kernel.py +83 -171
  8. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_device.py +1 -1
  9. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_dispatch.py +13 -24
  10. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_emulation.py +11 -4
  11. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_fake_cupy.py +2 -2
  12. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_kernel.py +1 -132
  13. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_mpi.py +6 -17
  14. cunumpy-0.6.0/src/cunumpy/arguments.py +42 -0
  15. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/__init__.py +10 -23
  16. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/array_view.cuh +114 -2
  17. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/kernel_testing.py +4 -4
  18. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/kernels.py +8 -9
  19. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/mpi.py +32 -11
  20. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/PKG-INFO +18 -30
  21. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/SOURCES.txt +1 -7
  22. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/requires.txt +1 -0
  23. cunumpy-0.5.0/src/cunumpy/_deprecated.py +0 -19
  24. cunumpy-0.5.0/src/cunumpy/_mpi_serial.py +0 -657
  25. cunumpy-0.5.0/src/cunumpy/cuda_kernel.py +0 -8
  26. cunumpy-0.5.0/src/cunumpy/dispatch.py +0 -8
  27. cunumpy-0.5.0/src/cunumpy/kernel.py +0 -8
  28. cunumpy-0.5.0/src/cunumpy/main.py +0 -8
  29. cunumpy-0.5.0/src/cunumpy/testing.py +0 -9
  30. {cunumpy-0.5.0 → cunumpy-0.6.0}/setup.cfg +0 -0
  31. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_algorithms.py +0 -0
  32. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_fake_cupy_impl.py +0 -0
  33. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_fusion.py +0 -0
  34. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_mirror.py +0 -0
  35. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_morton.py +0 -0
  36. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_philox.py +0 -0
  37. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_profiling.py +0 -0
  38. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_random_streams.py +0 -0
  39. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_scipy_backend.py +0 -0
  40. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_staging.py +0 -0
  41. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_streams.py +0 -0
  42. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/_transfers.py +0 -0
  43. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/algorithms.py +0 -0
  44. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/atomic.cuh +0 -0
  45. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/index.cuh +0 -0
  46. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/morton.cuh +0 -0
  47. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/random.cuh +0 -0
  48. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/reduce.cuh +0 -0
  49. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/cuda/include/cunumpy/scan.cuh +0 -0
  50. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/memory.py +0 -0
  51. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/petsc.py +0 -0
  52. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/profiling.py +0 -0
  53. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/py.typed +0 -0
  54. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/rng.py +0 -0
  55. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy/xp.py +0 -0
  56. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/dependency_links.txt +0 -0
  57. {cunumpy-0.5.0 → cunumpy-0.6.0}/src/cunumpy.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cunumpy
3
- Version: 0.5.0
3
+ Version: 0.6.0
4
4
  Summary: Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`.
5
5
  Author: Max
6
6
  Project-URL: Source, https://github.com/max-models/cunumpy
@@ -15,6 +15,7 @@ Classifier: Programming Language :: Python :: 3.14
15
15
  Requires-Python: >=3.10
16
16
  Description-Content-Type: text/markdown
17
17
  Requires-Dist: array-api-compat
18
+ Requires-Dist: maybempi>=0.1.2
18
19
  Requires-Dist: numpy
19
20
  Provides-Extra: dev
20
21
  Requires-Dist: ruff; extra == "dev"
@@ -63,8 +64,9 @@ never hide a NumPy name:
63
64
 
64
65
  | Submodule | Contents |
65
66
  |---|---|
66
- | `xp.cuda` | CUDA only: `CudaKernel`, `CudaStruct`, CUDA headers, devices, streams |
67
- | `xp.kernels` | `Kernel`, `KernelCatalog`, `PyccelKernel`, host implementations, `fuse` |
67
+ | `xp.kernels` | `Kernel`, `KernelCatalog`, `PyccelKernel`, `CudaKernel`, host implementations, `fuse` |
68
+ | `xp.arguments` | CUDA only: `CudaStruct`, `CudaStructArguments`, `CudaArguments` |
69
+ | `xp.cuda` | CUDA only: devices, streams, debug mode, CUDA headers |
68
70
  | `xp.rng` | `random_streams`, `get_rng`, `philox_*` |
69
71
  | `xp.algorithms` | `morton_*`, `sort_by_key`, `cell_offsets`, `segment_boundaries`, `segment_sum`, `SegmentPlan` |
70
72
  | `xp.mpi` | `mpi_buffer`, reusable `MPIStaging`, CUDA-aware MPI |
@@ -73,7 +75,7 @@ never hide a NumPy name:
73
75
  | `xp.petsc` | `petsc_vec` |
74
76
  | `cunumpy.kernel_testing` | pytest helpers for host/CUDA kernel pairs |
75
77
 
76
- Everything except `xp.cuda` works on both backends.
78
+ Everything except `xp.cuda`, `xp.arguments` and `CudaKernel` works on both backends.
77
79
 
78
80
  ## Install
79
81
 
@@ -372,7 +374,7 @@ def axpy(a, x, y, n): # host version, e.g. compiled with Pyccel
372
374
  y[i] += a * x[i]
373
375
 
374
376
 
375
- kernel = xp.kernels.Kernel(axpy, xp.cuda.CudaKernel(AXPY, "axpy"))
377
+ kernel = xp.kernels.Kernel(axpy, xp.kernels.CudaKernel(AXPY, "axpy"))
376
378
 
377
379
  with xp.use_backend("cupy"):
378
380
  x = xp.arange(1000, dtype=xp.float64)
@@ -395,8 +397,8 @@ the matching memory layout) and packs values into it, which the kernel takes
395
397
  as one parameter:
396
398
 
397
399
  ```python
398
- Vec = xp.cuda.CudaStruct("Vec", [("data", "double*"), ("n", "int")])
399
- scale = xp.cuda.CudaKernel(
400
+ Vec = xp.arguments.CudaStruct("Vec", [("data", "double*"), ("n", "int")])
401
+ scale = xp.kernels.CudaKernel(
400
402
  Vec.declaration
401
403
  + r"""
402
404
  extern "C" __global__ void scale(Vec v, double a) {
@@ -418,29 +420,15 @@ device copies. Call it once when the object is
418
420
  built, not per kernel call; on the NumPy backend it raises, so host data is
419
421
  never copied to the device implicitly.
420
422
  When the host kernel takes such a group as one object too (e.g. a Pyccel class
421
- holding NumPy arrays), give the group both forms with `KernelArguments`:
422
- `__host_args__()` returns the object for the host kernel, `__cuda_args__()`
423
- the flattened device arguments. `Kernel` and `PyccelKernel` resolve
424
- `__host_args__()` on the host path and `CudaKernel` flattens `__cuda_args__()`
425
- on the CUDA path, so the call site is the same on both backends and each form
426
- can be built lazily on first access (a CPU run never builds device arguments):
423
+ holding NumPy arrays), write a CUDA class with the same constructor and
424
+ attributes (a `CudaStructArguments`, see below) and let the owner of the arrays
425
+ build the one for the active backend. CuNumpy passes argument objects through
426
+ as they are and never converts one form into the other:
427
427
 
428
428
  ```python
429
- class ParticleArguments(xp.kernels.KernelArguments):
430
- def __init__(self, markers):
431
- self.markers = markers
432
- self._host = None
433
-
434
- def __host_args__(self):
435
- if self._host is None:
436
- self._host = MarkerArguments(self.markers) # Pyccel class
437
- return self._host
438
-
439
- def __cuda_args__(self):
440
- return (self.markers, self.markers.shape[0])
441
-
442
-
443
- kernel(particles.kernel_args, dt, n_threads=n) # host or CUDA kernel
429
+ args_class = CudaMarkerArguments if xp.is_gpu(markers) else MarkerArguments
430
+ particles.args_markers = args_class(markers, markers.shape[0])
431
+ kernel(particles.args_markers, dt) # host or CUDA kernel
444
432
  ```
445
433
 
446
434
  Kernels ported from pyccel index arrays like `markers[ip, j]`, which needs
@@ -456,11 +444,11 @@ class MarkerArguments:
456
444
  def __init__(self, markers: "float[:, :]", n_markers: int, valid: "bool[:]"): ...
457
445
 
458
446
 
459
- MarkerArgs = xp.cuda.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs")
447
+ MarkerArgs = xp.arguments.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs")
460
448
  MarkerArgs.to_header(
461
449
  "marker_args.cuh"
462
450
  ) # Array2D<double> markers; long long n_markers; ...
463
- push = xp.cuda.CudaKernel(
451
+ push = xp.kernels.CudaKernel(
464
452
  r"""
465
453
  #include "marker_args.cuh"
466
454
  #include <cunumpy/index.cuh>
@@ -24,8 +24,9 @@ never hide a NumPy name:
24
24
 
25
25
  | Submodule | Contents |
26
26
  |---|---|
27
- | `xp.cuda` | CUDA only: `CudaKernel`, `CudaStruct`, CUDA headers, devices, streams |
28
- | `xp.kernels` | `Kernel`, `KernelCatalog`, `PyccelKernel`, host implementations, `fuse` |
27
+ | `xp.kernels` | `Kernel`, `KernelCatalog`, `PyccelKernel`, `CudaKernel`, host implementations, `fuse` |
28
+ | `xp.arguments` | CUDA only: `CudaStruct`, `CudaStructArguments`, `CudaArguments` |
29
+ | `xp.cuda` | CUDA only: devices, streams, debug mode, CUDA headers |
29
30
  | `xp.rng` | `random_streams`, `get_rng`, `philox_*` |
30
31
  | `xp.algorithms` | `morton_*`, `sort_by_key`, `cell_offsets`, `segment_boundaries`, `segment_sum`, `SegmentPlan` |
31
32
  | `xp.mpi` | `mpi_buffer`, reusable `MPIStaging`, CUDA-aware MPI |
@@ -34,7 +35,7 @@ never hide a NumPy name:
34
35
  | `xp.petsc` | `petsc_vec` |
35
36
  | `cunumpy.kernel_testing` | pytest helpers for host/CUDA kernel pairs |
36
37
 
37
- Everything except `xp.cuda` works on both backends.
38
+ Everything except `xp.cuda`, `xp.arguments` and `CudaKernel` works on both backends.
38
39
 
39
40
  ## Install
40
41
 
@@ -333,7 +334,7 @@ def axpy(a, x, y, n): # host version, e.g. compiled with Pyccel
333
334
  y[i] += a * x[i]
334
335
 
335
336
 
336
- kernel = xp.kernels.Kernel(axpy, xp.cuda.CudaKernel(AXPY, "axpy"))
337
+ kernel = xp.kernels.Kernel(axpy, xp.kernels.CudaKernel(AXPY, "axpy"))
337
338
 
338
339
  with xp.use_backend("cupy"):
339
340
  x = xp.arange(1000, dtype=xp.float64)
@@ -356,8 +357,8 @@ the matching memory layout) and packs values into it, which the kernel takes
356
357
  as one parameter:
357
358
 
358
359
  ```python
359
- Vec = xp.cuda.CudaStruct("Vec", [("data", "double*"), ("n", "int")])
360
- scale = xp.cuda.CudaKernel(
360
+ Vec = xp.arguments.CudaStruct("Vec", [("data", "double*"), ("n", "int")])
361
+ scale = xp.kernels.CudaKernel(
361
362
  Vec.declaration
362
363
  + r"""
363
364
  extern "C" __global__ void scale(Vec v, double a) {
@@ -379,29 +380,15 @@ device copies. Call it once when the object is
379
380
  built, not per kernel call; on the NumPy backend it raises, so host data is
380
381
  never copied to the device implicitly.
381
382
  When the host kernel takes such a group as one object too (e.g. a Pyccel class
382
- holding NumPy arrays), give the group both forms with `KernelArguments`:
383
- `__host_args__()` returns the object for the host kernel, `__cuda_args__()`
384
- the flattened device arguments. `Kernel` and `PyccelKernel` resolve
385
- `__host_args__()` on the host path and `CudaKernel` flattens `__cuda_args__()`
386
- on the CUDA path, so the call site is the same on both backends and each form
387
- can be built lazily on first access (a CPU run never builds device arguments):
383
+ holding NumPy arrays), write a CUDA class with the same constructor and
384
+ attributes (a `CudaStructArguments`, see below) and let the owner of the arrays
385
+ build the one for the active backend. CuNumpy passes argument objects through
386
+ as they are and never converts one form into the other:
388
387
 
389
388
  ```python
390
- class ParticleArguments(xp.kernels.KernelArguments):
391
- def __init__(self, markers):
392
- self.markers = markers
393
- self._host = None
394
-
395
- def __host_args__(self):
396
- if self._host is None:
397
- self._host = MarkerArguments(self.markers) # Pyccel class
398
- return self._host
399
-
400
- def __cuda_args__(self):
401
- return (self.markers, self.markers.shape[0])
402
-
403
-
404
- kernel(particles.kernel_args, dt, n_threads=n) # host or CUDA kernel
389
+ args_class = CudaMarkerArguments if xp.is_gpu(markers) else MarkerArguments
390
+ particles.args_markers = args_class(markers, markers.shape[0])
391
+ kernel(particles.args_markers, dt) # host or CUDA kernel
405
392
  ```
406
393
 
407
394
  Kernels ported from pyccel index arrays like `markers[ip, j]`, which needs
@@ -417,11 +404,11 @@ class MarkerArguments:
417
404
  def __init__(self, markers: "float[:, :]", n_markers: int, valid: "bool[:]"): ...
418
405
 
419
406
 
420
- MarkerArgs = xp.cuda.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs")
407
+ MarkerArgs = xp.arguments.CudaStruct.from_signature(MarkerArguments.__init__, "MarkerArgs")
421
408
  MarkerArgs.to_header(
422
409
  "marker_args.cuh"
423
410
  ) # Array2D<double> markers; long long n_markers; ...
424
- push = xp.cuda.CudaKernel(
411
+ push = xp.kernels.CudaKernel(
425
412
  r"""
426
413
  #include "marker_args.cuh"
427
414
  #include <cunumpy/index.cuh>
@@ -5,7 +5,7 @@ requires = [ "setuptools", "wheel" ]
5
5
 
6
6
  [project]
7
7
  name = "cunumpy"
8
- version = "0.5.0"
8
+ version = "0.6.0"
9
9
  description = "Simple wrapper for numpy and cupy. Replace `import numpy as np` with `import cunumpy as xp`."
10
10
  readme = "README.md"
11
11
  keywords = [ "python" ]
@@ -23,6 +23,7 @@ classifiers = [
23
23
  ]
24
24
  dependencies = [
25
25
  "array-api-compat",
26
+ "maybempi>=0.1.2",
26
27
  "numpy",
27
28
  ]
28
29
 
@@ -18,7 +18,7 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository.
18
18
  with one rank per GPU, profiling.
19
19
  * A kernel layer for porting compiled CPU kernels (Pyccel, Numba, Python loops)
20
20
  to CUDA one at a time: `PyccelKernel`, `CudaKernel`, `Kernel`,
21
- `KernelCatalog`, `KernelArguments`, `CudaStruct`, `DeviceMirror`, and test
21
+ `KernelCatalog`, `CudaStruct`, `CudaStructArguments`, `DeviceMirror`, and test
22
22
  helpers in `cunumpy.kernel_testing`.
23
23
 
24
24
  ## Hard rules
@@ -54,12 +54,12 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository.
54
54
  An argument the kernel writes but that is not declared leaves stale device
55
55
  data, silently. If unsure, leave `outputs=None` (copies everything back).
56
56
  11. **Helpers are in submodules; the top level is NumPy plus backend control.**
57
- `xp.cuda.CudaKernel`, `xp.kernels.Kernel`, `xp.rng.random_streams`,
57
+ `xp.kernels.CudaKernel`, `xp.kernels.Kernel`, `xp.arguments.CudaStructArguments`,
58
+ `xp.rng.random_streams`,
58
59
  `xp.algorithms.morton_keys`, `xp.mpi.mpi_buffer`, `xp.profiling.timed_region`,
59
60
  `xp.memory.HostStaging`, `xp.petsc.petsc_vec`; kernel test helpers in
60
- `cunumpy.kernel_testing`. The old top-level names (`xp.CudaKernel`) and
61
- `cunumpy.testing` are deprecated (removed in 0.6); do not write new code
62
- with them. Modules starting with `_` (`cunumpy._cuda_kernel`, ...) are
61
+ `cunumpy.kernel_testing`. There is no `xp.CudaKernel`, `xp.cuda.CudaKernel`
62
+ or `cunumpy.testing`; use the submodule names above. Modules starting with `_` (`cunumpy._cuda_kernel`, ...) are
63
63
  private; never import from them.
64
64
 
65
65
  ## Decision guide
@@ -71,7 +71,7 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository.
71
71
  | normalize inputs at an API boundary | `xp.to_cunumpy(a)` once |
72
72
  | hand data to SciPy/matplotlib/h5py | `xp.to_numpy(a)` |
73
73
  | call an existing NumPy-only kernel with GPU arrays (slow, correct) | `xp.kernels.PyccelKernel(fn, outputs=(...))` |
74
- | launch a hand-written CUDA C kernel | `xp.cuda.CudaKernel(source, "name")` / `CudaKernel.from_file(path)` |
74
+ | launch a hand-written CUDA C kernel | `xp.kernels.CudaKernel(source, "name")` / `CudaKernel.from_file(path)` |
75
75
  | host kernel + CUDA port, chosen by backend | `xp.kernels.Kernel(host_fn, cuda_kernel_or_None)` |
76
76
  | many kernels in a package, ported incrementally | `xp.kernels.KernelCatalog.from_package(__name__, missing_cuda="fallback")` |
77
77
  | host kernels compiled at first call (your compile function), NumPy fallback | `from_package(..., host_suffix="_pyccel", compile_host=my_compile, host_fallback={...})` -> `xp.kernels.CompiledHostKernel` |
@@ -89,8 +89,8 @@ https://max-models.github.io/cunumpy/ and in `docs/source/` of the repository.
89
89
  | copy device arrays to the host for output without stalling | `xp.memory.HostStaging(shape, dtype)`: `c = staging.copy(a)` ... `c.result()` |
90
90
  | PIC recipes (compaction, sort by cell, MPI exchange, graphs) | docs guide "Particle codes" |
91
91
  | reproducible random numbers per MPI rank | `xp.rng.random_streams.seed(seed, rank=rank)`, then `xp.rng.random_streams.normal(...)` / `.generator()` |
92
- | group arrays/scalars into one kernel argument | `xp.cuda.CudaArguments` (device only), `xp.kernels.KernelArguments` (host object + device tuple), `xp.cuda.CudaStruct` (C struct), `xp.cuda.CudaStructArguments` (C struct as a class) |
93
- | CUDA struct from a Pyccel argument class | `xp.cuda.CudaStruct.from_signature(Cls.__init__, "Name")`, `xp.cuda.write_cuda_header(...)` |
92
+ | group arrays/scalars into one kernel argument | `xp.arguments.CudaArguments` (flattened), `xp.arguments.CudaStruct` (C struct), `xp.arguments.CudaStructArguments` (C struct as a class); host kernels take their own argument objects, the caller picks one per backend |
93
+ | CUDA struct from a Pyccel argument class | `xp.arguments.CudaStruct.from_signature(Cls.__init__, "Name")`, `xp.arguments.write_cuda_header(...)` |
94
94
  | SciPy (sparse, sparse.linalg, fft, special, ndimage, ...) on either backend | `xp.scipy.<subpackage>.<name>` (SciPy or `cupyx.scipy`); `xp.scipy.special.available(name)` |
95
95
  | chain of elementwise operations as one GPU kernel | `@xp.kernels.fuse` (`cupy.fuse` for CuPy arrays, plain call otherwise) |
96
96
  | PETSc solve on device arrays without copies | `xp.petsc.petsc_vec(array)` (CUDA/HIP petsc4py for CuPy arrays); `xp.synchronize()` around PETSc calls |
@@ -153,6 +153,7 @@ MPI = (
153
153
  xp.mpi.get_mpi()
154
154
  ) # mpi4py.MPI under mpirun/srun, else a serial stand-in (no MPI_Init)
155
155
  xp.mpi.launched_under_mpi() # from the launcher env, without importing mpi4py
156
+ xp.mpi.is_serial(MPI) # True for the stand-in (re-exported from maybempi; MAYBEMPI=0/1)
156
157
  xp.mpi.mpi_is_cuda_aware(comm) # collective, once at startup; remembered
157
158
  with xp.mpi.mpi_buffer(a) as buf:
158
159
  comm.Send(buf, ...) # host array, CUDA-aware device
@@ -263,17 +264,17 @@ CuPy. Does not compile anything.
263
264
  `CudaKernel`:
264
265
 
265
266
  ```python
266
- k = xp.cuda.CudaKernel(source, name, *, block_size=128, options=(), include_dirs=(),
267
+ k = xp.kernels.CudaKernel(source, name, *, block_size=128, options=(), include_dirs=(),
267
268
  source_dir=None, structs=(), template_args=None,
268
269
  check_signature=True, debug=None)
269
- k = xp.cuda.CudaKernel.from_file("push/push_cuda.cu") # name "push"
270
- ks = xp.cuda.CudaKernel.all_from_file("ops.cu") # dict name -> kernel
270
+ k = xp.kernels.CudaKernel.from_file("push/push_cuda.cu") # name "push"
271
+ ks = xp.kernels.CudaKernel.all_from_file("ops.cu") # dict name -> kernel
271
272
  k(*args, n_threads=None, grid=None, block=None, shared_mem=0, stream=None)
272
273
  k.compile(log_stream=None); k.recompile(log_stream=None)
273
274
  k.is_compiled # successful compilation on the current CUDA device
274
275
  k.launch_shape(n_threads=None, *, grid=None, block=None, args=None) -> (grid, block)
275
276
  k.included_headers; k.compile_options(); k.debug_active()
276
- xp.cuda.CudaKernelVariants(factory).get(*key); .compile_all(keys, jobs=1)
277
+ xp.kernels.CudaKernelVariants(factory).get(*key); .compile_all(keys, jobs=1)
277
278
  xp.cuda.ctype_of(np.float64) == "double"
278
279
  xp.cuda.cuda_kernel_names(source); xp.cuda.parse_cuda_signature(source, name)
279
280
  xp.cuda.cuda_include_dir()
@@ -339,20 +340,12 @@ function `<name>` (host); optional `pkg/<name>/<name>_cuda.cu` defines
339
340
  Argument objects:
340
341
 
341
342
  ```python
342
- class Dev(xp.cuda.CudaArguments): # flattened into several CUDA params
343
+ class Dev(xp.arguments.CudaArguments): # flattened into several CUDA params
343
344
  def __init__(self, x, n):
344
345
  super().__init__(x, n)
345
346
 
346
347
 
347
- class Args(xp.kernels.KernelArguments): # one object, host form + device form
348
- def __host_args__(self):
349
- return host_object # host kernel gets this
350
-
351
- def __cuda_args__(self):
352
- return (arr, n, ...) # CUDA kernel gets these, flattened
353
-
354
-
355
- S = xp.cuda.CudaStruct(
348
+ S = xp.arguments.CudaStruct(
356
349
  "S", [("x", "double*"), ("n", "long long"), ("a", "Array2D<double>")]
357
350
  )
358
351
  S.declaration
@@ -360,12 +353,12 @@ S.dtype
360
353
  S.to_header(path)
361
354
  value = S(x=..., n=..., a=...)
362
355
  S.verify_layout() # GPU test: compiler layout == S.dtype (also verify_layout("hdr.cuh"))
363
- S = xp.cuda.CudaStruct.from_signature(Cls.__init__, "S", int_type="long long")
364
- xp.cuda.write_cuda_header("args.cuh", [S1, S2])
356
+ S = xp.arguments.CudaStruct.from_signature(Cls.__init__, "S", int_type="long long")
357
+ xp.arguments.write_cuda_header("args.cuh", [S1, S2])
365
358
 
366
359
 
367
360
  class A(
368
- xp.cuda.CudaStructArguments
361
+ xp.arguments.CudaStructArguments
369
362
  ): # the struct as a class; A.struct is the CudaStruct
370
363
  struct_name = "A"
371
364
  fields = (("x", "double*"), ("n", "int"))
@@ -375,12 +368,14 @@ class A(
375
368
  self.pack() # repacks itself when a field changes; copies repack
376
369
 
377
370
 
378
- xp.cuda.CudaKernel(S.declaration + src, "k", structs=[S])
379
- xp.kernels.resolve_host_args(args, kwargs)
371
+ xp.kernels.CudaKernel(S.declaration + src, "k", structs=[S])
372
+
373
+ # host kernels: a pyccel class and a CudaStructArguments with the same
374
+ # constructor; the owner builds one per backend, cunumpy never converts them
375
+ args = (CudaMarkerArguments if xp.is_gpu(markers) else MarkerArguments)(markers, n)
380
376
  ```
381
377
 
382
- Only top-level arguments are resolved. Cache both forms lazily and invalidate
383
- them when the underlying arrays are replaced. A packed struct holds device
378
+ A packed struct holds device
384
379
  addresses: re-pack after replacing an array (a `CudaStructArguments` does this
385
380
  itself at the next launch; make its fields properties to follow an owner's arrays).
386
381
 
@@ -409,7 +404,7 @@ Debugging:
409
404
 
410
405
  ```python
411
406
  xp.cuda.set_cuda_debug(True); xp.cuda.get_cuda_debug(); with xp.cuda.cuda_debug(): ...
412
- xp.cuda.CudaKernel(..., debug=True) # env: CUNUMPY_CUDA_DEBUG=1
407
+ xp.kernels.CudaKernel(..., debug=True) # env: CUNUMPY_CUDA_DEBUG=1
413
408
  ```
414
409
 
415
410
  Debug mode adds `-lineinfo -DCUNUMPY_BOUNDS_CHECK` at compile time and
@@ -443,12 +438,8 @@ def test_parity(kernel): check_parity(kernel)
443
438
  # without a GPU: CUNUMPY_FAKE_CUPY=1 CUNUMPY_BACKEND=cupy pytest (fake CuPy: strict host
444
439
  # stand-in, no kernel launches; fake_cupy_active(); requires_cupy skips)
445
440
 
446
- # argument classes with a pyccel host class: one object on both backends
447
- class MarkerArguments(xp.kernels.PyccelStructArguments):
448
- struct_name = "MarkerArgs"; fields = (("markers", "Array2D<double>"), ("Np", "long long"))
449
- host_class = pusher_args_kernels.MarkerArguments # pyccel class; cannot inherit
450
- host_fields = ("markers", "Np") # its constructor args, in order
451
- MarkerArgs = xp.cuda.CudaStruct.from_pyccel_class("pusher_args_kernels.py", "MarkerArguments", "MarkerArgs")
441
+ # struct fields from a pyccel argument class (contiguous=True or names -> CArray2D<T>)
442
+ MarkerArgs = xp.arguments.CudaStruct.from_pyccel_class("pusher_args_kernels.py", "MarkerArguments", "MarkerArgs")
452
443
  kernel.n_threads_from = lambda args: args[0].n_markers # launch size from an argument
453
444
  kernel.check_finite = True # NaN/inf after each launch (debug)
454
445
  ```
@@ -497,7 +488,7 @@ def scale_host(x: "float[:]", a: float, n: int):
497
488
 
498
489
 
499
490
  scale = xp.kernels.Kernel(
500
- scale_host, xp.cuda.CudaKernel(SRC, "scale"), host_options={"outputs": (0,)}
491
+ scale_host, xp.kernels.CudaKernel(SRC, "scale"), host_options={"outputs": (0,)}
501
492
  )
502
493
  scale(x, 2.0, x.size, n_threads=x.size)
503
494
  ```
@@ -1,9 +1,19 @@
1
1
  # cunumpy/__init__.py
2
2
  import re as _re
3
- import warnings as _warnings
4
3
  from importlib.metadata import PackageNotFoundError, version
5
4
 
6
- from cunumpy import algorithms, cuda, kernels, memory, mpi, petsc, profiling, rng, xp
5
+ from cunumpy import (
6
+ algorithms,
7
+ arguments,
8
+ cuda,
9
+ kernels,
10
+ memory,
11
+ mpi,
12
+ petsc,
13
+ profiling,
14
+ rng,
15
+ xp,
16
+ )
7
17
  from cunumpy._scipy_backend import scipy
8
18
  from cunumpy.xp import (
9
19
  as_device_array,
@@ -25,42 +35,6 @@ from cunumpy.xp import (
25
35
  use_backend,
26
36
  )
27
37
 
28
- # Names that were at the top level before cunumpy 0.5, and the submodule each
29
- # moved to. They still resolve (with a DeprecationWarning) until cunumpy 0.6.
30
- _MOVED = {
31
- **dict.fromkeys(cuda.__all__, "cuda"),
32
- **dict.fromkeys(
33
- (
34
- name
35
- for name in kernels.__all__
36
- if not name.endswith(
37
- ("_host_kernel_implementation", "_device_kernel_implementation")
38
- )
39
- and name != "DEVICE_IMPLEMENTATIONS"
40
- ),
41
- "kernels",
42
- ),
43
- **dict.fromkeys(rng.__all__, "rng"),
44
- **dict.fromkeys(algorithms.__all__, "algorithms"),
45
- # the names of cunumpy.mpi that were at the top level (not the later ones)
46
- **dict.fromkeys(
47
- (
48
- "get_mpi_cuda_aware",
49
- "local_rank",
50
- "mpi_buffer",
51
- "mpi_is_cuda_aware",
52
- "require_cuda_aware_mpi",
53
- "set_mpi_cuda_aware",
54
- "synchronize_for_mpi",
55
- ),
56
- "mpi",
57
- ),
58
- **dict.fromkeys(profiling.__all__, "profiling"),
59
- **dict.fromkeys(memory.__all__, "memory"),
60
- "petsc_vec": "petsc",
61
- }
62
- _MOVED.pop("BIT_GENERATORS") # never was at the top level
63
-
64
38
  try:
65
39
  __version__ = version("cunumpy")
66
40
  except PackageNotFoundError:
@@ -97,6 +71,7 @@ def require_version(minimum: str) -> None:
97
71
  __all__ = [
98
72
  "__version__",
99
73
  "algorithms",
74
+ "arguments",
100
75
  "as_device_array",
101
76
  "assert_same_backend",
102
77
  "backend_info",
@@ -134,31 +109,19 @@ def __getattr__(name: str):
134
109
 
135
110
  The public names of the active backend are copied into this namespace (see
136
111
  `_sync_backend_namespace`), so this only runs for names missing from the
137
- backend's ``__all__``, for ``numpy_backend``/``cupy_backend``, and for the
138
- names moved to a submodule in cunumpy 0.5 (see ``_MOVED``), which still
139
- resolve with a ``DeprecationWarning``.
112
+ backend's ``__all__`` and for ``numpy_backend``/``cupy_backend``.
140
113
  """
141
114
  if name == "numpy_backend":
142
115
  return xp.numpy_backend
143
116
  if name == "cupy_backend":
144
117
  return xp.cupy_backend
145
- submodule = _MOVED.get(name)
146
- if submodule is not None:
147
- _warnings.warn(
148
- f"cunumpy.{name} moved to cunumpy.{submodule}.{name}; the top-level "
149
- "name is deprecated and will be removed in cunumpy 0.6",
150
- DeprecationWarning,
151
- stacklevel=2,
152
- )
153
- return getattr(globals()[submodule], name)
154
118
  return getattr(xp.xp, name)
155
119
 
156
120
 
157
121
  # `xp.zeros` must be as fast as `numpy.zeros`. A module-level __getattr__ runs
158
122
  # only after the normal lookup failed, which costs about 3 us per access, so the
159
123
  # public names of the active backend module are copied into this namespace, and
160
- # replaced whenever the backend changes. cunumpy's own names and the deprecated
161
- # names of _MOVED (e.g. `fuse`, which CuPy also has) are never overwritten.
124
+ # replaced whenever the backend changes. cunumpy's own names are never overwritten.
162
125
  _OWN_NAMES = frozenset(globals())
163
126
  _backend_names: dict[int, dict[str, object]] = {} # id(module) -> names to copy
164
127
  _switches: dict[tuple[int, int], tuple[tuple[str, ...], dict[str, object]]] = {}
@@ -173,7 +136,6 @@ def _names_of(module) -> dict[str, object]:
173
136
  for name in getattr(module, "__all__", ())
174
137
  if not name.startswith("_")
175
138
  and name not in _OWN_NAMES
176
- and name not in _MOVED
177
139
  and hasattr(module, name)
178
140
  }
179
141
  _backend_names[id(module)] = names
@@ -9,6 +9,7 @@ import numpy as np
9
9
  from numpy import *
10
10
 
11
11
  from cunumpy import algorithms as algorithms
12
+ from cunumpy import arguments as arguments
12
13
  from cunumpy import cuda as cuda
13
14
  from cunumpy import kernels as kernels
14
15
  from cunumpy import memory as memory