XuPy 1.7.3__tar.gz → 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- xupy-2.0.0/PKG-INFO +203 -0
- xupy-2.0.0/README.md +181 -0
- xupy-2.0.0/XuPy.egg-info/PKG-INFO +203 -0
- xupy-2.0.0/XuPy.egg-info/SOURCES.txt +47 -0
- xupy-2.0.0/XuPy.egg-info/entry_points.txt +2 -0
- xupy-2.0.0/XuPy.egg-info/requires.txt +14 -0
- xupy-2.0.0/pyproject.toml +41 -0
- {xupy-1.7.3 → xupy-2.0.0}/test/test_arithmetic_compatibility.py +50 -50
- {xupy-1.7.3 → xupy-2.0.0}/test/test_core.py +57 -55
- {xupy-1.7.3 → xupy-2.0.0}/test/test_cpu_memory_context.py +13 -0
- xupy-2.0.0/test/test_gpu_no_sync.py +234 -0
- xupy-2.0.0/test/test_import_and_install.py +408 -0
- {xupy-1.7.3 → xupy-2.0.0}/test/test_large_array_printing.py +6 -6
- {xupy-1.7.3 → xupy-2.0.0}/test/test_ma_core.py +177 -156
- {xupy-1.7.3 → xupy-2.0.0}/test/test_ma_extras.py +222 -216
- xupy-2.0.0/test/test_ma_parity_api_core.py +1168 -0
- xupy-2.0.0/test/test_ma_parity_api_extras2.py +1130 -0
- xupy-2.0.0/test/test_ma_parity_extras_repr.py +1500 -0
- xupy-2.0.0/test/test_ma_parity_ops.py +1656 -0
- xupy-2.0.0/test/test_ma_parity_reductions.py +1245 -0
- xupy-2.0.0/test/test_ma_parity_structure.py +2825 -0
- xupy-2.0.0/test/test_ma_perf_semantics.py +157 -0
- xupy-2.0.0/test/test_memory_context_and_hygiene.py +248 -0
- xupy-2.0.0/test/test_namespace.py +340 -0
- xupy-2.0.0/test/test_numpy2_compat.py +110 -0
- xupy-2.0.0/test/test_shims.py +598 -0
- xupy-2.0.0/test/test_switching.py +584 -0
- xupy-2.0.0/xupy/__init__.py +28 -0
- xupy-2.0.0/xupy/__init__.pyi +77 -0
- xupy-2.0.0/xupy/__version__.py +1 -0
- xupy-2.0.0/xupy/_core.py +1261 -0
- xupy-2.0.0/xupy/_shims.py +244 -0
- xupy-2.0.0/xupy/install_cupy.py +125 -0
- xupy-2.0.0/xupy/ma/_backend.py +107 -0
- xupy-2.0.0/xupy/ma/_domains.py +196 -0
- xupy-2.0.0/xupy/ma/_funcs.py +915 -0
- xupy-2.0.0/xupy/ma/_ops.py +1452 -0
- xupy-2.0.0/xupy/ma/_printing.py +240 -0
- xupy-2.0.0/xupy/ma/_reductions.py +713 -0
- xupy-2.0.0/xupy/ma/_singletons.py +353 -0
- xupy-2.0.0/xupy/ma/core.py +1341 -0
- xupy-2.0.0/xupy/ma/extras.py +1095 -0
- {xupy-1.7.3 → xupy-2.0.0}/xupy/typings.py +17 -2
- xupy-1.7.3/PKG-INFO +0 -132
- xupy-1.7.3/README.md +0 -116
- xupy-1.7.3/XuPy.egg-info/PKG-INFO +0 -132
- xupy-1.7.3/XuPy.egg-info/SOURCES.txt +0 -26
- xupy-1.7.3/XuPy.egg-info/requires.txt +0 -1
- xupy-1.7.3/pyproject.toml +0 -24
- xupy-1.7.3/setup.py +0 -34
- xupy-1.7.3/xupy/__init__.py +0 -7
- xupy-1.7.3/xupy/__version__.py +0 -1
- xupy-1.7.3/xupy/_core.py +0 -1209
- xupy-1.7.3/xupy/_cupy_install/__check_availability__.py +0 -86
- xupy-1.7.3/xupy/_cupy_install/__init__.py +0 -0
- xupy-1.7.3/xupy/_cupy_install/__install_cupy__.py +0 -94
- xupy-1.7.3/xupy/ma/core.py +0 -3033
- xupy-1.7.3/xupy/ma/extras.py +0 -940
- {xupy-1.7.3 → xupy-2.0.0}/LICENSE +0 -0
- {xupy-1.7.3 → xupy-2.0.0}/XuPy.egg-info/dependency_links.txt +0 -0
- {xupy-1.7.3 → xupy-2.0.0}/XuPy.egg-info/top_level.txt +0 -0
- {xupy-1.7.3 → xupy-2.0.0}/setup.cfg +0 -0
- {xupy-1.7.3 → xupy-2.0.0}/xupy/ma/__init__.py +0 -0
- {xupy-1.7.3 → xupy-2.0.0}/xupy/py.typed +0 -0
xupy-2.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: XuPy
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: GPU Accelerated masked arrays with automatic handling of CPU and GPU arrays.
|
|
5
|
+
Author-email: Pietro Ferraiuolo <pietro.ferraiuolo@inaf.it>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/pietroferraiuolo/XuPy
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Requires-Dist: numpy>=2.0
|
|
12
|
+
Provides-Extra: cuda12
|
|
13
|
+
Requires-Dist: cupy-cuda12x>=14; extra == "cuda12"
|
|
14
|
+
Provides-Extra: cuda13
|
|
15
|
+
Requires-Dist: cupy-cuda13x>=14; extra == "cuda13"
|
|
16
|
+
Provides-Extra: array-api
|
|
17
|
+
Requires-Dist: array-api-compat>=1.11; extra == "array-api"
|
|
18
|
+
Provides-Extra: test
|
|
19
|
+
Requires-Dist: pytest; extra == "test"
|
|
20
|
+
Requires-Dist: psutil; extra == "test"
|
|
21
|
+
Dynamic: license-file
|
|
22
|
+
|
|
23
|
+
# XuPy
|
|
24
|
+
|
|
25
|
+

|
|
26
|
+
|
|
27
|
+
XuPy is a comprehensive Python package that provides GPU-accelerated masked arrays and NumPy-compatible functionality using CuPy. It automatically handles GPU/CPU fallback and offers an intuitive interface for scientific computing with masked data.
|
|
28
|
+
|
|
29
|
+
## Features
|
|
30
|
+
|
|
31
|
+
- **GPU Acceleration**: Automatic GPU detection with CuPy fallback to NumPy
|
|
32
|
+
- **Masked Arrays**: GPU masked arrays with the full `numpy.ma` API (every name of `numpy.ma.__all__` for numeric and bool dtypes) and `numpy.ma` semantics
|
|
33
|
+
- **Statistical Functions**: Comprehensive statistical operations (mean, std, var, min, max, etc.)
|
|
34
|
+
- **Array Manipulation**: Reshape, transpose, squeeze, expand_dims, and more
|
|
35
|
+
- **Mathematical Functions**: Trigonometric, exponential, logarithmic, and rounding functions
|
|
36
|
+
- **Random Generation**: Various random number generators (normal, uniform, etc.)
|
|
37
|
+
- **Universal Functions**: Support for applying any CuPy/NumPy ufunc with mask preservation
|
|
38
|
+
- **Performance**: Optimized for large-scale data processing on GPU
|
|
39
|
+
|
|
40
|
+
## What's new in 2.0
|
|
41
|
+
|
|
42
|
+
- **NumPy 2 namespace.** `xp` is a NumPy >= 2.0 namespace resolved per backend (CuPy on GPU, NumPy on CPU). Names removed in NumPy 2 raise `AttributeError` with a hint; `xp.float` and `xp.cfloat` are gone (use `xp.float64` / `xp.complex128`).
|
|
43
|
+
- **`xupy.ma` behaves like `numpy.ma`** on NumPy and CuPy data: same masks, fill values, hard masks, `numpy.ma` domain rules (results outside a function domain are masked; no other NaN/Inf auto-masking), return kinds (numpy scalars or `masked`) and all the public functions, including `median`, `unique`/set operations, `cov`, `polyfit`, `clump_*`, `masked_where` & co. Structured/record/object dtypes are not supported (`NotImplementedError`).
|
|
44
|
+
- **Device rule.** `masked_array` / `MaskedArray` / `array` / `*_like` move host input (NumPy, `numpy.ma`, lists) to the active backend; CuPy data and existing XuPy masked arrays never move implicitly, and operations follow the device of their operands.
|
|
45
|
+
- **No host synchronisation** in element-wise operations, ufuncs, `@`, axis reductions, `sort`, `clip`, `where`, `concatenate`, `average`, ... (see `test/test_gpu_no_sync.py`). Only scalar results, `repr`, boolean-mask assignment and data-dependent output sizes (`unique`, `compressed`, `nonzero`, ...) synchronise.
|
|
46
|
+
- **`MemoryContext` fixes** (MiB units, device restore, safe and fast cleanup) and a typed package (`py.typed`, `xupy/__init__.pyi`).
|
|
47
|
+
|
|
48
|
+
## Installation
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install xupy # CPU only (NumPy >= 2.0)
|
|
52
|
+
pip install "xupy[cuda12]" # with CuPy for CUDA 12.x
|
|
53
|
+
pip install "xupy[cuda13]" # with CuPy for CUDA 13.x
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Pick the extra matching the "CUDA Version" reported by `nvidia-smi`.
|
|
57
|
+
|
|
58
|
+
Alternatively, install XuPy and then let the helper detect your CUDA version and install CuPy:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
pip install xupy
|
|
62
|
+
python -m xupy.install_cupy # or: xupy-install-cupy
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Options: `--dry-run` (show the command without running it), `-y/--yes` (do not ask for confirmation), `--package PKG` (install a specific CuPy package, e.g. `cupy-cuda12x`).
|
|
66
|
+
|
|
67
|
+
`import xupy` never prompts. If an NVIDIA GPU is present but CuPy is unusable, XuPy falls back to NumPy and emits a single warning; set `XUPY_NO_GPU_WARNING=1` to silence it.
|
|
68
|
+
|
|
69
|
+
## Quick Start
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
import xupy as xp
|
|
73
|
+
|
|
74
|
+
# Create arrays with automatic GPU detection
|
|
75
|
+
a = xp.random.normal(0, 1, (1000, 1000))
|
|
76
|
+
b = xp.random.normal(0, 1, (1000, 1000))
|
|
77
|
+
|
|
78
|
+
# Create masks
|
|
79
|
+
mask = xp.random.random((1000, 1000)) > 0.1
|
|
80
|
+
|
|
81
|
+
# Create masked arrays
|
|
82
|
+
am = xp.ma.masked_array(a, mask)
|
|
83
|
+
bm = xp.ma.masked_array(b, mask)
|
|
84
|
+
|
|
85
|
+
# Perform operations (masks are automatically handled)
|
|
86
|
+
result = am + bm
|
|
87
|
+
mean_val = am.mean()
|
|
88
|
+
std_val = am.std()
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
## Backends and Devices
|
|
92
|
+
|
|
93
|
+
`xp` is a NumPy 2.x namespace backed by CuPy (GPU) when it is usable, NumPy (CPU) otherwise. The namespace is resolved at every access, so it always reflects the active backend.
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
import xupy as xp
|
|
97
|
+
|
|
98
|
+
xp.use_cpu() # global default: NumPy (thread-safe, idempotent)
|
|
99
|
+
xp.use_gpu() # global default: CuPy (RuntimeError if CuPy is unusable)
|
|
100
|
+
|
|
101
|
+
with xp.backend("cpu"): # scoped, thread- and asyncio-local ("cpu"/"numpy"/"gpu"/"cupy")
|
|
102
|
+
a = xp.zeros(3) # NumPy array; other threads are unaffected
|
|
103
|
+
|
|
104
|
+
xp.on_gpu # live: reflects the active backend
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
- `xp.ma` follows the backend: XuPy's GPU masked arrays on GPU, `numpy.ma` on CPU. `xp.np` and `xp.npma` are always `numpy` and `numpy.ma`.
|
|
108
|
+
- `from xupy import on_gpu` (and `from xupy import *`) is a snapshot taken at import time; use `xp.on_gpu` for the live value.
|
|
109
|
+
- `use_cpu()`/`use_gpu()` change the default for *all* threads. A switch from another thread while a computation is running can split it across backends; for concurrent code prefer `with xp.backend(...)`, which only affects the current thread/task.
|
|
110
|
+
- `xupy.ma` masked arrays follow their data: creating one from host data (NumPy arrays, `numpy.ma` arrays, lists) places it on the active backend (the GPU in GPU mode); an existing masked array never moves implicitly, and operations run on the device of their operands (a GPU masked array stays on the GPU even inside `with xp.backend("cpu")`). Use `a.to_device("cpu")` / `"gpu"` to move one explicitly.
|
|
111
|
+
- Because `xupy.ma` follows the backend, `import xupy.ma.core as mc` yields `numpy.ma.core` on CPU; use `from xupy.ma import core` or `sys.modules["xupy.ma"]` to always reach XuPy's module.
|
|
112
|
+
- Names removed in NumPy 2 (`NaN`, `float_`, `in1d`, `trapz`, ...) raise `AttributeError` with a hint on both backends, e.g. `xupy has no attribute 'NaN': removed in NumPy 2.0, use 'nan'`.
|
|
113
|
+
- NumPy 2 names CuPy lacks are shimmed on GPU (`vecdot`, `unstack`, `sort(stable=, descending=)`, `unique(sorted=)`, `errstate`, `linalg.vector_norm`, ...). Host-only names with no CuPy equivalent (`emath`, `strings`, `char`, `rec`, ...) raise an `AttributeError` that points to `xp.backend("cpu")` / `xp.asnumpy()`.
|
|
114
|
+
- `xp.on_device(i)` is always a context manager. On GPU, `i` in `0 .. n_gpus-1` selects that CUDA device (also with a single GPU) and `-1` runs the block on the CPU; out-of-range values raise `ValueError`. On CPU it is a no-op for any `i`.
|
|
115
|
+
- `xp.set_device(i)` sets the current CUDA device (setting the current one is a silent no-op, an invalid id raises `ValueError`); it is a no-op on CPU.
|
|
116
|
+
- The GPU banner (on import) and switch messages (when `use_cpu()`/`use_gpu()` actually change the backend) are printed to stdout and also emitted on the `xupy` logger at INFO level; `xp.backend(...)` scopes are silent.
|
|
117
|
+
|
|
118
|
+
## Performance Benefits
|
|
119
|
+
|
|
120
|
+
XuPy automatically detects GPU availability and provides significant speedup for large arrays:
|
|
121
|
+
|
|
122
|
+
- **Small arrays (< 1000 elements)**: CPU (NumPy) may be faster due to GPU overhead
|
|
123
|
+
- **Medium arrays (1000-10000 elements)**: GPU provides 2-5x speedup
|
|
124
|
+
- **Large arrays (> 10000 elements)**: GPU provides 5-20x speedup depending on operation complexity
|
|
125
|
+
|
|
126
|
+
### Benchmarks
|
|
127
|
+
|
|
128
|
+
`python benchmarks/bench_ma.py` compares `xupy.ma` (GPU), raw CuPy and `numpy.ma` on a few array sizes (`--sizes 500,2000,4000`, `--repeat N`, `--no-numpy`); it is a standalone script and is not run in CI.
|
|
129
|
+
|
|
130
|
+
## GPU Requirements
|
|
131
|
+
|
|
132
|
+
- **A GPU supported by CuPy >= 14** (see the CuPy documentation)
|
|
133
|
+
- **CuPy >= 14** (optional) installed, e.g. `pip install "xupy[cuda12]"` or `pip install "xupy[cuda13]"`
|
|
134
|
+
- **Automatic fallback** to NumPy if GPU is unavailable
|
|
135
|
+
|
|
136
|
+
## Requirements
|
|
137
|
+
|
|
138
|
+
- Python >= 3.10
|
|
139
|
+
- NumPy >= 2.0
|
|
140
|
+
- CuPy >= 14 (optional, for GPU support)
|
|
141
|
+
|
|
142
|
+
## API Compatibility
|
|
143
|
+
|
|
144
|
+
XuPy maintains high compatibility with NumPy's masked array interface while leveraging CuPy's optimized operations:
|
|
145
|
+
|
|
146
|
+
- All standard properties (`shape`, `dtype`, `size`, `ndim`, `T`)
|
|
147
|
+
- Comprehensive arithmetic operations with mask propagation
|
|
148
|
+
- **Memory-optimized statistical methods** (`mean`, `std`, `var`, `min`, `max`) using CuPy's native operations
|
|
149
|
+
- Array manipulation methods (`reshape`, `transpose`, `squeeze`)
|
|
150
|
+
- Universal function support through `apply_ufunc`
|
|
151
|
+
- Conversion to NumPy masked arrays via `asmarray()`
|
|
152
|
+
- **GPU memory management** through `MemoryContext`
|
|
153
|
+
|
|
154
|
+
## GPU Memory Management
|
|
155
|
+
|
|
156
|
+
XuPy includes an advanced `MemoryContext` class for efficient GPU memory management:
|
|
157
|
+
|
|
158
|
+
```python
|
|
159
|
+
import xupy as xp
|
|
160
|
+
|
|
161
|
+
# Basic usage with automatic cleanup
|
|
162
|
+
with xp.MemoryContext() as ctx:
|
|
163
|
+
# GPU operations
|
|
164
|
+
data = xp.random.normal(0, 1, (10000, 10000))
|
|
165
|
+
result = data.mean()
|
|
166
|
+
# Memory automatically cleaned up on exit
|
|
167
|
+
|
|
168
|
+
# Advanced features
|
|
169
|
+
with xp.MemoryContext(memory_threshold=0.8, auto_cleanup=True) as ctx:
|
|
170
|
+
# Monitor memory usage
|
|
171
|
+
mem_info = ctx.get_memory_info()
|
|
172
|
+
# 'total', 'free' and 'used' are in MiB (1024**2 bytes)
|
|
173
|
+
print(f"GPU Memory: {mem_info['used']:.2f} MiB")
|
|
174
|
+
|
|
175
|
+
# Aggressive cleanup when needed
|
|
176
|
+
if ctx.check_memory_pressure():
|
|
177
|
+
ctx.aggressive_cleanup()
|
|
178
|
+
|
|
179
|
+
# Emergency cleanup for critical situations
|
|
180
|
+
ctx.emergency_cleanup()
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
### MemoryContext Features
|
|
184
|
+
|
|
185
|
+
- **Automatic Cleanup**: Memory freed automatically when exiting context
|
|
186
|
+
- **Memory Monitoring**: Real-time tracking of GPU memory usage
|
|
187
|
+
- **Pressure Detection**: Automatic cleanup when memory usage is high
|
|
188
|
+
- **Aggressive Cleanup**: Force garbage collection and cache clearing
|
|
189
|
+
- **Emergency Cleanup**: Nuclear option for out-of-memory situations
|
|
190
|
+
- **Safe Cleanup**: Only garbage collection and memory-pool freeing; user objects are never modified
|
|
191
|
+
- **Memory History**: Keep history of memory usage over time
|
|
192
|
+
|
|
193
|
+
All memory figures are binary: `MB` / `MiB` = 1024**2 bytes and `GB` / `GiB` = 1024**3 bytes
|
|
194
|
+
(this also holds for `xp.array_size(shape, dtype, out_unit='MB')`, which accepts `'B'`, `'KB'`, `'MB'`, `'GB'`).
|
|
195
|
+
The previous device is always restored when the context exits.
|
|
196
|
+
|
|
197
|
+
## Documentation
|
|
198
|
+
|
|
199
|
+
For detailed documentation, including comprehensive API reference and advanced usage examples, see [docs/source/index.md](docs/source/index.md).
|
|
200
|
+
|
|
201
|
+
## License
|
|
202
|
+
|
|
203
|
+
See [LICENSE](LICENSE).
|
xupy-2.0.0/README.md
ADDED
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
# XuPy
|
|
2
|
+
|
|
3
|
+

|
|
4
|
+
|
|
5
|
+
XuPy is a comprehensive Python package that provides GPU-accelerated masked arrays and NumPy-compatible functionality using CuPy. It automatically handles GPU/CPU fallback and offers an intuitive interface for scientific computing with masked data.
|
|
6
|
+
|
|
7
|
+
## Features
|
|
8
|
+
|
|
9
|
+
- **GPU Acceleration**: Automatic GPU detection with CuPy fallback to NumPy
|
|
10
|
+
- **Masked Arrays**: GPU masked arrays with the full `numpy.ma` API (every name of `numpy.ma.__all__` for numeric and bool dtypes) and `numpy.ma` semantics
|
|
11
|
+
- **Statistical Functions**: Comprehensive statistical operations (mean, std, var, min, max, etc.)
|
|
12
|
+
- **Array Manipulation**: Reshape, transpose, squeeze, expand_dims, and more
|
|
13
|
+
- **Mathematical Functions**: Trigonometric, exponential, logarithmic, and rounding functions
|
|
14
|
+
- **Random Generation**: Various random number generators (normal, uniform, etc.)
|
|
15
|
+
- **Universal Functions**: Support for applying any CuPy/NumPy ufunc with mask preservation
|
|
16
|
+
- **Performance**: Optimized for large-scale data processing on GPU
|
|
17
|
+
|
|
18
|
+
## What's new in 2.0
|
|
19
|
+
|
|
20
|
+
- **NumPy 2 namespace.** `xp` is a NumPy >= 2.0 namespace resolved per backend (CuPy on GPU, NumPy on CPU). Names removed in NumPy 2 raise `AttributeError` with a hint; `xp.float` and `xp.cfloat` are gone (use `xp.float64` / `xp.complex128`).
|
|
21
|
+
- **`xupy.ma` behaves like `numpy.ma`** on NumPy and CuPy data: same masks, fill values, hard masks, `numpy.ma` domain rules (results outside a function domain are masked; no other NaN/Inf auto-masking), return kinds (numpy scalars or `masked`) and all the public functions, including `median`, `unique`/set operations, `cov`, `polyfit`, `clump_*`, `masked_where` & co. Structured/record/object dtypes are not supported (`NotImplementedError`).
|
|
22
|
+
- **Device rule.** `masked_array` / `MaskedArray` / `array` / `*_like` move host input (NumPy, `numpy.ma`, lists) to the active backend; CuPy data and existing XuPy masked arrays never move implicitly, and operations follow the device of their operands.
|
|
23
|
+
- **No host synchronisation** in element-wise operations, ufuncs, `@`, axis reductions, `sort`, `clip`, `where`, `concatenate`, `average`, ... (see `test/test_gpu_no_sync.py`). Only scalar results, `repr`, boolean-mask assignment and data-dependent output sizes (`unique`, `compressed`, `nonzero`, ...) synchronise.
|
|
24
|
+
- **`MemoryContext` fixes** (MiB units, device restore, safe and fast cleanup) and a typed package (`py.typed`, `xupy/__init__.pyi`).
|
|
25
|
+
|
|
26
|
+
## Installation
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install xupy # CPU only (NumPy >= 2.0)
|
|
30
|
+
pip install "xupy[cuda12]" # with CuPy for CUDA 12.x
|
|
31
|
+
pip install "xupy[cuda13]" # with CuPy for CUDA 13.x
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Pick the extra matching the "CUDA Version" reported by `nvidia-smi`.
|
|
35
|
+
|
|
36
|
+
Alternatively, install XuPy and then let the helper detect your CUDA version and install CuPy:
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
pip install xupy
|
|
40
|
+
python -m xupy.install_cupy # or: xupy-install-cupy
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Options: `--dry-run` (show the command without running it), `-y/--yes` (do not ask for confirmation), `--package PKG` (install a specific CuPy package, e.g. `cupy-cuda12x`).
|
|
44
|
+
|
|
45
|
+
`import xupy` never prompts. If an NVIDIA GPU is present but CuPy is unusable, XuPy falls back to NumPy and emits a single warning; set `XUPY_NO_GPU_WARNING=1` to silence it.
|
|
46
|
+
|
|
47
|
+
## Quick Start
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
import xupy as xp
|
|
51
|
+
|
|
52
|
+
# Create arrays with automatic GPU detection
|
|
53
|
+
a = xp.random.normal(0, 1, (1000, 1000))
|
|
54
|
+
b = xp.random.normal(0, 1, (1000, 1000))
|
|
55
|
+
|
|
56
|
+
# Create masks
|
|
57
|
+
mask = xp.random.random((1000, 1000)) > 0.1
|
|
58
|
+
|
|
59
|
+
# Create masked arrays
|
|
60
|
+
am = xp.ma.masked_array(a, mask)
|
|
61
|
+
bm = xp.ma.masked_array(b, mask)
|
|
62
|
+
|
|
63
|
+
# Perform operations (masks are automatically handled)
|
|
64
|
+
result = am + bm
|
|
65
|
+
mean_val = am.mean()
|
|
66
|
+
std_val = am.std()
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
## Backends and Devices
|
|
70
|
+
|
|
71
|
+
`xp` is a NumPy 2.x namespace backed by CuPy (GPU) when it is usable, NumPy (CPU) otherwise. The namespace is resolved at every access, so it always reflects the active backend.
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
import xupy as xp
|
|
75
|
+
|
|
76
|
+
xp.use_cpu() # global default: NumPy (thread-safe, idempotent)
|
|
77
|
+
xp.use_gpu() # global default: CuPy (RuntimeError if CuPy is unusable)
|
|
78
|
+
|
|
79
|
+
with xp.backend("cpu"): # scoped, thread- and asyncio-local ("cpu"/"numpy"/"gpu"/"cupy")
|
|
80
|
+
a = xp.zeros(3) # NumPy array; other threads are unaffected
|
|
81
|
+
|
|
82
|
+
xp.on_gpu # live: reflects the active backend
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
- `xp.ma` follows the backend: XuPy's GPU masked arrays on GPU, `numpy.ma` on CPU. `xp.np` and `xp.npma` are always `numpy` and `numpy.ma`.
|
|
86
|
+
- `from xupy import on_gpu` (and `from xupy import *`) is a snapshot taken at import time; use `xp.on_gpu` for the live value.
|
|
87
|
+
- `use_cpu()`/`use_gpu()` change the default for *all* threads. A switch from another thread while a computation is running can split it across backends; for concurrent code prefer `with xp.backend(...)`, which only affects the current thread/task.
|
|
88
|
+
- `xupy.ma` masked arrays follow their data: creating one from host data (NumPy arrays, `numpy.ma` arrays, lists) places it on the active backend (the GPU in GPU mode); an existing masked array never moves implicitly, and operations run on the device of their operands (a GPU masked array stays on the GPU even inside `with xp.backend("cpu")`). Use `a.to_device("cpu")` / `"gpu"` to move one explicitly.
|
|
89
|
+
- Because `xupy.ma` follows the backend, `import xupy.ma.core as mc` yields `numpy.ma.core` on CPU; use `from xupy.ma import core` or `sys.modules["xupy.ma"]` to always reach XuPy's module.
|
|
90
|
+
- Names removed in NumPy 2 (`NaN`, `float_`, `in1d`, `trapz`, ...) raise `AttributeError` with a hint on both backends, e.g. `xupy has no attribute 'NaN': removed in NumPy 2.0, use 'nan'`.
|
|
91
|
+
- NumPy 2 names CuPy lacks are shimmed on GPU (`vecdot`, `unstack`, `sort(stable=, descending=)`, `unique(sorted=)`, `errstate`, `linalg.vector_norm`, ...). Host-only names with no CuPy equivalent (`emath`, `strings`, `char`, `rec`, ...) raise an `AttributeError` that points to `xp.backend("cpu")` / `xp.asnumpy()`.
|
|
92
|
+
- `xp.on_device(i)` is always a context manager. On GPU, `i` in `0 .. n_gpus-1` selects that CUDA device (also with a single GPU) and `-1` runs the block on the CPU; out-of-range values raise `ValueError`. On CPU it is a no-op for any `i`.
|
|
93
|
+
- `xp.set_device(i)` sets the current CUDA device (setting the current one is a silent no-op, an invalid id raises `ValueError`); it is a no-op on CPU.
|
|
94
|
+
- The GPU banner (on import) and switch messages (when `use_cpu()`/`use_gpu()` actually change the backend) are printed to stdout and also emitted on the `xupy` logger at INFO level; `xp.backend(...)` scopes are silent.
|
|
95
|
+
|
|
96
|
+
## Performance Benefits
|
|
97
|
+
|
|
98
|
+
XuPy automatically detects GPU availability and provides significant speedup for large arrays:
|
|
99
|
+
|
|
100
|
+
- **Small arrays (< 1000 elements)**: CPU (NumPy) may be faster due to GPU overhead
|
|
101
|
+
- **Medium arrays (1000-10000 elements)**: GPU provides 2-5x speedup
|
|
102
|
+
- **Large arrays (> 10000 elements)**: GPU provides 5-20x speedup depending on operation complexity
|
|
103
|
+
|
|
104
|
+
### Benchmarks
|
|
105
|
+
|
|
106
|
+
`python benchmarks/bench_ma.py` compares `xupy.ma` (GPU), raw CuPy and `numpy.ma` on a few array sizes (`--sizes 500,2000,4000`, `--repeat N`, `--no-numpy`); it is a standalone script and is not run in CI.
|
|
107
|
+
|
|
108
|
+
## GPU Requirements
|
|
109
|
+
|
|
110
|
+
- **A GPU supported by CuPy >= 14** (see the CuPy documentation)
|
|
111
|
+
- **CuPy >= 14** (optional) installed, e.g. `pip install "xupy[cuda12]"` or `pip install "xupy[cuda13]"`
|
|
112
|
+
- **Automatic fallback** to NumPy if GPU is unavailable
|
|
113
|
+
|
|
114
|
+
## Requirements
|
|
115
|
+
|
|
116
|
+
- Python >= 3.10
|
|
117
|
+
- NumPy >= 2.0
|
|
118
|
+
- CuPy >= 14 (optional, for GPU support)
|
|
119
|
+
|
|
120
|
+
## API Compatibility
|
|
121
|
+
|
|
122
|
+
XuPy maintains high compatibility with NumPy's masked array interface while leveraging CuPy's optimized operations:
|
|
123
|
+
|
|
124
|
+
- All standard properties (`shape`, `dtype`, `size`, `ndim`, `T`)
|
|
125
|
+
- Comprehensive arithmetic operations with mask propagation
|
|
126
|
+
- **Memory-optimized statistical methods** (`mean`, `std`, `var`, `min`, `max`) using CuPy's native operations
|
|
127
|
+
- Array manipulation methods (`reshape`, `transpose`, `squeeze`)
|
|
128
|
+
- Universal function support through `apply_ufunc`
|
|
129
|
+
- Conversion to NumPy masked arrays via `asmarray()`
|
|
130
|
+
- **GPU memory management** through `MemoryContext`
|
|
131
|
+
|
|
132
|
+
## GPU Memory Management
|
|
133
|
+
|
|
134
|
+
XuPy includes an advanced `MemoryContext` class for efficient GPU memory management:
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
import xupy as xp
|
|
138
|
+
|
|
139
|
+
# Basic usage with automatic cleanup
|
|
140
|
+
with xp.MemoryContext() as ctx:
|
|
141
|
+
# GPU operations
|
|
142
|
+
data = xp.random.normal(0, 1, (10000, 10000))
|
|
143
|
+
result = data.mean()
|
|
144
|
+
# Memory automatically cleaned up on exit
|
|
145
|
+
|
|
146
|
+
# Advanced features
|
|
147
|
+
with xp.MemoryContext(memory_threshold=0.8, auto_cleanup=True) as ctx:
|
|
148
|
+
# Monitor memory usage
|
|
149
|
+
mem_info = ctx.get_memory_info()
|
|
150
|
+
# 'total', 'free' and 'used' are in MiB (1024**2 bytes)
|
|
151
|
+
print(f"GPU Memory: {mem_info['used']:.2f} MiB")
|
|
152
|
+
|
|
153
|
+
# Aggressive cleanup when needed
|
|
154
|
+
if ctx.check_memory_pressure():
|
|
155
|
+
ctx.aggressive_cleanup()
|
|
156
|
+
|
|
157
|
+
# Emergency cleanup for critical situations
|
|
158
|
+
ctx.emergency_cleanup()
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
### MemoryContext Features
|
|
162
|
+
|
|
163
|
+
- **Automatic Cleanup**: Memory freed automatically when exiting context
|
|
164
|
+
- **Memory Monitoring**: Real-time tracking of GPU memory usage
|
|
165
|
+
- **Pressure Detection**: Automatic cleanup when memory usage is high
|
|
166
|
+
- **Aggressive Cleanup**: Force garbage collection and cache clearing
|
|
167
|
+
- **Emergency Cleanup**: Nuclear option for out-of-memory situations
|
|
168
|
+
- **Safe Cleanup**: Only garbage collection and memory-pool freeing; user objects are never modified
|
|
169
|
+
- **Memory History**: Keep history of memory usage over time
|
|
170
|
+
|
|
171
|
+
All memory figures are binary: `MB` / `MiB` = 1024**2 bytes and `GB` / `GiB` = 1024**3 bytes
|
|
172
|
+
(this also holds for `xp.array_size(shape, dtype, out_unit='MB')`, which accepts `'B'`, `'KB'`, `'MB'`, `'GB'`).
|
|
173
|
+
The previous device is always restored when the context exits.
|
|
174
|
+
|
|
175
|
+
## Documentation
|
|
176
|
+
|
|
177
|
+
For detailed documentation, including comprehensive API reference and advanced usage examples, see [docs/source/index.md](docs/source/index.md).
|
|
178
|
+
|
|
179
|
+
## License
|
|
180
|
+
|
|
181
|
+
See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: XuPy
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: GPU Accelerated masked arrays with automatic handling of CPU and GPU arrays.
|
|
5
|
+
Author-email: Pietro Ferraiuolo <pietro.ferraiuolo@inaf.it>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/pietroferraiuolo/XuPy
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Requires-Dist: numpy>=2.0
|
|
12
|
+
Provides-Extra: cuda12
|
|
13
|
+
Requires-Dist: cupy-cuda12x>=14; extra == "cuda12"
|
|
14
|
+
Provides-Extra: cuda13
|
|
15
|
+
Requires-Dist: cupy-cuda13x>=14; extra == "cuda13"
|
|
16
|
+
Provides-Extra: array-api
|
|
17
|
+
Requires-Dist: array-api-compat>=1.11; extra == "array-api"
|
|
18
|
+
Provides-Extra: test
|
|
19
|
+
Requires-Dist: pytest; extra == "test"
|
|
20
|
+
Requires-Dist: psutil; extra == "test"
|
|
21
|
+
Dynamic: license-file
|
|
22
|
+
|
|
23
|
+
# XuPy
|
|
24
|
+
|
|
25
|
+

|
|
26
|
+
|
|
27
|
+
XuPy is a comprehensive Python package that provides GPU-accelerated masked arrays and NumPy-compatible functionality using CuPy. It automatically handles GPU/CPU fallback and offers an intuitive interface for scientific computing with masked data.
|
|
28
|
+
|
|
29
|
+
## Features
|
|
30
|
+
|
|
31
|
+
- **GPU Acceleration**: Automatic GPU detection with CuPy fallback to NumPy
|
|
32
|
+
- **Masked Arrays**: GPU masked arrays with the full `numpy.ma` API (every name of `numpy.ma.__all__` for numeric and bool dtypes) and `numpy.ma` semantics
|
|
33
|
+
- **Statistical Functions**: Comprehensive statistical operations (mean, std, var, min, max, etc.)
|
|
34
|
+
- **Array Manipulation**: Reshape, transpose, squeeze, expand_dims, and more
|
|
35
|
+
- **Mathematical Functions**: Trigonometric, exponential, logarithmic, and rounding functions
|
|
36
|
+
- **Random Generation**: Various random number generators (normal, uniform, etc.)
|
|
37
|
+
- **Universal Functions**: Support for applying any CuPy/NumPy ufunc with mask preservation
|
|
38
|
+
- **Performance**: Optimized for large-scale data processing on GPU
|
|
39
|
+
|
|
40
|
+
## What's new in 2.0
|
|
41
|
+
|
|
42
|
+
- **NumPy 2 namespace.** `xp` is a NumPy >= 2.0 namespace resolved per backend (CuPy on GPU, NumPy on CPU). Names removed in NumPy 2 raise `AttributeError` with a hint; `xp.float` and `xp.cfloat` are gone (use `xp.float64` / `xp.complex128`).
|
|
43
|
+
- **`xupy.ma` behaves like `numpy.ma`** on NumPy and CuPy data: same masks, fill values, hard masks, `numpy.ma` domain rules (results outside a function domain are masked; no other NaN/Inf auto-masking), return kinds (numpy scalars or `masked`) and all the public functions, including `median`, `unique`/set operations, `cov`, `polyfit`, `clump_*`, `masked_where` & co. Structured/record/object dtypes are not supported (`NotImplementedError`).
|
|
44
|
+
- **Device rule.** `masked_array` / `MaskedArray` / `array` / `*_like` move host input (NumPy, `numpy.ma`, lists) to the active backend; CuPy data and existing XuPy masked arrays never move implicitly, and operations follow the device of their operands.
|
|
45
|
+
- **No host synchronisation** in element-wise operations, ufuncs, `@`, axis reductions, `sort`, `clip`, `where`, `concatenate`, `average`, ... (see `test/test_gpu_no_sync.py`). Only scalar results, `repr`, boolean-mask assignment and data-dependent output sizes (`unique`, `compressed`, `nonzero`, ...) synchronise.
|
|
46
|
+
- **`MemoryContext` fixes** (MiB units, device restore, safe and fast cleanup) and a typed package (`py.typed`, `xupy/__init__.pyi`).
|
|
47
|
+
|
|
48
|
+
## Installation
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install xupy # CPU only (NumPy >= 2.0)
|
|
52
|
+
pip install "xupy[cuda12]" # with CuPy for CUDA 12.x
|
|
53
|
+
pip install "xupy[cuda13]" # with CuPy for CUDA 13.x
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Pick the extra matching the "CUDA Version" reported by `nvidia-smi`.
|
|
57
|
+
|
|
58
|
+
Alternatively, install XuPy and then let the helper detect your CUDA version and install CuPy:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
pip install xupy
|
|
62
|
+
python -m xupy.install_cupy # or: xupy-install-cupy
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Options: `--dry-run` (show the command without running it), `-y/--yes` (do not ask for confirmation), `--package PKG` (install a specific CuPy package, e.g. `cupy-cuda12x`).
|
|
66
|
+
|
|
67
|
+
`import xupy` never prompts. If an NVIDIA GPU is present but CuPy is unusable, XuPy falls back to NumPy and emits a single warning; set `XUPY_NO_GPU_WARNING=1` to silence it.
|
|
68
|
+
|
|
69
|
+
## Quick Start
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
import xupy as xp
|
|
73
|
+
|
|
74
|
+
# Create arrays with automatic GPU detection
|
|
75
|
+
a = xp.random.normal(0, 1, (1000, 1000))
|
|
76
|
+
b = xp.random.normal(0, 1, (1000, 1000))
|
|
77
|
+
|
|
78
|
+
# Create masks
|
|
79
|
+
mask = xp.random.random((1000, 1000)) > 0.1
|
|
80
|
+
|
|
81
|
+
# Create masked arrays
|
|
82
|
+
am = xp.ma.masked_array(a, mask)
|
|
83
|
+
bm = xp.ma.masked_array(b, mask)
|
|
84
|
+
|
|
85
|
+
# Perform operations (masks are automatically handled)
|
|
86
|
+
result = am + bm
|
|
87
|
+
mean_val = am.mean()
|
|
88
|
+
std_val = am.std()
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
## Backends and Devices
|
|
92
|
+
|
|
93
|
+
`xp` is a NumPy 2.x namespace backed by CuPy (GPU) when it is usable, NumPy (CPU) otherwise. The namespace is resolved at every access, so it always reflects the active backend.
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
import xupy as xp
|
|
97
|
+
|
|
98
|
+
xp.use_cpu() # global default: NumPy (thread-safe, idempotent)
|
|
99
|
+
xp.use_gpu() # global default: CuPy (RuntimeError if CuPy is unusable)
|
|
100
|
+
|
|
101
|
+
with xp.backend("cpu"): # scoped, thread- and asyncio-local ("cpu"/"numpy"/"gpu"/"cupy")
|
|
102
|
+
a = xp.zeros(3) # NumPy array; other threads are unaffected
|
|
103
|
+
|
|
104
|
+
xp.on_gpu # live: reflects the active backend
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
- `xp.ma` follows the backend: XuPy's GPU masked arrays on GPU, `numpy.ma` on CPU. `xp.np` and `xp.npma` are always `numpy` and `numpy.ma`.
|
|
108
|
+
- `from xupy import on_gpu` (and `from xupy import *`) is a snapshot taken at import time; use `xp.on_gpu` for the live value.
|
|
109
|
+
- `use_cpu()`/`use_gpu()` change the default for *all* threads. A switch from another thread while a computation is running can split it across backends; for concurrent code prefer `with xp.backend(...)`, which only affects the current thread/task.
|
|
110
|
+
- `xupy.ma` masked arrays follow their data: creating one from host data (NumPy arrays, `numpy.ma` arrays, lists) places it on the active backend (the GPU in GPU mode); an existing masked array never moves implicitly, and operations run on the device of their operands (a GPU masked array stays on the GPU even inside `with xp.backend("cpu")`). Use `a.to_device("cpu")` / `"gpu"` to move one explicitly.
|
|
111
|
+
- Because `xupy.ma` follows the backend, `import xupy.ma.core as mc` yields `numpy.ma.core` on CPU; use `from xupy.ma import core` or `sys.modules["xupy.ma"]` to always reach XuPy's module.
|
|
112
|
+
- Names removed in NumPy 2 (`NaN`, `float_`, `in1d`, `trapz`, ...) raise `AttributeError` with a hint on both backends, e.g. `xupy has no attribute 'NaN': removed in NumPy 2.0, use 'nan'`.
|
|
113
|
+
- NumPy 2 names CuPy lacks are shimmed on GPU (`vecdot`, `unstack`, `sort(stable=, descending=)`, `unique(sorted=)`, `errstate`, `linalg.vector_norm`, ...). Host-only names with no CuPy equivalent (`emath`, `strings`, `char`, `rec`, ...) raise an `AttributeError` that points to `xp.backend("cpu")` / `xp.asnumpy()`.
|
|
114
|
+
- `xp.on_device(i)` is always a context manager. On GPU, `i` in `0 .. n_gpus-1` selects that CUDA device (also with a single GPU) and `-1` runs the block on the CPU; out-of-range values raise `ValueError`. On CPU it is a no-op for any `i`.
|
|
115
|
+
- `xp.set_device(i)` sets the current CUDA device (setting the current one is a silent no-op, an invalid id raises `ValueError`); it is a no-op on CPU.
|
|
116
|
+
- The GPU banner (on import) and switch messages (when `use_cpu()`/`use_gpu()` actually change the backend) are printed to stdout and also emitted on the `xupy` logger at INFO level; `xp.backend(...)` scopes are silent.
|
|
117
|
+
|
|
118
|
+
## Performance Benefits
|
|
119
|
+
|
|
120
|
+
XuPy automatically detects GPU availability and provides significant speedup for large arrays:
|
|
121
|
+
|
|
122
|
+
- **Small arrays (< 1000 elements)**: CPU (NumPy) may be faster due to GPU overhead
|
|
123
|
+
- **Medium arrays (1000-10000 elements)**: GPU provides 2-5x speedup
|
|
124
|
+
- **Large arrays (> 10000 elements)**: GPU provides 5-20x speedup depending on operation complexity
|
|
125
|
+
|
|
126
|
+
### Benchmarks
|
|
127
|
+
|
|
128
|
+
`python benchmarks/bench_ma.py` compares `xupy.ma` (GPU), raw CuPy and `numpy.ma` on a few array sizes (`--sizes 500,2000,4000`, `--repeat N`, `--no-numpy`); it is a standalone script and is not run in CI.
|
|
129
|
+
|
|
130
|
+
## GPU Requirements
|
|
131
|
+
|
|
132
|
+
- **A GPU supported by CuPy >= 14** (see the CuPy documentation)
|
|
133
|
+
- **CuPy >= 14** (optional) installed, e.g. `pip install "xupy[cuda12]"` or `pip install "xupy[cuda13]"`
|
|
134
|
+
- **Automatic fallback** to NumPy if GPU is unavailable
|
|
135
|
+
|
|
136
|
+
## Requirements
|
|
137
|
+
|
|
138
|
+
- Python >= 3.10
|
|
139
|
+
- NumPy >= 2.0
|
|
140
|
+
- CuPy >= 14 (optional, for GPU support)
|
|
141
|
+
|
|
142
|
+
## API Compatibility
|
|
143
|
+
|
|
144
|
+
XuPy maintains high compatibility with NumPy's masked array interface while leveraging CuPy's optimized operations:
|
|
145
|
+
|
|
146
|
+
- All standard properties (`shape`, `dtype`, `size`, `ndim`, `T`)
|
|
147
|
+
- Comprehensive arithmetic operations with mask propagation
|
|
148
|
+
- **Memory-optimized statistical methods** (`mean`, `std`, `var`, `min`, `max`) using CuPy's native operations
|
|
149
|
+
- Array manipulation methods (`reshape`, `transpose`, `squeeze`)
|
|
150
|
+
- Universal function support through `apply_ufunc`
|
|
151
|
+
- Conversion to NumPy masked arrays via `asmarray()`
|
|
152
|
+
- **GPU memory management** through `MemoryContext`
|
|
153
|
+
|
|
154
|
+
## GPU Memory Management
|
|
155
|
+
|
|
156
|
+
XuPy includes an advanced `MemoryContext` class for efficient GPU memory management:
|
|
157
|
+
|
|
158
|
+
```python
|
|
159
|
+
import xupy as xp
|
|
160
|
+
|
|
161
|
+
# Basic usage with automatic cleanup
|
|
162
|
+
with xp.MemoryContext() as ctx:
|
|
163
|
+
# GPU operations
|
|
164
|
+
data = xp.random.normal(0, 1, (10000, 10000))
|
|
165
|
+
result = data.mean()
|
|
166
|
+
# Memory automatically cleaned up on exit
|
|
167
|
+
|
|
168
|
+
# Advanced features
|
|
169
|
+
with xp.MemoryContext(memory_threshold=0.8, auto_cleanup=True) as ctx:
|
|
170
|
+
# Monitor memory usage
|
|
171
|
+
mem_info = ctx.get_memory_info()
|
|
172
|
+
# 'total', 'free' and 'used' are in MiB (1024**2 bytes)
|
|
173
|
+
print(f"GPU Memory: {mem_info['used']:.2f} MiB")
|
|
174
|
+
|
|
175
|
+
# Aggressive cleanup when needed
|
|
176
|
+
if ctx.check_memory_pressure():
|
|
177
|
+
ctx.aggressive_cleanup()
|
|
178
|
+
|
|
179
|
+
# Emergency cleanup for critical situations
|
|
180
|
+
ctx.emergency_cleanup()
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
### MemoryContext Features
|
|
184
|
+
|
|
185
|
+
- **Automatic Cleanup**: Memory freed automatically when exiting context
|
|
186
|
+
- **Memory Monitoring**: Real-time tracking of GPU memory usage
|
|
187
|
+
- **Pressure Detection**: Automatic cleanup when memory usage is high
|
|
188
|
+
- **Aggressive Cleanup**: Force garbage collection and cache clearing
|
|
189
|
+
- **Emergency Cleanup**: Nuclear option for out-of-memory situations
|
|
190
|
+
- **Safe Cleanup**: Only garbage collection and memory-pool freeing; user objects are never modified
|
|
191
|
+
- **Memory History**: Keep history of memory usage over time
|
|
192
|
+
|
|
193
|
+
All memory figures are binary: `MB` / `MiB` = 1024**2 bytes and `GB` / `GiB` = 1024**3 bytes
|
|
194
|
+
(this also holds for `xp.array_size(shape, dtype, out_unit='MB')`, which accepts `'B'`, `'KB'`, `'MB'`, `'GB'`).
|
|
195
|
+
The previous device is always restored when the context exits.
|
|
196
|
+
|
|
197
|
+
## Documentation
|
|
198
|
+
|
|
199
|
+
For detailed documentation, including comprehensive API reference and advanced usage examples, see [docs/source/index.md](docs/source/index.md).
|
|
200
|
+
|
|
201
|
+
## License
|
|
202
|
+
|
|
203
|
+
See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
XuPy.egg-info/PKG-INFO
|
|
5
|
+
XuPy.egg-info/SOURCES.txt
|
|
6
|
+
XuPy.egg-info/dependency_links.txt
|
|
7
|
+
XuPy.egg-info/entry_points.txt
|
|
8
|
+
XuPy.egg-info/requires.txt
|
|
9
|
+
XuPy.egg-info/top_level.txt
|
|
10
|
+
test/test_arithmetic_compatibility.py
|
|
11
|
+
test/test_core.py
|
|
12
|
+
test/test_cpu_memory_context.py
|
|
13
|
+
test/test_gpu_no_sync.py
|
|
14
|
+
test/test_import_and_install.py
|
|
15
|
+
test/test_large_array_printing.py
|
|
16
|
+
test/test_ma_core.py
|
|
17
|
+
test/test_ma_extras.py
|
|
18
|
+
test/test_ma_parity_api_core.py
|
|
19
|
+
test/test_ma_parity_api_extras2.py
|
|
20
|
+
test/test_ma_parity_extras_repr.py
|
|
21
|
+
test/test_ma_parity_ops.py
|
|
22
|
+
test/test_ma_parity_reductions.py
|
|
23
|
+
test/test_ma_parity_structure.py
|
|
24
|
+
test/test_ma_perf_semantics.py
|
|
25
|
+
test/test_memory_context_and_hygiene.py
|
|
26
|
+
test/test_namespace.py
|
|
27
|
+
test/test_numpy2_compat.py
|
|
28
|
+
test/test_shims.py
|
|
29
|
+
test/test_switching.py
|
|
30
|
+
xupy/__init__.py
|
|
31
|
+
xupy/__init__.pyi
|
|
32
|
+
xupy/__version__.py
|
|
33
|
+
xupy/_core.py
|
|
34
|
+
xupy/_shims.py
|
|
35
|
+
xupy/install_cupy.py
|
|
36
|
+
xupy/py.typed
|
|
37
|
+
xupy/typings.py
|
|
38
|
+
xupy/ma/__init__.py
|
|
39
|
+
xupy/ma/_backend.py
|
|
40
|
+
xupy/ma/_domains.py
|
|
41
|
+
xupy/ma/_funcs.py
|
|
42
|
+
xupy/ma/_ops.py
|
|
43
|
+
xupy/ma/_printing.py
|
|
44
|
+
xupy/ma/_reductions.py
|
|
45
|
+
xupy/ma/_singletons.py
|
|
46
|
+
xupy/ma/core.py
|
|
47
|
+
xupy/ma/extras.py
|