XuPy 1.7.0__tar.gz → 1.7.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {xupy-1.7.0 → xupy-1.7.2}/PKG-INFO +1 -1
- {xupy-1.7.0 → xupy-1.7.2}/XuPy.egg-info/PKG-INFO +1 -1
- {xupy-1.7.0 → xupy-1.7.2}/XuPy.egg-info/SOURCES.txt +1 -0
- {xupy-1.7.0 → xupy-1.7.2}/pyproject.toml +1 -1
- {xupy-1.7.0 → xupy-1.7.2}/setup.py +6 -5
- {xupy-1.7.0 → xupy-1.7.2}/test/test_core.py +1 -1
- xupy-1.7.2/test/test_cpu_memory_context.py +123 -0
- xupy-1.7.2/xupy/__version__.py +1 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/_core.py +394 -42
- xupy-1.7.0/xupy/__version__.py +0 -1
- {xupy-1.7.0 → xupy-1.7.2}/LICENSE +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/README.md +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/XuPy.egg-info/dependency_links.txt +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/XuPy.egg-info/requires.txt +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/XuPy.egg-info/top_level.txt +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/setup.cfg +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/test/test_arithmetic_compatibility.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/test/test_large_array_printing.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/test/test_ma_core.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/test/test_ma_extras.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/__init__.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/_cupy_install/__check_availability__.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/_cupy_install/__init__.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/_cupy_install/__install_cupy__.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/ma/__init__.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/ma/core.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/ma/extras.py +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/py.typed +0 -0
- {xupy-1.7.0 → xupy-1.7.2}/xupy/typings.py +0 -0
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "XuPy"
|
|
7
|
-
version = "1.7.
|
|
7
|
+
version = "1.7.2"
|
|
8
8
|
description = "GPU Accelerated masked arrays with automatic handling of CPU and GPU arrays."
|
|
9
9
|
authors = [
|
|
10
10
|
{ name = "Pietro Ferraiuolo", email = "pietro.ferraiuolo@inaf.it" }
|
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
from setuptools import setup, find_packages
|
|
2
2
|
from setuptools.command.install import install
|
|
3
|
+
import tomllib
|
|
3
4
|
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
5
|
+
with open('pyproject.toml', 'rb') as f:
|
|
6
|
+
data = tomllib.load(f)
|
|
7
|
+
__version__ = data['project']['version']
|
|
7
8
|
|
|
8
9
|
class CustomInstall(install):
|
|
9
10
|
def run(self):
|
|
@@ -20,11 +21,11 @@ class CustomInstall(install):
|
|
|
20
21
|
|
|
21
22
|
setup(
|
|
22
23
|
name="XuPy",
|
|
23
|
-
version=
|
|
24
|
+
version=__version__,
|
|
24
25
|
description="GPU Accelerated masked arrays with automatic handling of CPU and GPU arrays.",
|
|
25
26
|
author="Pietro Ferraiuolo",
|
|
26
27
|
author_email="pietro.ferraiuolo@inaf.it",
|
|
27
|
-
packages=find_packages(),
|
|
28
|
+
packages=find_packages(),
|
|
28
29
|
install_requires=["numpy"],
|
|
29
30
|
python_requires=">=3.10",
|
|
30
31
|
cmdclass={
|
|
@@ -19,7 +19,7 @@ except ImportError:
|
|
|
19
19
|
HAS_CUPY = False
|
|
20
20
|
|
|
21
21
|
import xupy as xp
|
|
22
|
-
from xupy._core import NumpyContext, on_gpu
|
|
22
|
+
from xupy._core import NumpyContext, _CPUMemoryContext, on_gpu
|
|
23
23
|
|
|
24
24
|
# Skip all tests if CuPy is not available
|
|
25
25
|
pytestmark = pytest.mark.skipif(not HAS_CUPY, reason="CuPy not available")
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Tests for _CPUMemoryContext.
|
|
3
|
+
|
|
4
|
+
These tests do NOT require CuPy/CUDA and run purely on CPU (NumPy).
|
|
5
|
+
They validate that the CPU memory context manager exposes the same
|
|
6
|
+
interface as the GPU MemoryContext and behaves as a well-behaved
|
|
7
|
+
no-op context on CPU.
|
|
8
|
+
"""
|
|
9
|
+
import pytest
|
|
10
|
+
import numpy as np
|
|
11
|
+
|
|
12
|
+
import xupy as xp
|
|
13
|
+
from xupy._core import _CPUMemoryContext, on_gpu
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class TestCPUMemoryContext:
|
|
17
|
+
"""Tests for _CPUMemoryContext (always available, no GPU required)."""
|
|
18
|
+
|
|
19
|
+
def test_cpu_memory_context_is_importable(self):
|
|
20
|
+
"""_CPUMemoryContext can be imported from _core and is a class."""
|
|
21
|
+
assert callable(_CPUMemoryContext)
|
|
22
|
+
|
|
23
|
+
def test_cpu_memory_context_enter_returns_self(self):
|
|
24
|
+
"""__enter__ returns the context object itself."""
|
|
25
|
+
ctx = _CPUMemoryContext()
|
|
26
|
+
with ctx as c:
|
|
27
|
+
assert c is ctx
|
|
28
|
+
|
|
29
|
+
def test_cpu_memory_context_basic_usage(self, capsys):
|
|
30
|
+
"""Context manager completes without raising and prints a summary."""
|
|
31
|
+
with _CPUMemoryContext() as ctx:
|
|
32
|
+
arr = np.array([1, 2, 3])
|
|
33
|
+
assert arr is not None
|
|
34
|
+
captured = capsys.readouterr()
|
|
35
|
+
assert "Session completed" in captured.out
|
|
36
|
+
|
|
37
|
+
def test_cpu_memory_context_default_parameters(self):
|
|
38
|
+
"""Default parameters match the GPU MemoryContext API."""
|
|
39
|
+
ctx = _CPUMemoryContext()
|
|
40
|
+
assert ctx.device_id is None
|
|
41
|
+
assert ctx.auto_cleanup is True
|
|
42
|
+
assert ctx.memory_threshold == 0.9
|
|
43
|
+
assert ctx.monitor_interval == 1.0
|
|
44
|
+
|
|
45
|
+
def test_cpu_memory_context_custom_parameters(self):
|
|
46
|
+
"""Custom parameters are stored correctly."""
|
|
47
|
+
ctx = _CPUMemoryContext(
|
|
48
|
+
device_id=0,
|
|
49
|
+
auto_cleanup=False,
|
|
50
|
+
memory_threshold=0.8,
|
|
51
|
+
monitor_interval=2.0,
|
|
52
|
+
)
|
|
53
|
+
assert ctx.device_id == 0
|
|
54
|
+
assert ctx.auto_cleanup is False
|
|
55
|
+
assert ctx.memory_threshold == 0.8
|
|
56
|
+
assert ctx.monitor_interval == 2.0
|
|
57
|
+
|
|
58
|
+
def test_cpu_memory_context_get_memory_info_returns_dict(self):
|
|
59
|
+
"""get_memory_info always returns a dict with at least 'device'."""
|
|
60
|
+
ctx = _CPUMemoryContext()
|
|
61
|
+
info = ctx.get_memory_info()
|
|
62
|
+
assert isinstance(info, dict)
|
|
63
|
+
assert "device" in info
|
|
64
|
+
assert info["device"] == "cpu"
|
|
65
|
+
|
|
66
|
+
def test_cpu_memory_context_get_memory_info_keys(self):
|
|
67
|
+
"""get_memory_info returns expected keys when psutil is available."""
|
|
68
|
+
try:
|
|
69
|
+
import psutil # noqa: F401
|
|
70
|
+
psutil_available = True
|
|
71
|
+
except ImportError:
|
|
72
|
+
psutil_available = False
|
|
73
|
+
|
|
74
|
+
ctx = _CPUMemoryContext()
|
|
75
|
+
info = ctx.get_memory_info()
|
|
76
|
+
if psutil_available:
|
|
77
|
+
for key in ("total", "free", "used", "memory_percent"):
|
|
78
|
+
assert key in info, f"Missing key '{key}' in memory info"
|
|
79
|
+
assert info["total"] > 0
|
|
80
|
+
assert 0.0 <= info["memory_percent"] <= 1.0
|
|
81
|
+
else:
|
|
82
|
+
assert "error" in info
|
|
83
|
+
|
|
84
|
+
def test_cpu_memory_context_no_op_methods(self):
|
|
85
|
+
"""All GPU-specific methods are no-ops (do not raise)."""
|
|
86
|
+
ctx = _CPUMemoryContext()
|
|
87
|
+
ctx.track_object(object())
|
|
88
|
+
ctx.clear_cache()
|
|
89
|
+
ctx.aggressive_cleanup()
|
|
90
|
+
ctx.emergency_cleanup()
|
|
91
|
+
ctx.auto_cleanup_if_needed()
|
|
92
|
+
ctx.monitor_memory(duration=0.0)
|
|
93
|
+
ctx.force_memory_deallocation()
|
|
94
|
+
ctx.force_memory_pool_reset()
|
|
95
|
+
|
|
96
|
+
def test_cpu_memory_context_check_memory_pressure(self):
|
|
97
|
+
"""check_memory_pressure returns a bool."""
|
|
98
|
+
ctx = _CPUMemoryContext()
|
|
99
|
+
result = ctx.check_memory_pressure()
|
|
100
|
+
assert isinstance(result, bool)
|
|
101
|
+
|
|
102
|
+
def test_cpu_memory_context_repr(self):
|
|
103
|
+
"""__repr__ contains 'MemoryContext' and 'cpu'."""
|
|
104
|
+
ctx = _CPUMemoryContext()
|
|
105
|
+
r = repr(ctx)
|
|
106
|
+
assert "MemoryContext" in r
|
|
107
|
+
assert "cpu" in r
|
|
108
|
+
|
|
109
|
+
def test_cpu_memory_context_exception_propagation(self):
|
|
110
|
+
"""Exceptions raised inside the context propagate normally."""
|
|
111
|
+
with pytest.raises(ValueError):
|
|
112
|
+
with _CPUMemoryContext():
|
|
113
|
+
raise ValueError("test error")
|
|
114
|
+
|
|
115
|
+
def test_cpu_memory_context_exposed_when_on_cpu(self):
|
|
116
|
+
"""When running in CPU mode, xp.MemoryContext is _CPUMemoryContext."""
|
|
117
|
+
if not on_gpu:
|
|
118
|
+
assert hasattr(xp, "MemoryContext")
|
|
119
|
+
assert xp.MemoryContext is _CPUMemoryContext
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
if __name__ == "__main__":
|
|
123
|
+
pytest.main([__file__, "-v"])
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "1.7.2"
|
|
@@ -11,10 +11,12 @@ import time as _time
|
|
|
11
11
|
import builtins as _b
|
|
12
12
|
import sys as _sys
|
|
13
13
|
from . import typings as _t
|
|
14
|
+
from contextlib import contextmanager as _contextmanager
|
|
14
15
|
from ._cupy_install import __check_availability__ as __check__
|
|
15
16
|
|
|
16
17
|
_GPU = False
|
|
17
18
|
_GPU_AVAILABLE = False
|
|
19
|
+
_MULTIGPU = False
|
|
18
20
|
|
|
19
21
|
__check__.xupy_init()
|
|
20
22
|
__cuda_version__ = __check__.get_cuda_version()
|
|
@@ -25,8 +27,10 @@ try:
|
|
|
25
27
|
import cupy as _xp # type: ignore
|
|
26
28
|
|
|
27
29
|
_B2mb_ = 1024 * 1000 # using MB = 1,000,000 bytes
|
|
30
|
+
_Btgb_ = 1024 * 1000 * 1000 # using GB = 1,000,000,000 bytes
|
|
28
31
|
n_gpus = _xp.cuda.runtime.getDeviceCount()
|
|
29
32
|
if n_gpus > 1:
|
|
33
|
+
_MULTIGPU = True
|
|
30
34
|
gpus = {}
|
|
31
35
|
line1 = """
|
|
32
36
|
[XuPy] Multiple GPUs detected:
|
|
@@ -68,6 +72,7 @@ except Exception as err:
|
|
|
68
72
|
from numpy import * # type: ignore
|
|
69
73
|
|
|
70
74
|
on_gpu = _GPU
|
|
75
|
+
has_multi_gpu = _MULTIGPU
|
|
71
76
|
|
|
72
77
|
# Capture the set of names brought in by the wildcard import
|
|
73
78
|
_backend_names = frozenset(
|
|
@@ -77,6 +82,48 @@ _backend_names = frozenset(
|
|
|
77
82
|
|
|
78
83
|
_mode_names: set[str] = set()
|
|
79
84
|
|
|
85
|
+
def _array_size(
|
|
86
|
+
shape: tuple[int] | list[tuple[int]],
|
|
87
|
+
dtype: _t.DTypeLike = _np.float32,
|
|
88
|
+
out_unit: str = 'MB',
|
|
89
|
+
) -> int:
|
|
90
|
+
"""
|
|
91
|
+
Computes the expected allocated size on GPU of an array with size `shape`
|
|
92
|
+
and data type `dtype`.
|
|
93
|
+
|
|
94
|
+
Parameters
|
|
95
|
+
----------
|
|
96
|
+
shape : tuple[int] | list[tuple[int]]
|
|
97
|
+
The shape of the array. Can input multiple shapes as a list, and the
|
|
98
|
+
result will be the total size of all arrays combined.
|
|
99
|
+
dtype : DTypeLike, optional
|
|
100
|
+
The data type of the array elements (default: float32).
|
|
101
|
+
out_unit : str, optional
|
|
102
|
+
The unit for the output size. Can be 'MB' or 'GB' (default: 'MB').
|
|
103
|
+
|
|
104
|
+
Returns
|
|
105
|
+
-------
|
|
106
|
+
size : int
|
|
107
|
+
The size of the array in the specified unit.
|
|
108
|
+
|
|
109
|
+
Examples
|
|
110
|
+
--------
|
|
111
|
+
>>> import xupy as xp
|
|
112
|
+
>>> arr = xp.array([1, 2, 3])
|
|
113
|
+
>>> xp.array_size(arr)
|
|
114
|
+
12 # 3 elements * 4 bytes per int32
|
|
115
|
+
"""
|
|
116
|
+
norm = _B2mb_ if out_unit == 'MB' else _Btgb_
|
|
117
|
+
if isinstance(shape, tuple):
|
|
118
|
+
if isinstance(shape[0], int):
|
|
119
|
+
shape = [shape] # single shape case
|
|
120
|
+
size = []
|
|
121
|
+
for s in shape:
|
|
122
|
+
itemsize = _np.dtype(dtype).itemsize
|
|
123
|
+
num_elements = _np.prod(s)
|
|
124
|
+
size_bytes = num_elements * itemsize
|
|
125
|
+
size.append(int(size_bytes / norm))
|
|
126
|
+
return int(_np.sum(size))
|
|
80
127
|
|
|
81
128
|
# --- NUMPY Context manager ---
|
|
82
129
|
class NumpyContext:
|
|
@@ -114,16 +161,282 @@ class NumpyContext:
|
|
|
114
161
|
def __repr__(self) -> str:
|
|
115
162
|
"""String representation of the context manager."""
|
|
116
163
|
if _GPU:
|
|
164
|
+
|
|
117
165
|
return f"NumpyContext(original_device={self.original_device})"
|
|
118
166
|
else:
|
|
119
167
|
return "NumpyContext(no_gpu=True)"
|
|
120
168
|
|
|
121
169
|
|
|
170
|
+
# ---------------------------------------------------------------------------
|
|
171
|
+
# CPU Memory Context Manager (always available, no-op mock for CPU mode)
|
|
172
|
+
# ---------------------------------------------------------------------------
|
|
173
|
+
|
|
174
|
+
class _CPUMemoryContext:
|
|
175
|
+
"""CPU memory context manager — a no-op counterpart to the GPU _MemoryContext.
|
|
176
|
+
|
|
177
|
+
Provides the same interface as the GPU ``MemoryContext`` so that code written
|
|
178
|
+
against ``xp.MemoryContext`` runs transparently on CPU (NumPy) without any
|
|
179
|
+
changes. All GPU-specific operations (pool cleanup, device synchronisation,
|
|
180
|
+
etc.) are silently skipped.
|
|
181
|
+
|
|
182
|
+
Example
|
|
183
|
+
-------
|
|
184
|
+
>>> import xupy as xp # running in CPU mode
|
|
185
|
+
>>> with xp.MemoryContext() as ctx:
|
|
186
|
+
... arr = xp.array([1, 2, 3])
|
|
187
|
+
... print(ctx.get_memory_info())
|
|
188
|
+
"""
|
|
189
|
+
|
|
190
|
+
def __init__(
|
|
191
|
+
self,
|
|
192
|
+
device_id: _t.Optional[int] = None,
|
|
193
|
+
auto_cleanup: bool = True,
|
|
194
|
+
force_cleanup: bool = False,
|
|
195
|
+
print_report: bool = True,
|
|
196
|
+
memory_threshold: float = 0.9,
|
|
197
|
+
monitor_interval: float = 1.0,
|
|
198
|
+
):
|
|
199
|
+
"""
|
|
200
|
+
Parameters
|
|
201
|
+
----------
|
|
202
|
+
device_id : int, optional
|
|
203
|
+
Ignored on CPU; present for API compatibility with the GPU version.
|
|
204
|
+
auto_cleanup : bool, optional
|
|
205
|
+
Kept for API compatibility; no cleanup is performed on CPU.
|
|
206
|
+
memory_threshold : float, optional
|
|
207
|
+
Kept for API compatibility; no threshold enforcement on CPU.
|
|
208
|
+
monitor_interval : float, optional
|
|
209
|
+
Kept for API compatibility; no monitoring is performed on CPU.
|
|
210
|
+
"""
|
|
211
|
+
self.device_id = device_id
|
|
212
|
+
self.auto_cleanup = auto_cleanup
|
|
213
|
+
self.force_cleanup = force_cleanup
|
|
214
|
+
self.print_report = print_report
|
|
215
|
+
self.memory_threshold = memory_threshold
|
|
216
|
+
self.monitor_interval = monitor_interval
|
|
217
|
+
|
|
218
|
+
self._start_time: _t.Optional[float] = None
|
|
219
|
+
|
|
220
|
+
def __enter__(self):
|
|
221
|
+
"""Enter the CPU memory context."""
|
|
222
|
+
self._start_time = _time.time()
|
|
223
|
+
return self
|
|
224
|
+
|
|
225
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
226
|
+
"""Exit the CPU memory context (no-op cleanup)."""
|
|
227
|
+
if self._start_time is not None:
|
|
228
|
+
duration = _time.time() - self._start_time
|
|
229
|
+
if self.print_report:
|
|
230
|
+
print(f"[MemoryContext] Session completed in {duration:.2f}s (CPU mode)")
|
|
231
|
+
|
|
232
|
+
def track_object(self, obj):
|
|
233
|
+
"""No-op: object tracking is not needed on CPU."""
|
|
234
|
+
pass
|
|
235
|
+
|
|
236
|
+
def clear_cache(self):
|
|
237
|
+
"""No-op: no GPU memory pool to clear on CPU."""
|
|
238
|
+
pass
|
|
239
|
+
|
|
240
|
+
def aggressive_cleanup(self):
|
|
241
|
+
"""No-op: no GPU memory to aggressively free on CPU."""
|
|
242
|
+
pass
|
|
243
|
+
|
|
244
|
+
def emergency_cleanup(self):
|
|
245
|
+
"""No-op: no GPU memory emergency cleanup needed on CPU."""
|
|
246
|
+
pass
|
|
247
|
+
|
|
248
|
+
def get_memory_info(self) -> dict:
|
|
249
|
+
"""Return basic CPU/RAM memory information where available.
|
|
250
|
+
|
|
251
|
+
Uses ``psutil`` when installed; otherwise returns a minimal dict.
|
|
252
|
+
"""
|
|
253
|
+
info: dict = {"device": "cpu"}
|
|
254
|
+
try:
|
|
255
|
+
import psutil # type: ignore
|
|
256
|
+
vm = psutil.virtual_memory()
|
|
257
|
+
info.update(
|
|
258
|
+
{
|
|
259
|
+
"total": vm.total,
|
|
260
|
+
"free": vm.available,
|
|
261
|
+
"used": vm.used,
|
|
262
|
+
"memory_percent": vm.percent / 100.0,
|
|
263
|
+
}
|
|
264
|
+
)
|
|
265
|
+
except ImportError:
|
|
266
|
+
info["error"] = "psutil not installed; install it for detailed CPU memory info"
|
|
267
|
+
return info
|
|
268
|
+
|
|
269
|
+
def check_memory_pressure(self) -> bool:
|
|
270
|
+
"""Check if RAM usage is above the threshold (requires psutil)."""
|
|
271
|
+
mem_info = self.get_memory_info()
|
|
272
|
+
if "memory_percent" in mem_info:
|
|
273
|
+
return mem_info["memory_percent"] > self.memory_threshold
|
|
274
|
+
return False
|
|
275
|
+
|
|
276
|
+
def auto_cleanup_if_needed(self):
|
|
277
|
+
"""No-op: no GPU pressure-based cleanup on CPU."""
|
|
278
|
+
pass
|
|
279
|
+
|
|
280
|
+
def monitor_memory(self, duration: float = 10.0):
|
|
281
|
+
"""No-op: memory monitoring is not performed in CPU mode."""
|
|
282
|
+
pass
|
|
283
|
+
|
|
284
|
+
def force_memory_deallocation(self):
|
|
285
|
+
"""No-op: forced GPU memory deallocation is not applicable on CPU."""
|
|
286
|
+
pass
|
|
287
|
+
|
|
288
|
+
def force_memory_pool_reset(self):
|
|
289
|
+
"""No-op: GPU memory pool reset is not applicable on CPU."""
|
|
290
|
+
pass
|
|
291
|
+
|
|
292
|
+
def __repr__(self) -> str:
|
|
293
|
+
"""String representation of the CPU memory context."""
|
|
294
|
+
mem_info = self.get_memory_info()
|
|
295
|
+
if "used" in mem_info:
|
|
296
|
+
used_mb = mem_info["used"] / (1024 * 1000)
|
|
297
|
+
total_mb = mem_info["total"] / (1024 * 1000)
|
|
298
|
+
percent = mem_info["memory_percent"] * 100
|
|
299
|
+
return f"MemoryContext(device=cpu, memory={used_mb:.2f}/{total_mb:.2f} MB ({percent:.1f}%))"
|
|
300
|
+
return "MemoryContext(device=cpu)"
|
|
301
|
+
|
|
302
|
+
|
|
122
303
|
# ---------------------------------------------------------------------------
|
|
123
304
|
# GPU-only definitions (only created when CuPy was successfully loaded)
|
|
124
305
|
# ---------------------------------------------------------------------------
|
|
125
306
|
|
|
126
307
|
if _GPU_AVAILABLE:
|
|
308
|
+
|
|
309
|
+
def _allocate_with_fallback_device(
|
|
310
|
+
array: _t.ArrayLike,
|
|
311
|
+
dtype: _t.Optional[_t.DTypeLike] = None,
|
|
312
|
+
preferred_device: _t.Optional[int] = None,
|
|
313
|
+
safety_factor: float = 1.10,
|
|
314
|
+
reserve_mb: int = 128,
|
|
315
|
+
) -> tuple[_t.NDArray[_t.Any], int]:
|
|
316
|
+
"""
|
|
317
|
+
Allocate an array on the current/preferred GPU if enough memory is available.
|
|
318
|
+
If not, try other GPUs when multi-GPU is available.
|
|
319
|
+
|
|
320
|
+
Parameters
|
|
321
|
+
----------
|
|
322
|
+
array : ArrayLike
|
|
323
|
+
Input data to allocate on GPU.
|
|
324
|
+
dtype : DTypeLike, optional
|
|
325
|
+
Target dtype for allocation. If None, uses input dtype when available.
|
|
326
|
+
preferred_device : int, optional
|
|
327
|
+
Device to try first. If None, uses current CUDA device.
|
|
328
|
+
safety_factor : float, optional
|
|
329
|
+
Extra multiplicative margin on top of estimated size.
|
|
330
|
+
reserve_mb : int, optional
|
|
331
|
+
Additional fixed memory cushion to reduce OOM risk.
|
|
332
|
+
|
|
333
|
+
Returns
|
|
334
|
+
-------
|
|
335
|
+
gpu_array : NDArray
|
|
336
|
+
Allocated array on the selected GPU.
|
|
337
|
+
device_id : int
|
|
338
|
+
GPU id where the allocation was performed.
|
|
339
|
+
|
|
340
|
+
Raises
|
|
341
|
+
------
|
|
342
|
+
RuntimeError
|
|
343
|
+
If GPU backend is not available.
|
|
344
|
+
MemoryError
|
|
345
|
+
If no device has enough free memory.
|
|
346
|
+
"""
|
|
347
|
+
if not _GPU_AVAILABLE:
|
|
348
|
+
raise RuntimeError("[XuPy] GPU backend is not available.")
|
|
349
|
+
|
|
350
|
+
# Keep behavior explicit: this helper is for GPU allocation.
|
|
351
|
+
if not on_gpu:
|
|
352
|
+
raise RuntimeError("[XuPy] XuPy is in CPU mode. Call use_gpu() first.")
|
|
353
|
+
|
|
354
|
+
# Resolve dtype used for memory estimate and allocation.
|
|
355
|
+
target_dtype = dtype if dtype is not None else getattr(array, "dtype", _xp.float32)
|
|
356
|
+
|
|
357
|
+
# Estimate required memory in MB using existing XuPy helper.
|
|
358
|
+
shape = getattr(array, "shape", None)
|
|
359
|
+
if shape is None:
|
|
360
|
+
arr_np = _np.asarray(array, dtype=target_dtype)
|
|
361
|
+
shape = arr_np.shape
|
|
362
|
+
required_mb = _array_size(tuple(shape), dtype=target_dtype, out_unit="MB")
|
|
363
|
+
required_mb = int(required_mb * safety_factor) + int(reserve_mb)
|
|
364
|
+
|
|
365
|
+
current_device = _get_device()
|
|
366
|
+
first_device = current_device if preferred_device is None else int(preferred_device)
|
|
367
|
+
|
|
368
|
+
# Device probe order: preferred/current first, then the others.
|
|
369
|
+
device_order = [first_device]
|
|
370
|
+
if has_multi_gpu:
|
|
371
|
+
device_order.extend([d for d in range(n_gpus) if d != first_device])
|
|
372
|
+
|
|
373
|
+
for dev_id in device_order:
|
|
374
|
+
try:
|
|
375
|
+
with _on_device(dev_id):
|
|
376
|
+
free_b, _ = _xp.cuda.runtime.memGetInfo()
|
|
377
|
+
free_mb = int(free_b / _B2mb_)
|
|
378
|
+
|
|
379
|
+
if free_mb >= required_mb:
|
|
380
|
+
with _on_device(dev_id):
|
|
381
|
+
gpu_array = _xp.asarray(array, dtype=target_dtype)
|
|
382
|
+
return gpu_array, dev_id
|
|
383
|
+
except Exception:
|
|
384
|
+
# Skip unavailable/busy devices and continue probing.
|
|
385
|
+
continue
|
|
386
|
+
|
|
387
|
+
raise MemoryError(
|
|
388
|
+
f"[XuPy] Cannot allocate array (~{required_mb} MB incl. margin) "
|
|
389
|
+
f"on probed devices {device_order}."
|
|
390
|
+
)
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
@_contextmanager
|
|
394
|
+
def _on_device(device_id: int):
|
|
395
|
+
"""
|
|
396
|
+
Context manager to temporarily set the CUDA device for computations (cupy).
|
|
397
|
+
|
|
398
|
+
Parameters
|
|
399
|
+
----------
|
|
400
|
+
device_id : int
|
|
401
|
+
The ID of the CUDA device to set as default within the context.
|
|
402
|
+
|
|
403
|
+
If ``-1`` is passed, it will switch to CPU mode within the context
|
|
404
|
+
and restore GPU mode on exit.
|
|
405
|
+
|
|
406
|
+
Raises
|
|
407
|
+
------
|
|
408
|
+
RuntimeError : If the device cannot be set or if the device is already
|
|
409
|
+
the current device.
|
|
410
|
+
|
|
411
|
+
Examples
|
|
412
|
+
--------
|
|
413
|
+
.. code-block:: python
|
|
414
|
+
with xp.on_device(0):
|
|
415
|
+
# computations here will use device 0
|
|
416
|
+
array = xp.array([1,2,3]) # array allocated on GPU 0
|
|
417
|
+
with xp.on_device(1):
|
|
418
|
+
# computations here will use device 1
|
|
419
|
+
array = xp.array([4,5,6]) # array allocated on GPU 1
|
|
420
|
+
with xp.on_device(-1):
|
|
421
|
+
# computations here will use CPU
|
|
422
|
+
array = xp.array([7,8,9]) # array allocated on CPU
|
|
423
|
+
"""
|
|
424
|
+
original_device = _xp.cuda.runtime.getDevice()
|
|
425
|
+
try:
|
|
426
|
+
if device_id == -1:
|
|
427
|
+
use_cpu()
|
|
428
|
+
else:
|
|
429
|
+
_set_device(device_id)
|
|
430
|
+
yield
|
|
431
|
+
finally:
|
|
432
|
+
# Restore original device
|
|
433
|
+
try:
|
|
434
|
+
if device_id == -1:
|
|
435
|
+
use_gpu()
|
|
436
|
+
_xp.cuda.runtime.setDevice(original_device)
|
|
437
|
+
except Exception as e:
|
|
438
|
+
print(f"Warning: Could not restore original device: {e}")
|
|
439
|
+
|
|
127
440
|
|
|
128
441
|
def _set_device(device_id: int) -> None:
|
|
129
442
|
"""
|
|
@@ -157,6 +470,28 @@ if _GPU_AVAILABLE:
|
|
|
157
470
|
warnings.warn(
|
|
158
471
|
f"[XuPy] Device {device_id} is already the current device", UserWarning
|
|
159
472
|
)
|
|
473
|
+
|
|
474
|
+
def _get_device() -> int:
|
|
475
|
+
"""
|
|
476
|
+
Get the current CUDA device ID.
|
|
477
|
+
|
|
478
|
+
Returns
|
|
479
|
+
-------
|
|
480
|
+
int
|
|
481
|
+
The ID of the current CUDA device.
|
|
482
|
+
|
|
483
|
+
Raises
|
|
484
|
+
------
|
|
485
|
+
RuntimeError : If the GPU backend is not available.
|
|
486
|
+
|
|
487
|
+
Examples
|
|
488
|
+
--------
|
|
489
|
+
>>> current_device = xp.get_device()
|
|
490
|
+
>>> print(f"Current device ID: {current_device}")
|
|
491
|
+
"""
|
|
492
|
+
if not _GPU_AVAILABLE:
|
|
493
|
+
return -1 # Indicate no GPU available
|
|
494
|
+
return int(_xp.cuda.runtime.getDevice())
|
|
160
495
|
|
|
161
496
|
# --- GPU Memory Management Context Manager ---
|
|
162
497
|
class _MemoryContext:
|
|
@@ -176,6 +511,8 @@ if _GPU_AVAILABLE:
|
|
|
176
511
|
self,
|
|
177
512
|
device_id: _t.Optional[int] = None,
|
|
178
513
|
auto_cleanup: bool = True,
|
|
514
|
+
force_cleanup: bool = False,
|
|
515
|
+
print_report: bool = True,
|
|
179
516
|
memory_threshold: float = 0.9,
|
|
180
517
|
monitor_interval: float = 1.0,
|
|
181
518
|
):
|
|
@@ -188,15 +525,21 @@ if _GPU_AVAILABLE:
|
|
|
188
525
|
GPU device ID to manage. If None, uses current device.
|
|
189
526
|
auto_cleanup : bool, optional
|
|
190
527
|
Whether to automatically cleanup memory on exit (default: True).
|
|
528
|
+
force_cleanup : bool, optional
|
|
529
|
+
Whether to force memory cleanup on exit (default: False).
|
|
530
|
+
print_report : bool, optional
|
|
531
|
+
Whether to print memory usage report on exit (default: True).
|
|
191
532
|
memory_threshold : float, optional
|
|
192
533
|
Memory usage threshold (0-1) for automatic cleanup (default: 0.9).
|
|
193
534
|
monitor_interval : float, optional
|
|
194
|
-
Interval in
|
|
535
|
+
Interval in milliseconds for memory monitoring (default: 1.0).
|
|
195
536
|
"""
|
|
196
537
|
self.device_id = device_id
|
|
197
538
|
self.auto_cleanup = auto_cleanup
|
|
539
|
+
self.force_cleanup = force_cleanup
|
|
198
540
|
self.memory_threshold = memory_threshold
|
|
199
|
-
self.monitor_interval = monitor_interval
|
|
541
|
+
self.monitor_interval = monitor_interval/1000
|
|
542
|
+
self._print_report = print_report
|
|
200
543
|
|
|
201
544
|
self._device_ctx = None
|
|
202
545
|
self._original_device = None
|
|
@@ -237,33 +580,34 @@ if _GPU_AVAILABLE:
|
|
|
237
580
|
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
238
581
|
"""Exit the memory context with cleanup."""
|
|
239
582
|
try:
|
|
240
|
-
if self.auto_cleanup:
|
|
583
|
+
if self.auto_cleanup or self.force_cleanup:
|
|
241
584
|
self.aggressive_cleanup()
|
|
242
585
|
|
|
243
|
-
|
|
244
|
-
|
|
586
|
+
# Cleanup tracked GPU objects
|
|
587
|
+
self._cleanup_gpu_objects()
|
|
245
588
|
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
589
|
+
# Restore original device
|
|
590
|
+
if _GPU and self._device_ctx is not None:
|
|
591
|
+
try:
|
|
592
|
+
self._device_ctx.__exit__(exc_type, exc_val, exc_tb)
|
|
593
|
+
except Exception as e:
|
|
594
|
+
print(f"Warning: Error restoring device context: {e}")
|
|
252
595
|
|
|
253
596
|
# Final memory report
|
|
254
|
-
if self.
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
f"[MemoryContext] Memory delta: {memory_delta / (_B2mb_):.2f} MB"
|
|
262
|
-
)
|
|
263
|
-
if self._cleanup_count > 0:
|
|
597
|
+
if self._print_report:
|
|
598
|
+
if self._start_time:
|
|
599
|
+
duration = _time.time() - self._start_time
|
|
600
|
+
final_mem = self.get_memory_info()
|
|
601
|
+
if "used" in final_mem:
|
|
602
|
+
memory_delta = final_mem["used"] - self._initial_memory
|
|
603
|
+
print(f"[MemoryContext] Session completed in {duration:.2f}s")
|
|
264
604
|
print(
|
|
265
|
-
f"[MemoryContext]
|
|
605
|
+
f"[MemoryContext] Memory delta: {memory_delta / (_B2mb_):.2f} MB"
|
|
266
606
|
)
|
|
607
|
+
if self._cleanup_count > 0:
|
|
608
|
+
print(
|
|
609
|
+
f"[MemoryContext] Cleanup operations: {self._cleanup_count}"
|
|
610
|
+
)
|
|
267
611
|
|
|
268
612
|
except Exception as e:
|
|
269
613
|
print(f"Warning: Error during memory context cleanup: {e}")
|
|
@@ -322,7 +666,8 @@ if _GPU_AVAILABLE:
|
|
|
322
666
|
if not _GPU:
|
|
323
667
|
return
|
|
324
668
|
|
|
325
|
-
|
|
669
|
+
if self._print_report:
|
|
670
|
+
print("[MemoryContext] Performing aggressive memory cleanup...")
|
|
326
671
|
self._cleanup_count += 1
|
|
327
672
|
|
|
328
673
|
# Force garbage collection
|
|
@@ -385,17 +730,18 @@ if _GPU_AVAILABLE:
|
|
|
385
730
|
except Exception as e:
|
|
386
731
|
print(f"Warning: Could not get CUDA memory info: {e}")
|
|
387
732
|
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
733
|
+
if self.force_cleanup:
|
|
734
|
+
# As a last resort, try memory pool reset
|
|
735
|
+
try:
|
|
736
|
+
self.force_memory_pool_reset()
|
|
737
|
+
except Exception as e:
|
|
738
|
+
print(f"Warning: Memory pool reset failed: {e}")
|
|
393
739
|
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
740
|
+
# Final attempt: force memory deallocation
|
|
741
|
+
try:
|
|
742
|
+
self.force_memory_deallocation()
|
|
743
|
+
except Exception as e:
|
|
744
|
+
print(f"Warning: Forced memory deallocation failed: {e}")
|
|
399
745
|
|
|
400
746
|
def emergency_cleanup(self):
|
|
401
747
|
"""Emergency cleanup for out-of-memory situations."""
|
|
@@ -495,9 +841,9 @@ if _GPU_AVAILABLE:
|
|
|
495
841
|
|
|
496
842
|
info = {
|
|
497
843
|
"device": int(device_to_query),
|
|
498
|
-
"total": int(total),
|
|
499
|
-
"free": int(free),
|
|
500
|
-
"used": int(used),
|
|
844
|
+
"total": int(total / _B2mb_),
|
|
845
|
+
"free": int(free / _B2mb_),
|
|
846
|
+
"used": int(used / _B2mb_),
|
|
501
847
|
"memory_percent": memory_percent,
|
|
502
848
|
"pool_used": pool_used,
|
|
503
849
|
"pool_capacity": pool_capacity,
|
|
@@ -612,10 +958,8 @@ if _GPU_AVAILABLE:
|
|
|
612
958
|
# Check memory after
|
|
613
959
|
free_after, _ = _xp.cuda.runtime.memGetInfo()
|
|
614
960
|
freed = free_after - free_before
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
# )
|
|
618
|
-
print(f"[MemoryContext] Memory freed: {freed/(_B2mb_):.2f} MB")
|
|
961
|
+
if self._print_report:
|
|
962
|
+
print(f"[MemoryContext] Memory freed: {freed/(_B2mb_):.2f} MB")
|
|
619
963
|
|
|
620
964
|
except Exception as e:
|
|
621
965
|
print(f"Warning: Could not force memory deallocation: {e}")
|
|
@@ -625,7 +969,8 @@ if _GPU_AVAILABLE:
|
|
|
625
969
|
if not _GPU:
|
|
626
970
|
return
|
|
627
971
|
|
|
628
|
-
|
|
972
|
+
if self._print_report:
|
|
973
|
+
print("[MemoryContext] Performing memory pool reset...")
|
|
629
974
|
try:
|
|
630
975
|
# Get current pool
|
|
631
976
|
old_pool = _xp.get_default_memory_pool()
|
|
@@ -647,7 +992,8 @@ if _GPU_AVAILABLE:
|
|
|
647
992
|
# Synchronize to ensure operations are complete
|
|
648
993
|
_xp.cuda.runtime.deviceSynchronize()
|
|
649
994
|
|
|
650
|
-
|
|
995
|
+
if self._print_report:
|
|
996
|
+
print("[MemoryContext] Memory pool reset completed")
|
|
651
997
|
|
|
652
998
|
except Exception as e:
|
|
653
999
|
print(f"Warning: Could not reset memory pool: {e}")
|
|
@@ -675,6 +1021,7 @@ if _GPU_AVAILABLE:
|
|
|
675
1021
|
|
|
676
1022
|
def _gpu_definitions() -> dict:
|
|
677
1023
|
"""Return dict of names exposed only in GPU mode."""
|
|
1024
|
+
|
|
678
1025
|
def asmarray(array: _t.NDArray[_t.Any]) -> _t.MaskedArray:
|
|
679
1026
|
"""
|
|
680
1027
|
Converts an object to a masked array on the GPU.
|
|
@@ -699,6 +1046,8 @@ def _gpu_definitions() -> dict:
|
|
|
699
1046
|
'npma': _np.ma,
|
|
700
1047
|
'asmarray': asmarray,
|
|
701
1048
|
'set_device': _set_device,
|
|
1049
|
+
'array_size': _array_size,
|
|
1050
|
+
'on_device': _on_device,
|
|
702
1051
|
'MemoryContext': _MemoryContext,
|
|
703
1052
|
}
|
|
704
1053
|
|
|
@@ -726,8 +1075,11 @@ def _cpu_definitions() -> dict:
|
|
|
726
1075
|
'double': _np.float64,
|
|
727
1076
|
'cfloat': _np.complex128,
|
|
728
1077
|
'cdouble': _np.complex128,
|
|
1078
|
+
'array_size': _array_size,
|
|
729
1079
|
'asnumpy': asnumpy,
|
|
730
1080
|
'asmarray': asmarray,
|
|
1081
|
+
'MemoryContext': _CPUMemoryContext,
|
|
1082
|
+
'on_device': lambda device_id: None, # No-op on CPU
|
|
731
1083
|
}
|
|
732
1084
|
|
|
733
1085
|
|
xupy-1.7.0/xupy/__version__.py
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
__version__ = "1.7.0"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|