XuPy 1.7.0__tar.gz → 1.7.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. {xupy-1.7.0 → xupy-1.7.1}/PKG-INFO +1 -1
  2. {xupy-1.7.0 → xupy-1.7.1}/XuPy.egg-info/PKG-INFO +1 -1
  3. {xupy-1.7.0 → xupy-1.7.1}/XuPy.egg-info/SOURCES.txt +1 -0
  4. {xupy-1.7.0 → xupy-1.7.1}/pyproject.toml +1 -1
  5. {xupy-1.7.0 → xupy-1.7.1}/test/test_core.py +1 -1
  6. xupy-1.7.1/test/test_cpu_memory_context.py +123 -0
  7. xupy-1.7.1/xupy/__version__.py +1 -0
  8. {xupy-1.7.0 → xupy-1.7.1}/xupy/_core.py +394 -42
  9. xupy-1.7.0/xupy/__version__.py +0 -1
  10. {xupy-1.7.0 → xupy-1.7.1}/LICENSE +0 -0
  11. {xupy-1.7.0 → xupy-1.7.1}/README.md +0 -0
  12. {xupy-1.7.0 → xupy-1.7.1}/XuPy.egg-info/dependency_links.txt +0 -0
  13. {xupy-1.7.0 → xupy-1.7.1}/XuPy.egg-info/requires.txt +0 -0
  14. {xupy-1.7.0 → xupy-1.7.1}/XuPy.egg-info/top_level.txt +0 -0
  15. {xupy-1.7.0 → xupy-1.7.1}/setup.cfg +0 -0
  16. {xupy-1.7.0 → xupy-1.7.1}/setup.py +0 -0
  17. {xupy-1.7.0 → xupy-1.7.1}/test/test_arithmetic_compatibility.py +0 -0
  18. {xupy-1.7.0 → xupy-1.7.1}/test/test_large_array_printing.py +0 -0
  19. {xupy-1.7.0 → xupy-1.7.1}/test/test_ma_core.py +0 -0
  20. {xupy-1.7.0 → xupy-1.7.1}/test/test_ma_extras.py +0 -0
  21. {xupy-1.7.0 → xupy-1.7.1}/xupy/__init__.py +0 -0
  22. {xupy-1.7.0 → xupy-1.7.1}/xupy/_cupy_install/__check_availability__.py +0 -0
  23. {xupy-1.7.0 → xupy-1.7.1}/xupy/_cupy_install/__init__.py +0 -0
  24. {xupy-1.7.0 → xupy-1.7.1}/xupy/_cupy_install/__install_cupy__.py +0 -0
  25. {xupy-1.7.0 → xupy-1.7.1}/xupy/ma/__init__.py +0 -0
  26. {xupy-1.7.0 → xupy-1.7.1}/xupy/ma/core.py +0 -0
  27. {xupy-1.7.0 → xupy-1.7.1}/xupy/ma/extras.py +0 -0
  28. {xupy-1.7.0 → xupy-1.7.1}/xupy/py.typed +0 -0
  29. {xupy-1.7.0 → xupy-1.7.1}/xupy/typings.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: XuPy
3
- Version: 1.7.0
3
+ Version: 1.7.1
4
4
  Summary: GPU Accelerated masked arrays with automatic handling of CPU and GPU arrays.
5
5
  Author: Pietro Ferraiuolo
6
6
  Author-email: Pietro Ferraiuolo <pietro.ferraiuolo@inaf.it>
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: XuPy
3
- Version: 1.7.0
3
+ Version: 1.7.1
4
4
  Summary: GPU Accelerated masked arrays with automatic handling of CPU and GPU arrays.
5
5
  Author: Pietro Ferraiuolo
6
6
  Author-email: Pietro Ferraiuolo <pietro.ferraiuolo@inaf.it>
@@ -9,6 +9,7 @@ XuPy.egg-info/requires.txt
9
9
  XuPy.egg-info/top_level.txt
10
10
  test/test_arithmetic_compatibility.py
11
11
  test/test_core.py
12
+ test/test_cpu_memory_context.py
12
13
  test/test_large_array_printing.py
13
14
  test/test_ma_core.py
14
15
  test/test_ma_extras.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "XuPy"
7
- version = "1.7.0"
7
+ version = "1.7.1"
8
8
  description = "GPU Accelerated masked arrays with automatic handling of CPU and GPU arrays."
9
9
  authors = [
10
10
  { name = "Pietro Ferraiuolo", email = "pietro.ferraiuolo@inaf.it" }
@@ -19,7 +19,7 @@ except ImportError:
19
19
  HAS_CUPY = False
20
20
 
21
21
  import xupy as xp
22
- from xupy._core import NumpyContext, on_gpu
22
+ from xupy._core import NumpyContext, _CPUMemoryContext, on_gpu
23
23
 
24
24
  # Skip all tests if CuPy is not available
25
25
  pytestmark = pytest.mark.skipif(not HAS_CUPY, reason="CuPy not available")
@@ -0,0 +1,123 @@
1
+ """
2
+ Tests for _CPUMemoryContext.
3
+
4
+ These tests do NOT require CuPy/CUDA and run purely on CPU (NumPy).
5
+ They validate that the CPU memory context manager exposes the same
6
+ interface as the GPU MemoryContext and behaves as a well-behaved
7
+ no-op context on CPU.
8
+ """
9
+ import pytest
10
+ import numpy as np
11
+
12
+ import xupy as xp
13
+ from xupy._core import _CPUMemoryContext, on_gpu
14
+
15
+
16
+ class TestCPUMemoryContext:
17
+ """Tests for _CPUMemoryContext (always available, no GPU required)."""
18
+
19
+ def test_cpu_memory_context_is_importable(self):
20
+ """_CPUMemoryContext can be imported from _core and is a class."""
21
+ assert callable(_CPUMemoryContext)
22
+
23
+ def test_cpu_memory_context_enter_returns_self(self):
24
+ """__enter__ returns the context object itself."""
25
+ ctx = _CPUMemoryContext()
26
+ with ctx as c:
27
+ assert c is ctx
28
+
29
+ def test_cpu_memory_context_basic_usage(self, capsys):
30
+ """Context manager completes without raising and prints a summary."""
31
+ with _CPUMemoryContext() as ctx:
32
+ arr = np.array([1, 2, 3])
33
+ assert arr is not None
34
+ captured = capsys.readouterr()
35
+ assert "Session completed" in captured.out
36
+
37
+ def test_cpu_memory_context_default_parameters(self):
38
+ """Default parameters match the GPU MemoryContext API."""
39
+ ctx = _CPUMemoryContext()
40
+ assert ctx.device_id is None
41
+ assert ctx.auto_cleanup is True
42
+ assert ctx.memory_threshold == 0.9
43
+ assert ctx.monitor_interval == 1.0
44
+
45
+ def test_cpu_memory_context_custom_parameters(self):
46
+ """Custom parameters are stored correctly."""
47
+ ctx = _CPUMemoryContext(
48
+ device_id=0,
49
+ auto_cleanup=False,
50
+ memory_threshold=0.8,
51
+ monitor_interval=2.0,
52
+ )
53
+ assert ctx.device_id == 0
54
+ assert ctx.auto_cleanup is False
55
+ assert ctx.memory_threshold == 0.8
56
+ assert ctx.monitor_interval == 2.0
57
+
58
+ def test_cpu_memory_context_get_memory_info_returns_dict(self):
59
+ """get_memory_info always returns a dict with at least 'device'."""
60
+ ctx = _CPUMemoryContext()
61
+ info = ctx.get_memory_info()
62
+ assert isinstance(info, dict)
63
+ assert "device" in info
64
+ assert info["device"] == "cpu"
65
+
66
+ def test_cpu_memory_context_get_memory_info_keys(self):
67
+ """get_memory_info returns expected keys when psutil is available."""
68
+ try:
69
+ import psutil # noqa: F401
70
+ psutil_available = True
71
+ except ImportError:
72
+ psutil_available = False
73
+
74
+ ctx = _CPUMemoryContext()
75
+ info = ctx.get_memory_info()
76
+ if psutil_available:
77
+ for key in ("total", "free", "used", "memory_percent"):
78
+ assert key in info, f"Missing key '{key}' in memory info"
79
+ assert info["total"] > 0
80
+ assert 0.0 <= info["memory_percent"] <= 1.0
81
+ else:
82
+ assert "error" in info
83
+
84
+ def test_cpu_memory_context_no_op_methods(self):
85
+ """All GPU-specific methods are no-ops (do not raise)."""
86
+ ctx = _CPUMemoryContext()
87
+ ctx.track_object(object())
88
+ ctx.clear_cache()
89
+ ctx.aggressive_cleanup()
90
+ ctx.emergency_cleanup()
91
+ ctx.auto_cleanup_if_needed()
92
+ ctx.monitor_memory(duration=0.0)
93
+ ctx.force_memory_deallocation()
94
+ ctx.force_memory_pool_reset()
95
+
96
+ def test_cpu_memory_context_check_memory_pressure(self):
97
+ """check_memory_pressure returns a bool."""
98
+ ctx = _CPUMemoryContext()
99
+ result = ctx.check_memory_pressure()
100
+ assert isinstance(result, bool)
101
+
102
+ def test_cpu_memory_context_repr(self):
103
+ """__repr__ contains 'MemoryContext' and 'cpu'."""
104
+ ctx = _CPUMemoryContext()
105
+ r = repr(ctx)
106
+ assert "MemoryContext" in r
107
+ assert "cpu" in r
108
+
109
+ def test_cpu_memory_context_exception_propagation(self):
110
+ """Exceptions raised inside the context propagate normally."""
111
+ with pytest.raises(ValueError):
112
+ with _CPUMemoryContext():
113
+ raise ValueError("test error")
114
+
115
+ def test_cpu_memory_context_exposed_when_on_cpu(self):
116
+ """When running in CPU mode, xp.MemoryContext is _CPUMemoryContext."""
117
+ if not on_gpu:
118
+ assert hasattr(xp, "MemoryContext")
119
+ assert xp.MemoryContext is _CPUMemoryContext
120
+
121
+
122
+ if __name__ == "__main__":
123
+ pytest.main([__file__, "-v"])
@@ -0,0 +1 @@
1
+ __version__ = "1.7.1"
@@ -11,10 +11,12 @@ import time as _time
11
11
  import builtins as _b
12
12
  import sys as _sys
13
13
  from . import typings as _t
14
+ from contextlib import contextmanager as _contextmanager
14
15
  from ._cupy_install import __check_availability__ as __check__
15
16
 
16
17
  _GPU = False
17
18
  _GPU_AVAILABLE = False
19
+ _MULTIGPU = False
18
20
 
19
21
  __check__.xupy_init()
20
22
  __cuda_version__ = __check__.get_cuda_version()
@@ -25,8 +27,10 @@ try:
25
27
  import cupy as _xp # type: ignore
26
28
 
27
29
  _B2mb_ = 1024 * 1000 # using MB = 1,000,000 bytes
30
+ _Btgb_ = 1024 * 1000 * 1000 # using GB = 1,000,000,000 bytes
28
31
  n_gpus = _xp.cuda.runtime.getDeviceCount()
29
32
  if n_gpus > 1:
33
+ _MULTIGPU = True
30
34
  gpus = {}
31
35
  line1 = """
32
36
  [XuPy] Multiple GPUs detected:
@@ -68,6 +72,7 @@ except Exception as err:
68
72
  from numpy import * # type: ignore
69
73
 
70
74
  on_gpu = _GPU
75
+ has_multi_gpu = _MULTIGPU
71
76
 
72
77
  # Capture the set of names brought in by the wildcard import
73
78
  _backend_names = frozenset(
@@ -77,6 +82,48 @@ _backend_names = frozenset(
77
82
 
78
83
  _mode_names: set[str] = set()
79
84
 
85
+ def _array_size(
86
+ shape: tuple[int] | list[tuple[int]],
87
+ dtype: _t.DTypeLike = _xp.float32,
88
+ out_unit: str = 'MB',
89
+ ) -> int:
90
+ """
91
+ Computes the expected allocated size on GPU of an array with size `shape`
92
+ and data type `dtype`.
93
+
94
+ Parameters
95
+ ----------
96
+ shape : tuple[int] | list[tuple[int]]
97
+ The shape of the array. Can input multiple shapes as a list, and the
98
+ result will be the total size of all arrays combined.
99
+ dtype : DTypeLike, optional
100
+ The data type of the array elements (default: float32).
101
+ out_unit : str, optional
102
+ The unit for the output size. Can be 'MB' or 'GB' (default: 'MB').
103
+
104
+ Returns
105
+ -------
106
+ size : int
107
+ The size of the array in the specified unit.
108
+
109
+ Examples
110
+ --------
111
+ >>> import xupy as xp
112
+ >>> arr = xp.array([1, 2, 3])
113
+ >>> xp.array_size(arr)
114
+ 12 # 3 elements * 4 bytes per int32
115
+ """
116
+ norm = _B2mb_ if out_unit == 'MB' else _Btgb_
117
+ if isinstance(shape, tuple):
118
+ if isinstance(shape[0], int):
119
+ shape = [shape] # single shape case
120
+ size = []
121
+ for s in shape:
122
+ itemsize = _np.dtype(dtype).itemsize
123
+ num_elements = _np.prod(s)
124
+ size_bytes = num_elements * itemsize
125
+ size.append(int(size_bytes / norm))
126
+ return int(_np.sum(size))
80
127
 
81
128
  # --- NUMPY Context manager ---
82
129
  class NumpyContext:
@@ -114,16 +161,282 @@ class NumpyContext:
114
161
  def __repr__(self) -> str:
115
162
  """String representation of the context manager."""
116
163
  if _GPU:
164
+
117
165
  return f"NumpyContext(original_device={self.original_device})"
118
166
  else:
119
167
  return "NumpyContext(no_gpu=True)"
120
168
 
121
169
 
170
+ # ---------------------------------------------------------------------------
171
+ # CPU Memory Context Manager (always available, no-op mock for CPU mode)
172
+ # ---------------------------------------------------------------------------
173
+
174
+ class _CPUMemoryContext:
175
+ """CPU memory context manager — a no-op counterpart to the GPU _MemoryContext.
176
+
177
+ Provides the same interface as the GPU ``MemoryContext`` so that code written
178
+ against ``xp.MemoryContext`` runs transparently on CPU (NumPy) without any
179
+ changes. All GPU-specific operations (pool cleanup, device synchronisation,
180
+ etc.) are silently skipped.
181
+
182
+ Example
183
+ -------
184
+ >>> import xupy as xp # running in CPU mode
185
+ >>> with xp.MemoryContext() as ctx:
186
+ ... arr = xp.array([1, 2, 3])
187
+ ... print(ctx.get_memory_info())
188
+ """
189
+
190
+ def __init__(
191
+ self,
192
+ device_id: _t.Optional[int] = None,
193
+ auto_cleanup: bool = True,
194
+ force_cleanup: bool = False,
195
+ print_report: bool = True,
196
+ memory_threshold: float = 0.9,
197
+ monitor_interval: float = 1.0,
198
+ ):
199
+ """
200
+ Parameters
201
+ ----------
202
+ device_id : int, optional
203
+ Ignored on CPU; present for API compatibility with the GPU version.
204
+ auto_cleanup : bool, optional
205
+ Kept for API compatibility; no cleanup is performed on CPU.
206
+ memory_threshold : float, optional
207
+ Kept for API compatibility; no threshold enforcement on CPU.
208
+ monitor_interval : float, optional
209
+ Kept for API compatibility; no monitoring is performed on CPU.
210
+ """
211
+ self.device_id = device_id
212
+ self.auto_cleanup = auto_cleanup
213
+ self.force_cleanup = force_cleanup
214
+ self.print_report = print_report
215
+ self.memory_threshold = memory_threshold
216
+ self.monitor_interval = monitor_interval
217
+
218
+ self._start_time: _t.Optional[float] = None
219
+
220
+ def __enter__(self):
221
+ """Enter the CPU memory context."""
222
+ self._start_time = _time.time()
223
+ return self
224
+
225
+ def __exit__(self, exc_type, exc_val, exc_tb):
226
+ """Exit the CPU memory context (no-op cleanup)."""
227
+ if self._start_time is not None:
228
+ duration = _time.time() - self._start_time
229
+ if self.print_report:
230
+ print(f"[MemoryContext] Session completed in {duration:.2f}s (CPU mode)")
231
+
232
+ def track_object(self, obj):
233
+ """No-op: object tracking is not needed on CPU."""
234
+ pass
235
+
236
+ def clear_cache(self):
237
+ """No-op: no GPU memory pool to clear on CPU."""
238
+ pass
239
+
240
+ def aggressive_cleanup(self):
241
+ """No-op: no GPU memory to aggressively free on CPU."""
242
+ pass
243
+
244
+ def emergency_cleanup(self):
245
+ """No-op: no GPU memory emergency cleanup needed on CPU."""
246
+ pass
247
+
248
+ def get_memory_info(self) -> dict:
249
+ """Return basic CPU/RAM memory information where available.
250
+
251
+ Uses ``psutil`` when installed; otherwise returns a minimal dict.
252
+ """
253
+ info: dict = {"device": "cpu"}
254
+ try:
255
+ import psutil # type: ignore
256
+ vm = psutil.virtual_memory()
257
+ info.update(
258
+ {
259
+ "total": vm.total,
260
+ "free": vm.available,
261
+ "used": vm.used,
262
+ "memory_percent": vm.percent / 100.0,
263
+ }
264
+ )
265
+ except ImportError:
266
+ info["error"] = "psutil not installed; install it for detailed CPU memory info"
267
+ return info
268
+
269
+ def check_memory_pressure(self) -> bool:
270
+ """Check if RAM usage is above the threshold (requires psutil)."""
271
+ mem_info = self.get_memory_info()
272
+ if "memory_percent" in mem_info:
273
+ return mem_info["memory_percent"] > self.memory_threshold
274
+ return False
275
+
276
+ def auto_cleanup_if_needed(self):
277
+ """No-op: no GPU pressure-based cleanup on CPU."""
278
+ pass
279
+
280
+ def monitor_memory(self, duration: float = 10.0):
281
+ """No-op: memory monitoring is not performed in CPU mode."""
282
+ pass
283
+
284
+ def force_memory_deallocation(self):
285
+ """No-op: forced GPU memory deallocation is not applicable on CPU."""
286
+ pass
287
+
288
+ def force_memory_pool_reset(self):
289
+ """No-op: GPU memory pool reset is not applicable on CPU."""
290
+ pass
291
+
292
+ def __repr__(self) -> str:
293
+ """String representation of the CPU memory context."""
294
+ mem_info = self.get_memory_info()
295
+ if "used" in mem_info:
296
+ used_mb = mem_info["used"] / (1024 * 1000)
297
+ total_mb = mem_info["total"] / (1024 * 1000)
298
+ percent = mem_info["memory_percent"] * 100
299
+ return f"MemoryContext(device=cpu, memory={used_mb:.2f}/{total_mb:.2f} MB ({percent:.1f}%))"
300
+ return "MemoryContext(device=cpu)"
301
+
302
+
122
303
  # ---------------------------------------------------------------------------
123
304
  # GPU-only definitions (only created when CuPy was successfully loaded)
124
305
  # ---------------------------------------------------------------------------
125
306
 
126
307
  if _GPU_AVAILABLE:
308
+
309
+ def _allocate_with_fallback_device(
310
+ array: _t.ArrayLike,
311
+ dtype: _t.Optional[_t.DTypeLike] = None,
312
+ preferred_device: _t.Optional[int] = None,
313
+ safety_factor: float = 1.10,
314
+ reserve_mb: int = 128,
315
+ ) -> tuple[_t.NDArray[_t.Any], int]:
316
+ """
317
+ Allocate an array on the current/preferred GPU if enough memory is available.
318
+ If not, try other GPUs when multi-GPU is available.
319
+
320
+ Parameters
321
+ ----------
322
+ array : ArrayLike
323
+ Input data to allocate on GPU.
324
+ dtype : DTypeLike, optional
325
+ Target dtype for allocation. If None, uses input dtype when available.
326
+ preferred_device : int, optional
327
+ Device to try first. If None, uses current CUDA device.
328
+ safety_factor : float, optional
329
+ Extra multiplicative margin on top of estimated size.
330
+ reserve_mb : int, optional
331
+ Additional fixed memory cushion to reduce OOM risk.
332
+
333
+ Returns
334
+ -------
335
+ gpu_array : NDArray
336
+ Allocated array on the selected GPU.
337
+ device_id : int
338
+ GPU id where the allocation was performed.
339
+
340
+ Raises
341
+ ------
342
+ RuntimeError
343
+ If GPU backend is not available.
344
+ MemoryError
345
+ If no device has enough free memory.
346
+ """
347
+ if not _GPU_AVAILABLE:
348
+ raise RuntimeError("[XuPy] GPU backend is not available.")
349
+
350
+ # Keep behavior explicit: this helper is for GPU allocation.
351
+ if not on_gpu:
352
+ raise RuntimeError("[XuPy] XuPy is in CPU mode. Call use_gpu() first.")
353
+
354
+ # Resolve dtype used for memory estimate and allocation.
355
+ target_dtype = dtype if dtype is not None else getattr(array, "dtype", _xp.float32)
356
+
357
+ # Estimate required memory in MB using existing XuPy helper.
358
+ shape = getattr(array, "shape", None)
359
+ if shape is None:
360
+ arr_np = _np.asarray(array, dtype=target_dtype)
361
+ shape = arr_np.shape
362
+ required_mb = _array_size(tuple(shape), dtype=target_dtype, out_unit="MB")
363
+ required_mb = int(required_mb * safety_factor) + int(reserve_mb)
364
+
365
+ current_device = _get_device()
366
+ first_device = current_device if preferred_device is None else int(preferred_device)
367
+
368
+ # Device probe order: preferred/current first, then the others.
369
+ device_order = [first_device]
370
+ if has_multi_gpu:
371
+ device_order.extend([d for d in range(n_gpus) if d != first_device])
372
+
373
+ for dev_id in device_order:
374
+ try:
375
+ with _on_device(dev_id):
376
+ free_b, _ = _xp.cuda.runtime.memGetInfo()
377
+ free_mb = int(free_b / _B2mb_)
378
+
379
+ if free_mb >= required_mb:
380
+ with _on_device(dev_id):
381
+ gpu_array = _xp.asarray(array, dtype=target_dtype)
382
+ return gpu_array, dev_id
383
+ except Exception:
384
+ # Skip unavailable/busy devices and continue probing.
385
+ continue
386
+
387
+ raise MemoryError(
388
+ f"[XuPy] Cannot allocate array (~{required_mb} MB incl. margin) "
389
+ f"on probed devices {device_order}."
390
+ )
391
+
392
+
393
+ @_contextmanager
394
+ def _on_device(device_id: int):
395
+ """
396
+ Context manager to temporarily set the CUDA device for computations (cupy).
397
+
398
+ Parameters
399
+ ----------
400
+ device_id : int
401
+ The ID of the CUDA device to set as default within the context.
402
+
403
+ If ``-1`` is passed, it will switch to CPU mode within the context
404
+ and restore GPU mode on exit.
405
+
406
+ Raises
407
+ ------
408
+ RuntimeError : If the device cannot be set or if the device is already
409
+ the current device.
410
+
411
+ Examples
412
+ --------
413
+ .. code-block:: python
414
+ with xp.on_device(0):
415
+ # computations here will use device 0
416
+ array = xp.array([1,2,3]) # array allocated on GPU 0
417
+ with xp.on_device(1):
418
+ # computations here will use device 1
419
+ array = xp.array([4,5,6]) # array allocated on GPU 1
420
+ with xp.on_device(-1):
421
+ # computations here will use CPU
422
+ array = xp.array([7,8,9]) # array allocated on CPU
423
+ """
424
+ original_device = _xp.cuda.runtime.getDevice()
425
+ try:
426
+ if device_id == -1:
427
+ use_cpu()
428
+ else:
429
+ _set_device(device_id)
430
+ yield
431
+ finally:
432
+ # Restore original device
433
+ try:
434
+ if device_id == -1:
435
+ use_gpu()
436
+ _xp.cuda.runtime.setDevice(original_device)
437
+ except Exception as e:
438
+ print(f"Warning: Could not restore original device: {e}")
439
+
127
440
 
128
441
  def _set_device(device_id: int) -> None:
129
442
  """
@@ -157,6 +470,28 @@ if _GPU_AVAILABLE:
157
470
  warnings.warn(
158
471
  f"[XuPy] Device {device_id} is already the current device", UserWarning
159
472
  )
473
+
474
+ def _get_device() -> int:
475
+ """
476
+ Get the current CUDA device ID.
477
+
478
+ Returns
479
+ -------
480
+ int
481
+ The ID of the current CUDA device.
482
+
483
+ Raises
484
+ ------
485
+ RuntimeError : If the GPU backend is not available.
486
+
487
+ Examples
488
+ --------
489
+ >>> current_device = xp.get_device()
490
+ >>> print(f"Current device ID: {current_device}")
491
+ """
492
+ if not _GPU_AVAILABLE:
493
+ return -1 # Indicate no GPU available
494
+ return int(_xp.cuda.runtime.getDevice())
160
495
 
161
496
  # --- GPU Memory Management Context Manager ---
162
497
  class _MemoryContext:
@@ -176,6 +511,8 @@ if _GPU_AVAILABLE:
176
511
  self,
177
512
  device_id: _t.Optional[int] = None,
178
513
  auto_cleanup: bool = True,
514
+ force_cleanup: bool = False,
515
+ print_report: bool = True,
179
516
  memory_threshold: float = 0.9,
180
517
  monitor_interval: float = 1.0,
181
518
  ):
@@ -188,15 +525,21 @@ if _GPU_AVAILABLE:
188
525
  GPU device ID to manage. If None, uses current device.
189
526
  auto_cleanup : bool, optional
190
527
  Whether to automatically cleanup memory on exit (default: True).
528
+ force_cleanup : bool, optional
529
+ Whether to force memory cleanup on exit (default: False).
530
+ print_report : bool, optional
531
+ Whether to print memory usage report on exit (default: True).
191
532
  memory_threshold : float, optional
192
533
  Memory usage threshold (0-1) for automatic cleanup (default: 0.9).
193
534
  monitor_interval : float, optional
194
- Interval in seconds for memory monitoring (default: 1.0).
535
+ Interval in milliseconds for memory monitoring (default: 1.0).
195
536
  """
196
537
  self.device_id = device_id
197
538
  self.auto_cleanup = auto_cleanup
539
+ self.force_cleanup = force_cleanup
198
540
  self.memory_threshold = memory_threshold
199
- self.monitor_interval = monitor_interval
541
+ self.monitor_interval = monitor_interval/1000
542
+ self._print_report = print_report
200
543
 
201
544
  self._device_ctx = None
202
545
  self._original_device = None
@@ -237,33 +580,34 @@ if _GPU_AVAILABLE:
237
580
  def __exit__(self, exc_type, exc_val, exc_tb):
238
581
  """Exit the memory context with cleanup."""
239
582
  try:
240
- if self.auto_cleanup:
583
+ if self.auto_cleanup or self.force_cleanup:
241
584
  self.aggressive_cleanup()
242
585
 
243
- # Cleanup tracked GPU objects
244
- self._cleanup_gpu_objects()
586
+ # Cleanup tracked GPU objects
587
+ self._cleanup_gpu_objects()
245
588
 
246
- # Restore original device
247
- if _GPU and self._device_ctx is not None:
248
- try:
249
- self._device_ctx.__exit__(exc_type, exc_val, exc_tb)
250
- except Exception as e:
251
- print(f"Warning: Error restoring device context: {e}")
589
+ # Restore original device
590
+ if _GPU and self._device_ctx is not None:
591
+ try:
592
+ self._device_ctx.__exit__(exc_type, exc_val, exc_tb)
593
+ except Exception as e:
594
+ print(f"Warning: Error restoring device context: {e}")
252
595
 
253
596
  # Final memory report
254
- if self._start_time:
255
- duration = _time.time() - self._start_time
256
- final_mem = self.get_memory_info()
257
- if "used" in final_mem:
258
- memory_delta = final_mem["used"] - self._initial_memory
259
- print(f"[MemoryContext] Session completed in {duration:.2f}s")
260
- print(
261
- f"[MemoryContext] Memory delta: {memory_delta / (_B2mb_):.2f} MB"
262
- )
263
- if self._cleanup_count > 0:
597
+ if self._print_report:
598
+ if self._start_time:
599
+ duration = _time.time() - self._start_time
600
+ final_mem = self.get_memory_info()
601
+ if "used" in final_mem:
602
+ memory_delta = final_mem["used"] - self._initial_memory
603
+ print(f"[MemoryContext] Session completed in {duration:.2f}s")
264
604
  print(
265
- f"[MemoryContext] Cleanup operations: {self._cleanup_count}"
605
+ f"[MemoryContext] Memory delta: {memory_delta / (_B2mb_):.2f} MB"
266
606
  )
607
+ if self._cleanup_count > 0:
608
+ print(
609
+ f"[MemoryContext] Cleanup operations: {self._cleanup_count}"
610
+ )
267
611
 
268
612
  except Exception as e:
269
613
  print(f"Warning: Error during memory context cleanup: {e}")
@@ -322,7 +666,8 @@ if _GPU_AVAILABLE:
322
666
  if not _GPU:
323
667
  return
324
668
 
325
- print("[MemoryContext] Performing aggressive memory cleanup...")
669
+ if self._print_report:
670
+ print("[MemoryContext] Performing aggressive memory cleanup...")
326
671
  self._cleanup_count += 1
327
672
 
328
673
  # Force garbage collection
@@ -385,17 +730,18 @@ if _GPU_AVAILABLE:
385
730
  except Exception as e:
386
731
  print(f"Warning: Could not get CUDA memory info: {e}")
387
732
 
388
- # As a last resort, try memory pool reset
389
- try:
390
- self.force_memory_pool_reset()
391
- except Exception as e:
392
- print(f"Warning: Memory pool reset failed: {e}")
733
+ if self.force_cleanup:
734
+ # As a last resort, try memory pool reset
735
+ try:
736
+ self.force_memory_pool_reset()
737
+ except Exception as e:
738
+ print(f"Warning: Memory pool reset failed: {e}")
393
739
 
394
- # Final attempt: force memory deallocation
395
- try:
396
- self.force_memory_deallocation()
397
- except Exception as e:
398
- print(f"Warning: Forced memory deallocation failed: {e}")
740
+ # Final attempt: force memory deallocation
741
+ try:
742
+ self.force_memory_deallocation()
743
+ except Exception as e:
744
+ print(f"Warning: Forced memory deallocation failed: {e}")
399
745
 
400
746
  def emergency_cleanup(self):
401
747
  """Emergency cleanup for out-of-memory situations."""
@@ -495,9 +841,9 @@ if _GPU_AVAILABLE:
495
841
 
496
842
  info = {
497
843
  "device": int(device_to_query),
498
- "total": int(total),
499
- "free": int(free),
500
- "used": int(used),
844
+ "total": int(total / _B2mb_),
845
+ "free": int(free / _B2mb_),
846
+ "used": int(used / _B2mb_),
501
847
  "memory_percent": memory_percent,
502
848
  "pool_used": pool_used,
503
849
  "pool_capacity": pool_capacity,
@@ -612,10 +958,8 @@ if _GPU_AVAILABLE:
612
958
  # Check memory after
613
959
  free_after, _ = _xp.cuda.runtime.memGetInfo()
614
960
  freed = free_after - free_before
615
- # print(
616
- # f"[MemoryContext] Memory after forced deallocation: {free_after/(_B2mb_):.2f}/{total/(_B2mb_):.2f} MB"
617
- # )
618
- print(f"[MemoryContext] Memory freed: {freed/(_B2mb_):.2f} MB")
961
+ if self._print_report:
962
+ print(f"[MemoryContext] Memory freed: {freed/(_B2mb_):.2f} MB")
619
963
 
620
964
  except Exception as e:
621
965
  print(f"Warning: Could not force memory deallocation: {e}")
@@ -625,7 +969,8 @@ if _GPU_AVAILABLE:
625
969
  if not _GPU:
626
970
  return
627
971
 
628
- print("[MemoryContext] Performing memory pool reset...")
972
+ if self._print_report:
973
+ print("[MemoryContext] Performing memory pool reset...")
629
974
  try:
630
975
  # Get current pool
631
976
  old_pool = _xp.get_default_memory_pool()
@@ -647,7 +992,8 @@ if _GPU_AVAILABLE:
647
992
  # Synchronize to ensure operations are complete
648
993
  _xp.cuda.runtime.deviceSynchronize()
649
994
 
650
- print("[MemoryContext] Memory pool reset completed")
995
+ if self._print_report:
996
+ print("[MemoryContext] Memory pool reset completed")
651
997
 
652
998
  except Exception as e:
653
999
  print(f"Warning: Could not reset memory pool: {e}")
@@ -675,6 +1021,7 @@ if _GPU_AVAILABLE:
675
1021
 
676
1022
  def _gpu_definitions() -> dict:
677
1023
  """Return dict of names exposed only in GPU mode."""
1024
+
678
1025
  def asmarray(array: _t.NDArray[_t.Any]) -> _t.MaskedArray:
679
1026
  """
680
1027
  Converts an object to a masked array on the GPU.
@@ -699,6 +1046,8 @@ def _gpu_definitions() -> dict:
699
1046
  'npma': _np.ma,
700
1047
  'asmarray': asmarray,
701
1048
  'set_device': _set_device,
1049
+ 'array_size': _array_size,
1050
+ 'on_device': _on_device,
702
1051
  'MemoryContext': _MemoryContext,
703
1052
  }
704
1053
 
@@ -726,8 +1075,11 @@ def _cpu_definitions() -> dict:
726
1075
  'double': _np.float64,
727
1076
  'cfloat': _np.complex128,
728
1077
  'cdouble': _np.complex128,
1078
+ 'array_size': _array_size,
729
1079
  'asnumpy': asnumpy,
730
1080
  'asmarray': asmarray,
1081
+ 'MemoryContext': _CPUMemoryContext,
1082
+ 'on_device': lambda device_id: None, # No-op on CPU
731
1083
  }
732
1084
 
733
1085
 
@@ -1 +0,0 @@
1
- __version__ = "1.7.0"
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes