gpustack-runtime 0.2.2.post1__tar.gz → 0.2.2.post3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/PKG-INFO +1 -1
  2. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/_version.py +2 -2
  3. gpustack_runtime-0.2.2.post3/gpustack_runtime/_version_appendix.py +1 -0
  4. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/nvidia.py +8 -334
  5. gpustack_runtime-0.2.2.post1/gpustack_runtime/_version_appendix.py +0 -1
  6. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/.codespelldict +0 -0
  7. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/.codespellrc +0 -0
  8. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/.dockerignore +0 -0
  9. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/.gitattributes +0 -0
  10. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/.gitignore +0 -0
  11. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/.pre-commit-config.yaml +0 -0
  12. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/.python-version +0 -0
  13. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/LICENSE +0 -0
  14. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/Makefile +0 -0
  15. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/README.md +0 -0
  16. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/deploy/manifests/docker-compose.yaml +0 -0
  17. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/deploy/manifests/kubernetes.yaml +0 -0
  18. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/docs/index.md +0 -0
  19. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/docs/modules/gpustack_runtime.deployer.md +0 -0
  20. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/docs/modules/gpustack_runtime.detector.md +0 -0
  21. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/docs/modules/gpustack_runtime.md +0 -0
  22. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/__init__.py +0 -0
  23. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/__main__.py +0 -0
  24. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/_version.pyi +0 -0
  25. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/__init__.py +0 -0
  26. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/__types__.py +0 -0
  27. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/deployer.py +0 -0
  28. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/detector.py +0 -0
  29. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/images.py +0 -0
  30. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/__init__.py +0 -0
  31. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/__patches__.py +0 -0
  32. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/__types__.py +0 -0
  33. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/__utils__.py +0 -0
  34. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/__init__.py +0 -0
  35. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/__types__.py +0 -0
  36. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/__utils__.py +0 -0
  37. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/amd.py +0 -0
  38. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/ascend.py +0 -0
  39. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/hygon.py +0 -0
  40. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/iluvatar.py +0 -0
  41. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/metax.py +0 -0
  42. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/thead.py +0 -0
  43. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/docker.py +0 -0
  44. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/k8s/devicemanager/__init__.py +0 -0
  45. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/kuberentes.py +0 -0
  46. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/podman.py +0 -0
  47. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/__init__.py +0 -0
  48. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/__types__.py +0 -0
  49. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/__utils__.py +0 -0
  50. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/amd.py +0 -0
  51. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/ascend.py +0 -0
  52. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/cambricon.py +0 -0
  53. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/hygon.py +0 -0
  54. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/iluvatar.py +0 -0
  55. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/metax.py +0 -0
  56. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/mthreads.py +0 -0
  57. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyamdgpu/__init__.py +0 -0
  58. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyamdsmi/__init__.py +0 -0
  59. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pydcmi/__init__.py +0 -0
  60. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyhgml/__init__.py +0 -0
  61. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyhgml/libhgml.so +0 -0
  62. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyhgml/libuki.so +0 -0
  63. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyhsa/__init__.py +0 -0
  64. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyixml/__init__.py +0 -0
  65. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pymtml/__init__.py +0 -0
  66. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pymxsml/__init__.py +0 -0
  67. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pynvml/__init__.py +0 -0
  68. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyrocmsmi/__init__.py +0 -0
  69. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/thead.py +0 -0
  70. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/envs.py +0 -0
  71. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/logging.py +0 -0
  72. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/hatch.toml +0 -0
  73. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/mkdocs.yml +0 -0
  74. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/pack/Dockerfile +0 -0
  75. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/pack/Dockerfile.dummy +0 -0
  76. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/pyproject.toml +0 -0
  77. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/pytest.ini +0 -0
  78. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/ruff.toml +0 -0
  79. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/__init__.py +0 -0
  80. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_compare_versions.json +0 -0
  81. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_correct_runner_image.json +0 -0
  82. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json.json +0 -0
  83. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_jsons.json +0 -0
  84. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_yamls.yaml +0 -0
  85. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_json.json +0 -0
  86. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_yaml.yaml +0 -0
  87. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_nginx_entrypoint.sh +0 -0
  88. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/test_utils.py +0 -0
  89. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/test_workload_status.py +0 -0
  90. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/fixtures/__init__.py +0 -0
  91. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/README.md +0 -0
  92. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi300x.json +0 -0
  93. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi308x.json +0 -0
  94. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_amd_rx7800xt.json +0 -0
  95. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_ascend_310p3.json +0 -0
  96. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_ascend_910b2.json +0 -0
  97. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_hygon_k100ai.json +0 -0
  98. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_metax_c500.json +0 -0
  99. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_gb10.json +0 -0
  100. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100.json +0 -0
  101. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100_mig.json +0 -0
  102. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h200.json +0 -0
  103. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4080super.json +0 -0
  104. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4090d.json +0 -0
  105. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx5090d.json +0 -0
  106. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_thead_ppu.json +0 -0
  107. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi300x.json +0 -0
  108. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi308x.json +0 -0
  109. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_amd_rx7800xt.json +0 -0
  110. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_ascend_310p3.json +0 -0
  111. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_ascend_910b2.json +0 -0
  112. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_hygon_k100ai.json +0 -0
  113. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_metax_c500.json +0 -0
  114. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_mthreads_s5000.json +0 -0
  115. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100.json +0 -0
  116. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100_mig.json +0 -0
  117. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h200.json +0 -0
  118. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4080super.json +0 -0
  119. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4090d.json +0 -0
  120. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx5090d.json +0 -0
  121. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_thead_ppu.json +0 -0
  122. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_amd.py +0 -0
  123. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_ascend.py +0 -0
  124. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_cambricon.py +0 -0
  125. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_detector_utils.py +0 -0
  126. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_hygon.py +0 -0
  127. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_iluvatar.py +0 -0
  128. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_metax.py +0 -0
  129. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_mthreads.py +0 -0
  130. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_nvidia.py +0 -0
  131. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/uv.lock +0 -0
  132. {gpustack_runtime-0.2.2.post1 → gpustack_runtime-0.2.2.post3}/uv.toml +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gpustack-runtime
3
- Version: 0.2.2.post1
3
+ Version: 0.2.2.post3
4
4
  Summary: GPUStack Runtime is library for detecting GPU resources and launching GPU workloads.
5
5
  Project-URL: Homepage, https://github.com/gpustack/runtime
6
6
  Project-URL: Bug Tracker, https://github.com/gpustack/gpustack/issues
@@ -27,8 +27,8 @@ version_tuple: VERSION_TUPLE
27
27
  __commit_id__: COMMIT_ID
28
28
  commit_id: COMMIT_ID
29
29
 
30
- __version__ = version = '0.2.2.post1'
31
- __version_tuple__ = version_tuple = (0, 2, 2, 'post1')
30
+ __version__ = version = '0.2.2.post3'
31
+ __version_tuple__ = version_tuple = (0, 2, 2, 'post3')
32
32
  try:
33
33
  from ._version_appendix import git_commit
34
34
  __commit_id__ = commit_id = git_commit
@@ -0,0 +1 @@
1
+ git_commit = "e93cb90"
@@ -3,12 +3,10 @@ from __future__ import annotations as __future_annotations__
3
3
  import contextlib
4
4
  import logging
5
5
  import math
6
- import re
7
6
  import threading
8
7
  import time
9
8
  from _ctypes import byref
10
9
  from functools import lru_cache
11
- from pathlib import Path
12
10
 
13
11
  from .. import envs
14
12
  from ..logging import debug_log_exception, debug_log_warning
@@ -76,7 +74,7 @@ class NVIDIADetector(Detector):
76
74
  def __init__(self):
77
75
  super().__init__(ManufacturerEnum.NVIDIA)
78
76
 
79
- def detect(self) -> Devices | None: # noqa: PLR0915
77
+ def detect(self) -> Devices | None:
80
78
  """
81
79
  Detect NVIDIA GPUs using pynvml.
82
80
 
@@ -160,9 +158,13 @@ class NVIDIADetector(Detector):
160
158
  with contextlib.suppress(pynvml.NVMLError):
161
159
  dev_index = pynvml.nvmlDeviceGetMinorNumber(dev)
162
160
 
163
- # With MIG disabled, treat as a single device.
161
+ # Report the physical card, whether or not MIG is enabled.
162
+ # MIG instances are partitioned on demand by the operator's
163
+ # device-manager; they are not separate allocatable devices
164
+ # in this inventory. A MIG-enabled card is marked ``mig``
165
+ # in the appendix instead.
164
166
 
165
- if dev_mig_mode == pynvml.NVML_DEVICE_MIG_DISABLE:
167
+ if True:
166
168
  dev_name = pynvml.nvmlDeviceGetName(dev)
167
169
 
168
170
  dev_uuid = pynvml.nvmlDeviceGetUUID(dev)
@@ -214,6 +216,7 @@ class NVIDIADetector(Detector):
214
216
  dev_appendix = {
215
217
  "arch_family": _get_arch_family(dev_cc_t),
216
218
  "vgpu": dev_is_vgpu,
219
+ "mig": dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE,
217
220
  "bdf": dev_bdf,
218
221
  }
219
222
  if dev_numa:
@@ -247,157 +250,6 @@ class NVIDIADetector(Detector):
247
250
 
248
251
  continue
249
252
 
250
- # Otherwise, get MIG devices,
251
- # inspired by https://github.com/NVIDIA/go-nvlib/blob/fdfe25d0ffc9d7a8c166f4639ef236da81116262/pkg/nvlib/device/mig_device.go#L61-L154.
252
-
253
- dev_mig_minors = _get_mig_minors()
254
-
255
- mdev_name = ""
256
- mdev_cores = None
257
- mdev_count = pynvml.nvmlDeviceGetMaxMigDeviceCount(dev)
258
- for mdev_idx in range(mdev_count):
259
- mdev = None
260
- with contextlib.suppress(pynvml.NVMLError):
261
- mdev = pynvml.nvmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
262
- if not mdev:
263
- continue
264
-
265
- mdev_index = mdev_idx + dev_count * (dev_idx + 1)
266
- mdev_uuid = pynvml.nvmlDeviceGetUUID(mdev)
267
-
268
- mdev_mem = 0
269
- mdev_mem_used = 0
270
- mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
271
- with contextlib.suppress(pynvml.NVMLError):
272
- mdev_mem_info = pynvml.nvmlDeviceGetMemoryInfo(mdev)
273
- mdev_mem = byte_to_mebibyte( # byte to MiB
274
- mdev_mem_info.total,
275
- )
276
- mdev_mem_used = byte_to_mebibyte( # byte to MiB
277
- mdev_mem_info.used,
278
- )
279
- if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
280
- mdev_mem_ecc_errors = (
281
- pynvml.nvmlDeviceGetMemoryErrorCounter(
282
- mdev,
283
- pynvml.NVML_MEMORY_ERROR_TYPE_UNCORRECTED,
284
- pynvml.NVML_AGGREGATE_ECC,
285
- pynvml.NVML_MEMORY_LOCATION_SRAM,
286
- )
287
- )
288
- if mdev_mem_ecc_errors > 0:
289
- mdev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
290
-
291
- mdev_appendix = {
292
- "arch_family": _get_arch_family(dev_cc_t),
293
- "vgpu": True,
294
- "sliced": True,
295
- "bdf": dev_bdf,
296
- }
297
- if dev_numa:
298
- mdev_appendix["numa"] = dev_numa
299
-
300
- mdev_gi_id = pynvml.nvmlDeviceGetGpuInstanceId(mdev)
301
- mdev_appendix["gpu_instance_id"] = mdev_gi_id
302
- mdev_ci_id = pynvml.nvmlDeviceGetComputeInstanceId(mdev)
303
- mdev_appendix["compute_instance_id"] = mdev_ci_id
304
- if envs.GPUSTACK_RUNTIME_DETECT_PHYSICAL_INDEX_PRIORITY:
305
- mdev_appendix["gpu_instance_index"] = dev_mig_minors.get(
306
- (dev_index, mdev_gi_id, None),
307
- )
308
- mdev_appendix["compute_instance_index"] = dev_mig_minors.get(
309
- (dev_index, mdev_gi_id, mdev_ci_id),
310
- )
311
-
312
- mdev_cores_util = _get_sm_util_from_gpm_metrics(dev, mdev_gi_id)
313
-
314
- mdev_gi = pynvml.nvmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
315
- mdev_ci = pynvml.nvmlGpuInstanceGetComputeInstanceById(
316
- mdev_gi,
317
- mdev_ci_id,
318
- )
319
- mdev_gi_info = pynvml.nvmlGpuInstanceGetInfo(mdev_gi)
320
- mdev_ci_info = pynvml.nvmlComputeInstanceGetInfo(mdev_ci)
321
- for dev_gi_prf_id in range(
322
- pynvml.NVML_GPU_INSTANCE_PROFILE_COUNT,
323
- ):
324
- try:
325
- dev_gi_prf = pynvml.nvmlDeviceGetGpuInstanceProfileInfo(
326
- dev,
327
- dev_gi_prf_id,
328
- )
329
- if dev_gi_prf.id != mdev_gi_info.profileId:
330
- continue
331
- except pynvml.NVMLError:
332
- continue
333
-
334
- for dev_ci_prf_id in range(
335
- pynvml.NVML_COMPUTE_INSTANCE_PROFILE_COUNT,
336
- ):
337
- for dev_cig_prf_id in range(
338
- pynvml.NVML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
339
- ):
340
- try:
341
- dev_ci_prf = pynvml.nvmlGpuInstanceGetComputeInstanceProfileInfo(
342
- mdev_gi,
343
- dev_ci_prf_id,
344
- dev_cig_prf_id,
345
- )
346
- if dev_ci_prf.id != mdev_ci_info.profileId:
347
- continue
348
- except pynvml.NVMLError:
349
- continue
350
-
351
- ci_slice = _get_compute_instance_slice(dev_ci_prf_id)
352
- gi_slice = _get_gpu_instance_slice(dev_gi_prf_id)
353
- if ci_slice == gi_slice:
354
- if hasattr(dev_gi_prf, "name"):
355
- mdev_name = dev_gi_prf.name
356
- else:
357
- gi_mem = round(
358
- math.ceil(dev_gi_prf.memorySizeMB >> 10),
359
- )
360
- mdev_name = f"{gi_slice}g.{gi_mem}gb"
361
- elif hasattr(dev_ci_prf, "name"):
362
- mdev_name = dev_ci_prf.name
363
- else:
364
- gi_mem = round(
365
- math.ceil(dev_gi_prf.memorySizeMB >> 10),
366
- )
367
- mdev_name = f"{ci_slice}c.{gi_slice}g.{gi_mem}gb"
368
- gi_attrs = _get_gpu_instance_attrs(dev_gi_prf_id)
369
- if gi_attrs:
370
- mdev_name += f"+{gi_attrs}"
371
- gi_neg_attrs = _get_gpu_instance_negattrs(dev_gi_prf_id)
372
- if gi_neg_attrs:
373
- mdev_name += f"-{gi_neg_attrs}"
374
-
375
- mdev_cores = dev_ci_prf.multiprocessorCount
376
-
377
- break
378
-
379
- ret.append(
380
- Device(
381
- manufacturer=self.manufacturer,
382
- index=mdev_index,
383
- name=mdev_name,
384
- uuid=mdev_uuid,
385
- driver_version=sys_driver_ver,
386
- runtime_version=sys_runtime_ver,
387
- runtime_version_original=sys_runtime_ver_original,
388
- compute_capability=dev_cc,
389
- cores=mdev_cores,
390
- cores_utilization=mdev_cores_util,
391
- memory=mdev_mem,
392
- memory_used=mdev_mem_used,
393
- memory_utilization=get_utilization(mdev_mem_used, mdev_mem),
394
- memory_status=mdev_mem_status,
395
- temperature=dev_temp,
396
- power=dev_power,
397
- power_used=dev_power_used,
398
- appendix=mdev_appendix,
399
- ),
400
- )
401
253
  except pynvml.NVMLError:
402
254
  debug_log_exception(logger, "Failed to fetch devices")
403
255
  raise
@@ -746,141 +598,6 @@ def _get_arch_family(dev_cc_t: list[int]) -> str:
746
598
  return "Unknown"
747
599
 
748
600
 
749
- def _get_gpu_instance_slice(dev_gi_prf_id: int) -> int:
750
- """
751
- Get the number of slice for a given GPU Instance Profile ID.
752
-
753
- Args:
754
- dev_gi_prf_id:
755
- The GPU Instance Profile ID.
756
-
757
- Returns:
758
- The number of slices.
759
-
760
- """
761
- match dev_gi_prf_id:
762
- case (
763
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE
764
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_REV1
765
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_REV2
766
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_GFX
767
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_NO_ME
768
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_ALL_ME
769
- ):
770
- return 1
771
- case (
772
- pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE
773
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_REV1
774
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_GFX
775
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_NO_ME
776
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_ALL_ME
777
- ):
778
- return 2
779
- case pynvml.NVML_GPU_INSTANCE_PROFILE_3_SLICE:
780
- return 3
781
- case (
782
- pynvml.NVML_GPU_INSTANCE_PROFILE_4_SLICE
783
- | pynvml.NVML_GPU_INSTANCE_PROFILE_4_SLICE_GFX
784
- ):
785
- return 4
786
- case pynvml.NVML_GPU_INSTANCE_PROFILE_6_SLICE:
787
- return 6
788
- case pynvml.NVML_GPU_INSTANCE_PROFILE_7_SLICE:
789
- return 7
790
- case pynvml.NVML_GPU_INSTANCE_PROFILE_8_SLICE:
791
- return 8
792
-
793
- msg = f"Invalid GPU Instance Profile ID: {dev_gi_prf_id}"
794
- raise AttributeError(msg)
795
-
796
-
797
- def _get_compute_instance_slice(dev_ci_prf_id: int) -> int:
798
- """
799
- Get the number of slice for a given Compute Instance Profile ID.
800
-
801
- Args:
802
- dev_ci_prf_id:
803
- The Compute Instance Profile ID.
804
-
805
- Returns:
806
- The number of slices.
807
-
808
- """
809
- match dev_ci_prf_id:
810
- case (
811
- pynvml.NVML_COMPUTE_INSTANCE_PROFILE_1_SLICE
812
- | pynvml.NVML_COMPUTE_INSTANCE_PROFILE_1_SLICE_REV1
813
- ):
814
- return 1
815
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_2_SLICE:
816
- return 2
817
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_3_SLICE:
818
- return 3
819
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_4_SLICE:
820
- return 4
821
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_6_SLICE:
822
- return 6
823
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_7_SLICE:
824
- return 7
825
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_8_SLICE:
826
- return 8
827
-
828
- msg = f"Invalid Compute Instance Profile ID: {dev_ci_prf_id}"
829
- raise AttributeError(msg)
830
-
831
-
832
- def _get_gpu_instance_attrs(dev_gi_prf_id: int) -> str:
833
- """
834
- Get attributes for a given GPU Instance Profile ID.
835
-
836
- Args:
837
- dev_gi_prf_id:
838
- The GPU Instance Profile ID.
839
-
840
- Returns:
841
- A string representing the attributes, or an empty string if none.
842
-
843
- """
844
- match dev_gi_prf_id:
845
- case (
846
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_REV1
847
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_REV1
848
- ):
849
- return "me"
850
- case (
851
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_ALL_ME
852
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_ALL_ME
853
- ):
854
- return "me.all"
855
- case (
856
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_GFX
857
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_GFX
858
- | pynvml.NVML_GPU_INSTANCE_PROFILE_4_SLICE_GFX
859
- ):
860
- return "gfx"
861
- return ""
862
-
863
-
864
- def _get_gpu_instance_negattrs(dev_gi_prf_id) -> str:
865
- """
866
- Get negative attributes for a given GPU Instance Profile ID.
867
-
868
- Args:
869
- dev_gi_prf_id:
870
- The GPU Instance Profile ID.
871
-
872
- Returns:
873
- A string representing the negative attributes, or an empty string if none.
874
-
875
- """
876
- if dev_gi_prf_id in [
877
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_NO_ME,
878
- pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_NO_ME,
879
- ]:
880
- return "me"
881
- return ""
882
-
883
-
884
601
  def _is_vgpu(dev_config: bytes) -> bool:
885
602
  """
886
603
  Determine if the device is a vGPU based on its PCI configuration space.
@@ -914,46 +631,3 @@ def _is_vgpu(dev_config: bytes) -> bool:
914
631
  # Check for vGPU signature,
915
632
  # which is either 0x56 (NVIDIA vGPU) or 0x46 (NVIDIA GRID).
916
633
  return dev_cap[3] == 0x56 or dev_cap[4] == 0x46
917
-
918
-
919
- def _get_mig_minors() -> dict[tuple, int] | None:
920
- """
921
- Get the minor mapping for MIG capability devices.
922
-
923
- Returns:
924
- A dict mapping (gpu_id, gi_id, ci_id) to minor number,
925
- or None if not supported.
926
-
927
- """
928
- mig_minors_path = Path("/proc/driver/nvidia-caps/mig-minors")
929
- if not mig_minors_path.exists():
930
- return None
931
-
932
- ret = {}
933
- for _line in mig_minors_path.read_text(encoding="utf-8").splitlines():
934
- line = _line.strip()
935
- if not line:
936
- continue
937
-
938
- # Scan lines like:
939
- # gpu%d/gi%d/ci%d/access %d
940
- m = re.match(r"gpu(\d+)/gi(\d+)/ci(\d+)/access (\d+)", line)
941
- if m:
942
- gpu_id = int(m.group(1))
943
- gi_id = int(m.group(2))
944
- ci_id = int(m.group(3))
945
- minor = int(m.group(4))
946
- ret[(gpu_id, gi_id, ci_id)] = minor
947
- continue
948
-
949
- # Scan lines like:
950
- # gpu%d/gi%d/access %d
951
- m = re.match(r"gpu(\d+)/gi(\d+)/access (\d+)", line)
952
- if m:
953
- gpu_id = int(m.group(1))
954
- gi_id = int(m.group(2))
955
- minor = int(m.group(3))
956
- ret[(gpu_id, gi_id, None)] = minor
957
- continue
958
-
959
- return ret
@@ -1 +0,0 @@
1
- git_commit = "4d942f0"