gpustack-runtime 0.2.2.post2__tar.gz → 0.2.2.post3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/PKG-INFO +1 -1
  2. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/_version.py +2 -2
  3. gpustack_runtime-0.2.2.post3/gpustack_runtime/_version_appendix.py +1 -0
  4. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/nvidia.py +8 -380
  5. gpustack_runtime-0.2.2.post2/gpustack_runtime/_version_appendix.py +0 -1
  6. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/.codespelldict +0 -0
  7. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/.codespellrc +0 -0
  8. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/.dockerignore +0 -0
  9. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/.gitattributes +0 -0
  10. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/.gitignore +0 -0
  11. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/.pre-commit-config.yaml +0 -0
  12. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/.python-version +0 -0
  13. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/LICENSE +0 -0
  14. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/Makefile +0 -0
  15. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/README.md +0 -0
  16. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/deploy/manifests/docker-compose.yaml +0 -0
  17. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/deploy/manifests/kubernetes.yaml +0 -0
  18. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/docs/index.md +0 -0
  19. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/docs/modules/gpustack_runtime.deployer.md +0 -0
  20. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/docs/modules/gpustack_runtime.detector.md +0 -0
  21. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/docs/modules/gpustack_runtime.md +0 -0
  22. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/__init__.py +0 -0
  23. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/__main__.py +0 -0
  24. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/_version.pyi +0 -0
  25. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/__init__.py +0 -0
  26. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/__types__.py +0 -0
  27. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/deployer.py +0 -0
  28. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/detector.py +0 -0
  29. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/cmds/images.py +0 -0
  30. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/__init__.py +0 -0
  31. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/__patches__.py +0 -0
  32. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/__types__.py +0 -0
  33. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/__utils__.py +0 -0
  34. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/__init__.py +0 -0
  35. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/__types__.py +0 -0
  36. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/__utils__.py +0 -0
  37. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/amd.py +0 -0
  38. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/ascend.py +0 -0
  39. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/hygon.py +0 -0
  40. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/iluvatar.py +0 -0
  41. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/metax.py +0 -0
  42. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/cdi/thead.py +0 -0
  43. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/docker.py +0 -0
  44. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/k8s/devicemanager/__init__.py +0 -0
  45. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/kuberentes.py +0 -0
  46. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/deployer/podman.py +0 -0
  47. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/__init__.py +0 -0
  48. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/__types__.py +0 -0
  49. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/__utils__.py +0 -0
  50. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/amd.py +0 -0
  51. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/ascend.py +0 -0
  52. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/cambricon.py +0 -0
  53. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/hygon.py +0 -0
  54. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/iluvatar.py +0 -0
  55. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/metax.py +0 -0
  56. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/mthreads.py +0 -0
  57. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyamdgpu/__init__.py +0 -0
  58. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyamdsmi/__init__.py +0 -0
  59. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pydcmi/__init__.py +0 -0
  60. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyhgml/__init__.py +0 -0
  61. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyhgml/libhgml.so +0 -0
  62. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyhgml/libuki.so +0 -0
  63. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyhsa/__init__.py +0 -0
  64. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyixml/__init__.py +0 -0
  65. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pymtml/__init__.py +0 -0
  66. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pymxsml/__init__.py +0 -0
  67. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pynvml/__init__.py +0 -0
  68. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/pyrocmsmi/__init__.py +0 -0
  69. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/detector/thead.py +0 -0
  70. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/envs.py +0 -0
  71. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/gpustack_runtime/logging.py +0 -0
  72. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/hatch.toml +0 -0
  73. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/mkdocs.yml +0 -0
  74. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/pack/Dockerfile +0 -0
  75. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/pack/Dockerfile.dummy +0 -0
  76. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/pyproject.toml +0 -0
  77. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/pytest.ini +0 -0
  78. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/ruff.toml +0 -0
  79. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/__init__.py +0 -0
  80. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_compare_versions.json +0 -0
  81. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_correct_runner_image.json +0 -0
  82. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json.json +0 -0
  83. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_jsons.json +0 -0
  84. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_yamls.yaml +0 -0
  85. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_json.json +0 -0
  86. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_yaml.yaml +0 -0
  87. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/fixtures/test_nginx_entrypoint.sh +0 -0
  88. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/test_utils.py +0 -0
  89. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/deployer/test_workload_status.py +0 -0
  90. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/fixtures/__init__.py +0 -0
  91. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/README.md +0 -0
  92. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi300x.json +0 -0
  93. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi308x.json +0 -0
  94. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_amd_rx7800xt.json +0 -0
  95. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_ascend_310p3.json +0 -0
  96. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_ascend_910b2.json +0 -0
  97. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_hygon_k100ai.json +0 -0
  98. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_metax_c500.json +0 -0
  99. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_gb10.json +0 -0
  100. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100.json +0 -0
  101. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100_mig.json +0 -0
  102. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h200.json +0 -0
  103. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4080super.json +0 -0
  104. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4090d.json +0 -0
  105. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx5090d.json +0 -0
  106. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/detect_output_thead_ppu.json +0 -0
  107. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi300x.json +0 -0
  108. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi308x.json +0 -0
  109. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_amd_rx7800xt.json +0 -0
  110. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_ascend_310p3.json +0 -0
  111. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_ascend_910b2.json +0 -0
  112. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_hygon_k100ai.json +0 -0
  113. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_metax_c500.json +0 -0
  114. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_mthreads_s5000.json +0 -0
  115. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100.json +0 -0
  116. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100_mig.json +0 -0
  117. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h200.json +0 -0
  118. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4080super.json +0 -0
  119. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4090d.json +0 -0
  120. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx5090d.json +0 -0
  121. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/samples/topology_output_thead_ppu.json +0 -0
  122. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_amd.py +0 -0
  123. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_ascend.py +0 -0
  124. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_cambricon.py +0 -0
  125. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_detector_utils.py +0 -0
  126. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_hygon.py +0 -0
  127. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_iluvatar.py +0 -0
  128. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_metax.py +0 -0
  129. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_mthreads.py +0 -0
  130. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/tests/gpustack_runtime/detector/test_nvidia.py +0 -0
  131. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/uv.lock +0 -0
  132. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post3}/uv.toml +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gpustack-runtime
3
- Version: 0.2.2.post2
3
+ Version: 0.2.2.post3
4
4
  Summary: GPUStack Runtime is library for detecting GPU resources and launching GPU workloads.
5
5
  Project-URL: Homepage, https://github.com/gpustack/runtime
6
6
  Project-URL: Bug Tracker, https://github.com/gpustack/gpustack/issues
@@ -27,8 +27,8 @@ version_tuple: VERSION_TUPLE
27
27
  __commit_id__: COMMIT_ID
28
28
  commit_id: COMMIT_ID
29
29
 
30
- __version__ = version = '0.2.2.post2'
31
- __version_tuple__ = version_tuple = (0, 2, 2, 'post2')
30
+ __version__ = version = '0.2.2.post3'
31
+ __version_tuple__ = version_tuple = (0, 2, 2, 'post3')
32
32
  try:
33
33
  from ._version_appendix import git_commit
34
34
  __commit_id__ = commit_id = git_commit
@@ -0,0 +1 @@
1
+ git_commit = "e93cb90"
@@ -3,12 +3,10 @@ from __future__ import annotations as __future_annotations__
3
3
  import contextlib
4
4
  import logging
5
5
  import math
6
- import re
7
6
  import threading
8
7
  import time
9
8
  from _ctypes import byref
10
9
  from functools import lru_cache
11
- from pathlib import Path
12
10
 
13
11
  from .. import envs
14
12
  from ..logging import debug_log_exception, debug_log_warning
@@ -76,7 +74,7 @@ class NVIDIADetector(Detector):
76
74
  def __init__(self):
77
75
  super().__init__(ManufacturerEnum.NVIDIA)
78
76
 
79
- def detect(self) -> Devices | None: # noqa: PLR0915
77
+ def detect(self) -> Devices | None:
80
78
  """
81
79
  Detect NVIDIA GPUs using pynvml.
82
80
 
@@ -160,9 +158,13 @@ class NVIDIADetector(Detector):
160
158
  with contextlib.suppress(pynvml.NVMLError):
161
159
  dev_index = pynvml.nvmlDeviceGetMinorNumber(dev)
162
160
 
163
- # With MIG disabled, treat as a single device.
161
+ # Report the physical card, whether or not MIG is enabled.
162
+ # MIG instances are partitioned on demand by the operator's
163
+ # device-manager; they are not separate allocatable devices
164
+ # in this inventory. A MIG-enabled card is marked ``mig``
165
+ # in the appendix instead.
164
166
 
165
- if dev_mig_mode == pynvml.NVML_DEVICE_MIG_DISABLE:
167
+ if True:
166
168
  dev_name = pynvml.nvmlDeviceGetName(dev)
167
169
 
168
170
  dev_uuid = pynvml.nvmlDeviceGetUUID(dev)
@@ -214,6 +216,7 @@ class NVIDIADetector(Detector):
214
216
  dev_appendix = {
215
217
  "arch_family": _get_arch_family(dev_cc_t),
216
218
  "vgpu": dev_is_vgpu,
219
+ "mig": dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE,
217
220
  "bdf": dev_bdf,
218
221
  }
219
222
  if dev_numa:
@@ -247,203 +250,6 @@ class NVIDIADetector(Detector):
247
250
 
248
251
  continue
249
252
 
250
- # Otherwise, get MIG devices,
251
- # inspired by https://github.com/NVIDIA/go-nvlib/blob/fdfe25d0ffc9d7a8c166f4639ef236da81116262/pkg/nvlib/device/mig_device.go#L61-L154.
252
-
253
- dev_mig_minors = _get_mig_minors()
254
-
255
- mdev_name = ""
256
- mdev_cores = None
257
- mdevs_before = len(ret)
258
- mdev_count = pynvml.nvmlDeviceGetMaxMigDeviceCount(dev)
259
- for mdev_idx in range(mdev_count):
260
- mdev = None
261
- with contextlib.suppress(pynvml.NVMLError):
262
- mdev = pynvml.nvmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
263
- if not mdev:
264
- continue
265
-
266
- mdev_index = mdev_idx + dev_count * (dev_idx + 1)
267
- mdev_uuid = pynvml.nvmlDeviceGetUUID(mdev)
268
-
269
- mdev_mem = 0
270
- mdev_mem_used = 0
271
- mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
272
- with contextlib.suppress(pynvml.NVMLError):
273
- mdev_mem_info = pynvml.nvmlDeviceGetMemoryInfo(mdev)
274
- mdev_mem = byte_to_mebibyte( # byte to MiB
275
- mdev_mem_info.total,
276
- )
277
- mdev_mem_used = byte_to_mebibyte( # byte to MiB
278
- mdev_mem_info.used,
279
- )
280
- if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
281
- mdev_mem_ecc_errors = (
282
- pynvml.nvmlDeviceGetMemoryErrorCounter(
283
- mdev,
284
- pynvml.NVML_MEMORY_ERROR_TYPE_UNCORRECTED,
285
- pynvml.NVML_AGGREGATE_ECC,
286
- pynvml.NVML_MEMORY_LOCATION_SRAM,
287
- )
288
- )
289
- if mdev_mem_ecc_errors > 0:
290
- mdev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
291
-
292
- mdev_appendix = {
293
- "arch_family": _get_arch_family(dev_cc_t),
294
- "vgpu": True,
295
- "sliced": True,
296
- "bdf": dev_bdf,
297
- }
298
- if dev_numa:
299
- mdev_appendix["numa"] = dev_numa
300
-
301
- mdev_gi_id = pynvml.nvmlDeviceGetGpuInstanceId(mdev)
302
- mdev_appendix["gpu_instance_id"] = mdev_gi_id
303
- mdev_ci_id = pynvml.nvmlDeviceGetComputeInstanceId(mdev)
304
- mdev_appendix["compute_instance_id"] = mdev_ci_id
305
- if envs.GPUSTACK_RUNTIME_DETECT_PHYSICAL_INDEX_PRIORITY:
306
- mdev_appendix["gpu_instance_index"] = dev_mig_minors.get(
307
- (dev_index, mdev_gi_id, None),
308
- )
309
- mdev_appendix["compute_instance_index"] = dev_mig_minors.get(
310
- (dev_index, mdev_gi_id, mdev_ci_id),
311
- )
312
-
313
- mdev_cores_util = _get_sm_util_from_gpm_metrics(dev, mdev_gi_id)
314
-
315
- mdev_gi = pynvml.nvmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
316
- mdev_ci = pynvml.nvmlGpuInstanceGetComputeInstanceById(
317
- mdev_gi,
318
- mdev_ci_id,
319
- )
320
- mdev_gi_info = pynvml.nvmlGpuInstanceGetInfo(mdev_gi)
321
- mdev_ci_info = pynvml.nvmlComputeInstanceGetInfo(mdev_ci)
322
- for dev_gi_prf_id in range(
323
- pynvml.NVML_GPU_INSTANCE_PROFILE_COUNT,
324
- ):
325
- try:
326
- dev_gi_prf = pynvml.nvmlDeviceGetGpuInstanceProfileInfo(
327
- dev,
328
- dev_gi_prf_id,
329
- )
330
- if dev_gi_prf.id != mdev_gi_info.profileId:
331
- continue
332
- except pynvml.NVMLError:
333
- continue
334
-
335
- for dev_ci_prf_id in range(
336
- pynvml.NVML_COMPUTE_INSTANCE_PROFILE_COUNT,
337
- ):
338
- for dev_cig_prf_id in range(
339
- pynvml.NVML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
340
- ):
341
- try:
342
- dev_ci_prf = pynvml.nvmlGpuInstanceGetComputeInstanceProfileInfo(
343
- mdev_gi,
344
- dev_ci_prf_id,
345
- dev_cig_prf_id,
346
- )
347
- if dev_ci_prf.id != mdev_ci_info.profileId:
348
- continue
349
- except pynvml.NVMLError:
350
- continue
351
-
352
- ci_slice = _get_compute_instance_slice(dev_ci_prf_id)
353
- gi_slice = _get_gpu_instance_slice(dev_gi_prf_id)
354
- if ci_slice == gi_slice:
355
- if hasattr(dev_gi_prf, "name"):
356
- mdev_name = dev_gi_prf.name
357
- else:
358
- gi_mem = round(
359
- math.ceil(dev_gi_prf.memorySizeMB >> 10),
360
- )
361
- mdev_name = f"{gi_slice}g.{gi_mem}gb"
362
- elif hasattr(dev_ci_prf, "name"):
363
- mdev_name = dev_ci_prf.name
364
- else:
365
- gi_mem = round(
366
- math.ceil(dev_gi_prf.memorySizeMB >> 10),
367
- )
368
- mdev_name = f"{ci_slice}c.{gi_slice}g.{gi_mem}gb"
369
- gi_attrs = _get_gpu_instance_attrs(dev_gi_prf_id)
370
- if gi_attrs:
371
- mdev_name += f"+{gi_attrs}"
372
- gi_neg_attrs = _get_gpu_instance_negattrs(dev_gi_prf_id)
373
- if gi_neg_attrs:
374
- mdev_name += f"-{gi_neg_attrs}"
375
-
376
- mdev_cores = dev_ci_prf.multiprocessorCount
377
-
378
- break
379
-
380
- ret.append(
381
- Device(
382
- manufacturer=self.manufacturer,
383
- index=mdev_index,
384
- name=mdev_name,
385
- uuid=mdev_uuid,
386
- driver_version=sys_driver_ver,
387
- runtime_version=sys_runtime_ver,
388
- runtime_version_original=sys_runtime_ver_original,
389
- compute_capability=dev_cc,
390
- cores=mdev_cores,
391
- cores_utilization=mdev_cores_util,
392
- memory=mdev_mem,
393
- memory_used=mdev_mem_used,
394
- memory_utilization=get_utilization(mdev_mem_used, mdev_mem),
395
- memory_status=mdev_mem_status,
396
- temperature=dev_temp,
397
- power=dev_power,
398
- power_used=dev_power_used,
399
- appendix=mdev_appendix,
400
- ),
401
- )
402
-
403
- if len(ret) == mdevs_before:
404
- # MIG is enabled but no GPU instances exist yet (an
405
- # operator/device-manager typically partitions the card
406
- # on demand). Report the physical card anyway so the
407
- # worker inventory keeps it visible for vendor/backend
408
- # matching; mark it MIG-managed so it is not mistaken
409
- # for a whole-card allocatable device.
410
- dev_name = pynvml.nvmlDeviceGetName(dev)
411
- dev_uuid = pynvml.nvmlDeviceGetUUID(dev)
412
- dev_mem = 0
413
- dev_mem_used = 0
414
- with contextlib.suppress(pynvml.NVMLError):
415
- dev_mem_info = pynvml.nvmlDeviceGetMemoryInfo(dev)
416
- dev_mem = byte_to_mebibyte(dev_mem_info.total)
417
- dev_mem_used = byte_to_mebibyte(dev_mem_info.used)
418
- dev_cores = None
419
- with contextlib.suppress(pynvml.NVMLError):
420
- dev_cores = pynvml.nvmlDeviceGetNumGpuCores(dev)
421
- ret.append(
422
- Device(
423
- manufacturer=self.manufacturer,
424
- index=dev_index,
425
- name=dev_name,
426
- uuid=dev_uuid,
427
- driver_version=sys_driver_ver,
428
- runtime_version=sys_runtime_ver,
429
- runtime_version_original=sys_runtime_ver_original,
430
- compute_capability=dev_cc,
431
- cores=dev_cores,
432
- cores_utilization=0,
433
- memory=dev_mem,
434
- memory_used=dev_mem_used,
435
- memory_utilization=get_utilization(dev_mem_used, dev_mem),
436
- temperature=dev_temp,
437
- power=dev_power,
438
- power_used=dev_power_used,
439
- appendix={
440
- "arch_family": _get_arch_family(dev_cc_t),
441
- "vgpu": False,
442
- "mig": True,
443
- "bdf": dev_bdf,
444
- },
445
- ),
446
- )
447
253
  except pynvml.NVMLError:
448
254
  debug_log_exception(logger, "Failed to fetch devices")
449
255
  raise
@@ -792,141 +598,6 @@ def _get_arch_family(dev_cc_t: list[int]) -> str:
792
598
  return "Unknown"
793
599
 
794
600
 
795
- def _get_gpu_instance_slice(dev_gi_prf_id: int) -> int:
796
- """
797
- Get the number of slice for a given GPU Instance Profile ID.
798
-
799
- Args:
800
- dev_gi_prf_id:
801
- The GPU Instance Profile ID.
802
-
803
- Returns:
804
- The number of slices.
805
-
806
- """
807
- match dev_gi_prf_id:
808
- case (
809
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE
810
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_REV1
811
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_REV2
812
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_GFX
813
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_NO_ME
814
- | pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_ALL_ME
815
- ):
816
- return 1
817
- case (
818
- pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE
819
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_REV1
820
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_GFX
821
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_NO_ME
822
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_ALL_ME
823
- ):
824
- return 2
825
- case pynvml.NVML_GPU_INSTANCE_PROFILE_3_SLICE:
826
- return 3
827
- case (
828
- pynvml.NVML_GPU_INSTANCE_PROFILE_4_SLICE
829
- | pynvml.NVML_GPU_INSTANCE_PROFILE_4_SLICE_GFX
830
- ):
831
- return 4
832
- case pynvml.NVML_GPU_INSTANCE_PROFILE_6_SLICE:
833
- return 6
834
- case pynvml.NVML_GPU_INSTANCE_PROFILE_7_SLICE:
835
- return 7
836
- case pynvml.NVML_GPU_INSTANCE_PROFILE_8_SLICE:
837
- return 8
838
-
839
- msg = f"Invalid GPU Instance Profile ID: {dev_gi_prf_id}"
840
- raise AttributeError(msg)
841
-
842
-
843
- def _get_compute_instance_slice(dev_ci_prf_id: int) -> int:
844
- """
845
- Get the number of slice for a given Compute Instance Profile ID.
846
-
847
- Args:
848
- dev_ci_prf_id:
849
- The Compute Instance Profile ID.
850
-
851
- Returns:
852
- The number of slices.
853
-
854
- """
855
- match dev_ci_prf_id:
856
- case (
857
- pynvml.NVML_COMPUTE_INSTANCE_PROFILE_1_SLICE
858
- | pynvml.NVML_COMPUTE_INSTANCE_PROFILE_1_SLICE_REV1
859
- ):
860
- return 1
861
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_2_SLICE:
862
- return 2
863
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_3_SLICE:
864
- return 3
865
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_4_SLICE:
866
- return 4
867
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_6_SLICE:
868
- return 6
869
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_7_SLICE:
870
- return 7
871
- case pynvml.NVML_COMPUTE_INSTANCE_PROFILE_8_SLICE:
872
- return 8
873
-
874
- msg = f"Invalid Compute Instance Profile ID: {dev_ci_prf_id}"
875
- raise AttributeError(msg)
876
-
877
-
878
- def _get_gpu_instance_attrs(dev_gi_prf_id: int) -> str:
879
- """
880
- Get attributes for a given GPU Instance Profile ID.
881
-
882
- Args:
883
- dev_gi_prf_id:
884
- The GPU Instance Profile ID.
885
-
886
- Returns:
887
- A string representing the attributes, or an empty string if none.
888
-
889
- """
890
- match dev_gi_prf_id:
891
- case (
892
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_REV1
893
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_REV1
894
- ):
895
- return "me"
896
- case (
897
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_ALL_ME
898
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_ALL_ME
899
- ):
900
- return "me.all"
901
- case (
902
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_GFX
903
- | pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_GFX
904
- | pynvml.NVML_GPU_INSTANCE_PROFILE_4_SLICE_GFX
905
- ):
906
- return "gfx"
907
- return ""
908
-
909
-
910
- def _get_gpu_instance_negattrs(dev_gi_prf_id) -> str:
911
- """
912
- Get negative attributes for a given GPU Instance Profile ID.
913
-
914
- Args:
915
- dev_gi_prf_id:
916
- The GPU Instance Profile ID.
917
-
918
- Returns:
919
- A string representing the negative attributes, or an empty string if none.
920
-
921
- """
922
- if dev_gi_prf_id in [
923
- pynvml.NVML_GPU_INSTANCE_PROFILE_1_SLICE_NO_ME,
924
- pynvml.NVML_GPU_INSTANCE_PROFILE_2_SLICE_NO_ME,
925
- ]:
926
- return "me"
927
- return ""
928
-
929
-
930
601
  def _is_vgpu(dev_config: bytes) -> bool:
931
602
  """
932
603
  Determine if the device is a vGPU based on its PCI configuration space.
@@ -960,46 +631,3 @@ def _is_vgpu(dev_config: bytes) -> bool:
960
631
  # Check for vGPU signature,
961
632
  # which is either 0x56 (NVIDIA vGPU) or 0x46 (NVIDIA GRID).
962
633
  return dev_cap[3] == 0x56 or dev_cap[4] == 0x46
963
-
964
-
965
- def _get_mig_minors() -> dict[tuple, int] | None:
966
- """
967
- Get the minor mapping for MIG capability devices.
968
-
969
- Returns:
970
- A dict mapping (gpu_id, gi_id, ci_id) to minor number,
971
- or None if not supported.
972
-
973
- """
974
- mig_minors_path = Path("/proc/driver/nvidia-caps/mig-minors")
975
- if not mig_minors_path.exists():
976
- return None
977
-
978
- ret = {}
979
- for _line in mig_minors_path.read_text(encoding="utf-8").splitlines():
980
- line = _line.strip()
981
- if not line:
982
- continue
983
-
984
- # Scan lines like:
985
- # gpu%d/gi%d/ci%d/access %d
986
- m = re.match(r"gpu(\d+)/gi(\d+)/ci(\d+)/access (\d+)", line)
987
- if m:
988
- gpu_id = int(m.group(1))
989
- gi_id = int(m.group(2))
990
- ci_id = int(m.group(3))
991
- minor = int(m.group(4))
992
- ret[(gpu_id, gi_id, ci_id)] = minor
993
- continue
994
-
995
- # Scan lines like:
996
- # gpu%d/gi%d/access %d
997
- m = re.match(r"gpu(\d+)/gi(\d+)/access (\d+)", line)
998
- if m:
999
- gpu_id = int(m.group(1))
1000
- gi_id = int(m.group(2))
1001
- minor = int(m.group(3))
1002
- ret[(gpu_id, gi_id, None)] = minor
1003
- continue
1004
-
1005
- return ret
@@ -1 +0,0 @@
1
- git_commit = "480e0fd"