gpustack-runtime 0.2.2.post4__tar.gz → 0.2.2.post6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/PKG-INFO +1 -1
  2. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/_version.py +2 -2
  3. gpustack_runtime-0.2.2.post6/gpustack_runtime/_version_appendix.py +1 -0
  4. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/__types__.py +18 -3
  5. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/kuberentes.py +9 -0
  6. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/__init__.py +37 -0
  7. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/__types__.py +36 -0
  8. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/nvidia.py +183 -1
  9. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/thead.py +257 -205
  10. gpustack_runtime-0.2.2.post6/tests/gpustack_runtime/detector/test_mig_devices.py +116 -0
  11. gpustack_runtime-0.2.2.post4/gpustack_runtime/_version_appendix.py +0 -1
  12. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.codespelldict +0 -0
  13. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.codespellrc +0 -0
  14. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.dockerignore +0 -0
  15. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.gitattributes +0 -0
  16. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.gitignore +0 -0
  17. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.pre-commit-config.yaml +0 -0
  18. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.python-version +0 -0
  19. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/LICENSE +0 -0
  20. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/Makefile +0 -0
  21. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/README.md +0 -0
  22. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/deploy/manifests/docker-compose.yaml +0 -0
  23. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/deploy/manifests/kubernetes.yaml +0 -0
  24. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/docs/index.md +0 -0
  25. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/docs/modules/gpustack_runtime.deployer.md +0 -0
  26. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/docs/modules/gpustack_runtime.detector.md +0 -0
  27. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/docs/modules/gpustack_runtime.md +0 -0
  28. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/__init__.py +0 -0
  29. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/__main__.py +0 -0
  30. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/_version.pyi +0 -0
  31. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/__init__.py +0 -0
  32. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/__types__.py +0 -0
  33. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/deployer.py +0 -0
  34. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/detector.py +0 -0
  35. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/images.py +0 -0
  36. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/__init__.py +0 -0
  37. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/__patches__.py +0 -0
  38. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/__utils__.py +0 -0
  39. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/__init__.py +0 -0
  40. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/__types__.py +0 -0
  41. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/__utils__.py +0 -0
  42. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/amd.py +0 -0
  43. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/ascend.py +0 -0
  44. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/hygon.py +0 -0
  45. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/iluvatar.py +0 -0
  46. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/metax.py +0 -0
  47. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/thead.py +0 -0
  48. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/docker.py +0 -0
  49. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/k8s/devicemanager/__init__.py +0 -0
  50. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/podman.py +0 -0
  51. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/__utils__.py +0 -0
  52. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/amd.py +0 -0
  53. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/ascend.py +0 -0
  54. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/cambricon.py +0 -0
  55. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/hygon.py +0 -0
  56. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/iluvatar.py +0 -0
  57. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/metax.py +0 -0
  58. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/mthreads.py +0 -0
  59. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyamdgpu/__init__.py +0 -0
  60. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyamdsmi/__init__.py +0 -0
  61. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pydcmi/__init__.py +0 -0
  62. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyhgml/__init__.py +0 -0
  63. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyhgml/libhgml.so +0 -0
  64. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyhgml/libuki.so +0 -0
  65. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyhsa/__init__.py +0 -0
  66. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyixml/__init__.py +0 -0
  67. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pymtml/__init__.py +0 -0
  68. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pymxsml/__init__.py +0 -0
  69. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pynvml/__init__.py +0 -0
  70. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyrocmsmi/__init__.py +0 -0
  71. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/envs.py +0 -0
  72. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/logging.py +0 -0
  73. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/hatch.toml +0 -0
  74. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/mkdocs.yml +0 -0
  75. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/pack/Dockerfile +0 -0
  76. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/pack/Dockerfile.dummy +0 -0
  77. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/pyproject.toml +0 -0
  78. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/pytest.ini +0 -0
  79. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/ruff.toml +0 -0
  80. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/__init__.py +0 -0
  81. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_compare_versions.json +0 -0
  82. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_correct_runner_image.json +0 -0
  83. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json.json +0 -0
  84. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_jsons.json +0 -0
  85. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_yamls.yaml +0 -0
  86. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_json.json +0 -0
  87. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_yaml.yaml +0 -0
  88. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_nginx_entrypoint.sh +0 -0
  89. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/test_utils.py +0 -0
  90. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/test_workload_status.py +0 -0
  91. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/fixtures/__init__.py +0 -0
  92. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/README.md +0 -0
  93. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi300x.json +0 -0
  94. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi308x.json +0 -0
  95. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_amd_rx7800xt.json +0 -0
  96. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_ascend_310p3.json +0 -0
  97. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_ascend_910b2.json +0 -0
  98. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_hygon_k100ai.json +0 -0
  99. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_metax_c500.json +0 -0
  100. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_gb10.json +0 -0
  101. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100.json +0 -0
  102. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100_mig.json +0 -0
  103. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h200.json +0 -0
  104. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4080super.json +0 -0
  105. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4090d.json +0 -0
  106. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx5090d.json +0 -0
  107. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_thead_ppu.json +0 -0
  108. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi300x.json +0 -0
  109. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi308x.json +0 -0
  110. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_amd_rx7800xt.json +0 -0
  111. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_ascend_310p3.json +0 -0
  112. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_ascend_910b2.json +0 -0
  113. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_hygon_k100ai.json +0 -0
  114. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_metax_c500.json +0 -0
  115. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_mthreads_s5000.json +0 -0
  116. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100.json +0 -0
  117. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100_mig.json +0 -0
  118. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h200.json +0 -0
  119. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4080super.json +0 -0
  120. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4090d.json +0 -0
  121. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx5090d.json +0 -0
  122. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_thead_ppu.json +0 -0
  123. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_amd.py +0 -0
  124. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_ascend.py +0 -0
  125. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_cambricon.py +0 -0
  126. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_detector_utils.py +0 -0
  127. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_hygon.py +0 -0
  128. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_iluvatar.py +0 -0
  129. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_metax.py +0 -0
  130. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_mthreads.py +0 -0
  131. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_nvidia.py +0 -0
  132. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/uv.lock +0 -0
  133. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/uv.toml +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gpustack-runtime
3
- Version: 0.2.2.post4
3
+ Version: 0.2.2.post6
4
4
  Summary: GPUStack Runtime is library for detecting GPU resources and launching GPU workloads.
5
5
  Project-URL: Homepage, https://github.com/gpustack/runtime
6
6
  Project-URL: Bug Tracker, https://github.com/gpustack/gpustack/issues
@@ -27,8 +27,8 @@ version_tuple: VERSION_TUPLE
27
27
  __commit_id__: COMMIT_ID
28
28
  commit_id: COMMIT_ID
29
29
 
30
- __version__ = version = '0.2.2.post4'
31
- __version_tuple__ = version_tuple = (0, 2, 2, 'post4')
30
+ __version__ = version = '0.2.2.post6'
31
+ __version_tuple__ = version_tuple = (0, 2, 2, 'post6')
32
32
  try:
33
33
  from ._version_appendix import git_commit
34
34
  __commit_id__ = commit_id = git_commit
@@ -0,0 +1 @@
1
+ git_commit = "79ee112"
@@ -16,6 +16,7 @@ from .. import envs
16
16
  from ..detector import (
17
17
  ManufacturerEnum,
18
18
  detect_devices,
19
+ expand_mig_devices,
19
20
  group_devices_by_manufacturer,
20
21
  manufacturer_to_backend,
21
22
  )
@@ -1373,9 +1374,11 @@ class Deployer(ABC):
1373
1374
 
1374
1375
  self._materials = {}
1375
1376
 
1376
- group_devices = group_devices_by_manufacturer(
1377
- detect_devices(fast=False),
1378
- )
1377
+ devices = detect_devices(fast=False)
1378
+ if self.allowed_mig_devices:
1379
+ devices = expand_mig_devices(devices)
1380
+
1381
+ group_devices = group_devices_by_manufacturer(devices)
1379
1382
 
1380
1383
  if group_devices:
1381
1384
  for manu, devs in group_devices.items():
@@ -1698,6 +1701,18 @@ class Deployer(ABC):
1698
1701
  """
1699
1702
  return True
1700
1703
 
1704
+ @property
1705
+ def allowed_mig_devices(self) -> bool:
1706
+ """
1707
+ Return whether the deployer addresses the MIG devices of a MIG-partitioned
1708
+ card, instead of the card itself.
1709
+
1710
+ Returns:
1711
+ True if addressed, False otherwise.
1712
+
1713
+ """
1714
+ return True
1715
+
1701
1716
  def close(self):
1702
1717
  if self._pool:
1703
1718
  self._pool.shutdown(cancel_futures=True)
@@ -1308,6 +1308,15 @@ class KubernetesDeployer(EndoscopicDeployer):
1308
1308
  def allowed_runtime_uuid_values(self) -> bool:
1309
1309
  return get_resource_injection_policy() != "kdp"
1310
1310
 
1311
+ @property
1312
+ def allowed_mig_devices(self) -> bool:
1313
+ # On Kubernetes a partitioner (e.g. the GPUStack Operator's device
1314
+ # manager) owns MIG: it creates and destroys the instances on demand,
1315
+ # so the card is the item to hold on to, and a pod asks for a slice of
1316
+ # it by resource rather than by addressing an instance that may not
1317
+ # outlive the request.
1318
+ return False
1319
+
1311
1320
  def _prepare_mirrored_deployment(self):
1312
1321
  """
1313
1322
  Prepare for mirrored deployment.
@@ -291,6 +291,42 @@ def filter_devices_by_manufacturer(
291
291
  return [dev for dev in devices or [] if dev.manufacturer == manufacturer]
292
292
 
293
293
 
294
+ def expand_mig_devices(devices: Devices | None) -> Devices:
295
+ """
296
+ Replace every MIG-partitioned card with its MIG devices.
297
+
298
+ Detection reports the physical card and carries its MIG devices in the
299
+ `mig_devices` appendix, because where a partitioner (e.g. the GPUStack
300
+ Operator's device manager) owns MIG, the instances come and go and the
301
+ card is the only stable inventory item. Callers that address MIG devices
302
+ themselves need them as devices instead: a MIG-enabled card cannot run a
303
+ workload, only its instances can.
304
+
305
+ A MIG-enabled card with no instances is kept as-is: there is nothing to
306
+ address yet, and dropping it would make the card disappear from the
307
+ inventory.
308
+
309
+ Args:
310
+ devices:
311
+ A list of devices to be expanded.
312
+
313
+ Returns:
314
+ A list of devices where MIG-partitioned cards are substituted by their
315
+ MIG devices.
316
+
317
+ """
318
+ expanded: Devices = []
319
+ for dev in devices or []:
320
+ mig_devs = (dev.appendix or {}).get("mig_devices")
321
+ if not mig_devs:
322
+ expanded.append(dev)
323
+ continue
324
+ expanded.extend(
325
+ Device(manufacturer=dev.manufacturer, **mig_dev) for mig_dev in mig_devs
326
+ )
327
+ return expanded
328
+
329
+
294
330
  __all__ = [
295
331
  "Device",
296
332
  "DeviceMemoryStatusEnum",
@@ -302,6 +338,7 @@ __all__ = [
302
338
  "backend_to_manufacturer",
303
339
  "detect_backend",
304
340
  "detect_devices",
341
+ "expand_mig_devices",
305
342
  "filter_devices_by_manufacturer",
306
343
  "get_devices_topologies",
307
344
  "group_devices_by_manufacturer",
@@ -461,6 +461,42 @@ def reduce_devices_distances(
461
461
  return result
462
462
 
463
463
 
464
+ def index_mig_devices(
465
+ devices: Devices,
466
+ mig_devices: dict[int, list[dict]],
467
+ slots: int,
468
+ ) -> None:
469
+ """
470
+ Number the given cards' MIG devices in place.
471
+
472
+ A MIG device carries no driver-side inventory index, so its index is
473
+ synthetic: every card owns a block of `slots` indexes, and the blocks
474
+ start above the largest index the physical cards report. The reported
475
+ index may be the card's minor number
476
+ (`GPUSTACK_RUNTIME_DETECT_PHYSICAL_INDEX_PRIORITY`), which is not bound by
477
+ the card count, hence the offset is measured from the reported indexes
478
+ instead of the count. Sizing a block by the slots a card can host, rather
479
+ than by the MIG devices it currently has, keeps a card's numbering
480
+ independent of its neighbours: partitioning one card never renumbers
481
+ another's MIG devices.
482
+
483
+ Args:
484
+ devices:
485
+ The detected physical cards, already indexed.
486
+ mig_devices:
487
+ The MIG devices to number, keyed by the enumeration index of the
488
+ card hosting them. Each entry's `index` is the driver slot it was
489
+ found at, which the block offset is added to.
490
+ slots:
491
+ The number of MIG devices a card can host, i.e. the block size.
492
+
493
+ """
494
+ base = max((dev.index for dev in devices), default=-1) + 1
495
+ for dev_idx, migs in mig_devices.items():
496
+ for mig in migs:
497
+ mig["index"] += base + dev_idx * slots
498
+
499
+
464
500
  class Detector(ABC):
465
501
  """
466
502
  Base class for all detectors.
@@ -11,7 +11,14 @@ from functools import lru_cache
11
11
  from .. import envs
12
12
  from ..logging import debug_log_exception, debug_log_warning
13
13
  from . import DeviceMemoryStatusEnum, Topology, pynvml
14
- from .__types__ import Detector, Device, Devices, ManufacturerEnum, TopologyDistanceEnum
14
+ from .__types__ import (
15
+ Detector,
16
+ Device,
17
+ Devices,
18
+ ManufacturerEnum,
19
+ TopologyDistanceEnum,
20
+ index_mig_devices,
21
+ )
15
22
  from .__utils__ import (
16
23
  PCIDevice,
17
24
  bitmask_to_str,
@@ -113,6 +120,13 @@ class NVIDIADetector(Detector):
113
120
  sys_runtime_ver_original,
114
121
  )
115
122
 
123
+ # MIG devices of every MIG-enabled card, keyed by the card's
124
+ # enumeration index, and the largest number of MIG devices a card
125
+ # can host: both are needed to number them once every card is
126
+ # detected, see index_mig_devices.
127
+ devs_mig_devices: dict[int, list[dict]] = {}
128
+ devs_mig_slots = 0
129
+
116
130
  dev_count = pynvml.nvmlDeviceGetCount()
117
131
  for dev_idx in range(dev_count):
118
132
  dev = pynvml.nvmlDeviceGetHandleByIndex(dev_idx)
@@ -218,6 +232,27 @@ class NVIDIADetector(Detector):
218
232
  "mig": dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE,
219
233
  "bdf": dev_bdf,
220
234
  }
235
+ if dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE:
236
+ dev_mig_slots = 0
237
+ with contextlib.suppress(pynvml.NVMLError):
238
+ dev_mig_slots = pynvml.nvmlDeviceGetMaxMigDeviceCount(dev)
239
+ devs_mig_slots = max(devs_mig_slots, dev_mig_slots)
240
+ dev_mig_devices = _get_mig_devices(
241
+ dev,
242
+ dev_mig_slots,
243
+ dev_cc_t,
244
+ sys_driver_ver,
245
+ sys_runtime_ver,
246
+ sys_runtime_ver_original,
247
+ dev_cc,
248
+ dev_temp,
249
+ dev_power,
250
+ dev_power_used,
251
+ dev_bdf,
252
+ dev_numa,
253
+ )
254
+ dev_appendix["mig_devices"] = dev_mig_devices
255
+ devs_mig_devices[dev_idx] = dev_mig_devices
221
256
  if dev_numa:
222
257
  dev_appendix["numa"] = dev_numa
223
258
 
@@ -246,6 +281,8 @@ class NVIDIADetector(Detector):
246
281
  appendix=dev_appendix,
247
282
  ),
248
283
  )
284
+
285
+ index_mig_devices(ret, devs_mig_devices, devs_mig_slots)
249
286
  except pynvml.NVMLError:
250
287
  debug_log_exception(logger, "Failed to fetch devices")
251
288
  raise
@@ -558,6 +595,151 @@ def _get_links_state(
558
595
  }
559
596
 
560
597
 
598
+ def _get_mig_devices(
599
+ dev,
600
+ dev_mig_slots: int,
601
+ dev_cc_t,
602
+ sys_driver_ver,
603
+ sys_runtime_ver,
604
+ sys_runtime_ver_original,
605
+ dev_cc,
606
+ dev_temp,
607
+ dev_power,
608
+ dev_power_used,
609
+ dev_bdf: str,
610
+ dev_numa,
611
+ ) -> list[dict]:
612
+ """
613
+ Enumerate the card's current MIG devices with the same detail a plain
614
+ device carries (profile name, uuid, compute/memory utilization, memory
615
+ health, temperature and power), returned as appendix entries of the
616
+ physical card rather than standalone devices. Empty when MIG is enabled
617
+ but no GPU instances exist yet.
618
+
619
+ Each entry's `index` is the driver slot the MIG device was found at:
620
+ index_mig_devices turns it into the device index once every card is
621
+ detected.
622
+ """
623
+ ret: list[dict] = []
624
+ with contextlib.suppress(pynvml.NVMLError):
625
+ for mdev_idx in range(dev_mig_slots):
626
+ mdev = None
627
+ with contextlib.suppress(pynvml.NVMLError):
628
+ mdev = pynvml.nvmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
629
+ if not mdev:
630
+ continue
631
+
632
+ mdev_uuid = pynvml.nvmlDeviceGetUUID(mdev)
633
+
634
+ mdev_mem = 0
635
+ mdev_mem_used = 0
636
+ mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
637
+ with contextlib.suppress(pynvml.NVMLError):
638
+ mdev_mem_info = pynvml.nvmlDeviceGetMemoryInfo(mdev)
639
+ mdev_mem = byte_to_mebibyte(mdev_mem_info.total)
640
+ mdev_mem_used = byte_to_mebibyte(mdev_mem_info.used)
641
+ if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
642
+ mdev_mem_ecc_errors = pynvml.nvmlDeviceGetMemoryErrorCounter(
643
+ mdev,
644
+ pynvml.NVML_MEMORY_ERROR_TYPE_UNCORRECTED,
645
+ pynvml.NVML_AGGREGATE_ECC,
646
+ pynvml.NVML_MEMORY_LOCATION_SRAM,
647
+ )
648
+ if mdev_mem_ecc_errors > 0:
649
+ mdev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
650
+
651
+ mdev_appendix = {
652
+ "arch_family": _get_arch_family(dev_cc_t),
653
+ "vgpu": True,
654
+ "sliced": True,
655
+ "mig": True,
656
+ "bdf": dev_bdf,
657
+ }
658
+ if dev_numa:
659
+ mdev_appendix["numa"] = dev_numa
660
+
661
+ mdev_gi_id = pynvml.nvmlDeviceGetGpuInstanceId(mdev)
662
+ mdev_appendix["gpu_instance_id"] = mdev_gi_id
663
+ mdev_ci_id = pynvml.nvmlDeviceGetComputeInstanceId(mdev)
664
+ mdev_appendix["compute_instance_id"] = mdev_ci_id
665
+
666
+ mdev_cores_util = _get_sm_util_from_gpm_metrics(dev, mdev_gi_id)
667
+
668
+ mdev_name = ""
669
+ mdev_cores = None
670
+ mdev_gi = pynvml.nvmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
671
+ mdev_ci = pynvml.nvmlGpuInstanceGetComputeInstanceById(
672
+ mdev_gi,
673
+ mdev_ci_id,
674
+ )
675
+ mdev_gi_info = pynvml.nvmlGpuInstanceGetInfo(mdev_gi)
676
+ mdev_ci_info = pynvml.nvmlComputeInstanceGetInfo(mdev_ci)
677
+ for dev_gi_prf_id in range(pynvml.NVML_GPU_INSTANCE_PROFILE_COUNT):
678
+ try:
679
+ dev_gi_prf = pynvml.nvmlDeviceGetGpuInstanceProfileInfo(
680
+ dev,
681
+ dev_gi_prf_id,
682
+ )
683
+ if dev_gi_prf.id != mdev_gi_info.profileId:
684
+ continue
685
+ except pynvml.NVMLError:
686
+ continue
687
+
688
+ gi_mem = round(math.ceil(dev_gi_prf.memorySizeMB >> 10))
689
+ gi_prf_name = getattr(dev_gi_prf, "name", None)
690
+ mdev_name = (
691
+ gi_prf_name.removeprefix("MIG ")
692
+ if gi_prf_name
693
+ else f"{dev_gi_prf.sliceCount}g.{gi_mem}gb"
694
+ )
695
+
696
+ for dev_ci_prf_id in range(
697
+ pynvml.NVML_COMPUTE_INSTANCE_PROFILE_COUNT,
698
+ ):
699
+ for dev_cig_prf_id in range(
700
+ pynvml.NVML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
701
+ ):
702
+ try:
703
+ mdev_ci_prf = (
704
+ pynvml.nvmlGpuInstanceGetComputeInstanceProfileInfo(
705
+ mdev_gi,
706
+ dev_ci_prf_id,
707
+ dev_cig_prf_id,
708
+ )
709
+ )
710
+ if mdev_ci_prf.id != mdev_ci_info.profileId:
711
+ continue
712
+ except pynvml.NVMLError:
713
+ continue
714
+ mdev_cores = mdev_ci_prf.multiprocessorCount
715
+ break
716
+
717
+ break
718
+
719
+ ret.append(
720
+ {
721
+ "index": mdev_idx,
722
+ "name": mdev_name,
723
+ "uuid": mdev_uuid,
724
+ "driver_version": sys_driver_ver,
725
+ "runtime_version": sys_runtime_ver,
726
+ "runtime_version_original": sys_runtime_ver_original,
727
+ "compute_capability": dev_cc,
728
+ "cores": mdev_cores,
729
+ "cores_utilization": mdev_cores_util,
730
+ "memory": mdev_mem,
731
+ "memory_used": mdev_mem_used,
732
+ "memory_utilization": get_utilization(mdev_mem_used, mdev_mem),
733
+ "memory_status": mdev_mem_status,
734
+ "temperature": dev_temp,
735
+ "power": dev_power,
736
+ "power_used": dev_power_used,
737
+ "appendix": mdev_appendix,
738
+ },
739
+ )
740
+ return ret
741
+
742
+
561
743
  def _get_arch_family(dev_cc_t: list[int]) -> str:
562
744
  """
563
745
  Get the architecture family based on the CUDA compute capability.