gpustack-runtime 0.2.2.post4__tar.gz → 0.2.2.post5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/PKG-INFO +1 -1
  2. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/_version.py +2 -2
  3. gpustack_runtime-0.2.2.post5/gpustack_runtime/_version_appendix.py +1 -0
  4. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/nvidia.py +158 -0
  5. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/thead.py +241 -205
  6. gpustack_runtime-0.2.2.post4/gpustack_runtime/_version_appendix.py +0 -1
  7. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.codespelldict +0 -0
  8. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.codespellrc +0 -0
  9. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.dockerignore +0 -0
  10. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.gitattributes +0 -0
  11. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.gitignore +0 -0
  12. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.pre-commit-config.yaml +0 -0
  13. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.python-version +0 -0
  14. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/LICENSE +0 -0
  15. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/Makefile +0 -0
  16. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/README.md +0 -0
  17. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/deploy/manifests/docker-compose.yaml +0 -0
  18. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/deploy/manifests/kubernetes.yaml +0 -0
  19. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/docs/index.md +0 -0
  20. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/docs/modules/gpustack_runtime.deployer.md +0 -0
  21. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/docs/modules/gpustack_runtime.detector.md +0 -0
  22. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/docs/modules/gpustack_runtime.md +0 -0
  23. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/__init__.py +0 -0
  24. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/__main__.py +0 -0
  25. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/_version.pyi +0 -0
  26. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/__init__.py +0 -0
  27. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/__types__.py +0 -0
  28. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/deployer.py +0 -0
  29. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/detector.py +0 -0
  30. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/images.py +0 -0
  31. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__init__.py +0 -0
  32. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__patches__.py +0 -0
  33. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__types__.py +0 -0
  34. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__utils__.py +0 -0
  35. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/__init__.py +0 -0
  36. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/__types__.py +0 -0
  37. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/__utils__.py +0 -0
  38. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/amd.py +0 -0
  39. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/ascend.py +0 -0
  40. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/hygon.py +0 -0
  41. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/iluvatar.py +0 -0
  42. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/metax.py +0 -0
  43. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/thead.py +0 -0
  44. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/docker.py +0 -0
  45. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/k8s/devicemanager/__init__.py +0 -0
  46. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/kuberentes.py +0 -0
  47. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/podman.py +0 -0
  48. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/__init__.py +0 -0
  49. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/__types__.py +0 -0
  50. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/__utils__.py +0 -0
  51. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/amd.py +0 -0
  52. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/ascend.py +0 -0
  53. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/cambricon.py +0 -0
  54. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/hygon.py +0 -0
  55. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/iluvatar.py +0 -0
  56. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/metax.py +0 -0
  57. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/mthreads.py +0 -0
  58. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyamdgpu/__init__.py +0 -0
  59. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyamdsmi/__init__.py +0 -0
  60. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pydcmi/__init__.py +0 -0
  61. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyhgml/__init__.py +0 -0
  62. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyhgml/libhgml.so +0 -0
  63. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyhgml/libuki.so +0 -0
  64. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyhsa/__init__.py +0 -0
  65. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyixml/__init__.py +0 -0
  66. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pymtml/__init__.py +0 -0
  67. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pymxsml/__init__.py +0 -0
  68. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pynvml/__init__.py +0 -0
  69. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyrocmsmi/__init__.py +0 -0
  70. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/envs.py +0 -0
  71. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/logging.py +0 -0
  72. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/hatch.toml +0 -0
  73. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/mkdocs.yml +0 -0
  74. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/pack/Dockerfile +0 -0
  75. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/pack/Dockerfile.dummy +0 -0
  76. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/pyproject.toml +0 -0
  77. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/pytest.ini +0 -0
  78. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/ruff.toml +0 -0
  79. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/__init__.py +0 -0
  80. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_compare_versions.json +0 -0
  81. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_correct_runner_image.json +0 -0
  82. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json.json +0 -0
  83. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_jsons.json +0 -0
  84. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_yamls.yaml +0 -0
  85. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_json.json +0 -0
  86. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_yaml.yaml +0 -0
  87. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_nginx_entrypoint.sh +0 -0
  88. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/test_utils.py +0 -0
  89. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/test_workload_status.py +0 -0
  90. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/fixtures/__init__.py +0 -0
  91. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/README.md +0 -0
  92. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi300x.json +0 -0
  93. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi308x.json +0 -0
  94. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_amd_rx7800xt.json +0 -0
  95. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_ascend_310p3.json +0 -0
  96. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_ascend_910b2.json +0 -0
  97. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_hygon_k100ai.json +0 -0
  98. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_metax_c500.json +0 -0
  99. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_gb10.json +0 -0
  100. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100.json +0 -0
  101. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100_mig.json +0 -0
  102. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h200.json +0 -0
  103. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4080super.json +0 -0
  104. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4090d.json +0 -0
  105. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx5090d.json +0 -0
  106. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_thead_ppu.json +0 -0
  107. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi300x.json +0 -0
  108. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi308x.json +0 -0
  109. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_amd_rx7800xt.json +0 -0
  110. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_ascend_310p3.json +0 -0
  111. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_ascend_910b2.json +0 -0
  112. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_hygon_k100ai.json +0 -0
  113. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_metax_c500.json +0 -0
  114. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_mthreads_s5000.json +0 -0
  115. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100.json +0 -0
  116. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100_mig.json +0 -0
  117. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h200.json +0 -0
  118. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4080super.json +0 -0
  119. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4090d.json +0 -0
  120. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx5090d.json +0 -0
  121. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_thead_ppu.json +0 -0
  122. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_amd.py +0 -0
  123. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_ascend.py +0 -0
  124. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_cambricon.py +0 -0
  125. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_detector_utils.py +0 -0
  126. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_hygon.py +0 -0
  127. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_iluvatar.py +0 -0
  128. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_metax.py +0 -0
  129. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_mthreads.py +0 -0
  130. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_nvidia.py +0 -0
  131. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/uv.lock +0 -0
  132. {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/uv.toml +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gpustack-runtime
3
- Version: 0.2.2.post4
3
+ Version: 0.2.2.post5
4
4
  Summary: GPUStack Runtime is library for detecting GPU resources and launching GPU workloads.
5
5
  Project-URL: Homepage, https://github.com/gpustack/runtime
6
6
  Project-URL: Bug Tracker, https://github.com/gpustack/gpustack/issues
@@ -27,8 +27,8 @@ version_tuple: VERSION_TUPLE
27
27
  __commit_id__: COMMIT_ID
28
28
  commit_id: COMMIT_ID
29
29
 
30
- __version__ = version = '0.2.2.post4'
31
- __version_tuple__ = version_tuple = (0, 2, 2, 'post4')
30
+ __version__ = version = '0.2.2.post5'
31
+ __version_tuple__ = version_tuple = (0, 2, 2, 'post5')
32
32
  try:
33
33
  from ._version_appendix import git_commit
34
34
  __commit_id__ = commit_id = git_commit
@@ -0,0 +1 @@
1
+ git_commit = "fd63e6b"
@@ -218,6 +218,22 @@ class NVIDIADetector(Detector):
218
218
  "mig": dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE,
219
219
  "bdf": dev_bdf,
220
220
  }
221
+ if dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE:
222
+ dev_appendix["mig_devices"] = _get_mig_devices(
223
+ dev,
224
+ dev_idx,
225
+ dev_count,
226
+ dev_cc_t,
227
+ sys_driver_ver,
228
+ sys_runtime_ver,
229
+ sys_runtime_ver_original,
230
+ dev_cc,
231
+ dev_temp,
232
+ dev_power,
233
+ dev_power_used,
234
+ dev_bdf,
235
+ dev_numa,
236
+ )
221
237
  if dev_numa:
222
238
  dev_appendix["numa"] = dev_numa
223
239
 
@@ -558,6 +574,148 @@ def _get_links_state(
558
574
  }
559
575
 
560
576
 
577
+ def _get_mig_devices(
578
+ dev,
579
+ dev_idx: int,
580
+ dev_count: int,
581
+ dev_cc_t,
582
+ sys_driver_ver,
583
+ sys_runtime_ver,
584
+ sys_runtime_ver_original,
585
+ dev_cc,
586
+ dev_temp,
587
+ dev_power,
588
+ dev_power_used,
589
+ dev_bdf: str,
590
+ dev_numa,
591
+ ) -> list[dict]:
592
+ """
593
+ Enumerate the card's current MIG devices with the same detail a plain
594
+ device carries (profile name, uuid, compute/memory utilization, memory
595
+ health, temperature and power), returned as appendix entries of the
596
+ physical card rather than standalone devices. Empty when MIG is enabled
597
+ but no GPU instances exist yet.
598
+ """
599
+ ret: list[dict] = []
600
+ with contextlib.suppress(pynvml.NVMLError):
601
+ for mdev_idx in range(pynvml.nvmlDeviceGetMaxMigDeviceCount(dev)):
602
+ mdev = None
603
+ with contextlib.suppress(pynvml.NVMLError):
604
+ mdev = pynvml.nvmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
605
+ if not mdev:
606
+ continue
607
+
608
+ mdev_uuid = pynvml.nvmlDeviceGetUUID(mdev)
609
+
610
+ mdev_mem = 0
611
+ mdev_mem_used = 0
612
+ mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
613
+ with contextlib.suppress(pynvml.NVMLError):
614
+ mdev_mem_info = pynvml.nvmlDeviceGetMemoryInfo(mdev)
615
+ mdev_mem = byte_to_mebibyte(mdev_mem_info.total)
616
+ mdev_mem_used = byte_to_mebibyte(mdev_mem_info.used)
617
+ if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
618
+ mdev_mem_ecc_errors = pynvml.nvmlDeviceGetMemoryErrorCounter(
619
+ mdev,
620
+ pynvml.NVML_MEMORY_ERROR_TYPE_UNCORRECTED,
621
+ pynvml.NVML_AGGREGATE_ECC,
622
+ pynvml.NVML_MEMORY_LOCATION_SRAM,
623
+ )
624
+ if mdev_mem_ecc_errors > 0:
625
+ mdev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
626
+
627
+ mdev_appendix = {
628
+ "arch_family": _get_arch_family(dev_cc_t),
629
+ "vgpu": True,
630
+ "sliced": True,
631
+ "mig": True,
632
+ "bdf": dev_bdf,
633
+ }
634
+ if dev_numa:
635
+ mdev_appendix["numa"] = dev_numa
636
+
637
+ mdev_gi_id = pynvml.nvmlDeviceGetGpuInstanceId(mdev)
638
+ mdev_appendix["gpu_instance_id"] = mdev_gi_id
639
+ mdev_ci_id = pynvml.nvmlDeviceGetComputeInstanceId(mdev)
640
+ mdev_appendix["compute_instance_id"] = mdev_ci_id
641
+
642
+ mdev_cores_util = _get_sm_util_from_gpm_metrics(dev, mdev_gi_id)
643
+
644
+ mdev_name = ""
645
+ mdev_cores = None
646
+ mdev_gi = pynvml.nvmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
647
+ mdev_ci = pynvml.nvmlGpuInstanceGetComputeInstanceById(
648
+ mdev_gi,
649
+ mdev_ci_id,
650
+ )
651
+ mdev_gi_info = pynvml.nvmlGpuInstanceGetInfo(mdev_gi)
652
+ mdev_ci_info = pynvml.nvmlComputeInstanceGetInfo(mdev_ci)
653
+ for dev_gi_prf_id in range(pynvml.NVML_GPU_INSTANCE_PROFILE_COUNT):
654
+ try:
655
+ dev_gi_prf = pynvml.nvmlDeviceGetGpuInstanceProfileInfo(
656
+ dev,
657
+ dev_gi_prf_id,
658
+ )
659
+ if dev_gi_prf.id != mdev_gi_info.profileId:
660
+ continue
661
+ except pynvml.NVMLError:
662
+ continue
663
+
664
+ gi_mem = round(math.ceil(dev_gi_prf.memorySizeMB >> 10))
665
+ gi_prf_name = getattr(dev_gi_prf, "name", None)
666
+ mdev_name = (
667
+ gi_prf_name.removeprefix("MIG ")
668
+ if gi_prf_name
669
+ else f"{dev_gi_prf.sliceCount}g.{gi_mem}gb"
670
+ )
671
+
672
+ for dev_ci_prf_id in range(
673
+ pynvml.NVML_COMPUTE_INSTANCE_PROFILE_COUNT,
674
+ ):
675
+ for dev_cig_prf_id in range(
676
+ pynvml.NVML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
677
+ ):
678
+ try:
679
+ mdev_ci_prf = (
680
+ pynvml.nvmlGpuInstanceGetComputeInstanceProfileInfo(
681
+ mdev_gi,
682
+ dev_ci_prf_id,
683
+ dev_cig_prf_id,
684
+ )
685
+ )
686
+ if mdev_ci_prf.id != mdev_ci_info.profileId:
687
+ continue
688
+ except pynvml.NVMLError:
689
+ continue
690
+ mdev_cores = mdev_ci_prf.multiprocessorCount
691
+ break
692
+
693
+ break
694
+
695
+ ret.append(
696
+ {
697
+ "index": mdev_idx + dev_count * (dev_idx + 1),
698
+ "name": mdev_name,
699
+ "uuid": mdev_uuid,
700
+ "driver_version": sys_driver_ver,
701
+ "runtime_version": sys_runtime_ver,
702
+ "runtime_version_original": sys_runtime_ver_original,
703
+ "compute_capability": dev_cc,
704
+ "cores": mdev_cores,
705
+ "cores_utilization": mdev_cores_util,
706
+ "memory": mdev_mem,
707
+ "memory_used": mdev_mem_used,
708
+ "memory_utilization": get_utilization(mdev_mem_used, mdev_mem),
709
+ "memory_status": mdev_mem_status,
710
+ "temperature": dev_temp,
711
+ "power": dev_power,
712
+ "power_used": dev_power_used,
713
+ "appendix": mdev_appendix,
714
+ },
715
+ )
716
+ return ret
717
+
718
+
561
719
  def _get_arch_family(dev_cc_t: list[int]) -> str:
562
720
  """
563
721
  Get the architecture family based on the CUDA compute capability.
@@ -162,220 +162,104 @@ class THeadDetector(Detector):
162
162
 
163
163
  dev_index = dev_idx
164
164
 
165
- # With MIG disabled, treat as a single device.
165
+ # Report the physical card, whether or not MIG is enabled.
166
+ # MIG instances are partitioned on demand by the operator's
167
+ # device-manager; they are not separate allocatable devices
168
+ # in this inventory. A MIG-enabled card is marked ``mig``
169
+ # in the appendix instead.
166
170
 
167
- if dev_mig_mode == pyhgml.HGML_DEVICE_MIG_DISABLE:
168
- dev_name = pyhgml.hgmlDeviceGetName(dev)
171
+ dev_name = pyhgml.hgmlDeviceGetName(dev)
169
172
 
170
- dev_uuid = pyhgml.hgmlDeviceGetUUID(dev)
173
+ dev_uuid = pyhgml.hgmlDeviceGetUUID(dev)
171
174
 
172
- dev_cores = None
173
- with contextlib.suppress(pyhgml.HGMLError):
174
- dev_cores = pyhgml.hgmlDeviceGetNumGpuCores(dev)
175
-
176
- dev_cores_util = None
177
- with contextlib.suppress(pyhgml.HGMLError):
178
- dev_util_rates = pyhgml.hgmlDeviceGetUtilizationRates(dev)
179
- dev_cores_util = dev_util_rates.gpu
180
- if dev_cores_util is None:
181
- debug_log_warning(
182
- logger,
183
- "Failed to get device %d cores utilization, setting to 0",
184
- dev_index,
185
- )
186
- dev_cores_util = 0
175
+ dev_cores = None
176
+ with contextlib.suppress(pyhgml.HGMLError):
177
+ dev_cores = pyhgml.hgmlDeviceGetNumGpuCores(dev)
187
178
 
188
- dev_mem = 0
189
- dev_mem_used = 0
190
- dev_mem_status = DeviceMemoryStatusEnum.HEALTHY
191
- with contextlib.suppress(pyhgml.HGMLError):
192
- dev_mem_info = pyhgml.hgmlDeviceGetMemoryInfo(dev)
193
- dev_mem = byte_to_mebibyte( # byte to MiB
194
- dev_mem_info.total,
195
- )
196
- dev_mem_used = byte_to_mebibyte( # byte to MiB
197
- dev_mem_info.used,
198
- )
199
- if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
200
- dev_mem_ecc_errors = pyhgml.hgmlDeviceGetMemoryErrorCounter(
201
- dev,
202
- pyhgml.HGML_MEMORY_ERROR_TYPE_UNCORRECTED,
203
- pyhgml.HGML_VOLATILE_ECC,
204
- pyhgml.HGML_MEMORY_LOCATION_DRAM,
205
- )
206
- if dev_mem_ecc_errors > 0:
207
- dev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
208
-
209
- dev_is_vgpu = False
210
- if dev_bdf:
211
- dev_is_vgpu = get_physical_function_by_bdf(dev_bdf) != dev_bdf
212
-
213
- dev_appendix = {
214
- "vgpu": dev_is_vgpu,
215
- "bdf": dev_bdf,
216
- }
217
- if dev_numa:
218
- dev_appendix["numa"] = dev_numa
219
-
220
- ret.append(
221
- Device(
222
- manufacturer=self.manufacturer,
223
- index=dev_index,
224
- name=dev_name,
225
- uuid=dev_uuid,
226
- driver_version=sys_driver_ver,
227
- runtime_version=sys_runtime_ver,
228
- runtime_version_original=sys_runtime_ver_original,
229
- compute_capability=dev_cc,
230
- cores=dev_cores,
231
- cores_utilization=dev_cores_util,
232
- memory=dev_mem,
233
- memory_used=dev_mem_used,
234
- memory_utilization=get_utilization(dev_mem_used, dev_mem),
235
- memory_status=dev_mem_status,
236
- temperature=dev_temp,
237
- power=dev_power,
238
- power_used=dev_power_used,
239
- appendix=dev_appendix,
240
- ),
179
+ dev_cores_util = None
180
+ with contextlib.suppress(pyhgml.HGMLError):
181
+ dev_util_rates = pyhgml.hgmlDeviceGetUtilizationRates(dev)
182
+ dev_cores_util = dev_util_rates.gpu
183
+ if dev_cores_util is None:
184
+ debug_log_warning(
185
+ logger,
186
+ "Failed to get device %d cores utilization, setting to 0",
187
+ dev_index,
241
188
  )
189
+ dev_cores_util = 0
242
190
 
243
- continue
244
-
245
- # Otherwise, get MIG devices.
246
-
247
- mdev_name = ""
248
- mdev_cores = None
249
- mdev_count = pyhgml.hgmlDeviceGetMaxMigDeviceCount(dev)
250
- for mdev_idx in range(mdev_count):
251
- mdev = None
252
- with contextlib.suppress(pyhgml.HGMLError):
253
- mdev = pyhgml.hgmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
254
- if not mdev:
255
- continue
256
-
257
- mdev_index = mdev_idx + dev_count * (dev_idx + 1)
258
- mdev_uuid = pyhgml.hgmlDeviceGetUUID(mdev)
259
-
260
- mdev_mem = 0
261
- mdev_mem_used = 0
262
- mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
263
- with contextlib.suppress(pyhgml.HGMLError):
264
- mdev_mem_info = pyhgml.hgmlDeviceGetMemoryInfo(mdev)
265
- mdev_mem = byte_to_mebibyte( # byte to MiB
266
- mdev_mem_info.total,
267
- )
268
- mdev_mem_used = byte_to_mebibyte( # byte to MiB
269
- mdev_mem_info.used,
191
+ dev_mem = 0
192
+ dev_mem_used = 0
193
+ dev_mem_status = DeviceMemoryStatusEnum.HEALTHY
194
+ with contextlib.suppress(pyhgml.HGMLError):
195
+ dev_mem_info = pyhgml.hgmlDeviceGetMemoryInfo(dev)
196
+ dev_mem = byte_to_mebibyte( # byte to MiB
197
+ dev_mem_info.total,
198
+ )
199
+ dev_mem_used = byte_to_mebibyte( # byte to MiB
200
+ dev_mem_info.used,
201
+ )
202
+ if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
203
+ dev_mem_ecc_errors = pyhgml.hgmlDeviceGetMemoryErrorCounter(
204
+ dev,
205
+ pyhgml.HGML_MEMORY_ERROR_TYPE_UNCORRECTED,
206
+ pyhgml.HGML_VOLATILE_ECC,
207
+ pyhgml.HGML_MEMORY_LOCATION_DRAM,
270
208
  )
271
- if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
272
- mdev_mem_ecc_errors = (
273
- pyhgml.hgmlDeviceGetMemoryErrorCounter(
274
- mdev,
275
- pyhgml.HGML_MEMORY_ERROR_TYPE_UNCORRECTED,
276
- pyhgml.HGML_AGGREGATE_ECC,
277
- pyhgml.HGML_MEMORY_LOCATION_SRAM,
278
- )
279
- )
280
- if mdev_mem_ecc_errors > 0:
281
- mdev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
282
-
283
- mdev_appendix = {
284
- "vgpu": True,
285
- "sliced": True,
286
- "bdf": dev_bdf,
287
- }
288
- if dev_numa:
289
- mdev_appendix["numa"] = dev_numa
290
-
291
- mdev_gi_id = pyhgml.hgmlDeviceGetGpuInstanceId(mdev)
292
- mdev_appendix["gpu_instance_id"] = mdev_gi_id
293
- mdev_ci_id = pyhgml.hgmlDeviceGetComputeInstanceId(mdev)
294
- mdev_appendix["compute_instance_id"] = mdev_ci_id
295
-
296
- mdev_cores_util = _get_sm_util_from_gpm_metrics(dev, mdev_gi_id)
297
-
298
- mdev_gi = pyhgml.hgmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
299
- mdev_ci = pyhgml.hgmlGpuInstanceGetComputeInstanceById(
300
- mdev_gi,
301
- mdev_ci_id,
209
+ if dev_mem_ecc_errors > 0:
210
+ dev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
211
+
212
+ dev_is_vgpu = False
213
+ if dev_bdf:
214
+ dev_is_vgpu = get_physical_function_by_bdf(dev_bdf) != dev_bdf
215
+
216
+ dev_appendix = {
217
+ "vgpu": dev_is_vgpu,
218
+ "mig": dev_mig_mode != pyhgml.HGML_DEVICE_MIG_DISABLE,
219
+ "bdf": dev_bdf,
220
+ }
221
+ if dev_mig_mode != pyhgml.HGML_DEVICE_MIG_DISABLE:
222
+ dev_appendix["mig_devices"] = _get_mig_devices(
223
+ dev,
224
+ dev_idx,
225
+ dev_count,
226
+ dev_cc_t,
227
+ sys_driver_ver,
228
+ sys_runtime_ver,
229
+ sys_runtime_ver_original,
230
+ dev_cc,
231
+ dev_temp,
232
+ dev_power,
233
+ dev_power_used,
234
+ dev_bdf,
235
+ dev_numa,
302
236
  )
303
- mdev_gi_info = pyhgml.hgmlGpuInstanceGetInfo(mdev_gi)
304
- mdev_ci_info = pyhgml.hgmlComputeInstanceGetInfo(mdev_ci)
305
- for dev_gi_prf_id in range(
306
- pyhgml.HGML_GPU_INSTANCE_PROFILE_COUNT,
307
- ):
308
- try:
309
- dev_gi_prf = pyhgml.hgmlDeviceGetGpuInstanceProfileInfo(
310
- dev,
311
- dev_gi_prf_id,
312
- )
313
- if dev_gi_prf.id != mdev_gi_info.profileId:
314
- continue
315
- except pyhgml.HGMLError:
316
- continue
237
+ if dev_numa:
238
+ dev_appendix["numa"] = dev_numa
239
+
240
+ ret.append(
241
+ Device(
242
+ manufacturer=self.manufacturer,
243
+ index=dev_index,
244
+ name=dev_name,
245
+ uuid=dev_uuid,
246
+ driver_version=sys_driver_ver,
247
+ runtime_version=sys_runtime_ver,
248
+ runtime_version_original=sys_runtime_ver_original,
249
+ compute_capability=dev_cc,
250
+ cores=dev_cores,
251
+ cores_utilization=dev_cores_util,
252
+ memory=dev_mem,
253
+ memory_used=dev_mem_used,
254
+ memory_utilization=get_utilization(dev_mem_used, dev_mem),
255
+ memory_status=dev_mem_status,
256
+ temperature=dev_temp,
257
+ power=dev_power,
258
+ power_used=dev_power_used,
259
+ appendix=dev_appendix,
260
+ ),
261
+ )
317
262
 
318
- for dev_ci_prf_id in range(
319
- pyhgml.HGML_COMPUTE_INSTANCE_PROFILE_COUNT,
320
- ):
321
- for dev_cig_prf_id in range(
322
- pyhgml.HGML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
323
- ):
324
- try:
325
- mdev_ci_prf = pyhgml.hgmlGpuInstanceGetComputeInstanceProfileInfo(
326
- mdev_gi,
327
- dev_ci_prf_id,
328
- dev_cig_prf_id,
329
- )
330
- if mdev_ci_prf.id != mdev_ci_info.profileId:
331
- continue
332
- except pyhgml.HGMLError:
333
- continue
334
-
335
- ci_slice = _get_compute_instance_slice(dev_ci_prf_id)
336
- gi_slice = _get_gpu_instance_slice(dev_gi_prf_id)
337
- if ci_slice == gi_slice:
338
- if hasattr(dev_gi_prf, "name"):
339
- mdev_name = dev_gi_prf.name
340
- else:
341
- gi_mem = round(
342
- math.ceil(dev_gi_prf.memorySizeMB >> 10),
343
- )
344
- mdev_name = f"{gi_slice}g.{gi_mem}gb"
345
- elif hasattr(mdev_ci_prf, "name"):
346
- mdev_name = mdev_ci_prf.name
347
- else:
348
- gi_mem = round(
349
- math.ceil(dev_gi_prf.memorySizeMB >> 10),
350
- )
351
- mdev_name = f"{ci_slice}u.{gi_slice}g.{gi_mem}gb"
352
-
353
- mdev_cores = mdev_ci_prf.multiprocessorCount
354
-
355
- break
356
-
357
- ret.append(
358
- Device(
359
- manufacturer=self.manufacturer,
360
- index=mdev_index,
361
- name=mdev_name,
362
- uuid=mdev_uuid,
363
- driver_version=sys_driver_ver,
364
- runtime_version=sys_runtime_ver,
365
- runtime_version_original=sys_runtime_ver_original,
366
- compute_capability=dev_cc,
367
- cores=mdev_cores,
368
- cores_utilization=mdev_cores_util,
369
- memory=mdev_mem,
370
- memory_used=mdev_mem_used,
371
- memory_utilization=get_utilization(mdev_mem_used, mdev_mem),
372
- memory_status=mdev_mem_status,
373
- temperature=dev_temp,
374
- power=dev_power,
375
- power_used=dev_power_used,
376
- appendix=mdev_appendix,
377
- ),
378
- )
379
263
  except pyhgml.HGMLError:
380
264
  debug_log_exception(logger, "Failed to fetch devices")
381
265
  raise
@@ -655,6 +539,158 @@ def _get_links_state(
655
539
  }
656
540
 
657
541
 
542
+ def _get_mig_devices(
543
+ dev,
544
+ dev_idx: int,
545
+ dev_count: int,
546
+ sys_driver_ver,
547
+ sys_runtime_ver,
548
+ sys_runtime_ver_original,
549
+ dev_cc,
550
+ dev_temp,
551
+ dev_power,
552
+ dev_power_used,
553
+ dev_bdf: str,
554
+ dev_numa,
555
+ ) -> list[dict]:
556
+ """
557
+ Enumerate the card's current MIG devices with the same detail a plain
558
+ device carries (profile name, uuid, compute/memory utilization, memory
559
+ health, temperature and power), returned as appendix entries of the
560
+ physical card rather than standalone devices. Empty when MIG is enabled
561
+ but no GPU instances exist yet.
562
+ """
563
+ ret: list[dict] = []
564
+ with contextlib.suppress(pyhgml.HGMLError):
565
+ for mdev_idx in range(pyhgml.hgmlDeviceGetMaxMigDeviceCount(dev)):
566
+ mdev = None
567
+ with contextlib.suppress(pyhgml.HGMLError):
568
+ mdev = pyhgml.hgmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
569
+ if not mdev:
570
+ continue
571
+
572
+ mdev_uuid = pyhgml.hgmlDeviceGetUUID(mdev)
573
+
574
+ mdev_mem = 0
575
+ mdev_mem_used = 0
576
+ mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
577
+ with contextlib.suppress(pyhgml.HGMLError):
578
+ mdev_mem_info = pyhgml.hgmlDeviceGetMemoryInfo(mdev)
579
+ mdev_mem = byte_to_mebibyte(mdev_mem_info.total)
580
+ mdev_mem_used = byte_to_mebibyte(mdev_mem_info.used)
581
+ if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
582
+ mdev_mem_ecc_errors = pyhgml.hgmlDeviceGetMemoryErrorCounter(
583
+ mdev,
584
+ pyhgml.HGML_MEMORY_ERROR_TYPE_UNCORRECTED,
585
+ pyhgml.HGML_AGGREGATE_ECC,
586
+ pyhgml.HGML_MEMORY_LOCATION_SRAM,
587
+ )
588
+ if mdev_mem_ecc_errors > 0:
589
+ mdev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
590
+
591
+ mdev_appendix = {
592
+ "vgpu": True,
593
+ "sliced": True,
594
+ "mig": True,
595
+ "bdf": dev_bdf,
596
+ }
597
+ if dev_numa:
598
+ mdev_appendix["numa"] = dev_numa
599
+
600
+ mdev_gi_id = pyhgml.hgmlDeviceGetGpuInstanceId(mdev)
601
+ mdev_appendix["gpu_instance_id"] = mdev_gi_id
602
+ mdev_ci_id = pyhgml.hgmlDeviceGetComputeInstanceId(mdev)
603
+ mdev_appendix["compute_instance_id"] = mdev_ci_id
604
+
605
+ mdev_cores_util = _get_sm_util_from_gpm_metrics(dev, mdev_gi_id)
606
+
607
+ mdev_name = ""
608
+ mdev_cores = None
609
+ mdev_gi = pyhgml.hgmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
610
+ mdev_ci = pyhgml.hgmlGpuInstanceGetComputeInstanceById(
611
+ mdev_gi,
612
+ mdev_ci_id,
613
+ )
614
+ mdev_gi_info = pyhgml.hgmlGpuInstanceGetInfo(mdev_gi)
615
+ mdev_ci_info = pyhgml.hgmlComputeInstanceGetInfo(mdev_ci)
616
+ for dev_gi_prf_id in range(pyhgml.HGML_GPU_INSTANCE_PROFILE_COUNT):
617
+ try:
618
+ dev_gi_prf = pyhgml.hgmlDeviceGetGpuInstanceProfileInfo(
619
+ dev,
620
+ dev_gi_prf_id,
621
+ )
622
+ if dev_gi_prf.id != mdev_gi_info.profileId:
623
+ continue
624
+ except pyhgml.HGMLError:
625
+ continue
626
+
627
+ for dev_ci_prf_id in range(
628
+ pyhgml.HGML_COMPUTE_INSTANCE_PROFILE_COUNT,
629
+ ):
630
+ for dev_cig_prf_id in range(
631
+ pyhgml.HGML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
632
+ ):
633
+ try:
634
+ mdev_ci_prf = (
635
+ pyhgml.hgmlGpuInstanceGetComputeInstanceProfileInfo(
636
+ mdev_gi,
637
+ dev_ci_prf_id,
638
+ dev_cig_prf_id,
639
+ )
640
+ )
641
+ if mdev_ci_prf.id != mdev_ci_info.profileId:
642
+ continue
643
+ except pyhgml.HGMLError:
644
+ continue
645
+
646
+ ci_slice = _get_compute_instance_slice(dev_ci_prf_id)
647
+ gi_slice = _get_gpu_instance_slice(dev_gi_prf_id)
648
+ if ci_slice == gi_slice:
649
+ if hasattr(dev_gi_prf, "name"):
650
+ mdev_name = dev_gi_prf.name
651
+ else:
652
+ gi_mem = round(
653
+ math.ceil(dev_gi_prf.memorySizeMB >> 10),
654
+ )
655
+ mdev_name = f"{gi_slice}g.{gi_mem}gb"
656
+ elif hasattr(mdev_ci_prf, "name"):
657
+ mdev_name = mdev_ci_prf.name
658
+ else:
659
+ gi_mem = round(
660
+ math.ceil(dev_gi_prf.memorySizeMB >> 10),
661
+ )
662
+ mdev_name = f"{ci_slice}u.{gi_slice}g.{gi_mem}gb"
663
+
664
+ mdev_cores = mdev_ci_prf.multiprocessorCount
665
+
666
+ break
667
+
668
+ break
669
+
670
+ ret.append(
671
+ {
672
+ "index": mdev_idx + dev_count * (dev_idx + 1),
673
+ "name": mdev_name,
674
+ "uuid": mdev_uuid,
675
+ "driver_version": sys_driver_ver,
676
+ "runtime_version": sys_runtime_ver,
677
+ "runtime_version_original": sys_runtime_ver_original,
678
+ "compute_capability": dev_cc,
679
+ "cores": mdev_cores,
680
+ "cores_utilization": mdev_cores_util,
681
+ "memory": mdev_mem,
682
+ "memory_used": mdev_mem_used,
683
+ "memory_utilization": get_utilization(mdev_mem_used, mdev_mem),
684
+ "memory_status": mdev_mem_status,
685
+ "temperature": dev_temp,
686
+ "power": dev_power,
687
+ "power_used": dev_power_used,
688
+ "appendix": mdev_appendix,
689
+ },
690
+ )
691
+ return ret
692
+
693
+
658
694
  def _get_gpu_instance_slice(dev_gi_prf_id: int) -> int:
659
695
  """
660
696
  Get the number of slices for a given GPU Instance Profile ID.
@@ -1 +0,0 @@
1
- git_commit = "c736814"