gpustack-runtime 0.2.2.post2__tar.gz → 0.2.2.post4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/PKG-INFO +1 -1
  2. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/_version.py +2 -2
  3. gpustack_runtime-0.2.2.post4/gpustack_runtime/_version_appendix.py +1 -0
  4. gpustack_runtime-0.2.2.post4/gpustack_runtime/detector/nvidia.py +629 -0
  5. gpustack_runtime-0.2.2.post2/gpustack_runtime/_version_appendix.py +0 -1
  6. gpustack_runtime-0.2.2.post2/gpustack_runtime/detector/nvidia.py +0 -1005
  7. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.codespelldict +0 -0
  8. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.codespellrc +0 -0
  9. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.dockerignore +0 -0
  10. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.gitattributes +0 -0
  11. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.gitignore +0 -0
  12. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.pre-commit-config.yaml +0 -0
  13. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.python-version +0 -0
  14. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/LICENSE +0 -0
  15. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/Makefile +0 -0
  16. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/README.md +0 -0
  17. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/deploy/manifests/docker-compose.yaml +0 -0
  18. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/deploy/manifests/kubernetes.yaml +0 -0
  19. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/docs/index.md +0 -0
  20. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/docs/modules/gpustack_runtime.deployer.md +0 -0
  21. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/docs/modules/gpustack_runtime.detector.md +0 -0
  22. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/docs/modules/gpustack_runtime.md +0 -0
  23. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/__init__.py +0 -0
  24. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/__main__.py +0 -0
  25. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/_version.pyi +0 -0
  26. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/__init__.py +0 -0
  27. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/__types__.py +0 -0
  28. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/deployer.py +0 -0
  29. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/detector.py +0 -0
  30. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/images.py +0 -0
  31. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/__init__.py +0 -0
  32. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/__patches__.py +0 -0
  33. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/__types__.py +0 -0
  34. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/__utils__.py +0 -0
  35. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/__init__.py +0 -0
  36. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/__types__.py +0 -0
  37. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/__utils__.py +0 -0
  38. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/amd.py +0 -0
  39. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/ascend.py +0 -0
  40. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/hygon.py +0 -0
  41. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/iluvatar.py +0 -0
  42. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/metax.py +0 -0
  43. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/thead.py +0 -0
  44. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/docker.py +0 -0
  45. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/k8s/devicemanager/__init__.py +0 -0
  46. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/kuberentes.py +0 -0
  47. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/podman.py +0 -0
  48. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/__init__.py +0 -0
  49. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/__types__.py +0 -0
  50. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/__utils__.py +0 -0
  51. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/amd.py +0 -0
  52. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/ascend.py +0 -0
  53. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/cambricon.py +0 -0
  54. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/hygon.py +0 -0
  55. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/iluvatar.py +0 -0
  56. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/metax.py +0 -0
  57. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/mthreads.py +0 -0
  58. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyamdgpu/__init__.py +0 -0
  59. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyamdsmi/__init__.py +0 -0
  60. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pydcmi/__init__.py +0 -0
  61. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyhgml/__init__.py +0 -0
  62. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyhgml/libhgml.so +0 -0
  63. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyhgml/libuki.so +0 -0
  64. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyhsa/__init__.py +0 -0
  65. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyixml/__init__.py +0 -0
  66. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pymtml/__init__.py +0 -0
  67. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pymxsml/__init__.py +0 -0
  68. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pynvml/__init__.py +0 -0
  69. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyrocmsmi/__init__.py +0 -0
  70. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/thead.py +0 -0
  71. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/envs.py +0 -0
  72. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/logging.py +0 -0
  73. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/hatch.toml +0 -0
  74. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/mkdocs.yml +0 -0
  75. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/pack/Dockerfile +0 -0
  76. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/pack/Dockerfile.dummy +0 -0
  77. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/pyproject.toml +0 -0
  78. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/pytest.ini +0 -0
  79. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/ruff.toml +0 -0
  80. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/__init__.py +0 -0
  81. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_compare_versions.json +0 -0
  82. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_correct_runner_image.json +0 -0
  83. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json.json +0 -0
  84. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_jsons.json +0 -0
  85. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_yamls.yaml +0 -0
  86. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_json.json +0 -0
  87. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_yaml.yaml +0 -0
  88. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_nginx_entrypoint.sh +0 -0
  89. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/test_utils.py +0 -0
  90. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/test_workload_status.py +0 -0
  91. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/fixtures/__init__.py +0 -0
  92. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/README.md +0 -0
  93. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi300x.json +0 -0
  94. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi308x.json +0 -0
  95. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_amd_rx7800xt.json +0 -0
  96. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_ascend_310p3.json +0 -0
  97. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_ascend_910b2.json +0 -0
  98. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_hygon_k100ai.json +0 -0
  99. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_metax_c500.json +0 -0
  100. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_gb10.json +0 -0
  101. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100.json +0 -0
  102. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100_mig.json +0 -0
  103. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h200.json +0 -0
  104. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4080super.json +0 -0
  105. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4090d.json +0 -0
  106. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx5090d.json +0 -0
  107. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_thead_ppu.json +0 -0
  108. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi300x.json +0 -0
  109. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi308x.json +0 -0
  110. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_amd_rx7800xt.json +0 -0
  111. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_ascend_310p3.json +0 -0
  112. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_ascend_910b2.json +0 -0
  113. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_hygon_k100ai.json +0 -0
  114. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_metax_c500.json +0 -0
  115. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_mthreads_s5000.json +0 -0
  116. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100.json +0 -0
  117. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100_mig.json +0 -0
  118. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h200.json +0 -0
  119. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4080super.json +0 -0
  120. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4090d.json +0 -0
  121. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx5090d.json +0 -0
  122. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_thead_ppu.json +0 -0
  123. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_amd.py +0 -0
  124. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_ascend.py +0 -0
  125. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_cambricon.py +0 -0
  126. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_detector_utils.py +0 -0
  127. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_hygon.py +0 -0
  128. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_iluvatar.py +0 -0
  129. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_metax.py +0 -0
  130. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_mthreads.py +0 -0
  131. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_nvidia.py +0 -0
  132. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/uv.lock +0 -0
  133. {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/uv.toml +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gpustack-runtime
3
- Version: 0.2.2.post2
3
+ Version: 0.2.2.post4
4
4
  Summary: GPUStack Runtime is library for detecting GPU resources and launching GPU workloads.
5
5
  Project-URL: Homepage, https://github.com/gpustack/runtime
6
6
  Project-URL: Bug Tracker, https://github.com/gpustack/gpustack/issues
@@ -27,8 +27,8 @@ version_tuple: VERSION_TUPLE
27
27
  __commit_id__: COMMIT_ID
28
28
  commit_id: COMMIT_ID
29
29
 
30
- __version__ = version = '0.2.2.post2'
31
- __version_tuple__ = version_tuple = (0, 2, 2, 'post2')
30
+ __version__ = version = '0.2.2.post4'
31
+ __version_tuple__ = version_tuple = (0, 2, 2, 'post4')
32
32
  try:
33
33
  from ._version_appendix import git_commit
34
34
  __commit_id__ = commit_id = git_commit
@@ -0,0 +1 @@
1
+ git_commit = "c736814"
@@ -0,0 +1,629 @@
1
+ from __future__ import annotations as __future_annotations__
2
+
3
+ import contextlib
4
+ import logging
5
+ import math
6
+ import threading
7
+ import time
8
+ from _ctypes import byref
9
+ from functools import lru_cache
10
+
11
+ from .. import envs
12
+ from ..logging import debug_log_exception, debug_log_warning
13
+ from . import DeviceMemoryStatusEnum, Topology, pynvml
14
+ from .__types__ import Detector, Device, Devices, ManufacturerEnum, TopologyDistanceEnum
15
+ from .__utils__ import (
16
+ PCIDevice,
17
+ bitmask_to_str,
18
+ byte_to_mebibyte,
19
+ get_brief_version,
20
+ get_memory,
21
+ get_numa_node_by_bdf,
22
+ get_numa_nodeset_size,
23
+ get_pci_devices,
24
+ get_utilization,
25
+ map_numa_node_to_cpu_affinity,
26
+ stringify_uuid,
27
+ )
28
+
29
+ logger = logging.getLogger(__name__)
30
+
31
+
32
+ class NVIDIADetector(Detector):
33
+ """
34
+ Detect NVIDIA GPUs.
35
+ """
36
+
37
+ @staticmethod
38
+ @lru_cache(maxsize=1)
39
+ def is_supported() -> bool:
40
+ """
41
+ Check if NVIDIA detection is supported.
42
+
43
+ Returns:
44
+ True if supported, False otherwise.
45
+
46
+ """
47
+ supported = False
48
+ if envs.GPUSTACK_RUNTIME_DETECT.lower() not in ("auto", "nvidia"):
49
+ logger.debug("NVIDIA detection is disabled by environment variable")
50
+ return supported
51
+
52
+ pci_devs = NVIDIADetector.detect_pci_devices()
53
+ if not pci_devs and not envs.GPUSTACK_RUNTIME_DETECT_NO_PCI_CHECK:
54
+ logger.debug("No NVIDIA PCI devices found")
55
+ return supported
56
+
57
+ try:
58
+ pynvml.nvmlInit()
59
+ supported = True
60
+ except Exception:
61
+ debug_log_exception(logger, "Failed to initialize NVML")
62
+
63
+ return supported
64
+
65
+ @staticmethod
66
+ @lru_cache(maxsize=1)
67
+ def detect_pci_devices() -> dict[str, PCIDevice]:
68
+ # See https://pcisig.com/membership/member-companies?combine=NVIDIA.
69
+ pci_devs = get_pci_devices(vendor="0x10de")
70
+ if not pci_devs:
71
+ return {}
72
+ return {dev.address: dev for dev in pci_devs}
73
+
74
+ def __init__(self):
75
+ super().__init__(ManufacturerEnum.NVIDIA)
76
+
77
+ def detect(self) -> Devices | None:
78
+ """
79
+ Detect NVIDIA GPUs using pynvml.
80
+
81
+ Returns:
82
+ A list of detected NVIDIA GPU devices,
83
+ or None if not supported.
84
+
85
+ Raises:
86
+ If there is an error during detection.
87
+
88
+ """
89
+ if not self.is_supported():
90
+ return None
91
+
92
+ ret: Devices = []
93
+
94
+ try:
95
+ pci_devs = NVIDIADetector.detect_pci_devices()
96
+
97
+ pynvml.nvmlInit()
98
+
99
+ sys_driver_ver = pynvml.nvmlSystemGetDriverVersion()
100
+
101
+ sys_runtime_ver_original = pynvml.nvmlSystemGetCudaDriverVersion()
102
+ sys_runtime_ver_original = ".".join(
103
+ map(
104
+ str,
105
+ [
106
+ sys_runtime_ver_original // 1000,
107
+ (sys_runtime_ver_original % 1000) // 10,
108
+ (sys_runtime_ver_original % 10),
109
+ ],
110
+ ),
111
+ )
112
+ sys_runtime_ver = get_brief_version(
113
+ sys_runtime_ver_original,
114
+ )
115
+
116
+ dev_count = pynvml.nvmlDeviceGetCount()
117
+ for dev_idx in range(dev_count):
118
+ dev = pynvml.nvmlDeviceGetHandleByIndex(dev_idx)
119
+
120
+ dev_cc_t = pynvml.nvmlDeviceGetCudaComputeCapability(dev)
121
+ dev_cc = ".".join(map(str, dev_cc_t))
122
+
123
+ dev_pci_info = pynvml.nvmlDeviceGetPciInfo(dev)
124
+ dev_bdf = str(dev_pci_info.busIdLegacy).lower()
125
+
126
+ dev_numa = get_numa_node_by_bdf(dev_bdf)
127
+ if not dev_numa:
128
+ with contextlib.suppress(pynvml.NVMLError):
129
+ dev_node_affinity = pynvml.nvmlDeviceGetMemoryAffinity(
130
+ dev,
131
+ get_numa_nodeset_size(),
132
+ pynvml.NVML_AFFINITY_SCOPE_NODE,
133
+ )
134
+ dev_numa = bitmask_to_str(list(dev_node_affinity))
135
+
136
+ dev_temp = None
137
+ with contextlib.suppress(pynvml.NVMLError):
138
+ dev_temp = pynvml.nvmlDeviceGetTemperature(
139
+ dev,
140
+ pynvml.NVML_TEMPERATURE_GPU,
141
+ )
142
+
143
+ dev_power = None
144
+ dev_power_used = None
145
+ with contextlib.suppress(pynvml.NVMLError):
146
+ dev_power = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(dev)
147
+ dev_power = dev_power // 1000 # mW to W
148
+ dev_power_used = (
149
+ pynvml.nvmlDeviceGetPowerUsage(dev) // 1000
150
+ ) # mW to W
151
+
152
+ dev_mig_mode = pynvml.NVML_DEVICE_MIG_DISABLE
153
+ with contextlib.suppress(pynvml.NVMLError):
154
+ dev_mig_mode, _ = pynvml.nvmlDeviceGetMigMode(dev)
155
+
156
+ dev_index = dev_idx
157
+ if envs.GPUSTACK_RUNTIME_DETECT_PHYSICAL_INDEX_PRIORITY:
158
+ with contextlib.suppress(pynvml.NVMLError):
159
+ dev_index = pynvml.nvmlDeviceGetMinorNumber(dev)
160
+
161
+ # Report the physical card, whether or not MIG is enabled.
162
+ # MIG instances are partitioned on demand by the operator's
163
+ # device-manager; they are not separate allocatable devices
164
+ # in this inventory. A MIG-enabled card is marked ``mig``
165
+ # in the appendix instead.
166
+
167
+ dev_name = pynvml.nvmlDeviceGetName(dev)
168
+
169
+ dev_uuid = pynvml.nvmlDeviceGetUUID(dev)
170
+
171
+ dev_cores = None
172
+ with contextlib.suppress(pynvml.NVMLError):
173
+ dev_cores = pynvml.nvmlDeviceGetNumGpuCores(dev)
174
+
175
+ dev_cores_util = _get_sm_util_from_gpm_metrics(dev)
176
+ if dev_cores_util is None:
177
+ with contextlib.suppress(pynvml.NVMLError):
178
+ dev_util_rates = pynvml.nvmlDeviceGetUtilizationRates(dev)
179
+ dev_cores_util = dev_util_rates.gpu
180
+ if dev_cores_util is None:
181
+ debug_log_warning(
182
+ logger,
183
+ "Failed to get device %d cores utilization, setting to 0",
184
+ dev_index,
185
+ )
186
+ dev_cores_util = 0
187
+
188
+ dev_mem = 0
189
+ dev_mem_used = 0
190
+ dev_mem_status = DeviceMemoryStatusEnum.HEALTHY
191
+ with contextlib.suppress(pynvml.NVMLError):
192
+ dev_mem_info = pynvml.nvmlDeviceGetMemoryInfo(dev)
193
+ dev_mem = byte_to_mebibyte( # byte to MiB
194
+ dev_mem_info.total,
195
+ )
196
+ dev_mem_used = byte_to_mebibyte( # byte to MiB
197
+ dev_mem_info.used,
198
+ )
199
+ if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
200
+ dev_mem_ecc_errors = pynvml.nvmlDeviceGetMemoryErrorCounter(
201
+ dev,
202
+ pynvml.NVML_MEMORY_ERROR_TYPE_UNCORRECTED,
203
+ pynvml.NVML_VOLATILE_ECC,
204
+ pynvml.NVML_MEMORY_LOCATION_DRAM,
205
+ )
206
+ if dev_mem_ecc_errors > 0:
207
+ dev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
208
+ if dev_mem == 0:
209
+ dev_mem, dev_mem_used = get_memory()
210
+
211
+ dev_is_vgpu = False
212
+ if dev_bdf in pci_devs:
213
+ dev_is_vgpu = _is_vgpu(pci_devs[dev_bdf].config)
214
+
215
+ dev_appendix = {
216
+ "arch_family": _get_arch_family(dev_cc_t),
217
+ "vgpu": dev_is_vgpu,
218
+ "mig": dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE,
219
+ "bdf": dev_bdf,
220
+ }
221
+ if dev_numa:
222
+ dev_appendix["numa"] = dev_numa
223
+
224
+ if dev_fabric_info := _get_fabric_info(dev):
225
+ dev_appendix.update(dev_fabric_info)
226
+
227
+ ret.append(
228
+ Device(
229
+ manufacturer=self.manufacturer,
230
+ index=dev_index,
231
+ name=dev_name,
232
+ uuid=dev_uuid,
233
+ driver_version=sys_driver_ver,
234
+ runtime_version=sys_runtime_ver,
235
+ runtime_version_original=sys_runtime_ver_original,
236
+ compute_capability=dev_cc,
237
+ cores=dev_cores,
238
+ cores_utilization=dev_cores_util,
239
+ memory=dev_mem,
240
+ memory_used=dev_mem_used,
241
+ memory_utilization=get_utilization(dev_mem_used, dev_mem),
242
+ memory_status=dev_mem_status,
243
+ temperature=dev_temp,
244
+ power=dev_power,
245
+ power_used=dev_power_used,
246
+ appendix=dev_appendix,
247
+ ),
248
+ )
249
+ except pynvml.NVMLError:
250
+ debug_log_exception(logger, "Failed to fetch devices")
251
+ raise
252
+ except Exception:
253
+ debug_log_exception(logger, "Failed to process devices fetching")
254
+ raise
255
+
256
+ return ret
257
+
258
+ def get_topology(self, devices: Devices | None = None) -> Topology | None:
259
+ """
260
+ Get the Topology object between NVIDIA GPUs.
261
+
262
+ Args:
263
+ devices:
264
+ The list of detected NVIDIA devices.
265
+ If None, detect topology for all available devices.
266
+
267
+ Returns:
268
+ The Topology object, or None if not supported.
269
+
270
+ """
271
+ if devices is None:
272
+ devices = self.detect()
273
+ if devices is None:
274
+ return None
275
+
276
+ ret = Topology(
277
+ manufacturer=self.manufacturer,
278
+ devices_count=len(devices),
279
+ )
280
+
281
+ get_links_cache = {}
282
+
283
+ try:
284
+ pynvml.nvmlInit()
285
+
286
+ for i, dev_i in enumerate(devices):
287
+ dev_i_bdf = dev_i.appendix.get("bdf")
288
+ if dev_i.appendix.get("sliced", False):
289
+ dev_i_handle = pynvml.nvmlDeviceGetHandleByPciBusId(dev_i_bdf)
290
+ else:
291
+ dev_i_handle = pynvml.nvmlDeviceGetHandleByUUID(dev_i.uuid)
292
+
293
+ # Get NUMA and CPU affinities.
294
+ ret.devices_numa_affinities[i] = dev_i.appendix.get("numa", "")
295
+ ret.devices_cpu_affinities[i] = map_numa_node_to_cpu_affinity(
296
+ ret.devices_numa_affinities[i],
297
+ )
298
+
299
+ # Get links state if applicable.
300
+ if dev_i_bdf in get_links_cache:
301
+ dev_i_links_state = get_links_cache[dev_i_bdf]
302
+ else:
303
+ dev_i_links_state = _get_links_state(dev_i_handle)
304
+ get_links_cache[dev_i_bdf] = dev_i_links_state
305
+ if dev_i_links_state:
306
+ ret.appendices[i].update(dev_i_links_state)
307
+ # In practice, if a card has an active *Link,
308
+ # then other cards in the same machine should be interconnected with it through the *Link.
309
+ if dev_i_links_state.get("links_active_count", 0) > 0:
310
+ for j, dev_j in enumerate(devices):
311
+ if dev_i.index == dev_j.index:
312
+ continue
313
+ ret.devices_distances[i][j] = TopologyDistanceEnum.LINK
314
+ ret.devices_distances[j][i] = TopologyDistanceEnum.LINK
315
+ continue
316
+
317
+ # Get distances to other devices.
318
+ for j, dev_j in enumerate(devices):
319
+ if dev_i.index == dev_j.index or ret.devices_distances[i][j] != 0:
320
+ continue
321
+
322
+ dev_j_bdf = dev_j.appendix.get("bdf")
323
+ if dev_i_bdf == dev_j_bdf:
324
+ distance = TopologyDistanceEnum.SELF
325
+ else:
326
+ if dev_j.appendix.get("sliced", False):
327
+ dev_j_handle = pynvml.nvmlDeviceGetHandleByPciBusId(
328
+ dev_j_bdf,
329
+ )
330
+ else:
331
+ dev_j_handle = pynvml.nvmlDeviceGetHandleByUUID(dev_j.uuid)
332
+
333
+ distance = TopologyDistanceEnum.UNK
334
+ try:
335
+ distance = pynvml.nvmlDeviceGetTopologyCommonAncestor(
336
+ dev_i_handle,
337
+ dev_j_handle,
338
+ )
339
+ except pynvml.NVMLError:
340
+ debug_log_exception(
341
+ logger,
342
+ "Failed to get distance between device %d and %d",
343
+ dev_i.index,
344
+ dev_j.index,
345
+ )
346
+
347
+ ret.devices_distances[i][j] = distance
348
+ ret.devices_distances[j][i] = distance
349
+ except Exception:
350
+ debug_log_exception(logger, "Failed to process topology fetching")
351
+ raise
352
+
353
+ return ret
354
+
355
+
356
+ def _get_gpm_metrics(
357
+ metrics: list[int],
358
+ dev: pynvml.c_nvmlDevice_t,
359
+ gpu_instance_id: int | None = None,
360
+ interval: float = 0.1,
361
+ ) -> list[pynvml.c_nvmlGpmMetric_t] | None:
362
+ """
363
+ Get GPM metrics for a device or a MIG GPU instance.
364
+
365
+ Args:
366
+ metrics:
367
+ A list of GPM metric IDs to query.
368
+ dev:
369
+ The NVML device handle.
370
+ gpu_instance_id:
371
+ The GPU instance ID for MIG devices.
372
+ interval:
373
+ Interval in seconds between two samples.
374
+
375
+ Returns:
376
+ A list of GPM metric structures, or None if failed.
377
+
378
+ """
379
+ try:
380
+ dev_gpm_support = pynvml.nvmlGpmQueryDeviceSupport(dev)
381
+ if not bool(dev_gpm_support.isSupportedDevice):
382
+ return None
383
+ except pynvml.NVMLError:
384
+ debug_log_warning(logger, "Unsupported GPM query")
385
+ return None
386
+
387
+ dev_gpm_metrics = pynvml.c_nvmlGpmMetricsGet_t()
388
+ try:
389
+ dev_gpm_metrics.sample1 = pynvml.nvmlGpmSampleAlloc()
390
+ dev_gpm_metrics.sample2 = pynvml.nvmlGpmSampleAlloc()
391
+ if gpu_instance_id is None:
392
+ pynvml.nvmlGpmSampleGet(dev, dev_gpm_metrics.sample1)
393
+ time.sleep(interval)
394
+ pynvml.nvmlGpmSampleGet(dev, dev_gpm_metrics.sample2)
395
+ else:
396
+ pynvml.nvmlGpmMigSampleGet(dev, gpu_instance_id, dev_gpm_metrics.sample1)
397
+ time.sleep(interval)
398
+ pynvml.nvmlGpmMigSampleGet(dev, gpu_instance_id, dev_gpm_metrics.sample2)
399
+ dev_gpm_metrics.version = pynvml.NVML_GPM_METRICS_GET_VERSION
400
+ dev_gpm_metrics.numMetrics = len(metrics)
401
+ for metric_idx, metric in enumerate(metrics):
402
+ dev_gpm_metrics.metrics[metric_idx].metricId = metric
403
+ pynvml.nvmlGpmMetricsGet(dev_gpm_metrics)
404
+ except pynvml.NVMLError:
405
+ debug_log_exception(logger, "Failed to get GPM metrics")
406
+ return None
407
+ finally:
408
+ if dev_gpm_metrics.sample1:
409
+ pynvml.nvmlGpmSampleFree(dev_gpm_metrics.sample1)
410
+ if dev_gpm_metrics.sample2:
411
+ pynvml.nvmlGpmSampleFree(dev_gpm_metrics.sample2)
412
+ return list(dev_gpm_metrics.metrics)
413
+
414
+
415
+ _gpm_metrics_lock = threading.Lock()
416
+
417
+
418
+ def _get_sm_util_from_gpm_metrics(
419
+ dev: pynvml.c_nvmlDevice_t,
420
+ gpu_instance_id: int | None = None,
421
+ interval: float = 0.1,
422
+ ) -> int | None:
423
+ """
424
+ Get SM utilization from GPM metrics.
425
+
426
+ Args:
427
+ dev:
428
+ The NVML device handle.
429
+ gpu_instance_id:
430
+ The GPU instance ID for MIG devices.
431
+ interval:
432
+ Interval in seconds between two samples.
433
+
434
+ Returns:
435
+ The SM utilization as an integer percentage, or None if failed.
436
+
437
+ """
438
+ with _gpm_metrics_lock:
439
+ dev_gpm_metrics = _get_gpm_metrics(
440
+ metrics=[pynvml.NVML_GPM_METRIC_SM_UTIL],
441
+ dev=dev,
442
+ gpu_instance_id=gpu_instance_id,
443
+ interval=interval,
444
+ )
445
+
446
+ if dev_gpm_metrics and not math.isnan(dev_gpm_metrics[0].value):
447
+ return int(dev_gpm_metrics[0].value)
448
+
449
+ return None
450
+
451
+
452
+ def _extract_field_value(
453
+ field_value: pynvml.c_nvmlFieldValue_t,
454
+ ) -> int | float | None:
455
+ """
456
+ Extract the value from a NVML field value structure.
457
+
458
+ Args:
459
+ field_value:
460
+ The NVML field value structure.
461
+
462
+ Returns:
463
+ The extracted value as int, float, or None if unknown.
464
+
465
+ """
466
+ if field_value.nvmlReturn != pynvml.NVML_SUCCESS:
467
+ return None
468
+ match field_value.valueType:
469
+ case pynvml.NVML_VALUE_TYPE_DOUBLE:
470
+ return field_value.value.dVal
471
+ case pynvml.NVML_VALUE_TYPE_UNSIGNED_INT:
472
+ return field_value.value.uiVal
473
+ case pynvml.NVML_VALUE_TYPE_UNSIGNED_LONG:
474
+ return field_value.value.ulVal
475
+ case pynvml.NVML_VALUE_TYPE_UNSIGNED_LONG_LONG:
476
+ return field_value.value.ullVal
477
+ case pynvml.NVML_VALUE_TYPE_SIGNED_LONG_LONG:
478
+ return field_value.value.sllVal
479
+ case pynvml.NVML_VALUE_TYPE_SIGNED_INT:
480
+ return field_value.value.siVal
481
+ case pynvml.NVML_VALUE_TYPE_UNSIGNED_SHORT:
482
+ return field_value.value.usVal
483
+ return None
484
+
485
+
486
+ def _get_fabric_info(
487
+ dev: pynvml.c_nvmlDevice_t,
488
+ ) -> dict | None:
489
+ """
490
+ Get the NVSwitch fabric information for a device.
491
+
492
+ Args:
493
+ dev:
494
+ The NVML device handle.
495
+
496
+ Returns:
497
+ A dict includes fabric info or None if failed.
498
+
499
+ """
500
+ try:
501
+ dev_fabric = pynvml.c_nvmlGpuFabricInfoV_t()
502
+ ret = pynvml.nvmlDeviceGetGpuFabricInfoV(dev, byref(dev_fabric))
503
+ if ret != pynvml.NVML_SUCCESS:
504
+ return None
505
+ if dev_fabric.state != pynvml.NVML_GPU_FABRIC_STATE_COMPLETED:
506
+ return None
507
+ return {
508
+ "fabric_cluster_uuid": stringify_uuid(bytes(dev_fabric.clusterUuid)),
509
+ "fabric_clique_id": dev_fabric.cliqueId,
510
+ }
511
+ except pynvml.NVMLError:
512
+ debug_log_warning(logger, "Failed to get NVSwitch fabric info")
513
+
514
+ return None
515
+
516
+
517
+ def _get_links_state(
518
+ dev: pynvml.c_nvmlDevice_t,
519
+ ) -> dict | None:
520
+ """
521
+ Get the NVLink links count and state for a device.
522
+
523
+ Args:
524
+ dev:
525
+ The NVML device handle.
526
+
527
+ Returns:
528
+ A dict includes links state or None if failed.
529
+
530
+ """
531
+ dev_links_count = 0
532
+ try:
533
+ dev_fields = pynvml.nvmlDeviceGetFieldValues(
534
+ dev,
535
+ fieldIds=[pynvml.NVML_FI_DEV_NVLINK_LINK_COUNT],
536
+ )
537
+ dev_links_count = _extract_field_value(dev_fields[0])
538
+ except pynvml.NVMLError:
539
+ debug_log_warning(logger, "Failed to get NVLink links count")
540
+ if not dev_links_count:
541
+ return None
542
+
543
+ dev_links_state = 0
544
+ dev_links_active_count = 0
545
+ try:
546
+ for link_idx in range(int(dev_links_count)):
547
+ dev_link_state = pynvml.nvmlDeviceGetNvLinkState(dev, link_idx)
548
+ if dev_link_state:
549
+ dev_links_state |= 1 << link_idx
550
+ dev_links_active_count += 1
551
+ except pynvml.NVMLError:
552
+ debug_log_warning(logger, "Failed to get NVLink link state")
553
+
554
+ return {
555
+ "links_count": dev_links_count,
556
+ "links_state": dev_links_state,
557
+ "links_active_count": dev_links_active_count,
558
+ }
559
+
560
+
561
+ def _get_arch_family(dev_cc_t: list[int]) -> str:
562
+ """
563
+ Get the architecture family based on the CUDA compute capability.
564
+
565
+ Args:
566
+ dev_cc_t:
567
+ The CUDA compute capability as a list of two integers.
568
+
569
+ Returns:
570
+ The architecture family as a string.
571
+
572
+ """
573
+ match dev_cc_t[0]:
574
+ case 1:
575
+ return "Tesla"
576
+ case 2:
577
+ return "Fermi"
578
+ case 3:
579
+ return "Kepler"
580
+ case 5:
581
+ return "Maxwell"
582
+ case 6:
583
+ return "Pascal"
584
+ case 7:
585
+ return "Volta" if dev_cc_t[1] < 5 else "Turing"
586
+ case 8:
587
+ if dev_cc_t[1] < 9:
588
+ return "Ampere"
589
+ return "Ada-Lovelace"
590
+ case 9:
591
+ return "Hopper"
592
+ case 10 | 12:
593
+ return "Blackwell"
594
+ return "Unknown"
595
+
596
+
597
+ def _is_vgpu(dev_config: bytes) -> bool:
598
+ """
599
+ Determine if the device is a vGPU based on its PCI configuration space.
600
+
601
+ """
602
+ status = 0x06
603
+ cap_supported = 0x10
604
+ cap_start = 0x34
605
+ cap_vendor_specific_id = 0x09
606
+
607
+ if dev_config[status] & cap_supported == 0:
608
+ return False
609
+
610
+ # Find the capability list
611
+ dev_cap: bytes | None = None
612
+ visited = set()
613
+ pos = dev_config[cap_start]
614
+ while pos != 0 and pos not in visited and pos < len(dev_config) - 2:
615
+ visited.add(pos)
616
+ ptr = dev_config[pos : pos + 3] # id, next, length
617
+ if ptr[0] == 0xFF:
618
+ break
619
+ if ptr[0] == cap_vendor_specific_id:
620
+ dev_cap = dev_config[pos : pos + ptr[2]]
621
+ break
622
+ pos = ptr[1]
623
+
624
+ if not dev_cap or len(dev_cap) < 5:
625
+ return False
626
+
627
+ # Check for vGPU signature,
628
+ # which is either 0x56 (NVIDIA vGPU) or 0x46 (NVIDIA GRID).
629
+ return dev_cap[3] == 0x56 or dev_cap[4] == 0x46
@@ -1 +0,0 @@
1
- git_commit = "480e0fd"