gpustack-runtime 0.2.2.post2__tar.gz → 0.2.2.post4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/PKG-INFO +1 -1
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/_version.py +2 -2
- gpustack_runtime-0.2.2.post4/gpustack_runtime/_version_appendix.py +1 -0
- gpustack_runtime-0.2.2.post4/gpustack_runtime/detector/nvidia.py +629 -0
- gpustack_runtime-0.2.2.post2/gpustack_runtime/_version_appendix.py +0 -1
- gpustack_runtime-0.2.2.post2/gpustack_runtime/detector/nvidia.py +0 -1005
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.codespelldict +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.codespellrc +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.dockerignore +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.gitattributes +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.gitignore +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.pre-commit-config.yaml +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/.python-version +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/LICENSE +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/Makefile +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/README.md +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/deploy/manifests/docker-compose.yaml +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/deploy/manifests/kubernetes.yaml +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/docs/index.md +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/docs/modules/gpustack_runtime.deployer.md +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/docs/modules/gpustack_runtime.detector.md +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/docs/modules/gpustack_runtime.md +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/__main__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/_version.pyi +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/deployer.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/detector.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/cmds/images.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/__patches__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/amd.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/ascend.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/hygon.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/metax.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/cdi/thead.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/docker.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/k8s/devicemanager/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/kuberentes.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/deployer/podman.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/amd.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/ascend.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/cambricon.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/hygon.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/metax.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/mthreads.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyamdgpu/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyamdsmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pydcmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyhgml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyhgml/libhgml.so +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyhgml/libuki.so +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyhsa/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyixml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pymtml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pymxsml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pynvml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/pyrocmsmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/detector/thead.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/envs.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/gpustack_runtime/logging.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/hatch.toml +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/mkdocs.yml +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/pack/Dockerfile +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/pack/Dockerfile.dummy +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/pyproject.toml +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/pytest.ini +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/ruff.toml +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_compare_versions.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_correct_runner_image.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_jsons.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_yamls.yaml +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_json.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_yaml.yaml +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/fixtures/test_nginx_entrypoint.sh +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/test_utils.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/deployer/test_workload_status.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/fixtures/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/README.md +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi300x.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi308x.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_amd_rx7800xt.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_ascend_310p3.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_ascend_910b2.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_hygon_k100ai.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_metax_c500.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_gb10.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100_mig.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h200.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4080super.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4090d.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx5090d.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/detect_output_thead_ppu.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi300x.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi308x.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_amd_rx7800xt.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_ascend_310p3.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_ascend_910b2.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_hygon_k100ai.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_metax_c500.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_mthreads_s5000.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100_mig.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h200.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4080super.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4090d.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx5090d.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/samples/topology_output_thead_ppu.json +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_amd.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_ascend.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_cambricon.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_detector_utils.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_hygon.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_metax.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_mthreads.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/tests/gpustack_runtime/detector/test_nvidia.py +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/uv.lock +0 -0
- {gpustack_runtime-0.2.2.post2 → gpustack_runtime-0.2.2.post4}/uv.toml +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gpustack-runtime
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.post4
|
|
4
4
|
Summary: GPUStack Runtime is library for detecting GPU resources and launching GPU workloads.
|
|
5
5
|
Project-URL: Homepage, https://github.com/gpustack/runtime
|
|
6
6
|
Project-URL: Bug Tracker, https://github.com/gpustack/gpustack/issues
|
|
@@ -27,8 +27,8 @@ version_tuple: VERSION_TUPLE
|
|
|
27
27
|
__commit_id__: COMMIT_ID
|
|
28
28
|
commit_id: COMMIT_ID
|
|
29
29
|
|
|
30
|
-
__version__ = version = '0.2.2.
|
|
31
|
-
__version_tuple__ = version_tuple = (0, 2, 2, '
|
|
30
|
+
__version__ = version = '0.2.2.post4'
|
|
31
|
+
__version_tuple__ = version_tuple = (0, 2, 2, 'post4')
|
|
32
32
|
try:
|
|
33
33
|
from ._version_appendix import git_commit
|
|
34
34
|
__commit_id__ = commit_id = git_commit
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
git_commit = "c736814"
|
|
@@ -0,0 +1,629 @@
|
|
|
1
|
+
from __future__ import annotations as __future_annotations__
|
|
2
|
+
|
|
3
|
+
import contextlib
|
|
4
|
+
import logging
|
|
5
|
+
import math
|
|
6
|
+
import threading
|
|
7
|
+
import time
|
|
8
|
+
from _ctypes import byref
|
|
9
|
+
from functools import lru_cache
|
|
10
|
+
|
|
11
|
+
from .. import envs
|
|
12
|
+
from ..logging import debug_log_exception, debug_log_warning
|
|
13
|
+
from . import DeviceMemoryStatusEnum, Topology, pynvml
|
|
14
|
+
from .__types__ import Detector, Device, Devices, ManufacturerEnum, TopologyDistanceEnum
|
|
15
|
+
from .__utils__ import (
|
|
16
|
+
PCIDevice,
|
|
17
|
+
bitmask_to_str,
|
|
18
|
+
byte_to_mebibyte,
|
|
19
|
+
get_brief_version,
|
|
20
|
+
get_memory,
|
|
21
|
+
get_numa_node_by_bdf,
|
|
22
|
+
get_numa_nodeset_size,
|
|
23
|
+
get_pci_devices,
|
|
24
|
+
get_utilization,
|
|
25
|
+
map_numa_node_to_cpu_affinity,
|
|
26
|
+
stringify_uuid,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
logger = logging.getLogger(__name__)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class NVIDIADetector(Detector):
|
|
33
|
+
"""
|
|
34
|
+
Detect NVIDIA GPUs.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
@staticmethod
|
|
38
|
+
@lru_cache(maxsize=1)
|
|
39
|
+
def is_supported() -> bool:
|
|
40
|
+
"""
|
|
41
|
+
Check if NVIDIA detection is supported.
|
|
42
|
+
|
|
43
|
+
Returns:
|
|
44
|
+
True if supported, False otherwise.
|
|
45
|
+
|
|
46
|
+
"""
|
|
47
|
+
supported = False
|
|
48
|
+
if envs.GPUSTACK_RUNTIME_DETECT.lower() not in ("auto", "nvidia"):
|
|
49
|
+
logger.debug("NVIDIA detection is disabled by environment variable")
|
|
50
|
+
return supported
|
|
51
|
+
|
|
52
|
+
pci_devs = NVIDIADetector.detect_pci_devices()
|
|
53
|
+
if not pci_devs and not envs.GPUSTACK_RUNTIME_DETECT_NO_PCI_CHECK:
|
|
54
|
+
logger.debug("No NVIDIA PCI devices found")
|
|
55
|
+
return supported
|
|
56
|
+
|
|
57
|
+
try:
|
|
58
|
+
pynvml.nvmlInit()
|
|
59
|
+
supported = True
|
|
60
|
+
except Exception:
|
|
61
|
+
debug_log_exception(logger, "Failed to initialize NVML")
|
|
62
|
+
|
|
63
|
+
return supported
|
|
64
|
+
|
|
65
|
+
@staticmethod
|
|
66
|
+
@lru_cache(maxsize=1)
|
|
67
|
+
def detect_pci_devices() -> dict[str, PCIDevice]:
|
|
68
|
+
# See https://pcisig.com/membership/member-companies?combine=NVIDIA.
|
|
69
|
+
pci_devs = get_pci_devices(vendor="0x10de")
|
|
70
|
+
if not pci_devs:
|
|
71
|
+
return {}
|
|
72
|
+
return {dev.address: dev for dev in pci_devs}
|
|
73
|
+
|
|
74
|
+
def __init__(self):
|
|
75
|
+
super().__init__(ManufacturerEnum.NVIDIA)
|
|
76
|
+
|
|
77
|
+
def detect(self) -> Devices | None:
|
|
78
|
+
"""
|
|
79
|
+
Detect NVIDIA GPUs using pynvml.
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
A list of detected NVIDIA GPU devices,
|
|
83
|
+
or None if not supported.
|
|
84
|
+
|
|
85
|
+
Raises:
|
|
86
|
+
If there is an error during detection.
|
|
87
|
+
|
|
88
|
+
"""
|
|
89
|
+
if not self.is_supported():
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
ret: Devices = []
|
|
93
|
+
|
|
94
|
+
try:
|
|
95
|
+
pci_devs = NVIDIADetector.detect_pci_devices()
|
|
96
|
+
|
|
97
|
+
pynvml.nvmlInit()
|
|
98
|
+
|
|
99
|
+
sys_driver_ver = pynvml.nvmlSystemGetDriverVersion()
|
|
100
|
+
|
|
101
|
+
sys_runtime_ver_original = pynvml.nvmlSystemGetCudaDriverVersion()
|
|
102
|
+
sys_runtime_ver_original = ".".join(
|
|
103
|
+
map(
|
|
104
|
+
str,
|
|
105
|
+
[
|
|
106
|
+
sys_runtime_ver_original // 1000,
|
|
107
|
+
(sys_runtime_ver_original % 1000) // 10,
|
|
108
|
+
(sys_runtime_ver_original % 10),
|
|
109
|
+
],
|
|
110
|
+
),
|
|
111
|
+
)
|
|
112
|
+
sys_runtime_ver = get_brief_version(
|
|
113
|
+
sys_runtime_ver_original,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
dev_count = pynvml.nvmlDeviceGetCount()
|
|
117
|
+
for dev_idx in range(dev_count):
|
|
118
|
+
dev = pynvml.nvmlDeviceGetHandleByIndex(dev_idx)
|
|
119
|
+
|
|
120
|
+
dev_cc_t = pynvml.nvmlDeviceGetCudaComputeCapability(dev)
|
|
121
|
+
dev_cc = ".".join(map(str, dev_cc_t))
|
|
122
|
+
|
|
123
|
+
dev_pci_info = pynvml.nvmlDeviceGetPciInfo(dev)
|
|
124
|
+
dev_bdf = str(dev_pci_info.busIdLegacy).lower()
|
|
125
|
+
|
|
126
|
+
dev_numa = get_numa_node_by_bdf(dev_bdf)
|
|
127
|
+
if not dev_numa:
|
|
128
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
129
|
+
dev_node_affinity = pynvml.nvmlDeviceGetMemoryAffinity(
|
|
130
|
+
dev,
|
|
131
|
+
get_numa_nodeset_size(),
|
|
132
|
+
pynvml.NVML_AFFINITY_SCOPE_NODE,
|
|
133
|
+
)
|
|
134
|
+
dev_numa = bitmask_to_str(list(dev_node_affinity))
|
|
135
|
+
|
|
136
|
+
dev_temp = None
|
|
137
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
138
|
+
dev_temp = pynvml.nvmlDeviceGetTemperature(
|
|
139
|
+
dev,
|
|
140
|
+
pynvml.NVML_TEMPERATURE_GPU,
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
dev_power = None
|
|
144
|
+
dev_power_used = None
|
|
145
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
146
|
+
dev_power = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(dev)
|
|
147
|
+
dev_power = dev_power // 1000 # mW to W
|
|
148
|
+
dev_power_used = (
|
|
149
|
+
pynvml.nvmlDeviceGetPowerUsage(dev) // 1000
|
|
150
|
+
) # mW to W
|
|
151
|
+
|
|
152
|
+
dev_mig_mode = pynvml.NVML_DEVICE_MIG_DISABLE
|
|
153
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
154
|
+
dev_mig_mode, _ = pynvml.nvmlDeviceGetMigMode(dev)
|
|
155
|
+
|
|
156
|
+
dev_index = dev_idx
|
|
157
|
+
if envs.GPUSTACK_RUNTIME_DETECT_PHYSICAL_INDEX_PRIORITY:
|
|
158
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
159
|
+
dev_index = pynvml.nvmlDeviceGetMinorNumber(dev)
|
|
160
|
+
|
|
161
|
+
# Report the physical card, whether or not MIG is enabled.
|
|
162
|
+
# MIG instances are partitioned on demand by the operator's
|
|
163
|
+
# device-manager; they are not separate allocatable devices
|
|
164
|
+
# in this inventory. A MIG-enabled card is marked ``mig``
|
|
165
|
+
# in the appendix instead.
|
|
166
|
+
|
|
167
|
+
dev_name = pynvml.nvmlDeviceGetName(dev)
|
|
168
|
+
|
|
169
|
+
dev_uuid = pynvml.nvmlDeviceGetUUID(dev)
|
|
170
|
+
|
|
171
|
+
dev_cores = None
|
|
172
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
173
|
+
dev_cores = pynvml.nvmlDeviceGetNumGpuCores(dev)
|
|
174
|
+
|
|
175
|
+
dev_cores_util = _get_sm_util_from_gpm_metrics(dev)
|
|
176
|
+
if dev_cores_util is None:
|
|
177
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
178
|
+
dev_util_rates = pynvml.nvmlDeviceGetUtilizationRates(dev)
|
|
179
|
+
dev_cores_util = dev_util_rates.gpu
|
|
180
|
+
if dev_cores_util is None:
|
|
181
|
+
debug_log_warning(
|
|
182
|
+
logger,
|
|
183
|
+
"Failed to get device %d cores utilization, setting to 0",
|
|
184
|
+
dev_index,
|
|
185
|
+
)
|
|
186
|
+
dev_cores_util = 0
|
|
187
|
+
|
|
188
|
+
dev_mem = 0
|
|
189
|
+
dev_mem_used = 0
|
|
190
|
+
dev_mem_status = DeviceMemoryStatusEnum.HEALTHY
|
|
191
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
192
|
+
dev_mem_info = pynvml.nvmlDeviceGetMemoryInfo(dev)
|
|
193
|
+
dev_mem = byte_to_mebibyte( # byte to MiB
|
|
194
|
+
dev_mem_info.total,
|
|
195
|
+
)
|
|
196
|
+
dev_mem_used = byte_to_mebibyte( # byte to MiB
|
|
197
|
+
dev_mem_info.used,
|
|
198
|
+
)
|
|
199
|
+
if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
|
|
200
|
+
dev_mem_ecc_errors = pynvml.nvmlDeviceGetMemoryErrorCounter(
|
|
201
|
+
dev,
|
|
202
|
+
pynvml.NVML_MEMORY_ERROR_TYPE_UNCORRECTED,
|
|
203
|
+
pynvml.NVML_VOLATILE_ECC,
|
|
204
|
+
pynvml.NVML_MEMORY_LOCATION_DRAM,
|
|
205
|
+
)
|
|
206
|
+
if dev_mem_ecc_errors > 0:
|
|
207
|
+
dev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
|
|
208
|
+
if dev_mem == 0:
|
|
209
|
+
dev_mem, dev_mem_used = get_memory()
|
|
210
|
+
|
|
211
|
+
dev_is_vgpu = False
|
|
212
|
+
if dev_bdf in pci_devs:
|
|
213
|
+
dev_is_vgpu = _is_vgpu(pci_devs[dev_bdf].config)
|
|
214
|
+
|
|
215
|
+
dev_appendix = {
|
|
216
|
+
"arch_family": _get_arch_family(dev_cc_t),
|
|
217
|
+
"vgpu": dev_is_vgpu,
|
|
218
|
+
"mig": dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE,
|
|
219
|
+
"bdf": dev_bdf,
|
|
220
|
+
}
|
|
221
|
+
if dev_numa:
|
|
222
|
+
dev_appendix["numa"] = dev_numa
|
|
223
|
+
|
|
224
|
+
if dev_fabric_info := _get_fabric_info(dev):
|
|
225
|
+
dev_appendix.update(dev_fabric_info)
|
|
226
|
+
|
|
227
|
+
ret.append(
|
|
228
|
+
Device(
|
|
229
|
+
manufacturer=self.manufacturer,
|
|
230
|
+
index=dev_index,
|
|
231
|
+
name=dev_name,
|
|
232
|
+
uuid=dev_uuid,
|
|
233
|
+
driver_version=sys_driver_ver,
|
|
234
|
+
runtime_version=sys_runtime_ver,
|
|
235
|
+
runtime_version_original=sys_runtime_ver_original,
|
|
236
|
+
compute_capability=dev_cc,
|
|
237
|
+
cores=dev_cores,
|
|
238
|
+
cores_utilization=dev_cores_util,
|
|
239
|
+
memory=dev_mem,
|
|
240
|
+
memory_used=dev_mem_used,
|
|
241
|
+
memory_utilization=get_utilization(dev_mem_used, dev_mem),
|
|
242
|
+
memory_status=dev_mem_status,
|
|
243
|
+
temperature=dev_temp,
|
|
244
|
+
power=dev_power,
|
|
245
|
+
power_used=dev_power_used,
|
|
246
|
+
appendix=dev_appendix,
|
|
247
|
+
),
|
|
248
|
+
)
|
|
249
|
+
except pynvml.NVMLError:
|
|
250
|
+
debug_log_exception(logger, "Failed to fetch devices")
|
|
251
|
+
raise
|
|
252
|
+
except Exception:
|
|
253
|
+
debug_log_exception(logger, "Failed to process devices fetching")
|
|
254
|
+
raise
|
|
255
|
+
|
|
256
|
+
return ret
|
|
257
|
+
|
|
258
|
+
def get_topology(self, devices: Devices | None = None) -> Topology | None:
|
|
259
|
+
"""
|
|
260
|
+
Get the Topology object between NVIDIA GPUs.
|
|
261
|
+
|
|
262
|
+
Args:
|
|
263
|
+
devices:
|
|
264
|
+
The list of detected NVIDIA devices.
|
|
265
|
+
If None, detect topology for all available devices.
|
|
266
|
+
|
|
267
|
+
Returns:
|
|
268
|
+
The Topology object, or None if not supported.
|
|
269
|
+
|
|
270
|
+
"""
|
|
271
|
+
if devices is None:
|
|
272
|
+
devices = self.detect()
|
|
273
|
+
if devices is None:
|
|
274
|
+
return None
|
|
275
|
+
|
|
276
|
+
ret = Topology(
|
|
277
|
+
manufacturer=self.manufacturer,
|
|
278
|
+
devices_count=len(devices),
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
get_links_cache = {}
|
|
282
|
+
|
|
283
|
+
try:
|
|
284
|
+
pynvml.nvmlInit()
|
|
285
|
+
|
|
286
|
+
for i, dev_i in enumerate(devices):
|
|
287
|
+
dev_i_bdf = dev_i.appendix.get("bdf")
|
|
288
|
+
if dev_i.appendix.get("sliced", False):
|
|
289
|
+
dev_i_handle = pynvml.nvmlDeviceGetHandleByPciBusId(dev_i_bdf)
|
|
290
|
+
else:
|
|
291
|
+
dev_i_handle = pynvml.nvmlDeviceGetHandleByUUID(dev_i.uuid)
|
|
292
|
+
|
|
293
|
+
# Get NUMA and CPU affinities.
|
|
294
|
+
ret.devices_numa_affinities[i] = dev_i.appendix.get("numa", "")
|
|
295
|
+
ret.devices_cpu_affinities[i] = map_numa_node_to_cpu_affinity(
|
|
296
|
+
ret.devices_numa_affinities[i],
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
# Get links state if applicable.
|
|
300
|
+
if dev_i_bdf in get_links_cache:
|
|
301
|
+
dev_i_links_state = get_links_cache[dev_i_bdf]
|
|
302
|
+
else:
|
|
303
|
+
dev_i_links_state = _get_links_state(dev_i_handle)
|
|
304
|
+
get_links_cache[dev_i_bdf] = dev_i_links_state
|
|
305
|
+
if dev_i_links_state:
|
|
306
|
+
ret.appendices[i].update(dev_i_links_state)
|
|
307
|
+
# In practice, if a card has an active *Link,
|
|
308
|
+
# then other cards in the same machine should be interconnected with it through the *Link.
|
|
309
|
+
if dev_i_links_state.get("links_active_count", 0) > 0:
|
|
310
|
+
for j, dev_j in enumerate(devices):
|
|
311
|
+
if dev_i.index == dev_j.index:
|
|
312
|
+
continue
|
|
313
|
+
ret.devices_distances[i][j] = TopologyDistanceEnum.LINK
|
|
314
|
+
ret.devices_distances[j][i] = TopologyDistanceEnum.LINK
|
|
315
|
+
continue
|
|
316
|
+
|
|
317
|
+
# Get distances to other devices.
|
|
318
|
+
for j, dev_j in enumerate(devices):
|
|
319
|
+
if dev_i.index == dev_j.index or ret.devices_distances[i][j] != 0:
|
|
320
|
+
continue
|
|
321
|
+
|
|
322
|
+
dev_j_bdf = dev_j.appendix.get("bdf")
|
|
323
|
+
if dev_i_bdf == dev_j_bdf:
|
|
324
|
+
distance = TopologyDistanceEnum.SELF
|
|
325
|
+
else:
|
|
326
|
+
if dev_j.appendix.get("sliced", False):
|
|
327
|
+
dev_j_handle = pynvml.nvmlDeviceGetHandleByPciBusId(
|
|
328
|
+
dev_j_bdf,
|
|
329
|
+
)
|
|
330
|
+
else:
|
|
331
|
+
dev_j_handle = pynvml.nvmlDeviceGetHandleByUUID(dev_j.uuid)
|
|
332
|
+
|
|
333
|
+
distance = TopologyDistanceEnum.UNK
|
|
334
|
+
try:
|
|
335
|
+
distance = pynvml.nvmlDeviceGetTopologyCommonAncestor(
|
|
336
|
+
dev_i_handle,
|
|
337
|
+
dev_j_handle,
|
|
338
|
+
)
|
|
339
|
+
except pynvml.NVMLError:
|
|
340
|
+
debug_log_exception(
|
|
341
|
+
logger,
|
|
342
|
+
"Failed to get distance between device %d and %d",
|
|
343
|
+
dev_i.index,
|
|
344
|
+
dev_j.index,
|
|
345
|
+
)
|
|
346
|
+
|
|
347
|
+
ret.devices_distances[i][j] = distance
|
|
348
|
+
ret.devices_distances[j][i] = distance
|
|
349
|
+
except Exception:
|
|
350
|
+
debug_log_exception(logger, "Failed to process topology fetching")
|
|
351
|
+
raise
|
|
352
|
+
|
|
353
|
+
return ret
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _get_gpm_metrics(
|
|
357
|
+
metrics: list[int],
|
|
358
|
+
dev: pynvml.c_nvmlDevice_t,
|
|
359
|
+
gpu_instance_id: int | None = None,
|
|
360
|
+
interval: float = 0.1,
|
|
361
|
+
) -> list[pynvml.c_nvmlGpmMetric_t] | None:
|
|
362
|
+
"""
|
|
363
|
+
Get GPM metrics for a device or a MIG GPU instance.
|
|
364
|
+
|
|
365
|
+
Args:
|
|
366
|
+
metrics:
|
|
367
|
+
A list of GPM metric IDs to query.
|
|
368
|
+
dev:
|
|
369
|
+
The NVML device handle.
|
|
370
|
+
gpu_instance_id:
|
|
371
|
+
The GPU instance ID for MIG devices.
|
|
372
|
+
interval:
|
|
373
|
+
Interval in seconds between two samples.
|
|
374
|
+
|
|
375
|
+
Returns:
|
|
376
|
+
A list of GPM metric structures, or None if failed.
|
|
377
|
+
|
|
378
|
+
"""
|
|
379
|
+
try:
|
|
380
|
+
dev_gpm_support = pynvml.nvmlGpmQueryDeviceSupport(dev)
|
|
381
|
+
if not bool(dev_gpm_support.isSupportedDevice):
|
|
382
|
+
return None
|
|
383
|
+
except pynvml.NVMLError:
|
|
384
|
+
debug_log_warning(logger, "Unsupported GPM query")
|
|
385
|
+
return None
|
|
386
|
+
|
|
387
|
+
dev_gpm_metrics = pynvml.c_nvmlGpmMetricsGet_t()
|
|
388
|
+
try:
|
|
389
|
+
dev_gpm_metrics.sample1 = pynvml.nvmlGpmSampleAlloc()
|
|
390
|
+
dev_gpm_metrics.sample2 = pynvml.nvmlGpmSampleAlloc()
|
|
391
|
+
if gpu_instance_id is None:
|
|
392
|
+
pynvml.nvmlGpmSampleGet(dev, dev_gpm_metrics.sample1)
|
|
393
|
+
time.sleep(interval)
|
|
394
|
+
pynvml.nvmlGpmSampleGet(dev, dev_gpm_metrics.sample2)
|
|
395
|
+
else:
|
|
396
|
+
pynvml.nvmlGpmMigSampleGet(dev, gpu_instance_id, dev_gpm_metrics.sample1)
|
|
397
|
+
time.sleep(interval)
|
|
398
|
+
pynvml.nvmlGpmMigSampleGet(dev, gpu_instance_id, dev_gpm_metrics.sample2)
|
|
399
|
+
dev_gpm_metrics.version = pynvml.NVML_GPM_METRICS_GET_VERSION
|
|
400
|
+
dev_gpm_metrics.numMetrics = len(metrics)
|
|
401
|
+
for metric_idx, metric in enumerate(metrics):
|
|
402
|
+
dev_gpm_metrics.metrics[metric_idx].metricId = metric
|
|
403
|
+
pynvml.nvmlGpmMetricsGet(dev_gpm_metrics)
|
|
404
|
+
except pynvml.NVMLError:
|
|
405
|
+
debug_log_exception(logger, "Failed to get GPM metrics")
|
|
406
|
+
return None
|
|
407
|
+
finally:
|
|
408
|
+
if dev_gpm_metrics.sample1:
|
|
409
|
+
pynvml.nvmlGpmSampleFree(dev_gpm_metrics.sample1)
|
|
410
|
+
if dev_gpm_metrics.sample2:
|
|
411
|
+
pynvml.nvmlGpmSampleFree(dev_gpm_metrics.sample2)
|
|
412
|
+
return list(dev_gpm_metrics.metrics)
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
_gpm_metrics_lock = threading.Lock()
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _get_sm_util_from_gpm_metrics(
|
|
419
|
+
dev: pynvml.c_nvmlDevice_t,
|
|
420
|
+
gpu_instance_id: int | None = None,
|
|
421
|
+
interval: float = 0.1,
|
|
422
|
+
) -> int | None:
|
|
423
|
+
"""
|
|
424
|
+
Get SM utilization from GPM metrics.
|
|
425
|
+
|
|
426
|
+
Args:
|
|
427
|
+
dev:
|
|
428
|
+
The NVML device handle.
|
|
429
|
+
gpu_instance_id:
|
|
430
|
+
The GPU instance ID for MIG devices.
|
|
431
|
+
interval:
|
|
432
|
+
Interval in seconds between two samples.
|
|
433
|
+
|
|
434
|
+
Returns:
|
|
435
|
+
The SM utilization as an integer percentage, or None if failed.
|
|
436
|
+
|
|
437
|
+
"""
|
|
438
|
+
with _gpm_metrics_lock:
|
|
439
|
+
dev_gpm_metrics = _get_gpm_metrics(
|
|
440
|
+
metrics=[pynvml.NVML_GPM_METRIC_SM_UTIL],
|
|
441
|
+
dev=dev,
|
|
442
|
+
gpu_instance_id=gpu_instance_id,
|
|
443
|
+
interval=interval,
|
|
444
|
+
)
|
|
445
|
+
|
|
446
|
+
if dev_gpm_metrics and not math.isnan(dev_gpm_metrics[0].value):
|
|
447
|
+
return int(dev_gpm_metrics[0].value)
|
|
448
|
+
|
|
449
|
+
return None
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def _extract_field_value(
|
|
453
|
+
field_value: pynvml.c_nvmlFieldValue_t,
|
|
454
|
+
) -> int | float | None:
|
|
455
|
+
"""
|
|
456
|
+
Extract the value from a NVML field value structure.
|
|
457
|
+
|
|
458
|
+
Args:
|
|
459
|
+
field_value:
|
|
460
|
+
The NVML field value structure.
|
|
461
|
+
|
|
462
|
+
Returns:
|
|
463
|
+
The extracted value as int, float, or None if unknown.
|
|
464
|
+
|
|
465
|
+
"""
|
|
466
|
+
if field_value.nvmlReturn != pynvml.NVML_SUCCESS:
|
|
467
|
+
return None
|
|
468
|
+
match field_value.valueType:
|
|
469
|
+
case pynvml.NVML_VALUE_TYPE_DOUBLE:
|
|
470
|
+
return field_value.value.dVal
|
|
471
|
+
case pynvml.NVML_VALUE_TYPE_UNSIGNED_INT:
|
|
472
|
+
return field_value.value.uiVal
|
|
473
|
+
case pynvml.NVML_VALUE_TYPE_UNSIGNED_LONG:
|
|
474
|
+
return field_value.value.ulVal
|
|
475
|
+
case pynvml.NVML_VALUE_TYPE_UNSIGNED_LONG_LONG:
|
|
476
|
+
return field_value.value.ullVal
|
|
477
|
+
case pynvml.NVML_VALUE_TYPE_SIGNED_LONG_LONG:
|
|
478
|
+
return field_value.value.sllVal
|
|
479
|
+
case pynvml.NVML_VALUE_TYPE_SIGNED_INT:
|
|
480
|
+
return field_value.value.siVal
|
|
481
|
+
case pynvml.NVML_VALUE_TYPE_UNSIGNED_SHORT:
|
|
482
|
+
return field_value.value.usVal
|
|
483
|
+
return None
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def _get_fabric_info(
|
|
487
|
+
dev: pynvml.c_nvmlDevice_t,
|
|
488
|
+
) -> dict | None:
|
|
489
|
+
"""
|
|
490
|
+
Get the NVSwitch fabric information for a device.
|
|
491
|
+
|
|
492
|
+
Args:
|
|
493
|
+
dev:
|
|
494
|
+
The NVML device handle.
|
|
495
|
+
|
|
496
|
+
Returns:
|
|
497
|
+
A dict includes fabric info or None if failed.
|
|
498
|
+
|
|
499
|
+
"""
|
|
500
|
+
try:
|
|
501
|
+
dev_fabric = pynvml.c_nvmlGpuFabricInfoV_t()
|
|
502
|
+
ret = pynvml.nvmlDeviceGetGpuFabricInfoV(dev, byref(dev_fabric))
|
|
503
|
+
if ret != pynvml.NVML_SUCCESS:
|
|
504
|
+
return None
|
|
505
|
+
if dev_fabric.state != pynvml.NVML_GPU_FABRIC_STATE_COMPLETED:
|
|
506
|
+
return None
|
|
507
|
+
return {
|
|
508
|
+
"fabric_cluster_uuid": stringify_uuid(bytes(dev_fabric.clusterUuid)),
|
|
509
|
+
"fabric_clique_id": dev_fabric.cliqueId,
|
|
510
|
+
}
|
|
511
|
+
except pynvml.NVMLError:
|
|
512
|
+
debug_log_warning(logger, "Failed to get NVSwitch fabric info")
|
|
513
|
+
|
|
514
|
+
return None
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
def _get_links_state(
|
|
518
|
+
dev: pynvml.c_nvmlDevice_t,
|
|
519
|
+
) -> dict | None:
|
|
520
|
+
"""
|
|
521
|
+
Get the NVLink links count and state for a device.
|
|
522
|
+
|
|
523
|
+
Args:
|
|
524
|
+
dev:
|
|
525
|
+
The NVML device handle.
|
|
526
|
+
|
|
527
|
+
Returns:
|
|
528
|
+
A dict includes links state or None if failed.
|
|
529
|
+
|
|
530
|
+
"""
|
|
531
|
+
dev_links_count = 0
|
|
532
|
+
try:
|
|
533
|
+
dev_fields = pynvml.nvmlDeviceGetFieldValues(
|
|
534
|
+
dev,
|
|
535
|
+
fieldIds=[pynvml.NVML_FI_DEV_NVLINK_LINK_COUNT],
|
|
536
|
+
)
|
|
537
|
+
dev_links_count = _extract_field_value(dev_fields[0])
|
|
538
|
+
except pynvml.NVMLError:
|
|
539
|
+
debug_log_warning(logger, "Failed to get NVLink links count")
|
|
540
|
+
if not dev_links_count:
|
|
541
|
+
return None
|
|
542
|
+
|
|
543
|
+
dev_links_state = 0
|
|
544
|
+
dev_links_active_count = 0
|
|
545
|
+
try:
|
|
546
|
+
for link_idx in range(int(dev_links_count)):
|
|
547
|
+
dev_link_state = pynvml.nvmlDeviceGetNvLinkState(dev, link_idx)
|
|
548
|
+
if dev_link_state:
|
|
549
|
+
dev_links_state |= 1 << link_idx
|
|
550
|
+
dev_links_active_count += 1
|
|
551
|
+
except pynvml.NVMLError:
|
|
552
|
+
debug_log_warning(logger, "Failed to get NVLink link state")
|
|
553
|
+
|
|
554
|
+
return {
|
|
555
|
+
"links_count": dev_links_count,
|
|
556
|
+
"links_state": dev_links_state,
|
|
557
|
+
"links_active_count": dev_links_active_count,
|
|
558
|
+
}
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _get_arch_family(dev_cc_t: list[int]) -> str:
|
|
562
|
+
"""
|
|
563
|
+
Get the architecture family based on the CUDA compute capability.
|
|
564
|
+
|
|
565
|
+
Args:
|
|
566
|
+
dev_cc_t:
|
|
567
|
+
The CUDA compute capability as a list of two integers.
|
|
568
|
+
|
|
569
|
+
Returns:
|
|
570
|
+
The architecture family as a string.
|
|
571
|
+
|
|
572
|
+
"""
|
|
573
|
+
match dev_cc_t[0]:
|
|
574
|
+
case 1:
|
|
575
|
+
return "Tesla"
|
|
576
|
+
case 2:
|
|
577
|
+
return "Fermi"
|
|
578
|
+
case 3:
|
|
579
|
+
return "Kepler"
|
|
580
|
+
case 5:
|
|
581
|
+
return "Maxwell"
|
|
582
|
+
case 6:
|
|
583
|
+
return "Pascal"
|
|
584
|
+
case 7:
|
|
585
|
+
return "Volta" if dev_cc_t[1] < 5 else "Turing"
|
|
586
|
+
case 8:
|
|
587
|
+
if dev_cc_t[1] < 9:
|
|
588
|
+
return "Ampere"
|
|
589
|
+
return "Ada-Lovelace"
|
|
590
|
+
case 9:
|
|
591
|
+
return "Hopper"
|
|
592
|
+
case 10 | 12:
|
|
593
|
+
return "Blackwell"
|
|
594
|
+
return "Unknown"
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
def _is_vgpu(dev_config: bytes) -> bool:
|
|
598
|
+
"""
|
|
599
|
+
Determine if the device is a vGPU based on its PCI configuration space.
|
|
600
|
+
|
|
601
|
+
"""
|
|
602
|
+
status = 0x06
|
|
603
|
+
cap_supported = 0x10
|
|
604
|
+
cap_start = 0x34
|
|
605
|
+
cap_vendor_specific_id = 0x09
|
|
606
|
+
|
|
607
|
+
if dev_config[status] & cap_supported == 0:
|
|
608
|
+
return False
|
|
609
|
+
|
|
610
|
+
# Find the capability list
|
|
611
|
+
dev_cap: bytes | None = None
|
|
612
|
+
visited = set()
|
|
613
|
+
pos = dev_config[cap_start]
|
|
614
|
+
while pos != 0 and pos not in visited and pos < len(dev_config) - 2:
|
|
615
|
+
visited.add(pos)
|
|
616
|
+
ptr = dev_config[pos : pos + 3] # id, next, length
|
|
617
|
+
if ptr[0] == 0xFF:
|
|
618
|
+
break
|
|
619
|
+
if ptr[0] == cap_vendor_specific_id:
|
|
620
|
+
dev_cap = dev_config[pos : pos + ptr[2]]
|
|
621
|
+
break
|
|
622
|
+
pos = ptr[1]
|
|
623
|
+
|
|
624
|
+
if not dev_cap or len(dev_cap) < 5:
|
|
625
|
+
return False
|
|
626
|
+
|
|
627
|
+
# Check for vGPU signature,
|
|
628
|
+
# which is either 0x56 (NVIDIA vGPU) or 0x46 (NVIDIA GRID).
|
|
629
|
+
return dev_cap[3] == 0x56 or dev_cap[4] == 0x46
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
git_commit = "480e0fd"
|