gpustack-runtime 0.2.2.post4__tar.gz → 0.2.2.post5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/PKG-INFO +1 -1
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/_version.py +2 -2
- gpustack_runtime-0.2.2.post5/gpustack_runtime/_version_appendix.py +1 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/nvidia.py +158 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/thead.py +241 -205
- gpustack_runtime-0.2.2.post4/gpustack_runtime/_version_appendix.py +0 -1
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.codespelldict +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.codespellrc +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.dockerignore +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.gitattributes +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.gitignore +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.pre-commit-config.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/.python-version +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/LICENSE +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/Makefile +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/README.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/deploy/manifests/docker-compose.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/deploy/manifests/kubernetes.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/docs/index.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/docs/modules/gpustack_runtime.deployer.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/docs/modules/gpustack_runtime.detector.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/docs/modules/gpustack_runtime.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/__main__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/_version.pyi +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/deployer.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/detector.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/images.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__patches__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/amd.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/ascend.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/hygon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/metax.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/thead.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/docker.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/k8s/devicemanager/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/kuberentes.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/podman.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/amd.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/ascend.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/cambricon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/hygon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/metax.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/mthreads.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyamdgpu/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyamdsmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pydcmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyhgml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyhgml/libhgml.so +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyhgml/libuki.so +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyhsa/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyixml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pymtml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pymxsml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pynvml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/pyrocmsmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/envs.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/logging.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/hatch.toml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/mkdocs.yml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/pack/Dockerfile +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/pack/Dockerfile.dummy +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/pyproject.toml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/pytest.ini +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/ruff.toml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_compare_versions.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_correct_runner_image.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_jsons.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_yamls.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_json.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_yaml.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/fixtures/test_nginx_entrypoint.sh +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/test_utils.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/deployer/test_workload_status.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/fixtures/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/README.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi300x.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi308x.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_amd_rx7800xt.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_ascend_310p3.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_ascend_910b2.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_hygon_k100ai.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_metax_c500.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_gb10.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100_mig.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h200.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4080super.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4090d.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx5090d.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/detect_output_thead_ppu.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi300x.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi308x.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_amd_rx7800xt.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_ascend_310p3.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_ascend_910b2.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_hygon_k100ai.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_metax_c500.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_mthreads_s5000.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100_mig.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h200.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4080super.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4090d.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx5090d.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/samples/topology_output_thead_ppu.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_amd.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_ascend.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_cambricon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_detector_utils.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_hygon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_metax.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_mthreads.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/tests/gpustack_runtime/detector/test_nvidia.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/uv.lock +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/uv.toml +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gpustack-runtime
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.post5
|
|
4
4
|
Summary: GPUStack Runtime is library for detecting GPU resources and launching GPU workloads.
|
|
5
5
|
Project-URL: Homepage, https://github.com/gpustack/runtime
|
|
6
6
|
Project-URL: Bug Tracker, https://github.com/gpustack/gpustack/issues
|
|
@@ -27,8 +27,8 @@ version_tuple: VERSION_TUPLE
|
|
|
27
27
|
__commit_id__: COMMIT_ID
|
|
28
28
|
commit_id: COMMIT_ID
|
|
29
29
|
|
|
30
|
-
__version__ = version = '0.2.2.
|
|
31
|
-
__version_tuple__ = version_tuple = (0, 2, 2, '
|
|
30
|
+
__version__ = version = '0.2.2.post5'
|
|
31
|
+
__version_tuple__ = version_tuple = (0, 2, 2, 'post5')
|
|
32
32
|
try:
|
|
33
33
|
from ._version_appendix import git_commit
|
|
34
34
|
__commit_id__ = commit_id = git_commit
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
git_commit = "fd63e6b"
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/nvidia.py
RENAMED
|
@@ -218,6 +218,22 @@ class NVIDIADetector(Detector):
|
|
|
218
218
|
"mig": dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE,
|
|
219
219
|
"bdf": dev_bdf,
|
|
220
220
|
}
|
|
221
|
+
if dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE:
|
|
222
|
+
dev_appendix["mig_devices"] = _get_mig_devices(
|
|
223
|
+
dev,
|
|
224
|
+
dev_idx,
|
|
225
|
+
dev_count,
|
|
226
|
+
dev_cc_t,
|
|
227
|
+
sys_driver_ver,
|
|
228
|
+
sys_runtime_ver,
|
|
229
|
+
sys_runtime_ver_original,
|
|
230
|
+
dev_cc,
|
|
231
|
+
dev_temp,
|
|
232
|
+
dev_power,
|
|
233
|
+
dev_power_used,
|
|
234
|
+
dev_bdf,
|
|
235
|
+
dev_numa,
|
|
236
|
+
)
|
|
221
237
|
if dev_numa:
|
|
222
238
|
dev_appendix["numa"] = dev_numa
|
|
223
239
|
|
|
@@ -558,6 +574,148 @@ def _get_links_state(
|
|
|
558
574
|
}
|
|
559
575
|
|
|
560
576
|
|
|
577
|
+
def _get_mig_devices(
|
|
578
|
+
dev,
|
|
579
|
+
dev_idx: int,
|
|
580
|
+
dev_count: int,
|
|
581
|
+
dev_cc_t,
|
|
582
|
+
sys_driver_ver,
|
|
583
|
+
sys_runtime_ver,
|
|
584
|
+
sys_runtime_ver_original,
|
|
585
|
+
dev_cc,
|
|
586
|
+
dev_temp,
|
|
587
|
+
dev_power,
|
|
588
|
+
dev_power_used,
|
|
589
|
+
dev_bdf: str,
|
|
590
|
+
dev_numa,
|
|
591
|
+
) -> list[dict]:
|
|
592
|
+
"""
|
|
593
|
+
Enumerate the card's current MIG devices with the same detail a plain
|
|
594
|
+
device carries (profile name, uuid, compute/memory utilization, memory
|
|
595
|
+
health, temperature and power), returned as appendix entries of the
|
|
596
|
+
physical card rather than standalone devices. Empty when MIG is enabled
|
|
597
|
+
but no GPU instances exist yet.
|
|
598
|
+
"""
|
|
599
|
+
ret: list[dict] = []
|
|
600
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
601
|
+
for mdev_idx in range(pynvml.nvmlDeviceGetMaxMigDeviceCount(dev)):
|
|
602
|
+
mdev = None
|
|
603
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
604
|
+
mdev = pynvml.nvmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
|
|
605
|
+
if not mdev:
|
|
606
|
+
continue
|
|
607
|
+
|
|
608
|
+
mdev_uuid = pynvml.nvmlDeviceGetUUID(mdev)
|
|
609
|
+
|
|
610
|
+
mdev_mem = 0
|
|
611
|
+
mdev_mem_used = 0
|
|
612
|
+
mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
|
|
613
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
614
|
+
mdev_mem_info = pynvml.nvmlDeviceGetMemoryInfo(mdev)
|
|
615
|
+
mdev_mem = byte_to_mebibyte(mdev_mem_info.total)
|
|
616
|
+
mdev_mem_used = byte_to_mebibyte(mdev_mem_info.used)
|
|
617
|
+
if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
|
|
618
|
+
mdev_mem_ecc_errors = pynvml.nvmlDeviceGetMemoryErrorCounter(
|
|
619
|
+
mdev,
|
|
620
|
+
pynvml.NVML_MEMORY_ERROR_TYPE_UNCORRECTED,
|
|
621
|
+
pynvml.NVML_AGGREGATE_ECC,
|
|
622
|
+
pynvml.NVML_MEMORY_LOCATION_SRAM,
|
|
623
|
+
)
|
|
624
|
+
if mdev_mem_ecc_errors > 0:
|
|
625
|
+
mdev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
|
|
626
|
+
|
|
627
|
+
mdev_appendix = {
|
|
628
|
+
"arch_family": _get_arch_family(dev_cc_t),
|
|
629
|
+
"vgpu": True,
|
|
630
|
+
"sliced": True,
|
|
631
|
+
"mig": True,
|
|
632
|
+
"bdf": dev_bdf,
|
|
633
|
+
}
|
|
634
|
+
if dev_numa:
|
|
635
|
+
mdev_appendix["numa"] = dev_numa
|
|
636
|
+
|
|
637
|
+
mdev_gi_id = pynvml.nvmlDeviceGetGpuInstanceId(mdev)
|
|
638
|
+
mdev_appendix["gpu_instance_id"] = mdev_gi_id
|
|
639
|
+
mdev_ci_id = pynvml.nvmlDeviceGetComputeInstanceId(mdev)
|
|
640
|
+
mdev_appendix["compute_instance_id"] = mdev_ci_id
|
|
641
|
+
|
|
642
|
+
mdev_cores_util = _get_sm_util_from_gpm_metrics(dev, mdev_gi_id)
|
|
643
|
+
|
|
644
|
+
mdev_name = ""
|
|
645
|
+
mdev_cores = None
|
|
646
|
+
mdev_gi = pynvml.nvmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
|
|
647
|
+
mdev_ci = pynvml.nvmlGpuInstanceGetComputeInstanceById(
|
|
648
|
+
mdev_gi,
|
|
649
|
+
mdev_ci_id,
|
|
650
|
+
)
|
|
651
|
+
mdev_gi_info = pynvml.nvmlGpuInstanceGetInfo(mdev_gi)
|
|
652
|
+
mdev_ci_info = pynvml.nvmlComputeInstanceGetInfo(mdev_ci)
|
|
653
|
+
for dev_gi_prf_id in range(pynvml.NVML_GPU_INSTANCE_PROFILE_COUNT):
|
|
654
|
+
try:
|
|
655
|
+
dev_gi_prf = pynvml.nvmlDeviceGetGpuInstanceProfileInfo(
|
|
656
|
+
dev,
|
|
657
|
+
dev_gi_prf_id,
|
|
658
|
+
)
|
|
659
|
+
if dev_gi_prf.id != mdev_gi_info.profileId:
|
|
660
|
+
continue
|
|
661
|
+
except pynvml.NVMLError:
|
|
662
|
+
continue
|
|
663
|
+
|
|
664
|
+
gi_mem = round(math.ceil(dev_gi_prf.memorySizeMB >> 10))
|
|
665
|
+
gi_prf_name = getattr(dev_gi_prf, "name", None)
|
|
666
|
+
mdev_name = (
|
|
667
|
+
gi_prf_name.removeprefix("MIG ")
|
|
668
|
+
if gi_prf_name
|
|
669
|
+
else f"{dev_gi_prf.sliceCount}g.{gi_mem}gb"
|
|
670
|
+
)
|
|
671
|
+
|
|
672
|
+
for dev_ci_prf_id in range(
|
|
673
|
+
pynvml.NVML_COMPUTE_INSTANCE_PROFILE_COUNT,
|
|
674
|
+
):
|
|
675
|
+
for dev_cig_prf_id in range(
|
|
676
|
+
pynvml.NVML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
|
|
677
|
+
):
|
|
678
|
+
try:
|
|
679
|
+
mdev_ci_prf = (
|
|
680
|
+
pynvml.nvmlGpuInstanceGetComputeInstanceProfileInfo(
|
|
681
|
+
mdev_gi,
|
|
682
|
+
dev_ci_prf_id,
|
|
683
|
+
dev_cig_prf_id,
|
|
684
|
+
)
|
|
685
|
+
)
|
|
686
|
+
if mdev_ci_prf.id != mdev_ci_info.profileId:
|
|
687
|
+
continue
|
|
688
|
+
except pynvml.NVMLError:
|
|
689
|
+
continue
|
|
690
|
+
mdev_cores = mdev_ci_prf.multiprocessorCount
|
|
691
|
+
break
|
|
692
|
+
|
|
693
|
+
break
|
|
694
|
+
|
|
695
|
+
ret.append(
|
|
696
|
+
{
|
|
697
|
+
"index": mdev_idx + dev_count * (dev_idx + 1),
|
|
698
|
+
"name": mdev_name,
|
|
699
|
+
"uuid": mdev_uuid,
|
|
700
|
+
"driver_version": sys_driver_ver,
|
|
701
|
+
"runtime_version": sys_runtime_ver,
|
|
702
|
+
"runtime_version_original": sys_runtime_ver_original,
|
|
703
|
+
"compute_capability": dev_cc,
|
|
704
|
+
"cores": mdev_cores,
|
|
705
|
+
"cores_utilization": mdev_cores_util,
|
|
706
|
+
"memory": mdev_mem,
|
|
707
|
+
"memory_used": mdev_mem_used,
|
|
708
|
+
"memory_utilization": get_utilization(mdev_mem_used, mdev_mem),
|
|
709
|
+
"memory_status": mdev_mem_status,
|
|
710
|
+
"temperature": dev_temp,
|
|
711
|
+
"power": dev_power,
|
|
712
|
+
"power_used": dev_power_used,
|
|
713
|
+
"appendix": mdev_appendix,
|
|
714
|
+
},
|
|
715
|
+
)
|
|
716
|
+
return ret
|
|
717
|
+
|
|
718
|
+
|
|
561
719
|
def _get_arch_family(dev_cc_t: list[int]) -> str:
|
|
562
720
|
"""
|
|
563
721
|
Get the architecture family based on the CUDA compute capability.
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/thead.py
RENAMED
|
@@ -162,220 +162,104 @@ class THeadDetector(Detector):
|
|
|
162
162
|
|
|
163
163
|
dev_index = dev_idx
|
|
164
164
|
|
|
165
|
-
#
|
|
165
|
+
# Report the physical card, whether or not MIG is enabled.
|
|
166
|
+
# MIG instances are partitioned on demand by the operator's
|
|
167
|
+
# device-manager; they are not separate allocatable devices
|
|
168
|
+
# in this inventory. A MIG-enabled card is marked ``mig``
|
|
169
|
+
# in the appendix instead.
|
|
166
170
|
|
|
167
|
-
|
|
168
|
-
dev_name = pyhgml.hgmlDeviceGetName(dev)
|
|
171
|
+
dev_name = pyhgml.hgmlDeviceGetName(dev)
|
|
169
172
|
|
|
170
|
-
|
|
173
|
+
dev_uuid = pyhgml.hgmlDeviceGetUUID(dev)
|
|
171
174
|
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
dev_cores_util = None
|
|
177
|
-
with contextlib.suppress(pyhgml.HGMLError):
|
|
178
|
-
dev_util_rates = pyhgml.hgmlDeviceGetUtilizationRates(dev)
|
|
179
|
-
dev_cores_util = dev_util_rates.gpu
|
|
180
|
-
if dev_cores_util is None:
|
|
181
|
-
debug_log_warning(
|
|
182
|
-
logger,
|
|
183
|
-
"Failed to get device %d cores utilization, setting to 0",
|
|
184
|
-
dev_index,
|
|
185
|
-
)
|
|
186
|
-
dev_cores_util = 0
|
|
175
|
+
dev_cores = None
|
|
176
|
+
with contextlib.suppress(pyhgml.HGMLError):
|
|
177
|
+
dev_cores = pyhgml.hgmlDeviceGetNumGpuCores(dev)
|
|
187
178
|
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
dev_mem_info.used,
|
|
198
|
-
)
|
|
199
|
-
if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
|
|
200
|
-
dev_mem_ecc_errors = pyhgml.hgmlDeviceGetMemoryErrorCounter(
|
|
201
|
-
dev,
|
|
202
|
-
pyhgml.HGML_MEMORY_ERROR_TYPE_UNCORRECTED,
|
|
203
|
-
pyhgml.HGML_VOLATILE_ECC,
|
|
204
|
-
pyhgml.HGML_MEMORY_LOCATION_DRAM,
|
|
205
|
-
)
|
|
206
|
-
if dev_mem_ecc_errors > 0:
|
|
207
|
-
dev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
|
|
208
|
-
|
|
209
|
-
dev_is_vgpu = False
|
|
210
|
-
if dev_bdf:
|
|
211
|
-
dev_is_vgpu = get_physical_function_by_bdf(dev_bdf) != dev_bdf
|
|
212
|
-
|
|
213
|
-
dev_appendix = {
|
|
214
|
-
"vgpu": dev_is_vgpu,
|
|
215
|
-
"bdf": dev_bdf,
|
|
216
|
-
}
|
|
217
|
-
if dev_numa:
|
|
218
|
-
dev_appendix["numa"] = dev_numa
|
|
219
|
-
|
|
220
|
-
ret.append(
|
|
221
|
-
Device(
|
|
222
|
-
manufacturer=self.manufacturer,
|
|
223
|
-
index=dev_index,
|
|
224
|
-
name=dev_name,
|
|
225
|
-
uuid=dev_uuid,
|
|
226
|
-
driver_version=sys_driver_ver,
|
|
227
|
-
runtime_version=sys_runtime_ver,
|
|
228
|
-
runtime_version_original=sys_runtime_ver_original,
|
|
229
|
-
compute_capability=dev_cc,
|
|
230
|
-
cores=dev_cores,
|
|
231
|
-
cores_utilization=dev_cores_util,
|
|
232
|
-
memory=dev_mem,
|
|
233
|
-
memory_used=dev_mem_used,
|
|
234
|
-
memory_utilization=get_utilization(dev_mem_used, dev_mem),
|
|
235
|
-
memory_status=dev_mem_status,
|
|
236
|
-
temperature=dev_temp,
|
|
237
|
-
power=dev_power,
|
|
238
|
-
power_used=dev_power_used,
|
|
239
|
-
appendix=dev_appendix,
|
|
240
|
-
),
|
|
179
|
+
dev_cores_util = None
|
|
180
|
+
with contextlib.suppress(pyhgml.HGMLError):
|
|
181
|
+
dev_util_rates = pyhgml.hgmlDeviceGetUtilizationRates(dev)
|
|
182
|
+
dev_cores_util = dev_util_rates.gpu
|
|
183
|
+
if dev_cores_util is None:
|
|
184
|
+
debug_log_warning(
|
|
185
|
+
logger,
|
|
186
|
+
"Failed to get device %d cores utilization, setting to 0",
|
|
187
|
+
dev_index,
|
|
241
188
|
)
|
|
189
|
+
dev_cores_util = 0
|
|
242
190
|
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
if not
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
mdev_mem = 0
|
|
261
|
-
mdev_mem_used = 0
|
|
262
|
-
mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
|
|
263
|
-
with contextlib.suppress(pyhgml.HGMLError):
|
|
264
|
-
mdev_mem_info = pyhgml.hgmlDeviceGetMemoryInfo(mdev)
|
|
265
|
-
mdev_mem = byte_to_mebibyte( # byte to MiB
|
|
266
|
-
mdev_mem_info.total,
|
|
267
|
-
)
|
|
268
|
-
mdev_mem_used = byte_to_mebibyte( # byte to MiB
|
|
269
|
-
mdev_mem_info.used,
|
|
191
|
+
dev_mem = 0
|
|
192
|
+
dev_mem_used = 0
|
|
193
|
+
dev_mem_status = DeviceMemoryStatusEnum.HEALTHY
|
|
194
|
+
with contextlib.suppress(pyhgml.HGMLError):
|
|
195
|
+
dev_mem_info = pyhgml.hgmlDeviceGetMemoryInfo(dev)
|
|
196
|
+
dev_mem = byte_to_mebibyte( # byte to MiB
|
|
197
|
+
dev_mem_info.total,
|
|
198
|
+
)
|
|
199
|
+
dev_mem_used = byte_to_mebibyte( # byte to MiB
|
|
200
|
+
dev_mem_info.used,
|
|
201
|
+
)
|
|
202
|
+
if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
|
|
203
|
+
dev_mem_ecc_errors = pyhgml.hgmlDeviceGetMemoryErrorCounter(
|
|
204
|
+
dev,
|
|
205
|
+
pyhgml.HGML_MEMORY_ERROR_TYPE_UNCORRECTED,
|
|
206
|
+
pyhgml.HGML_VOLATILE_ECC,
|
|
207
|
+
pyhgml.HGML_MEMORY_LOCATION_DRAM,
|
|
270
208
|
)
|
|
271
|
-
if
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
mdev_gi = pyhgml.hgmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
|
|
299
|
-
mdev_ci = pyhgml.hgmlGpuInstanceGetComputeInstanceById(
|
|
300
|
-
mdev_gi,
|
|
301
|
-
mdev_ci_id,
|
|
209
|
+
if dev_mem_ecc_errors > 0:
|
|
210
|
+
dev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
|
|
211
|
+
|
|
212
|
+
dev_is_vgpu = False
|
|
213
|
+
if dev_bdf:
|
|
214
|
+
dev_is_vgpu = get_physical_function_by_bdf(dev_bdf) != dev_bdf
|
|
215
|
+
|
|
216
|
+
dev_appendix = {
|
|
217
|
+
"vgpu": dev_is_vgpu,
|
|
218
|
+
"mig": dev_mig_mode != pyhgml.HGML_DEVICE_MIG_DISABLE,
|
|
219
|
+
"bdf": dev_bdf,
|
|
220
|
+
}
|
|
221
|
+
if dev_mig_mode != pyhgml.HGML_DEVICE_MIG_DISABLE:
|
|
222
|
+
dev_appendix["mig_devices"] = _get_mig_devices(
|
|
223
|
+
dev,
|
|
224
|
+
dev_idx,
|
|
225
|
+
dev_count,
|
|
226
|
+
dev_cc_t,
|
|
227
|
+
sys_driver_ver,
|
|
228
|
+
sys_runtime_ver,
|
|
229
|
+
sys_runtime_ver_original,
|
|
230
|
+
dev_cc,
|
|
231
|
+
dev_temp,
|
|
232
|
+
dev_power,
|
|
233
|
+
dev_power_used,
|
|
234
|
+
dev_bdf,
|
|
235
|
+
dev_numa,
|
|
302
236
|
)
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
237
|
+
if dev_numa:
|
|
238
|
+
dev_appendix["numa"] = dev_numa
|
|
239
|
+
|
|
240
|
+
ret.append(
|
|
241
|
+
Device(
|
|
242
|
+
manufacturer=self.manufacturer,
|
|
243
|
+
index=dev_index,
|
|
244
|
+
name=dev_name,
|
|
245
|
+
uuid=dev_uuid,
|
|
246
|
+
driver_version=sys_driver_ver,
|
|
247
|
+
runtime_version=sys_runtime_ver,
|
|
248
|
+
runtime_version_original=sys_runtime_ver_original,
|
|
249
|
+
compute_capability=dev_cc,
|
|
250
|
+
cores=dev_cores,
|
|
251
|
+
cores_utilization=dev_cores_util,
|
|
252
|
+
memory=dev_mem,
|
|
253
|
+
memory_used=dev_mem_used,
|
|
254
|
+
memory_utilization=get_utilization(dev_mem_used, dev_mem),
|
|
255
|
+
memory_status=dev_mem_status,
|
|
256
|
+
temperature=dev_temp,
|
|
257
|
+
power=dev_power,
|
|
258
|
+
power_used=dev_power_used,
|
|
259
|
+
appendix=dev_appendix,
|
|
260
|
+
),
|
|
261
|
+
)
|
|
317
262
|
|
|
318
|
-
for dev_ci_prf_id in range(
|
|
319
|
-
pyhgml.HGML_COMPUTE_INSTANCE_PROFILE_COUNT,
|
|
320
|
-
):
|
|
321
|
-
for dev_cig_prf_id in range(
|
|
322
|
-
pyhgml.HGML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
|
|
323
|
-
):
|
|
324
|
-
try:
|
|
325
|
-
mdev_ci_prf = pyhgml.hgmlGpuInstanceGetComputeInstanceProfileInfo(
|
|
326
|
-
mdev_gi,
|
|
327
|
-
dev_ci_prf_id,
|
|
328
|
-
dev_cig_prf_id,
|
|
329
|
-
)
|
|
330
|
-
if mdev_ci_prf.id != mdev_ci_info.profileId:
|
|
331
|
-
continue
|
|
332
|
-
except pyhgml.HGMLError:
|
|
333
|
-
continue
|
|
334
|
-
|
|
335
|
-
ci_slice = _get_compute_instance_slice(dev_ci_prf_id)
|
|
336
|
-
gi_slice = _get_gpu_instance_slice(dev_gi_prf_id)
|
|
337
|
-
if ci_slice == gi_slice:
|
|
338
|
-
if hasattr(dev_gi_prf, "name"):
|
|
339
|
-
mdev_name = dev_gi_prf.name
|
|
340
|
-
else:
|
|
341
|
-
gi_mem = round(
|
|
342
|
-
math.ceil(dev_gi_prf.memorySizeMB >> 10),
|
|
343
|
-
)
|
|
344
|
-
mdev_name = f"{gi_slice}g.{gi_mem}gb"
|
|
345
|
-
elif hasattr(mdev_ci_prf, "name"):
|
|
346
|
-
mdev_name = mdev_ci_prf.name
|
|
347
|
-
else:
|
|
348
|
-
gi_mem = round(
|
|
349
|
-
math.ceil(dev_gi_prf.memorySizeMB >> 10),
|
|
350
|
-
)
|
|
351
|
-
mdev_name = f"{ci_slice}u.{gi_slice}g.{gi_mem}gb"
|
|
352
|
-
|
|
353
|
-
mdev_cores = mdev_ci_prf.multiprocessorCount
|
|
354
|
-
|
|
355
|
-
break
|
|
356
|
-
|
|
357
|
-
ret.append(
|
|
358
|
-
Device(
|
|
359
|
-
manufacturer=self.manufacturer,
|
|
360
|
-
index=mdev_index,
|
|
361
|
-
name=mdev_name,
|
|
362
|
-
uuid=mdev_uuid,
|
|
363
|
-
driver_version=sys_driver_ver,
|
|
364
|
-
runtime_version=sys_runtime_ver,
|
|
365
|
-
runtime_version_original=sys_runtime_ver_original,
|
|
366
|
-
compute_capability=dev_cc,
|
|
367
|
-
cores=mdev_cores,
|
|
368
|
-
cores_utilization=mdev_cores_util,
|
|
369
|
-
memory=mdev_mem,
|
|
370
|
-
memory_used=mdev_mem_used,
|
|
371
|
-
memory_utilization=get_utilization(mdev_mem_used, mdev_mem),
|
|
372
|
-
memory_status=mdev_mem_status,
|
|
373
|
-
temperature=dev_temp,
|
|
374
|
-
power=dev_power,
|
|
375
|
-
power_used=dev_power_used,
|
|
376
|
-
appendix=mdev_appendix,
|
|
377
|
-
),
|
|
378
|
-
)
|
|
379
263
|
except pyhgml.HGMLError:
|
|
380
264
|
debug_log_exception(logger, "Failed to fetch devices")
|
|
381
265
|
raise
|
|
@@ -655,6 +539,158 @@ def _get_links_state(
|
|
|
655
539
|
}
|
|
656
540
|
|
|
657
541
|
|
|
542
|
+
def _get_mig_devices(
|
|
543
|
+
dev,
|
|
544
|
+
dev_idx: int,
|
|
545
|
+
dev_count: int,
|
|
546
|
+
sys_driver_ver,
|
|
547
|
+
sys_runtime_ver,
|
|
548
|
+
sys_runtime_ver_original,
|
|
549
|
+
dev_cc,
|
|
550
|
+
dev_temp,
|
|
551
|
+
dev_power,
|
|
552
|
+
dev_power_used,
|
|
553
|
+
dev_bdf: str,
|
|
554
|
+
dev_numa,
|
|
555
|
+
) -> list[dict]:
|
|
556
|
+
"""
|
|
557
|
+
Enumerate the card's current MIG devices with the same detail a plain
|
|
558
|
+
device carries (profile name, uuid, compute/memory utilization, memory
|
|
559
|
+
health, temperature and power), returned as appendix entries of the
|
|
560
|
+
physical card rather than standalone devices. Empty when MIG is enabled
|
|
561
|
+
but no GPU instances exist yet.
|
|
562
|
+
"""
|
|
563
|
+
ret: list[dict] = []
|
|
564
|
+
with contextlib.suppress(pyhgml.HGMLError):
|
|
565
|
+
for mdev_idx in range(pyhgml.hgmlDeviceGetMaxMigDeviceCount(dev)):
|
|
566
|
+
mdev = None
|
|
567
|
+
with contextlib.suppress(pyhgml.HGMLError):
|
|
568
|
+
mdev = pyhgml.hgmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
|
|
569
|
+
if not mdev:
|
|
570
|
+
continue
|
|
571
|
+
|
|
572
|
+
mdev_uuid = pyhgml.hgmlDeviceGetUUID(mdev)
|
|
573
|
+
|
|
574
|
+
mdev_mem = 0
|
|
575
|
+
mdev_mem_used = 0
|
|
576
|
+
mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
|
|
577
|
+
with contextlib.suppress(pyhgml.HGMLError):
|
|
578
|
+
mdev_mem_info = pyhgml.hgmlDeviceGetMemoryInfo(mdev)
|
|
579
|
+
mdev_mem = byte_to_mebibyte(mdev_mem_info.total)
|
|
580
|
+
mdev_mem_used = byte_to_mebibyte(mdev_mem_info.used)
|
|
581
|
+
if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
|
|
582
|
+
mdev_mem_ecc_errors = pyhgml.hgmlDeviceGetMemoryErrorCounter(
|
|
583
|
+
mdev,
|
|
584
|
+
pyhgml.HGML_MEMORY_ERROR_TYPE_UNCORRECTED,
|
|
585
|
+
pyhgml.HGML_AGGREGATE_ECC,
|
|
586
|
+
pyhgml.HGML_MEMORY_LOCATION_SRAM,
|
|
587
|
+
)
|
|
588
|
+
if mdev_mem_ecc_errors > 0:
|
|
589
|
+
mdev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
|
|
590
|
+
|
|
591
|
+
mdev_appendix = {
|
|
592
|
+
"vgpu": True,
|
|
593
|
+
"sliced": True,
|
|
594
|
+
"mig": True,
|
|
595
|
+
"bdf": dev_bdf,
|
|
596
|
+
}
|
|
597
|
+
if dev_numa:
|
|
598
|
+
mdev_appendix["numa"] = dev_numa
|
|
599
|
+
|
|
600
|
+
mdev_gi_id = pyhgml.hgmlDeviceGetGpuInstanceId(mdev)
|
|
601
|
+
mdev_appendix["gpu_instance_id"] = mdev_gi_id
|
|
602
|
+
mdev_ci_id = pyhgml.hgmlDeviceGetComputeInstanceId(mdev)
|
|
603
|
+
mdev_appendix["compute_instance_id"] = mdev_ci_id
|
|
604
|
+
|
|
605
|
+
mdev_cores_util = _get_sm_util_from_gpm_metrics(dev, mdev_gi_id)
|
|
606
|
+
|
|
607
|
+
mdev_name = ""
|
|
608
|
+
mdev_cores = None
|
|
609
|
+
mdev_gi = pyhgml.hgmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
|
|
610
|
+
mdev_ci = pyhgml.hgmlGpuInstanceGetComputeInstanceById(
|
|
611
|
+
mdev_gi,
|
|
612
|
+
mdev_ci_id,
|
|
613
|
+
)
|
|
614
|
+
mdev_gi_info = pyhgml.hgmlGpuInstanceGetInfo(mdev_gi)
|
|
615
|
+
mdev_ci_info = pyhgml.hgmlComputeInstanceGetInfo(mdev_ci)
|
|
616
|
+
for dev_gi_prf_id in range(pyhgml.HGML_GPU_INSTANCE_PROFILE_COUNT):
|
|
617
|
+
try:
|
|
618
|
+
dev_gi_prf = pyhgml.hgmlDeviceGetGpuInstanceProfileInfo(
|
|
619
|
+
dev,
|
|
620
|
+
dev_gi_prf_id,
|
|
621
|
+
)
|
|
622
|
+
if dev_gi_prf.id != mdev_gi_info.profileId:
|
|
623
|
+
continue
|
|
624
|
+
except pyhgml.HGMLError:
|
|
625
|
+
continue
|
|
626
|
+
|
|
627
|
+
for dev_ci_prf_id in range(
|
|
628
|
+
pyhgml.HGML_COMPUTE_INSTANCE_PROFILE_COUNT,
|
|
629
|
+
):
|
|
630
|
+
for dev_cig_prf_id in range(
|
|
631
|
+
pyhgml.HGML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
|
|
632
|
+
):
|
|
633
|
+
try:
|
|
634
|
+
mdev_ci_prf = (
|
|
635
|
+
pyhgml.hgmlGpuInstanceGetComputeInstanceProfileInfo(
|
|
636
|
+
mdev_gi,
|
|
637
|
+
dev_ci_prf_id,
|
|
638
|
+
dev_cig_prf_id,
|
|
639
|
+
)
|
|
640
|
+
)
|
|
641
|
+
if mdev_ci_prf.id != mdev_ci_info.profileId:
|
|
642
|
+
continue
|
|
643
|
+
except pyhgml.HGMLError:
|
|
644
|
+
continue
|
|
645
|
+
|
|
646
|
+
ci_slice = _get_compute_instance_slice(dev_ci_prf_id)
|
|
647
|
+
gi_slice = _get_gpu_instance_slice(dev_gi_prf_id)
|
|
648
|
+
if ci_slice == gi_slice:
|
|
649
|
+
if hasattr(dev_gi_prf, "name"):
|
|
650
|
+
mdev_name = dev_gi_prf.name
|
|
651
|
+
else:
|
|
652
|
+
gi_mem = round(
|
|
653
|
+
math.ceil(dev_gi_prf.memorySizeMB >> 10),
|
|
654
|
+
)
|
|
655
|
+
mdev_name = f"{gi_slice}g.{gi_mem}gb"
|
|
656
|
+
elif hasattr(mdev_ci_prf, "name"):
|
|
657
|
+
mdev_name = mdev_ci_prf.name
|
|
658
|
+
else:
|
|
659
|
+
gi_mem = round(
|
|
660
|
+
math.ceil(dev_gi_prf.memorySizeMB >> 10),
|
|
661
|
+
)
|
|
662
|
+
mdev_name = f"{ci_slice}u.{gi_slice}g.{gi_mem}gb"
|
|
663
|
+
|
|
664
|
+
mdev_cores = mdev_ci_prf.multiprocessorCount
|
|
665
|
+
|
|
666
|
+
break
|
|
667
|
+
|
|
668
|
+
break
|
|
669
|
+
|
|
670
|
+
ret.append(
|
|
671
|
+
{
|
|
672
|
+
"index": mdev_idx + dev_count * (dev_idx + 1),
|
|
673
|
+
"name": mdev_name,
|
|
674
|
+
"uuid": mdev_uuid,
|
|
675
|
+
"driver_version": sys_driver_ver,
|
|
676
|
+
"runtime_version": sys_runtime_ver,
|
|
677
|
+
"runtime_version_original": sys_runtime_ver_original,
|
|
678
|
+
"compute_capability": dev_cc,
|
|
679
|
+
"cores": mdev_cores,
|
|
680
|
+
"cores_utilization": mdev_cores_util,
|
|
681
|
+
"memory": mdev_mem,
|
|
682
|
+
"memory_used": mdev_mem_used,
|
|
683
|
+
"memory_utilization": get_utilization(mdev_mem_used, mdev_mem),
|
|
684
|
+
"memory_status": mdev_mem_status,
|
|
685
|
+
"temperature": dev_temp,
|
|
686
|
+
"power": dev_power,
|
|
687
|
+
"power_used": dev_power_used,
|
|
688
|
+
"appendix": mdev_appendix,
|
|
689
|
+
},
|
|
690
|
+
)
|
|
691
|
+
return ret
|
|
692
|
+
|
|
693
|
+
|
|
658
694
|
def _get_gpu_instance_slice(dev_gi_prf_id: int) -> int:
|
|
659
695
|
"""
|
|
660
696
|
Get the number of slices for a given GPU Instance Profile ID.
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
git_commit = "c736814"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/deploy/manifests/docker-compose.yaml
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/deploy/manifests/kubernetes.yaml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/docs/modules/gpustack_runtime.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/__init__.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/__types__.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/deployer.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/detector.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/cmds/images.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__types__.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/__utils__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/amd.py
RENAMED
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/hygon.py
RENAMED
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/metax.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/cdi/thead.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/docker.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/deployer/podman.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/__init__.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/__types__.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/__utils__.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/amd.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/ascend.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/cambricon.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/hygon.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/iluvatar.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/metax.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post5}/gpustack_runtime/detector/mthreads.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|