gpustack-runtime 0.2.2.post4__tar.gz → 0.2.2.post6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/PKG-INFO +1 -1
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/_version.py +2 -2
- gpustack_runtime-0.2.2.post6/gpustack_runtime/_version_appendix.py +1 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/__types__.py +18 -3
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/kuberentes.py +9 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/__init__.py +37 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/__types__.py +36 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/nvidia.py +183 -1
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/thead.py +257 -205
- gpustack_runtime-0.2.2.post6/tests/gpustack_runtime/detector/test_mig_devices.py +116 -0
- gpustack_runtime-0.2.2.post4/gpustack_runtime/_version_appendix.py +0 -1
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.codespelldict +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.codespellrc +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.dockerignore +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.gitattributes +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.gitignore +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.pre-commit-config.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/.python-version +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/LICENSE +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/Makefile +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/README.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/deploy/manifests/docker-compose.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/deploy/manifests/kubernetes.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/docs/index.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/docs/modules/gpustack_runtime.deployer.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/docs/modules/gpustack_runtime.detector.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/docs/modules/gpustack_runtime.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/__main__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/_version.pyi +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/deployer.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/detector.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/cmds/images.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/__patches__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/amd.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/ascend.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/hygon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/metax.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/cdi/thead.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/docker.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/k8s/devicemanager/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/podman.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/amd.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/ascend.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/cambricon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/hygon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/metax.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/mthreads.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyamdgpu/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyamdsmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pydcmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyhgml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyhgml/libhgml.so +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyhgml/libuki.so +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyhsa/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyixml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pymtml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pymxsml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pynvml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/pyrocmsmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/envs.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/logging.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/hatch.toml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/mkdocs.yml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/pack/Dockerfile +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/pack/Dockerfile.dummy +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/pyproject.toml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/pytest.ini +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/ruff.toml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_compare_versions.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_correct_runner_image.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_jsons.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_yamls.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_json.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_yaml.yaml +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/fixtures/test_nginx_entrypoint.sh +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/test_utils.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/deployer/test_workload_status.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/fixtures/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/README.md +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi300x.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi308x.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_amd_rx7800xt.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_ascend_310p3.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_ascend_910b2.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_hygon_k100ai.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_metax_c500.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_gb10.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100_mig.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h200.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4080super.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4090d.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx5090d.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/detect_output_thead_ppu.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi300x.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi308x.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_amd_rx7800xt.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_ascend_310p3.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_ascend_910b2.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_hygon_k100ai.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_metax_c500.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_mthreads_s5000.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100_mig.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h200.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4080super.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4090d.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx5090d.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/samples/topology_output_thead_ppu.json +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_amd.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_ascend.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_cambricon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_detector_utils.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_hygon.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_metax.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_mthreads.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/tests/gpustack_runtime/detector/test_nvidia.py +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/uv.lock +0 -0
- {gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/uv.toml +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gpustack-runtime
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.post6
|
|
4
4
|
Summary: GPUStack Runtime is library for detecting GPU resources and launching GPU workloads.
|
|
5
5
|
Project-URL: Homepage, https://github.com/gpustack/runtime
|
|
6
6
|
Project-URL: Bug Tracker, https://github.com/gpustack/gpustack/issues
|
|
@@ -27,8 +27,8 @@ version_tuple: VERSION_TUPLE
|
|
|
27
27
|
__commit_id__: COMMIT_ID
|
|
28
28
|
commit_id: COMMIT_ID
|
|
29
29
|
|
|
30
|
-
__version__ = version = '0.2.2.
|
|
31
|
-
__version_tuple__ = version_tuple = (0, 2, 2, '
|
|
30
|
+
__version__ = version = '0.2.2.post6'
|
|
31
|
+
__version_tuple__ = version_tuple = (0, 2, 2, 'post6')
|
|
32
32
|
try:
|
|
33
33
|
from ._version_appendix import git_commit
|
|
34
34
|
__commit_id__ = commit_id = git_commit
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
git_commit = "79ee112"
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/deployer/__types__.py
RENAMED
|
@@ -16,6 +16,7 @@ from .. import envs
|
|
|
16
16
|
from ..detector import (
|
|
17
17
|
ManufacturerEnum,
|
|
18
18
|
detect_devices,
|
|
19
|
+
expand_mig_devices,
|
|
19
20
|
group_devices_by_manufacturer,
|
|
20
21
|
manufacturer_to_backend,
|
|
21
22
|
)
|
|
@@ -1373,9 +1374,11 @@ class Deployer(ABC):
|
|
|
1373
1374
|
|
|
1374
1375
|
self._materials = {}
|
|
1375
1376
|
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1377
|
+
devices = detect_devices(fast=False)
|
|
1378
|
+
if self.allowed_mig_devices:
|
|
1379
|
+
devices = expand_mig_devices(devices)
|
|
1380
|
+
|
|
1381
|
+
group_devices = group_devices_by_manufacturer(devices)
|
|
1379
1382
|
|
|
1380
1383
|
if group_devices:
|
|
1381
1384
|
for manu, devs in group_devices.items():
|
|
@@ -1698,6 +1701,18 @@ class Deployer(ABC):
|
|
|
1698
1701
|
"""
|
|
1699
1702
|
return True
|
|
1700
1703
|
|
|
1704
|
+
@property
|
|
1705
|
+
def allowed_mig_devices(self) -> bool:
|
|
1706
|
+
"""
|
|
1707
|
+
Return whether the deployer addresses the MIG devices of a MIG-partitioned
|
|
1708
|
+
card, instead of the card itself.
|
|
1709
|
+
|
|
1710
|
+
Returns:
|
|
1711
|
+
True if addressed, False otherwise.
|
|
1712
|
+
|
|
1713
|
+
"""
|
|
1714
|
+
return True
|
|
1715
|
+
|
|
1701
1716
|
def close(self):
|
|
1702
1717
|
if self._pool:
|
|
1703
1718
|
self._pool.shutdown(cancel_futures=True)
|
|
@@ -1308,6 +1308,15 @@ class KubernetesDeployer(EndoscopicDeployer):
|
|
|
1308
1308
|
def allowed_runtime_uuid_values(self) -> bool:
|
|
1309
1309
|
return get_resource_injection_policy() != "kdp"
|
|
1310
1310
|
|
|
1311
|
+
@property
|
|
1312
|
+
def allowed_mig_devices(self) -> bool:
|
|
1313
|
+
# On Kubernetes a partitioner (e.g. the GPUStack Operator's device
|
|
1314
|
+
# manager) owns MIG: it creates and destroys the instances on demand,
|
|
1315
|
+
# so the card is the item to hold on to, and a pod asks for a slice of
|
|
1316
|
+
# it by resource rather than by addressing an instance that may not
|
|
1317
|
+
# outlive the request.
|
|
1318
|
+
return False
|
|
1319
|
+
|
|
1311
1320
|
def _prepare_mirrored_deployment(self):
|
|
1312
1321
|
"""
|
|
1313
1322
|
Prepare for mirrored deployment.
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/__init__.py
RENAMED
|
@@ -291,6 +291,42 @@ def filter_devices_by_manufacturer(
|
|
|
291
291
|
return [dev for dev in devices or [] if dev.manufacturer == manufacturer]
|
|
292
292
|
|
|
293
293
|
|
|
294
|
+
def expand_mig_devices(devices: Devices | None) -> Devices:
|
|
295
|
+
"""
|
|
296
|
+
Replace every MIG-partitioned card with its MIG devices.
|
|
297
|
+
|
|
298
|
+
Detection reports the physical card and carries its MIG devices in the
|
|
299
|
+
`mig_devices` appendix, because where a partitioner (e.g. the GPUStack
|
|
300
|
+
Operator's device manager) owns MIG, the instances come and go and the
|
|
301
|
+
card is the only stable inventory item. Callers that address MIG devices
|
|
302
|
+
themselves need them as devices instead: a MIG-enabled card cannot run a
|
|
303
|
+
workload, only its instances can.
|
|
304
|
+
|
|
305
|
+
A MIG-enabled card with no instances is kept as-is: there is nothing to
|
|
306
|
+
address yet, and dropping it would make the card disappear from the
|
|
307
|
+
inventory.
|
|
308
|
+
|
|
309
|
+
Args:
|
|
310
|
+
devices:
|
|
311
|
+
A list of devices to be expanded.
|
|
312
|
+
|
|
313
|
+
Returns:
|
|
314
|
+
A list of devices where MIG-partitioned cards are substituted by their
|
|
315
|
+
MIG devices.
|
|
316
|
+
|
|
317
|
+
"""
|
|
318
|
+
expanded: Devices = []
|
|
319
|
+
for dev in devices or []:
|
|
320
|
+
mig_devs = (dev.appendix or {}).get("mig_devices")
|
|
321
|
+
if not mig_devs:
|
|
322
|
+
expanded.append(dev)
|
|
323
|
+
continue
|
|
324
|
+
expanded.extend(
|
|
325
|
+
Device(manufacturer=dev.manufacturer, **mig_dev) for mig_dev in mig_devs
|
|
326
|
+
)
|
|
327
|
+
return expanded
|
|
328
|
+
|
|
329
|
+
|
|
294
330
|
__all__ = [
|
|
295
331
|
"Device",
|
|
296
332
|
"DeviceMemoryStatusEnum",
|
|
@@ -302,6 +338,7 @@ __all__ = [
|
|
|
302
338
|
"backend_to_manufacturer",
|
|
303
339
|
"detect_backend",
|
|
304
340
|
"detect_devices",
|
|
341
|
+
"expand_mig_devices",
|
|
305
342
|
"filter_devices_by_manufacturer",
|
|
306
343
|
"get_devices_topologies",
|
|
307
344
|
"group_devices_by_manufacturer",
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/__types__.py
RENAMED
|
@@ -461,6 +461,42 @@ def reduce_devices_distances(
|
|
|
461
461
|
return result
|
|
462
462
|
|
|
463
463
|
|
|
464
|
+
def index_mig_devices(
|
|
465
|
+
devices: Devices,
|
|
466
|
+
mig_devices: dict[int, list[dict]],
|
|
467
|
+
slots: int,
|
|
468
|
+
) -> None:
|
|
469
|
+
"""
|
|
470
|
+
Number the given cards' MIG devices in place.
|
|
471
|
+
|
|
472
|
+
A MIG device carries no driver-side inventory index, so its index is
|
|
473
|
+
synthetic: every card owns a block of `slots` indexes, and the blocks
|
|
474
|
+
start above the largest index the physical cards report. The reported
|
|
475
|
+
index may be the card's minor number
|
|
476
|
+
(`GPUSTACK_RUNTIME_DETECT_PHYSICAL_INDEX_PRIORITY`), which is not bound by
|
|
477
|
+
the card count, hence the offset is measured from the reported indexes
|
|
478
|
+
instead of the count. Sizing a block by the slots a card can host, rather
|
|
479
|
+
than by the MIG devices it currently has, keeps a card's numbering
|
|
480
|
+
independent of its neighbours: partitioning one card never renumbers
|
|
481
|
+
another's MIG devices.
|
|
482
|
+
|
|
483
|
+
Args:
|
|
484
|
+
devices:
|
|
485
|
+
The detected physical cards, already indexed.
|
|
486
|
+
mig_devices:
|
|
487
|
+
The MIG devices to number, keyed by the enumeration index of the
|
|
488
|
+
card hosting them. Each entry's `index` is the driver slot it was
|
|
489
|
+
found at, which the block offset is added to.
|
|
490
|
+
slots:
|
|
491
|
+
The number of MIG devices a card can host, i.e. the block size.
|
|
492
|
+
|
|
493
|
+
"""
|
|
494
|
+
base = max((dev.index for dev in devices), default=-1) + 1
|
|
495
|
+
for dev_idx, migs in mig_devices.items():
|
|
496
|
+
for mig in migs:
|
|
497
|
+
mig["index"] += base + dev_idx * slots
|
|
498
|
+
|
|
499
|
+
|
|
464
500
|
class Detector(ABC):
|
|
465
501
|
"""
|
|
466
502
|
Base class for all detectors.
|
{gpustack_runtime-0.2.2.post4 → gpustack_runtime-0.2.2.post6}/gpustack_runtime/detector/nvidia.py
RENAMED
|
@@ -11,7 +11,14 @@ from functools import lru_cache
|
|
|
11
11
|
from .. import envs
|
|
12
12
|
from ..logging import debug_log_exception, debug_log_warning
|
|
13
13
|
from . import DeviceMemoryStatusEnum, Topology, pynvml
|
|
14
|
-
from .__types__ import
|
|
14
|
+
from .__types__ import (
|
|
15
|
+
Detector,
|
|
16
|
+
Device,
|
|
17
|
+
Devices,
|
|
18
|
+
ManufacturerEnum,
|
|
19
|
+
TopologyDistanceEnum,
|
|
20
|
+
index_mig_devices,
|
|
21
|
+
)
|
|
15
22
|
from .__utils__ import (
|
|
16
23
|
PCIDevice,
|
|
17
24
|
bitmask_to_str,
|
|
@@ -113,6 +120,13 @@ class NVIDIADetector(Detector):
|
|
|
113
120
|
sys_runtime_ver_original,
|
|
114
121
|
)
|
|
115
122
|
|
|
123
|
+
# MIG devices of every MIG-enabled card, keyed by the card's
|
|
124
|
+
# enumeration index, and the largest number of MIG devices a card
|
|
125
|
+
# can host: both are needed to number them once every card is
|
|
126
|
+
# detected, see index_mig_devices.
|
|
127
|
+
devs_mig_devices: dict[int, list[dict]] = {}
|
|
128
|
+
devs_mig_slots = 0
|
|
129
|
+
|
|
116
130
|
dev_count = pynvml.nvmlDeviceGetCount()
|
|
117
131
|
for dev_idx in range(dev_count):
|
|
118
132
|
dev = pynvml.nvmlDeviceGetHandleByIndex(dev_idx)
|
|
@@ -218,6 +232,27 @@ class NVIDIADetector(Detector):
|
|
|
218
232
|
"mig": dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE,
|
|
219
233
|
"bdf": dev_bdf,
|
|
220
234
|
}
|
|
235
|
+
if dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE:
|
|
236
|
+
dev_mig_slots = 0
|
|
237
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
238
|
+
dev_mig_slots = pynvml.nvmlDeviceGetMaxMigDeviceCount(dev)
|
|
239
|
+
devs_mig_slots = max(devs_mig_slots, dev_mig_slots)
|
|
240
|
+
dev_mig_devices = _get_mig_devices(
|
|
241
|
+
dev,
|
|
242
|
+
dev_mig_slots,
|
|
243
|
+
dev_cc_t,
|
|
244
|
+
sys_driver_ver,
|
|
245
|
+
sys_runtime_ver,
|
|
246
|
+
sys_runtime_ver_original,
|
|
247
|
+
dev_cc,
|
|
248
|
+
dev_temp,
|
|
249
|
+
dev_power,
|
|
250
|
+
dev_power_used,
|
|
251
|
+
dev_bdf,
|
|
252
|
+
dev_numa,
|
|
253
|
+
)
|
|
254
|
+
dev_appendix["mig_devices"] = dev_mig_devices
|
|
255
|
+
devs_mig_devices[dev_idx] = dev_mig_devices
|
|
221
256
|
if dev_numa:
|
|
222
257
|
dev_appendix["numa"] = dev_numa
|
|
223
258
|
|
|
@@ -246,6 +281,8 @@ class NVIDIADetector(Detector):
|
|
|
246
281
|
appendix=dev_appendix,
|
|
247
282
|
),
|
|
248
283
|
)
|
|
284
|
+
|
|
285
|
+
index_mig_devices(ret, devs_mig_devices, devs_mig_slots)
|
|
249
286
|
except pynvml.NVMLError:
|
|
250
287
|
debug_log_exception(logger, "Failed to fetch devices")
|
|
251
288
|
raise
|
|
@@ -558,6 +595,151 @@ def _get_links_state(
|
|
|
558
595
|
}
|
|
559
596
|
|
|
560
597
|
|
|
598
|
+
def _get_mig_devices(
|
|
599
|
+
dev,
|
|
600
|
+
dev_mig_slots: int,
|
|
601
|
+
dev_cc_t,
|
|
602
|
+
sys_driver_ver,
|
|
603
|
+
sys_runtime_ver,
|
|
604
|
+
sys_runtime_ver_original,
|
|
605
|
+
dev_cc,
|
|
606
|
+
dev_temp,
|
|
607
|
+
dev_power,
|
|
608
|
+
dev_power_used,
|
|
609
|
+
dev_bdf: str,
|
|
610
|
+
dev_numa,
|
|
611
|
+
) -> list[dict]:
|
|
612
|
+
"""
|
|
613
|
+
Enumerate the card's current MIG devices with the same detail a plain
|
|
614
|
+
device carries (profile name, uuid, compute/memory utilization, memory
|
|
615
|
+
health, temperature and power), returned as appendix entries of the
|
|
616
|
+
physical card rather than standalone devices. Empty when MIG is enabled
|
|
617
|
+
but no GPU instances exist yet.
|
|
618
|
+
|
|
619
|
+
Each entry's `index` is the driver slot the MIG device was found at:
|
|
620
|
+
index_mig_devices turns it into the device index once every card is
|
|
621
|
+
detected.
|
|
622
|
+
"""
|
|
623
|
+
ret: list[dict] = []
|
|
624
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
625
|
+
for mdev_idx in range(dev_mig_slots):
|
|
626
|
+
mdev = None
|
|
627
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
628
|
+
mdev = pynvml.nvmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
|
|
629
|
+
if not mdev:
|
|
630
|
+
continue
|
|
631
|
+
|
|
632
|
+
mdev_uuid = pynvml.nvmlDeviceGetUUID(mdev)
|
|
633
|
+
|
|
634
|
+
mdev_mem = 0
|
|
635
|
+
mdev_mem_used = 0
|
|
636
|
+
mdev_mem_status = DeviceMemoryStatusEnum.HEALTHY
|
|
637
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
638
|
+
mdev_mem_info = pynvml.nvmlDeviceGetMemoryInfo(mdev)
|
|
639
|
+
mdev_mem = byte_to_mebibyte(mdev_mem_info.total)
|
|
640
|
+
mdev_mem_used = byte_to_mebibyte(mdev_mem_info.used)
|
|
641
|
+
if not envs.GPUSTACK_RUNTIME_DETECT_NO_HEALTH_CHECK:
|
|
642
|
+
mdev_mem_ecc_errors = pynvml.nvmlDeviceGetMemoryErrorCounter(
|
|
643
|
+
mdev,
|
|
644
|
+
pynvml.NVML_MEMORY_ERROR_TYPE_UNCORRECTED,
|
|
645
|
+
pynvml.NVML_AGGREGATE_ECC,
|
|
646
|
+
pynvml.NVML_MEMORY_LOCATION_SRAM,
|
|
647
|
+
)
|
|
648
|
+
if mdev_mem_ecc_errors > 0:
|
|
649
|
+
mdev_mem_status = DeviceMemoryStatusEnum.UNHEALTHY
|
|
650
|
+
|
|
651
|
+
mdev_appendix = {
|
|
652
|
+
"arch_family": _get_arch_family(dev_cc_t),
|
|
653
|
+
"vgpu": True,
|
|
654
|
+
"sliced": True,
|
|
655
|
+
"mig": True,
|
|
656
|
+
"bdf": dev_bdf,
|
|
657
|
+
}
|
|
658
|
+
if dev_numa:
|
|
659
|
+
mdev_appendix["numa"] = dev_numa
|
|
660
|
+
|
|
661
|
+
mdev_gi_id = pynvml.nvmlDeviceGetGpuInstanceId(mdev)
|
|
662
|
+
mdev_appendix["gpu_instance_id"] = mdev_gi_id
|
|
663
|
+
mdev_ci_id = pynvml.nvmlDeviceGetComputeInstanceId(mdev)
|
|
664
|
+
mdev_appendix["compute_instance_id"] = mdev_ci_id
|
|
665
|
+
|
|
666
|
+
mdev_cores_util = _get_sm_util_from_gpm_metrics(dev, mdev_gi_id)
|
|
667
|
+
|
|
668
|
+
mdev_name = ""
|
|
669
|
+
mdev_cores = None
|
|
670
|
+
mdev_gi = pynvml.nvmlDeviceGetGpuInstanceById(dev, mdev_gi_id)
|
|
671
|
+
mdev_ci = pynvml.nvmlGpuInstanceGetComputeInstanceById(
|
|
672
|
+
mdev_gi,
|
|
673
|
+
mdev_ci_id,
|
|
674
|
+
)
|
|
675
|
+
mdev_gi_info = pynvml.nvmlGpuInstanceGetInfo(mdev_gi)
|
|
676
|
+
mdev_ci_info = pynvml.nvmlComputeInstanceGetInfo(mdev_ci)
|
|
677
|
+
for dev_gi_prf_id in range(pynvml.NVML_GPU_INSTANCE_PROFILE_COUNT):
|
|
678
|
+
try:
|
|
679
|
+
dev_gi_prf = pynvml.nvmlDeviceGetGpuInstanceProfileInfo(
|
|
680
|
+
dev,
|
|
681
|
+
dev_gi_prf_id,
|
|
682
|
+
)
|
|
683
|
+
if dev_gi_prf.id != mdev_gi_info.profileId:
|
|
684
|
+
continue
|
|
685
|
+
except pynvml.NVMLError:
|
|
686
|
+
continue
|
|
687
|
+
|
|
688
|
+
gi_mem = round(math.ceil(dev_gi_prf.memorySizeMB >> 10))
|
|
689
|
+
gi_prf_name = getattr(dev_gi_prf, "name", None)
|
|
690
|
+
mdev_name = (
|
|
691
|
+
gi_prf_name.removeprefix("MIG ")
|
|
692
|
+
if gi_prf_name
|
|
693
|
+
else f"{dev_gi_prf.sliceCount}g.{gi_mem}gb"
|
|
694
|
+
)
|
|
695
|
+
|
|
696
|
+
for dev_ci_prf_id in range(
|
|
697
|
+
pynvml.NVML_COMPUTE_INSTANCE_PROFILE_COUNT,
|
|
698
|
+
):
|
|
699
|
+
for dev_cig_prf_id in range(
|
|
700
|
+
pynvml.NVML_COMPUTE_INSTANCE_ENGINE_PROFILE_COUNT,
|
|
701
|
+
):
|
|
702
|
+
try:
|
|
703
|
+
mdev_ci_prf = (
|
|
704
|
+
pynvml.nvmlGpuInstanceGetComputeInstanceProfileInfo(
|
|
705
|
+
mdev_gi,
|
|
706
|
+
dev_ci_prf_id,
|
|
707
|
+
dev_cig_prf_id,
|
|
708
|
+
)
|
|
709
|
+
)
|
|
710
|
+
if mdev_ci_prf.id != mdev_ci_info.profileId:
|
|
711
|
+
continue
|
|
712
|
+
except pynvml.NVMLError:
|
|
713
|
+
continue
|
|
714
|
+
mdev_cores = mdev_ci_prf.multiprocessorCount
|
|
715
|
+
break
|
|
716
|
+
|
|
717
|
+
break
|
|
718
|
+
|
|
719
|
+
ret.append(
|
|
720
|
+
{
|
|
721
|
+
"index": mdev_idx,
|
|
722
|
+
"name": mdev_name,
|
|
723
|
+
"uuid": mdev_uuid,
|
|
724
|
+
"driver_version": sys_driver_ver,
|
|
725
|
+
"runtime_version": sys_runtime_ver,
|
|
726
|
+
"runtime_version_original": sys_runtime_ver_original,
|
|
727
|
+
"compute_capability": dev_cc,
|
|
728
|
+
"cores": mdev_cores,
|
|
729
|
+
"cores_utilization": mdev_cores_util,
|
|
730
|
+
"memory": mdev_mem,
|
|
731
|
+
"memory_used": mdev_mem_used,
|
|
732
|
+
"memory_utilization": get_utilization(mdev_mem_used, mdev_mem),
|
|
733
|
+
"memory_status": mdev_mem_status,
|
|
734
|
+
"temperature": dev_temp,
|
|
735
|
+
"power": dev_power,
|
|
736
|
+
"power_used": dev_power_used,
|
|
737
|
+
"appendix": mdev_appendix,
|
|
738
|
+
},
|
|
739
|
+
)
|
|
740
|
+
return ret
|
|
741
|
+
|
|
742
|
+
|
|
561
743
|
def _get_arch_family(dev_cc_t: list[int]) -> str:
|
|
562
744
|
"""
|
|
563
745
|
Get the architecture family based on the CUDA compute capability.
|