gpustack-runtime 0.2.2.post5__tar.gz → 0.2.2.post7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/PKG-INFO +1 -1
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/_version.py +2 -2
- gpustack_runtime-0.2.2.post7/gpustack_runtime/_version_appendix.py +1 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/__types__.py +18 -3
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/kuberentes.py +65 -1
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/__init__.py +37 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/__types__.py +36 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/nvidia.py +32 -8
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/thead.py +24 -8
- gpustack_runtime-0.2.2.post7/tests/gpustack_runtime/deployer/test_privileged.py +137 -0
- gpustack_runtime-0.2.2.post7/tests/gpustack_runtime/detector/test_mig_devices.py +116 -0
- gpustack_runtime-0.2.2.post5/gpustack_runtime/_version_appendix.py +0 -1
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/.codespelldict +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/.codespellrc +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/.dockerignore +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/.gitattributes +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/.gitignore +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/.pre-commit-config.yaml +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/.python-version +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/LICENSE +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/Makefile +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/README.md +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/deploy/manifests/docker-compose.yaml +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/deploy/manifests/kubernetes.yaml +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/docs/index.md +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/docs/modules/gpustack_runtime.deployer.md +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/docs/modules/gpustack_runtime.detector.md +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/docs/modules/gpustack_runtime.md +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/__main__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/_version.pyi +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/deployer.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/detector.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/images.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/__patches__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/__types__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/amd.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/ascend.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/hygon.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/metax.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/thead.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/docker.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/k8s/devicemanager/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/podman.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/__utils__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/amd.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/ascend.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/cambricon.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/hygon.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/metax.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/mthreads.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pyamdgpu/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pyamdsmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pydcmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pyhgml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pyhgml/libhgml.so +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pyhgml/libuki.so +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pyhsa/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pyixml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pymtml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pymxsml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pynvml/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/pyrocmsmi/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/envs.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/logging.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/hatch.toml +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/mkdocs.yml +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/pack/Dockerfile +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/pack/Dockerfile.dummy +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/pyproject.toml +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/pytest.ini +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/ruff.toml +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/fixtures/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/fixtures/test_compare_versions.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/fixtures/test_correct_runner_image.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_jsons.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_multiple_yamls.yaml +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_json.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/fixtures/test_load_yaml_or_json_single_yaml.yaml +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/fixtures/test_nginx_entrypoint.sh +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/test_utils.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/deployer/test_workload_status.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/fixtures/__init__.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/README.md +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi300x.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_amd_mi308x.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_amd_rx7800xt.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_ascend_310p3.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_ascend_910b2.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_hygon_k100ai.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_metax_c500.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_gb10.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h100_mig.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_h200.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4080super.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx4090d.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_nvidia_rtx5090d.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/detect_output_thead_ppu.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi300x.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_amd_mi308x.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_amd_rx7800xt.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_ascend_310p3.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_ascend_910b2.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_hygon_k100ai.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_metax_c500.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_mthreads_s5000.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h100_mig.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_h200.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4080super.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx4090d.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_nvidia_rtx5090d.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/samples/topology_output_thead_ppu.json +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/test_amd.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/test_ascend.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/test_cambricon.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/test_detector_utils.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/test_hygon.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/test_iluvatar.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/test_metax.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/test_mthreads.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/tests/gpustack_runtime/detector/test_nvidia.py +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/uv.lock +0 -0
- {gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/uv.toml +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gpustack-runtime
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.post7
|
|
4
4
|
Summary: GPUStack Runtime is library for detecting GPU resources and launching GPU workloads.
|
|
5
5
|
Project-URL: Homepage, https://github.com/gpustack/runtime
|
|
6
6
|
Project-URL: Bug Tracker, https://github.com/gpustack/gpustack/issues
|
|
@@ -27,8 +27,8 @@ version_tuple: VERSION_TUPLE
|
|
|
27
27
|
__commit_id__: COMMIT_ID
|
|
28
28
|
commit_id: COMMIT_ID
|
|
29
29
|
|
|
30
|
-
__version__ = version = '0.2.2.
|
|
31
|
-
__version_tuple__ = version_tuple = (0, 2, 2, '
|
|
30
|
+
__version__ = version = '0.2.2.post7'
|
|
31
|
+
__version_tuple__ = version_tuple = (0, 2, 2, 'post7')
|
|
32
32
|
try:
|
|
33
33
|
from ._version_appendix import git_commit
|
|
34
34
|
__commit_id__ = commit_id = git_commit
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
git_commit = "9147850"
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/__types__.py
RENAMED
|
@@ -16,6 +16,7 @@ from .. import envs
|
|
|
16
16
|
from ..detector import (
|
|
17
17
|
ManufacturerEnum,
|
|
18
18
|
detect_devices,
|
|
19
|
+
expand_mig_devices,
|
|
19
20
|
group_devices_by_manufacturer,
|
|
20
21
|
manufacturer_to_backend,
|
|
21
22
|
)
|
|
@@ -1373,9 +1374,11 @@ class Deployer(ABC):
|
|
|
1373
1374
|
|
|
1374
1375
|
self._materials = {}
|
|
1375
1376
|
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1377
|
+
devices = detect_devices(fast=False)
|
|
1378
|
+
if self.allowed_mig_devices:
|
|
1379
|
+
devices = expand_mig_devices(devices)
|
|
1380
|
+
|
|
1381
|
+
group_devices = group_devices_by_manufacturer(devices)
|
|
1379
1382
|
|
|
1380
1383
|
if group_devices:
|
|
1381
1384
|
for manu, devs in group_devices.items():
|
|
@@ -1698,6 +1701,18 @@ class Deployer(ABC):
|
|
|
1698
1701
|
"""
|
|
1699
1702
|
return True
|
|
1700
1703
|
|
|
1704
|
+
@property
|
|
1705
|
+
def allowed_mig_devices(self) -> bool:
|
|
1706
|
+
"""
|
|
1707
|
+
Return whether the deployer addresses the MIG devices of a MIG-partitioned
|
|
1708
|
+
card, instead of the card itself.
|
|
1709
|
+
|
|
1710
|
+
Returns:
|
|
1711
|
+
True if addressed, False otherwise.
|
|
1712
|
+
|
|
1713
|
+
"""
|
|
1714
|
+
return True
|
|
1715
|
+
|
|
1701
1716
|
def close(self):
|
|
1702
1717
|
if self._pool:
|
|
1703
1718
|
self._pool.shutdown(cancel_futures=True)
|
|
@@ -325,6 +325,61 @@ def _pin_pod_for_kueue(
|
|
|
325
325
|
}
|
|
326
326
|
|
|
327
327
|
|
|
328
|
+
def _is_device_plugin_resource(resource_key: str) -> bool:
|
|
329
|
+
"""
|
|
330
|
+
Report whether a resource key belongs to a device plugin resource family,
|
|
331
|
+
which is either a CDI kind ("nvidia.com/gpu") or one of its suffixed
|
|
332
|
+
variants ("nvidia.com/gpu.shared", "nvidia.com/gpu.sliced.units",
|
|
333
|
+
"nvidia.com/gpu.partitioned.mig-1g.20gb").
|
|
334
|
+
"""
|
|
335
|
+
return any(
|
|
336
|
+
resource_key == cdi or resource_key.startswith(f"{cdi}.")
|
|
337
|
+
for cdi in envs.GPUSTACK_RUNTIME_DEPLOY_RESOURCE_KEY_MAP_CDI.values()
|
|
338
|
+
)
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def _resolve_privileged(container: Container) -> bool:
|
|
342
|
+
"""
|
|
343
|
+
Resolve whether a container runs privileged.
|
|
344
|
+
|
|
345
|
+
Privilege is dropped when the container's devices are handed out by a
|
|
346
|
+
device plugin, which is the case for every device plugin resource family
|
|
347
|
+
and, under the KDP injection policy, for every mapped device request.
|
|
348
|
+
|
|
349
|
+
A privileged container receives all device nodes of the host, so it
|
|
350
|
+
enumerates -- and can use -- every accelerator on the node, no matter
|
|
351
|
+
which one the device plugin allocated to it. That silently undoes
|
|
352
|
+
slicing: a workload holding a single MIG device or a single memory slice
|
|
353
|
+
still sees the untouched cards next to it, and a soft-slicing limit
|
|
354
|
+
lands on whichever device comes first instead of the allocated one.
|
|
355
|
+
"""
|
|
356
|
+
if not container.execution or not container.execution.privileged:
|
|
357
|
+
return False
|
|
358
|
+
if not container.resources:
|
|
359
|
+
return True
|
|
360
|
+
|
|
361
|
+
kdp = get_resource_injection_policy() == "kdp"
|
|
362
|
+
for r_k in container.resources:
|
|
363
|
+
if r_k in ("cpu", "memory"):
|
|
364
|
+
continue
|
|
365
|
+
if _is_device_plugin_resource(r_k) or (
|
|
366
|
+
kdp
|
|
367
|
+
and (
|
|
368
|
+
r_k
|
|
369
|
+
in envs.GPUSTACK_RUNTIME_DEPLOY_RESOURCE_KEY_MAP_RUNTIME_VISIBLE_DEVICES
|
|
370
|
+
or r_k == envs.GPUSTACK_RUNTIME_DEPLOY_AUTOMAP_RESOURCE_KEY
|
|
371
|
+
)
|
|
372
|
+
):
|
|
373
|
+
clogger.info(
|
|
374
|
+
"Dropping privilege of container '%s', "
|
|
375
|
+
"as its device request '%s' is allocated by a device plugin",
|
|
376
|
+
container.name,
|
|
377
|
+
r_k,
|
|
378
|
+
)
|
|
379
|
+
return False
|
|
380
|
+
return True
|
|
381
|
+
|
|
382
|
+
|
|
328
383
|
class KubernetesDeployer(EndoscopicDeployer):
|
|
329
384
|
"""
|
|
330
385
|
Deployer implementation for Kubernetes.
|
|
@@ -1048,7 +1103,7 @@ class KubernetesDeployer(EndoscopicDeployer):
|
|
|
1048
1103
|
run_as_user=c.execution.run_as_user,
|
|
1049
1104
|
run_as_group=c.execution.run_as_group,
|
|
1050
1105
|
read_only_root_filesystem=c.execution.readonly_rootfs,
|
|
1051
|
-
privileged=c
|
|
1106
|
+
privileged=_resolve_privileged(c),
|
|
1052
1107
|
capabilities=(
|
|
1053
1108
|
kubernetes.client.V1Capabilities(
|
|
1054
1109
|
add=c.execution.capabilities.add,
|
|
@@ -1308,6 +1363,15 @@ class KubernetesDeployer(EndoscopicDeployer):
|
|
|
1308
1363
|
def allowed_runtime_uuid_values(self) -> bool:
|
|
1309
1364
|
return get_resource_injection_policy() != "kdp"
|
|
1310
1365
|
|
|
1366
|
+
@property
|
|
1367
|
+
def allowed_mig_devices(self) -> bool:
|
|
1368
|
+
# On Kubernetes a partitioner (e.g. the GPUStack Operator's device
|
|
1369
|
+
# manager) owns MIG: it creates and destroys the instances on demand,
|
|
1370
|
+
# so the card is the item to hold on to, and a pod asks for a slice of
|
|
1371
|
+
# it by resource rather than by addressing an instance that may not
|
|
1372
|
+
# outlive the request.
|
|
1373
|
+
return False
|
|
1374
|
+
|
|
1311
1375
|
def _prepare_mirrored_deployment(self):
|
|
1312
1376
|
"""
|
|
1313
1377
|
Prepare for mirrored deployment.
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/__init__.py
RENAMED
|
@@ -291,6 +291,42 @@ def filter_devices_by_manufacturer(
|
|
|
291
291
|
return [dev for dev in devices or [] if dev.manufacturer == manufacturer]
|
|
292
292
|
|
|
293
293
|
|
|
294
|
+
def expand_mig_devices(devices: Devices | None) -> Devices:
|
|
295
|
+
"""
|
|
296
|
+
Replace every MIG-partitioned card with its MIG devices.
|
|
297
|
+
|
|
298
|
+
Detection reports the physical card and carries its MIG devices in the
|
|
299
|
+
`mig_devices` appendix, because where a partitioner (e.g. the GPUStack
|
|
300
|
+
Operator's device manager) owns MIG, the instances come and go and the
|
|
301
|
+
card is the only stable inventory item. Callers that address MIG devices
|
|
302
|
+
themselves need them as devices instead: a MIG-enabled card cannot run a
|
|
303
|
+
workload, only its instances can.
|
|
304
|
+
|
|
305
|
+
A MIG-enabled card with no instances is kept as-is: there is nothing to
|
|
306
|
+
address yet, and dropping it would make the card disappear from the
|
|
307
|
+
inventory.
|
|
308
|
+
|
|
309
|
+
Args:
|
|
310
|
+
devices:
|
|
311
|
+
A list of devices to be expanded.
|
|
312
|
+
|
|
313
|
+
Returns:
|
|
314
|
+
A list of devices where MIG-partitioned cards are substituted by their
|
|
315
|
+
MIG devices.
|
|
316
|
+
|
|
317
|
+
"""
|
|
318
|
+
expanded: Devices = []
|
|
319
|
+
for dev in devices or []:
|
|
320
|
+
mig_devs = (dev.appendix or {}).get("mig_devices")
|
|
321
|
+
if not mig_devs:
|
|
322
|
+
expanded.append(dev)
|
|
323
|
+
continue
|
|
324
|
+
expanded.extend(
|
|
325
|
+
Device(manufacturer=dev.manufacturer, **mig_dev) for mig_dev in mig_devs
|
|
326
|
+
)
|
|
327
|
+
return expanded
|
|
328
|
+
|
|
329
|
+
|
|
294
330
|
__all__ = [
|
|
295
331
|
"Device",
|
|
296
332
|
"DeviceMemoryStatusEnum",
|
|
@@ -302,6 +338,7 @@ __all__ = [
|
|
|
302
338
|
"backend_to_manufacturer",
|
|
303
339
|
"detect_backend",
|
|
304
340
|
"detect_devices",
|
|
341
|
+
"expand_mig_devices",
|
|
305
342
|
"filter_devices_by_manufacturer",
|
|
306
343
|
"get_devices_topologies",
|
|
307
344
|
"group_devices_by_manufacturer",
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/__types__.py
RENAMED
|
@@ -461,6 +461,42 @@ def reduce_devices_distances(
|
|
|
461
461
|
return result
|
|
462
462
|
|
|
463
463
|
|
|
464
|
+
def index_mig_devices(
|
|
465
|
+
devices: Devices,
|
|
466
|
+
mig_devices: dict[int, list[dict]],
|
|
467
|
+
slots: int,
|
|
468
|
+
) -> None:
|
|
469
|
+
"""
|
|
470
|
+
Number the given cards' MIG devices in place.
|
|
471
|
+
|
|
472
|
+
A MIG device carries no driver-side inventory index, so its index is
|
|
473
|
+
synthetic: every card owns a block of `slots` indexes, and the blocks
|
|
474
|
+
start above the largest index the physical cards report. The reported
|
|
475
|
+
index may be the card's minor number
|
|
476
|
+
(`GPUSTACK_RUNTIME_DETECT_PHYSICAL_INDEX_PRIORITY`), which is not bound by
|
|
477
|
+
the card count, hence the offset is measured from the reported indexes
|
|
478
|
+
instead of the count. Sizing a block by the slots a card can host, rather
|
|
479
|
+
than by the MIG devices it currently has, keeps a card's numbering
|
|
480
|
+
independent of its neighbours: partitioning one card never renumbers
|
|
481
|
+
another's MIG devices.
|
|
482
|
+
|
|
483
|
+
Args:
|
|
484
|
+
devices:
|
|
485
|
+
The detected physical cards, already indexed.
|
|
486
|
+
mig_devices:
|
|
487
|
+
The MIG devices to number, keyed by the enumeration index of the
|
|
488
|
+
card hosting them. Each entry's `index` is the driver slot it was
|
|
489
|
+
found at, which the block offset is added to.
|
|
490
|
+
slots:
|
|
491
|
+
The number of MIG devices a card can host, i.e. the block size.
|
|
492
|
+
|
|
493
|
+
"""
|
|
494
|
+
base = max((dev.index for dev in devices), default=-1) + 1
|
|
495
|
+
for dev_idx, migs in mig_devices.items():
|
|
496
|
+
for mig in migs:
|
|
497
|
+
mig["index"] += base + dev_idx * slots
|
|
498
|
+
|
|
499
|
+
|
|
464
500
|
class Detector(ABC):
|
|
465
501
|
"""
|
|
466
502
|
Base class for all detectors.
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/nvidia.py
RENAMED
|
@@ -11,7 +11,14 @@ from functools import lru_cache
|
|
|
11
11
|
from .. import envs
|
|
12
12
|
from ..logging import debug_log_exception, debug_log_warning
|
|
13
13
|
from . import DeviceMemoryStatusEnum, Topology, pynvml
|
|
14
|
-
from .__types__ import
|
|
14
|
+
from .__types__ import (
|
|
15
|
+
Detector,
|
|
16
|
+
Device,
|
|
17
|
+
Devices,
|
|
18
|
+
ManufacturerEnum,
|
|
19
|
+
TopologyDistanceEnum,
|
|
20
|
+
index_mig_devices,
|
|
21
|
+
)
|
|
15
22
|
from .__utils__ import (
|
|
16
23
|
PCIDevice,
|
|
17
24
|
bitmask_to_str,
|
|
@@ -113,6 +120,13 @@ class NVIDIADetector(Detector):
|
|
|
113
120
|
sys_runtime_ver_original,
|
|
114
121
|
)
|
|
115
122
|
|
|
123
|
+
# MIG devices of every MIG-enabled card, keyed by the card's
|
|
124
|
+
# enumeration index, and the largest number of MIG devices a card
|
|
125
|
+
# can host: both are needed to number them once every card is
|
|
126
|
+
# detected, see index_mig_devices.
|
|
127
|
+
devs_mig_devices: dict[int, list[dict]] = {}
|
|
128
|
+
devs_mig_slots = 0
|
|
129
|
+
|
|
116
130
|
dev_count = pynvml.nvmlDeviceGetCount()
|
|
117
131
|
for dev_idx in range(dev_count):
|
|
118
132
|
dev = pynvml.nvmlDeviceGetHandleByIndex(dev_idx)
|
|
@@ -219,10 +233,13 @@ class NVIDIADetector(Detector):
|
|
|
219
233
|
"bdf": dev_bdf,
|
|
220
234
|
}
|
|
221
235
|
if dev_mig_mode != pynvml.NVML_DEVICE_MIG_DISABLE:
|
|
222
|
-
|
|
236
|
+
dev_mig_slots = 0
|
|
237
|
+
with contextlib.suppress(pynvml.NVMLError):
|
|
238
|
+
dev_mig_slots = pynvml.nvmlDeviceGetMaxMigDeviceCount(dev)
|
|
239
|
+
devs_mig_slots = max(devs_mig_slots, dev_mig_slots)
|
|
240
|
+
dev_mig_devices = _get_mig_devices(
|
|
223
241
|
dev,
|
|
224
|
-
|
|
225
|
-
dev_count,
|
|
242
|
+
dev_mig_slots,
|
|
226
243
|
dev_cc_t,
|
|
227
244
|
sys_driver_ver,
|
|
228
245
|
sys_runtime_ver,
|
|
@@ -234,6 +251,8 @@ class NVIDIADetector(Detector):
|
|
|
234
251
|
dev_bdf,
|
|
235
252
|
dev_numa,
|
|
236
253
|
)
|
|
254
|
+
dev_appendix["mig_devices"] = dev_mig_devices
|
|
255
|
+
devs_mig_devices[dev_idx] = dev_mig_devices
|
|
237
256
|
if dev_numa:
|
|
238
257
|
dev_appendix["numa"] = dev_numa
|
|
239
258
|
|
|
@@ -262,6 +281,8 @@ class NVIDIADetector(Detector):
|
|
|
262
281
|
appendix=dev_appendix,
|
|
263
282
|
),
|
|
264
283
|
)
|
|
284
|
+
|
|
285
|
+
index_mig_devices(ret, devs_mig_devices, devs_mig_slots)
|
|
265
286
|
except pynvml.NVMLError:
|
|
266
287
|
debug_log_exception(logger, "Failed to fetch devices")
|
|
267
288
|
raise
|
|
@@ -576,8 +597,7 @@ def _get_links_state(
|
|
|
576
597
|
|
|
577
598
|
def _get_mig_devices(
|
|
578
599
|
dev,
|
|
579
|
-
|
|
580
|
-
dev_count: int,
|
|
600
|
+
dev_mig_slots: int,
|
|
581
601
|
dev_cc_t,
|
|
582
602
|
sys_driver_ver,
|
|
583
603
|
sys_runtime_ver,
|
|
@@ -595,10 +615,14 @@ def _get_mig_devices(
|
|
|
595
615
|
health, temperature and power), returned as appendix entries of the
|
|
596
616
|
physical card rather than standalone devices. Empty when MIG is enabled
|
|
597
617
|
but no GPU instances exist yet.
|
|
618
|
+
|
|
619
|
+
Each entry's `index` is the driver slot the MIG device was found at:
|
|
620
|
+
index_mig_devices turns it into the device index once every card is
|
|
621
|
+
detected.
|
|
598
622
|
"""
|
|
599
623
|
ret: list[dict] = []
|
|
600
624
|
with contextlib.suppress(pynvml.NVMLError):
|
|
601
|
-
for mdev_idx in range(
|
|
625
|
+
for mdev_idx in range(dev_mig_slots):
|
|
602
626
|
mdev = None
|
|
603
627
|
with contextlib.suppress(pynvml.NVMLError):
|
|
604
628
|
mdev = pynvml.nvmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
|
|
@@ -694,7 +718,7 @@ def _get_mig_devices(
|
|
|
694
718
|
|
|
695
719
|
ret.append(
|
|
696
720
|
{
|
|
697
|
-
"index": mdev_idx
|
|
721
|
+
"index": mdev_idx,
|
|
698
722
|
"name": mdev_name,
|
|
699
723
|
"uuid": mdev_uuid,
|
|
700
724
|
"driver_version": sys_driver_ver,
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/thead.py
RENAMED
|
@@ -18,6 +18,7 @@ from .__types__ import (
|
|
|
18
18
|
ManufacturerEnum,
|
|
19
19
|
Topology,
|
|
20
20
|
TopologyDistanceEnum,
|
|
21
|
+
index_mig_devices,
|
|
21
22
|
)
|
|
22
23
|
from .__utils__ import (
|
|
23
24
|
PCIDevice,
|
|
@@ -120,6 +121,13 @@ class THeadDetector(Detector):
|
|
|
120
121
|
sys_runtime_ver_original,
|
|
121
122
|
)
|
|
122
123
|
|
|
124
|
+
# MIG devices of every MIG-enabled card, keyed by the card's
|
|
125
|
+
# enumeration index, and the largest number of MIG devices a card
|
|
126
|
+
# can host: both are needed to number them once every card is
|
|
127
|
+
# detected, see index_mig_devices.
|
|
128
|
+
devs_mig_devices: dict[int, list[dict]] = {}
|
|
129
|
+
devs_mig_slots = 0
|
|
130
|
+
|
|
123
131
|
dev_count = pyhgml.hgmlDeviceGetCount()
|
|
124
132
|
for dev_idx in range(dev_count):
|
|
125
133
|
dev = pyhgml.hgmlDeviceGetHandleByIndex(dev_idx)
|
|
@@ -219,11 +227,13 @@ class THeadDetector(Detector):
|
|
|
219
227
|
"bdf": dev_bdf,
|
|
220
228
|
}
|
|
221
229
|
if dev_mig_mode != pyhgml.HGML_DEVICE_MIG_DISABLE:
|
|
222
|
-
|
|
230
|
+
dev_mig_slots = 0
|
|
231
|
+
with contextlib.suppress(pyhgml.HGMLError):
|
|
232
|
+
dev_mig_slots = pyhgml.hgmlDeviceGetMaxMigDeviceCount(dev)
|
|
233
|
+
devs_mig_slots = max(devs_mig_slots, dev_mig_slots)
|
|
234
|
+
dev_mig_devices = _get_mig_devices(
|
|
223
235
|
dev,
|
|
224
|
-
|
|
225
|
-
dev_count,
|
|
226
|
-
dev_cc_t,
|
|
236
|
+
dev_mig_slots,
|
|
227
237
|
sys_driver_ver,
|
|
228
238
|
sys_runtime_ver,
|
|
229
239
|
sys_runtime_ver_original,
|
|
@@ -234,6 +244,8 @@ class THeadDetector(Detector):
|
|
|
234
244
|
dev_bdf,
|
|
235
245
|
dev_numa,
|
|
236
246
|
)
|
|
247
|
+
dev_appendix["mig_devices"] = dev_mig_devices
|
|
248
|
+
devs_mig_devices[dev_idx] = dev_mig_devices
|
|
237
249
|
if dev_numa:
|
|
238
250
|
dev_appendix["numa"] = dev_numa
|
|
239
251
|
|
|
@@ -260,6 +272,7 @@ class THeadDetector(Detector):
|
|
|
260
272
|
),
|
|
261
273
|
)
|
|
262
274
|
|
|
275
|
+
index_mig_devices(ret, devs_mig_devices, devs_mig_slots)
|
|
263
276
|
except pyhgml.HGMLError:
|
|
264
277
|
debug_log_exception(logger, "Failed to fetch devices")
|
|
265
278
|
raise
|
|
@@ -541,8 +554,7 @@ def _get_links_state(
|
|
|
541
554
|
|
|
542
555
|
def _get_mig_devices(
|
|
543
556
|
dev,
|
|
544
|
-
|
|
545
|
-
dev_count: int,
|
|
557
|
+
dev_mig_slots: int,
|
|
546
558
|
sys_driver_ver,
|
|
547
559
|
sys_runtime_ver,
|
|
548
560
|
sys_runtime_ver_original,
|
|
@@ -559,10 +571,14 @@ def _get_mig_devices(
|
|
|
559
571
|
health, temperature and power), returned as appendix entries of the
|
|
560
572
|
physical card rather than standalone devices. Empty when MIG is enabled
|
|
561
573
|
but no GPU instances exist yet.
|
|
574
|
+
|
|
575
|
+
Each entry's `index` is the driver slot the MIG device was found at:
|
|
576
|
+
index_mig_devices turns it into the device index once every card is
|
|
577
|
+
detected.
|
|
562
578
|
"""
|
|
563
579
|
ret: list[dict] = []
|
|
564
580
|
with contextlib.suppress(pyhgml.HGMLError):
|
|
565
|
-
for mdev_idx in range(
|
|
581
|
+
for mdev_idx in range(dev_mig_slots):
|
|
566
582
|
mdev = None
|
|
567
583
|
with contextlib.suppress(pyhgml.HGMLError):
|
|
568
584
|
mdev = pyhgml.hgmlDeviceGetMigDeviceHandleByIndex(dev, mdev_idx)
|
|
@@ -669,7 +685,7 @@ def _get_mig_devices(
|
|
|
669
685
|
|
|
670
686
|
ret.append(
|
|
671
687
|
{
|
|
672
|
-
"index": mdev_idx
|
|
688
|
+
"index": mdev_idx,
|
|
673
689
|
"name": mdev_name,
|
|
674
690
|
"uuid": mdev_uuid,
|
|
675
691
|
"driver_version": sys_driver_ver,
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
import pytest
|
|
2
|
+
|
|
3
|
+
from gpustack_runtime.deployer.__types__ import (
|
|
4
|
+
Container,
|
|
5
|
+
ContainerExecution,
|
|
6
|
+
ContainerResources,
|
|
7
|
+
)
|
|
8
|
+
from gpustack_runtime.deployer.kuberentes import _resolve_privileged
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _container(privileged: bool | None, resources: dict | None = None) -> Container:
|
|
12
|
+
container_resources = None
|
|
13
|
+
if resources is not None:
|
|
14
|
+
container_resources = ContainerResources()
|
|
15
|
+
container_resources.update(resources)
|
|
16
|
+
return Container(
|
|
17
|
+
name="default",
|
|
18
|
+
image="gpustack/runner:latest",
|
|
19
|
+
execution=(
|
|
20
|
+
ContainerExecution(privileged=privileged)
|
|
21
|
+
if privileged is not None
|
|
22
|
+
else None
|
|
23
|
+
),
|
|
24
|
+
resources=container_resources,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@pytest.mark.parametrize(
|
|
29
|
+
"name, privileged, resources, policy, expected",
|
|
30
|
+
[
|
|
31
|
+
(
|
|
32
|
+
"no execution",
|
|
33
|
+
None,
|
|
34
|
+
None,
|
|
35
|
+
"env",
|
|
36
|
+
False,
|
|
37
|
+
),
|
|
38
|
+
(
|
|
39
|
+
"not requested",
|
|
40
|
+
False,
|
|
41
|
+
{"nvidia.com/devices": "0"},
|
|
42
|
+
"env",
|
|
43
|
+
False,
|
|
44
|
+
),
|
|
45
|
+
(
|
|
46
|
+
"no resources",
|
|
47
|
+
True,
|
|
48
|
+
None,
|
|
49
|
+
"env",
|
|
50
|
+
True,
|
|
51
|
+
),
|
|
52
|
+
(
|
|
53
|
+
"non-device resources only",
|
|
54
|
+
True,
|
|
55
|
+
{"cpu": "2", "memory": "4Gi"},
|
|
56
|
+
"env",
|
|
57
|
+
True,
|
|
58
|
+
),
|
|
59
|
+
(
|
|
60
|
+
"specific whole cards, env injection",
|
|
61
|
+
True,
|
|
62
|
+
{"cpu": "2", "nvidia.com/devices": "0,1"},
|
|
63
|
+
"env",
|
|
64
|
+
True,
|
|
65
|
+
),
|
|
66
|
+
(
|
|
67
|
+
"all devices, env injection",
|
|
68
|
+
True,
|
|
69
|
+
{"nvidia.com/devices": "all"},
|
|
70
|
+
"env",
|
|
71
|
+
True,
|
|
72
|
+
),
|
|
73
|
+
(
|
|
74
|
+
"specific whole cards, kdp injection",
|
|
75
|
+
True,
|
|
76
|
+
{"nvidia.com/devices": "0,1"},
|
|
77
|
+
"kdp",
|
|
78
|
+
False,
|
|
79
|
+
),
|
|
80
|
+
(
|
|
81
|
+
"auto-mapped devices, kdp injection",
|
|
82
|
+
True,
|
|
83
|
+
{"gpustack.ai/devices": "0"},
|
|
84
|
+
"kdp",
|
|
85
|
+
False,
|
|
86
|
+
),
|
|
87
|
+
(
|
|
88
|
+
"exclusive whole card",
|
|
89
|
+
True,
|
|
90
|
+
{"nvidia.com/gpu": "1"},
|
|
91
|
+
"env",
|
|
92
|
+
False,
|
|
93
|
+
),
|
|
94
|
+
(
|
|
95
|
+
"soft slice",
|
|
96
|
+
True,
|
|
97
|
+
{
|
|
98
|
+
"nvidia.com/gpu.sliced": "1",
|
|
99
|
+
"nvidia.com/gpu.sliced.memory-percentage": "50",
|
|
100
|
+
"nvidia.com/gpu.sliced.cores-percentage": "50",
|
|
101
|
+
},
|
|
102
|
+
"env",
|
|
103
|
+
False,
|
|
104
|
+
),
|
|
105
|
+
(
|
|
106
|
+
"hard partition",
|
|
107
|
+
True,
|
|
108
|
+
{
|
|
109
|
+
"nvidia.com/gpu.partitioned": "1",
|
|
110
|
+
"nvidia.com/gpu.partitioned.mig-1g.20gb": "1",
|
|
111
|
+
},
|
|
112
|
+
"env",
|
|
113
|
+
False,
|
|
114
|
+
),
|
|
115
|
+
(
|
|
116
|
+
"non-NVIDIA soft slice",
|
|
117
|
+
True,
|
|
118
|
+
{"amd.com/gpu.sliced": "1"},
|
|
119
|
+
"env",
|
|
120
|
+
False,
|
|
121
|
+
),
|
|
122
|
+
(
|
|
123
|
+
"resource key merely prefixed by a CDI kind",
|
|
124
|
+
True,
|
|
125
|
+
{"nvidia.com/gpu-alike": "1"},
|
|
126
|
+
"env",
|
|
127
|
+
True,
|
|
128
|
+
),
|
|
129
|
+
],
|
|
130
|
+
)
|
|
131
|
+
def test_resolve_privileged(name, privileged, resources, policy, expected, monkeypatch):
|
|
132
|
+
monkeypatch.setattr(
|
|
133
|
+
"gpustack_runtime.deployer.kuberentes.get_resource_injection_policy",
|
|
134
|
+
lambda: policy,
|
|
135
|
+
)
|
|
136
|
+
actual = _resolve_privileged(_container(privileged, resources))
|
|
137
|
+
assert actual == expected, f"case {name} expected {expected}, but got {actual}"
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from gpustack_runtime.detector import (
|
|
4
|
+
Device,
|
|
5
|
+
ManufacturerEnum,
|
|
6
|
+
expand_mig_devices,
|
|
7
|
+
)
|
|
8
|
+
from gpustack_runtime.detector.__types__ import index_mig_devices
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _card(index: int, uuid: str, mig_devices: list[dict] | None = None) -> Device:
|
|
12
|
+
appendix: dict = {"mig": mig_devices is not None}
|
|
13
|
+
if mig_devices is not None:
|
|
14
|
+
appendix["mig_devices"] = mig_devices
|
|
15
|
+
return Device(
|
|
16
|
+
manufacturer=ManufacturerEnum.NVIDIA,
|
|
17
|
+
index=index,
|
|
18
|
+
name="NVIDIA A100-SXM4-40GB",
|
|
19
|
+
uuid=uuid,
|
|
20
|
+
memory=40960,
|
|
21
|
+
appendix=appendix,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _mig(slot: int, uuid: str) -> dict:
|
|
26
|
+
# As detection reports it: "index" is the driver slot, numbered later.
|
|
27
|
+
return {
|
|
28
|
+
"index": slot,
|
|
29
|
+
"name": "1g.5gb",
|
|
30
|
+
"uuid": uuid,
|
|
31
|
+
"memory": 4864,
|
|
32
|
+
"appendix": {"vgpu": True, "sliced": True, "mig": True},
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_index_mig_devices_numbers_above_the_cards():
|
|
37
|
+
cards = [_card(0, "GPU-0"), _card(1, "GPU-1")]
|
|
38
|
+
mig_devices = {0: [_mig(0, "MIG-0-0"), _mig(1, "MIG-0-1")]}
|
|
39
|
+
|
|
40
|
+
index_mig_devices(cards, mig_devices, 8)
|
|
41
|
+
|
|
42
|
+
assert [m["index"] for m in mig_devices[0]] == [2, 3]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def test_index_mig_devices_keeps_the_blocks_apart():
|
|
46
|
+
# Every card partitioned to the maximum: with a block per card, no two MIG
|
|
47
|
+
# devices can land on the same index however many cards there are.
|
|
48
|
+
cards = [_card(i, f"GPU-{i}") for i in range(2)]
|
|
49
|
+
mig_devices = {
|
|
50
|
+
dev_idx: [_mig(slot, f"MIG-{dev_idx}-{slot}") for slot in range(7)]
|
|
51
|
+
for dev_idx in range(2)
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
index_mig_devices(cards, mig_devices, 8)
|
|
55
|
+
|
|
56
|
+
indexes = [card.index for card in cards] + [
|
|
57
|
+
mig["index"] for migs in mig_devices.values() for mig in migs
|
|
58
|
+
]
|
|
59
|
+
assert len(indexes) == len(set(indexes))
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_index_mig_devices_clears_physical_indexes():
|
|
63
|
+
# Physical indexes are minor numbers when physical index priority is on,
|
|
64
|
+
# so they are not bound by the card count: numbering from the count would
|
|
65
|
+
# collide with the cards themselves.
|
|
66
|
+
cards = [_card(2, "GPU-2"), _card(3, "GPU-3")]
|
|
67
|
+
mig_devices = {
|
|
68
|
+
0: [_mig(0, "MIG-2-0"), _mig(1, "MIG-2-1")],
|
|
69
|
+
1: [_mig(0, "MIG-3-0")],
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
index_mig_devices(cards, mig_devices, 8)
|
|
73
|
+
|
|
74
|
+
assert [m["index"] for m in mig_devices[0]] == [4, 5]
|
|
75
|
+
assert [m["index"] for m in mig_devices[1]] == [12]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def test_index_mig_devices_isolates_a_cards_numbering():
|
|
79
|
+
# Destroying an instance on one card must not renumber another's.
|
|
80
|
+
cards = [_card(0, "GPU-0"), _card(1, "GPU-1")]
|
|
81
|
+
full = {
|
|
82
|
+
0: [_mig(slot, f"MIG-0-{slot}") for slot in range(3)],
|
|
83
|
+
1: [_mig(0, "MIG-1-0")],
|
|
84
|
+
}
|
|
85
|
+
partitioned = {0: [_mig(0, "MIG-0-0")], 1: [_mig(0, "MIG-1-0")]}
|
|
86
|
+
|
|
87
|
+
index_mig_devices(cards, full, 8)
|
|
88
|
+
index_mig_devices(cards, partitioned, 8)
|
|
89
|
+
|
|
90
|
+
assert full[1][0]["index"] == partitioned[1][0]["index"]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def test_expand_mig_devices_substitutes_the_card():
|
|
94
|
+
cards = [
|
|
95
|
+
_card(0, "GPU-0", [_mig(0, "MIG-0-0"), _mig(1, "MIG-0-1")]),
|
|
96
|
+
_card(1, "GPU-1"),
|
|
97
|
+
]
|
|
98
|
+
index_mig_devices(cards, {0: cards[0].appendix["mig_devices"]}, 8)
|
|
99
|
+
|
|
100
|
+
expanded = expand_mig_devices(cards)
|
|
101
|
+
|
|
102
|
+
assert [dev.uuid for dev in expanded] == ["MIG-0-0", "MIG-0-1", "GPU-1"]
|
|
103
|
+
assert [dev.index for dev in expanded] == [2, 3, 1]
|
|
104
|
+
assert all(dev.manufacturer == ManufacturerEnum.NVIDIA for dev in expanded)
|
|
105
|
+
assert expanded[0].memory == 4864
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def test_expand_mig_devices_keeps_an_unpartitioned_card():
|
|
109
|
+
# MIG enabled, no GPU instance yet: nothing to address, keep the card.
|
|
110
|
+
cards = [_card(0, "GPU-0", [])]
|
|
111
|
+
|
|
112
|
+
assert expand_mig_devices(cards) == cards
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def test_expand_mig_devices_tolerates_no_devices():
|
|
116
|
+
assert expand_mig_devices(None) == []
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
git_commit = "fd63e6b"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/deploy/manifests/docker-compose.yaml
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/deploy/manifests/kubernetes.yaml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/docs/modules/gpustack_runtime.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/__init__.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/__types__.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/deployer.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/detector.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/cmds/images.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/__utils__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/amd.py
RENAMED
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/hygon.py
RENAMED
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/metax.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/cdi/thead.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/docker.py
RENAMED
|
File without changes
|
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/deployer/podman.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/__utils__.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/amd.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/ascend.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/cambricon.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/hygon.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/iluvatar.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/metax.py
RENAMED
|
File without changes
|
{gpustack_runtime-0.2.2.post5 → gpustack_runtime-0.2.2.post7}/gpustack_runtime/detector/mthreads.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|