videoflow 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- videoflow/__init__.py +1 -0
- videoflow/backends/__init__.py +28 -0
- videoflow/backends/allocation.py +290 -0
- videoflow/backends/capabilities.py +652 -0
- videoflow/backends/faults.py +304 -0
- videoflow/backends/identity.py +205 -0
- videoflow/backends/memory/__init__.py +11 -0
- videoflow/backends/memory/allocation.py +571 -0
- videoflow/backends/memory/clock.py +45 -0
- videoflow/backends/memory/messaging.py +729 -0
- videoflow/backends/memory/mig_geometry.py +124 -0
- videoflow/backends/memory/payload.py +408 -0
- videoflow/backends/memory/runtime_store.py +171 -0
- videoflow/backends/messaging.py +322 -0
- videoflow/backends/observation.py +97 -0
- videoflow/backends/outcomes.py +152 -0
- videoflow/backends/payload.py +209 -0
- videoflow/backends/payload_bridge.py +119 -0
- videoflow/backends/runtime.py +884 -0
- videoflow/cli.py +13 -0
- videoflow/compile.py +14 -0
- videoflow/components/__init__.py +0 -0
- videoflow/components/descriptor.py +306 -0
- videoflow/components/oci.py +147 -0
- videoflow/consumers/__init__.py +6 -0
- videoflow/consumers/basic.py +80 -0
- videoflow/consumers/video.py +65 -0
- videoflow/core/__init__.py +15 -0
- videoflow/core/compiler.py +399 -0
- videoflow/core/constants.py +11 -0
- videoflow/core/context.py +124 -0
- videoflow/core/engine.py +260 -0
- videoflow/core/errors.py +576 -0
- videoflow/core/flow.py +147 -0
- videoflow/core/graph.py +193 -0
- videoflow/core/node.py +836 -0
- videoflow/core/policies.py +542 -0
- videoflow/core/provenance.py +308 -0
- videoflow/core/remote.py +276 -0
- videoflow/core/supervision.py +400 -0
- videoflow/core/task.py +566 -0
- videoflow/deploy/__init__.py +0 -0
- videoflow/deploy/admission.py +672 -0
- videoflow/deploy/allocation_dra.py +349 -0
- videoflow/deploy/allocation_kubernetes.py +567 -0
- videoflow/deploy/allocation_local.py +391 -0
- videoflow/deploy/broker_profiles.py +281 -0
- videoflow/deploy/build.py +354 -0
- videoflow/deploy/cli.py +2132 -0
- videoflow/deploy/cluster.py +765 -0
- videoflow/deploy/compile.py +153 -0
- videoflow/deploy/gpu.py +2287 -0
- videoflow/deploy/images.py +84 -0
- videoflow/deploy/infra.py +528 -0
- videoflow/deploy/localinfra.py +160 -0
- videoflow/deploy/manifests.py +1333 -0
- videoflow/deploy/mig.py +470 -0
- videoflow/deploy/profiles.py +217 -0
- videoflow/deploy/solution.py +431 -0
- videoflow/engines/__init__.py +0 -0
- videoflow/engines/kubernetes.py +704 -0
- videoflow/engines/local.py +934 -0
- videoflow/messaging/__init__.py +0 -0
- videoflow/messaging/grouping.py +565 -0
- videoflow/messaging/jetstream_backend.py +1109 -0
- videoflow/messaging/nats_messenger.py +2161 -0
- videoflow/messaging/obligations.py +105 -0
- videoflow/messaging/topology.py +1060 -0
- videoflow/processors/__init__.py +2 -0
- videoflow/processors/aggregators.py +97 -0
- videoflow/processors/basic.py +55 -0
- videoflow/processors/vision/__init__.py +5 -0
- videoflow/processors/vision/annotators.py +221 -0
- videoflow/processors/vision/counters.py +1 -0
- videoflow/processors/vision/detectors.py +38 -0
- videoflow/processors/vision/pose.py +1 -0
- videoflow/processors/vision/segmentation.py +33 -0
- videoflow/processors/vision/trackers.py +31 -0
- videoflow/processors/vision/transformers.py +171 -0
- videoflow/producers/__init__.py +6 -0
- videoflow/producers/basic.py +40 -0
- videoflow/producers/video.py +260 -0
- videoflow/provision.py +13 -0
- videoflow/runtime/__init__.py +0 -0
- videoflow/runtime/assetcheck.py +89 -0
- videoflow/runtime/gpucheck.py +239 -0
- videoflow/runtime/health.py +531 -0
- videoflow/runtime/idempotency.py +63 -0
- videoflow/runtime/logging_config.py +40 -0
- videoflow/runtime/provision.py +183 -0
- videoflow/runtime/redis_runtime_store.py +195 -0
- videoflow/runtime/runtime_stores.py +71 -0
- videoflow/runtime/scaling.py +597 -0
- videoflow/runtime/watchdog.py +151 -0
- videoflow/runtime/worker.py +751 -0
- videoflow/serialization.py +9 -0
- videoflow/utils/__init__.py +0 -0
- videoflow/utils/downloader.py +304 -0
- videoflow/utils/generic_utils.py +208 -0
- videoflow/utils/graph.py +100 -0
- videoflow/utils/parsers.py +26 -0
- videoflow/utils/plugins.py +56 -0
- videoflow/utils/system.py +278 -0
- videoflow/utils/transforms.py +22 -0
- videoflow/v1/__init__.py +1 -0
- videoflow/v1/envelope_pb2.py +44 -0
- videoflow/v1/envelope_pb2.pyi +61 -0
- videoflow/v1/error_pb2.py +42 -0
- videoflow/v1/error_pb2.pyi +45 -0
- videoflow/v1/payloads_pb2.py +44 -0
- videoflow/v1/payloads_pb2.pyi +50 -0
- videoflow/v1/value_pb2.py +47 -0
- videoflow/v1/value_pb2.pyi +54 -0
- videoflow/version.py +1 -0
- videoflow/wire/__init__.py +0 -0
- videoflow/wire/redis_payload_store.py +846 -0
- videoflow/wire/serialization.py +775 -0
- videoflow/worker.py +16 -0
- videoflow-1.0.1.dist-info/METADATA +1048 -0
- videoflow-1.0.1.dist-info/RECORD +123 -0
- videoflow-1.0.1.dist-info/WHEEL +4 -0
- videoflow-1.0.1.dist-info/entry_points.txt +2 -0
- videoflow-1.0.1.dist-info/licenses/LICENSE +21 -0
videoflow/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from .version import __version__
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
'''
|
|
2
|
+
Backend contracts: the responsibility boundaries behind a running flow.
|
|
3
|
+
|
|
4
|
+
Videoflow builds a graph on one machine and executes it on many. Between the graph
|
|
5
|
+
and the workers sit a handful of *backends* — a message transport, a payload store,
|
|
6
|
+
an accelerator allocator, an execution environment — and a runtime that decides what
|
|
7
|
+
correct processing means regardless of which concrete backend is underneath. This
|
|
8
|
+
package holds the contracts for those boundaries, the truthful outcome types they
|
|
9
|
+
share, and reference in-memory implementations that double as the executable
|
|
10
|
+
specification every real adapter is tested against.
|
|
11
|
+
|
|
12
|
+
- ``MessagingBackend`` (``messaging``): how does an envelope reach its required consumers?
|
|
13
|
+
- ``PayloadStore`` (``payload``): where are the image bytes, and how long must they survive?
|
|
14
|
+
- ``AcceleratorAllocationBackend`` (``allocation``): which accelerators may this workload use, \
|
|
15
|
+
under what guarantees? (Implemented by ``deploy.allocation_local`` for ``run-local``, \
|
|
16
|
+
``deploy.allocation_kubernetes`` over the GPU strategies, and render-only \
|
|
17
|
+
``deploy.allocation_dra``.)
|
|
18
|
+
- ``FlowRuntime`` (``runtime``): what constitutes correct processing and recovery?
|
|
19
|
+
- the composition planner (``capabilities``): can this graph meet its requested contract?
|
|
20
|
+
|
|
21
|
+
Two rules run through all of it. Policy lives above adapters: an adapter reports
|
|
22
|
+
what it can guarantee, and the planner rejects a request it cannot meet instead
|
|
23
|
+
of quietly weakening it. And observations are truthful: a read that failed is
|
|
24
|
+
``Unknown``, never zero, never empty, never complete.
|
|
25
|
+
|
|
26
|
+
Nothing here imports a broker, a store or a cluster client at module scope, so the
|
|
27
|
+
contracts can be imported wherever the graph can.
|
|
28
|
+
'''
|
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
'''
|
|
2
|
+
The ``AcceleratorAllocationBackend`` contract: which accelerator resources may a
|
|
3
|
+
workload use, and under what guarantees.
|
|
4
|
+
|
|
5
|
+
Named for what it decides, not for one mechanism: a Kubernetes DRA driver, the
|
|
6
|
+
device plugin, videoflow's managed MIG geometry and a local ``nvidia-smi`` walk
|
|
7
|
+
are all implementations. What every implementation must share:
|
|
8
|
+
|
|
9
|
+
- **Inventory and plans are advisory.** A feasible plan is a snapshot, never a
|
|
10
|
+
reservation; authoritative allocation must survive concurrent demand.
|
|
11
|
+
- **Reservation is a compare-and-swap.** Ownership is written under a server
|
|
12
|
+
enforced precondition (a Kubernetes ``resourceVersion`` or a JSON-patch ``test``
|
|
13
|
+
op), and a stale rollback cannot remove a newer owner.
|
|
14
|
+
- **Readiness is observed, not inferred.** "Allocated", "prepared" and
|
|
15
|
+
"application-ready" are different states; a historical ``success`` label is
|
|
16
|
+
not evidence for a new operation.
|
|
17
|
+
- **Reads that failed are Unknown.** An occupancy listing the API refused does
|
|
18
|
+
not prove a device idle.
|
|
19
|
+
- **Release is idempotent and generation-fenced**, and a retained workload keeps
|
|
20
|
+
its allocation until an explicit later release.
|
|
21
|
+
|
|
22
|
+
Memory is expressed in bytes with separate meanings — usable minimum, reserved,
|
|
23
|
+
hard limit, declared peak — because one ``gpu_memory_gb`` with three meanings
|
|
24
|
+
was how a scheduler reservation got mistaken for isolation.
|
|
25
|
+
'''
|
|
26
|
+
from __future__ import absolute_import, division, print_function
|
|
27
|
+
|
|
28
|
+
import abc
|
|
29
|
+
from dataclasses import dataclass, field
|
|
30
|
+
from typing import Any, Mapping, Sequence, Union
|
|
31
|
+
|
|
32
|
+
from .capabilities import ENFORCEMENT_NONE, AllocationCapabilities
|
|
33
|
+
from .outcomes import Observation
|
|
34
|
+
|
|
35
|
+
SHARING_EXCLUSIVE = 'exclusive'
|
|
36
|
+
SHARING_ISOLATED_MIG = 'isolated_mig'
|
|
37
|
+
SHARING_COOPERATIVE = 'cooperative'
|
|
38
|
+
SHARING_ACCOUNTING_ONLY = 'accounting_only'
|
|
39
|
+
SHARING_MODES = (SHARING_EXCLUSIVE, SHARING_ISOLATED_MIG, SHARING_COOPERATIVE, SHARING_ACCOUNTING_ONLY)
|
|
40
|
+
|
|
41
|
+
ELASTICITY_FIXED = 'fixed'
|
|
42
|
+
ELASTICITY_ELASTIC = 'elastic'
|
|
43
|
+
|
|
44
|
+
CLAIM_PENDING = 'pending'
|
|
45
|
+
CLAIM_ALLOCATED = 'allocated'
|
|
46
|
+
CLAIM_PREPARED = 'prepared'
|
|
47
|
+
CLAIM_READY = 'ready'
|
|
48
|
+
CLAIM_FAILED = 'failed'
|
|
49
|
+
CLAIM_RELEASING = 'releasing'
|
|
50
|
+
CLAIM_STATES = (CLAIM_PENDING, CLAIM_ALLOCATED, CLAIM_PREPARED, CLAIM_READY, CLAIM_FAILED, CLAIM_RELEASING)
|
|
51
|
+
|
|
52
|
+
RELEASE_RELEASED = 'released'
|
|
53
|
+
RELEASE_PENDING_RECOVERY = 'pending_recovery'
|
|
54
|
+
RELEASE_STALE = 'stale'
|
|
55
|
+
|
|
56
|
+
COMPLETENESS_COMPLETE = 'complete'
|
|
57
|
+
COMPLETENESS_PARTIAL = 'partial'
|
|
58
|
+
|
|
59
|
+
PROVENANCE_EXPLICIT = 'explicit'
|
|
60
|
+
PROVENANCE_DESCRIPTOR = 'descriptor'
|
|
61
|
+
PROVENANCE_DEFAULT = 'default'
|
|
62
|
+
|
|
63
|
+
@dataclass(frozen = True)
|
|
64
|
+
class DeviceIdentity:
|
|
65
|
+
'''
|
|
66
|
+
One accelerator as three different identifier spaces: a host ordinal, a
|
|
67
|
+
physical GPU UUID, and (for a slice) a MIG UUID. They are not interchangeable;
|
|
68
|
+
a worker maps whatever it was granted onto CUDA-local ordinals itself.
|
|
69
|
+
'''
|
|
70
|
+
node : str | None
|
|
71
|
+
ordinal : int | None
|
|
72
|
+
uuid : str | None
|
|
73
|
+
mig_uuid : str | None
|
|
74
|
+
product : str
|
|
75
|
+
memory_bytes : int | None
|
|
76
|
+
mig_profile : str | None = None
|
|
77
|
+
|
|
78
|
+
@dataclass(frozen = True)
|
|
79
|
+
class InventorySnapshot:
|
|
80
|
+
'''
|
|
81
|
+
- Arguments:
|
|
82
|
+
- occupancy: device key -> units in use, as far as the reads could see.
|
|
83
|
+
- sharing: node -> classification (``physical``, ``mig``, ``time-sliced``, \
|
|
84
|
+
``mps``, ``unknown``).
|
|
85
|
+
- owners: node -> owner recorded on it (videoflow's own stamp), if any.
|
|
86
|
+
- completeness: ``complete`` when every read succeeded, ``partial`` when \
|
|
87
|
+
some read failed — a partial snapshot may plan but may not admit.
|
|
88
|
+
'''
|
|
89
|
+
devices : tuple[DeviceIdentity, ...]
|
|
90
|
+
occupancy : Mapping[str, int]
|
|
91
|
+
sharing : Mapping[str, str]
|
|
92
|
+
owners : Mapping[str, str]
|
|
93
|
+
completeness : str
|
|
94
|
+
observed_at : float
|
|
95
|
+
generation : str
|
|
96
|
+
|
|
97
|
+
@dataclass(frozen = True)
|
|
98
|
+
class Constraint:
|
|
99
|
+
'''A hard requirement or a soft preference on device or node attributes.'''
|
|
100
|
+
key : str
|
|
101
|
+
operator : str
|
|
102
|
+
values : tuple[str, ...]
|
|
103
|
+
hard : bool = True
|
|
104
|
+
|
|
105
|
+
@dataclass(frozen = True)
|
|
106
|
+
class WorkloadRequest:
|
|
107
|
+
flow_id : str
|
|
108
|
+
run_id : str
|
|
109
|
+
workload_id : str
|
|
110
|
+
device_count : int
|
|
111
|
+
sharing : str
|
|
112
|
+
minimum_usable_memory_bytes : int | None = None
|
|
113
|
+
reserved_memory_bytes : int | None = None
|
|
114
|
+
hard_memory_limit_bytes : int | None = None
|
|
115
|
+
declared_peak_memory_bytes : int | None = None
|
|
116
|
+
features : frozenset[str] = frozenset()
|
|
117
|
+
constraints : tuple[Constraint, ...] = ()
|
|
118
|
+
elasticity : str = ELASTICITY_FIXED
|
|
119
|
+
host_cpu : str | None = None
|
|
120
|
+
host_memory : str | None = None
|
|
121
|
+
provenance : Mapping[str, str] = field(default_factory = dict)
|
|
122
|
+
|
|
123
|
+
@dataclass(frozen = True)
|
|
124
|
+
class FeasiblePlan:
|
|
125
|
+
'''Advisory. ``plan_id`` and ``snapshot_generation`` tie a later reservation to the inventory it was planned on.'''
|
|
126
|
+
assignments : Mapping[str, tuple[DeviceIdentity, ...]]
|
|
127
|
+
geometry : Mapping[str, str]
|
|
128
|
+
snapshot_generation : str
|
|
129
|
+
plan_id : str
|
|
130
|
+
notes : tuple[str, ...] = ()
|
|
131
|
+
|
|
132
|
+
@dataclass(frozen = True)
|
|
133
|
+
class Infeasible:
|
|
134
|
+
reasons : tuple[str, ...]
|
|
135
|
+
|
|
136
|
+
PlanOutcome = Union[FeasiblePlan, Infeasible]
|
|
137
|
+
|
|
138
|
+
@dataclass(frozen = True)
|
|
139
|
+
class ClaimObservation:
|
|
140
|
+
claim_id : str
|
|
141
|
+
owner : str
|
|
142
|
+
desired_generation : str
|
|
143
|
+
observed_generation : str | None
|
|
144
|
+
status : str
|
|
145
|
+
grant : tuple[DeviceIdentity, ...] | None
|
|
146
|
+
evidence : Mapping[str, Any] = field(default_factory = dict)
|
|
147
|
+
|
|
148
|
+
@dataclass(frozen = True)
|
|
149
|
+
class WorkloadBindings:
|
|
150
|
+
'''
|
|
151
|
+
How a granted claim reaches a workload: environment for a process, and
|
|
152
|
+
Kubernetes fragments (plain dicts — external API schema) for a pod.
|
|
153
|
+
'''
|
|
154
|
+
env : Mapping[str, str]
|
|
155
|
+
pod_fragment : dict
|
|
156
|
+
container_fragment : dict
|
|
157
|
+
claim_manifests : list[dict] = field(default_factory = list)
|
|
158
|
+
node_constraints : dict = field(default_factory = dict)
|
|
159
|
+
|
|
160
|
+
@dataclass(frozen = True)
|
|
161
|
+
class ReleaseObservation:
|
|
162
|
+
claim_id : str
|
|
163
|
+
status : str
|
|
164
|
+
remaining : tuple[str, ...] = ()
|
|
165
|
+
reason : str = ''
|
|
166
|
+
|
|
167
|
+
@dataclass(frozen = True)
|
|
168
|
+
class DeliveredGrant:
|
|
169
|
+
'''What a workload actually received, distinct from what it requested.'''
|
|
170
|
+
workload_id : str
|
|
171
|
+
devices : tuple[DeviceIdentity, ...]
|
|
172
|
+
exclusive : bool
|
|
173
|
+
requested : int
|
|
174
|
+
policy : str
|
|
175
|
+
#: ``observed`` when the grant reflects a successful host read; ``unobserved``
|
|
176
|
+
#: when discovery failed and the launcher went ahead without one — an empty
|
|
177
|
+
#: device list then means "could not tell", never "a zero-GPU machine".
|
|
178
|
+
host : str = 'observed'
|
|
179
|
+
|
|
180
|
+
def to_dict(self) -> dict[str, Any]:
|
|
181
|
+
return {
|
|
182
|
+
'workload_id': self.workload_id,
|
|
183
|
+
'devices': [{'node': d.node, 'ordinal': d.ordinal, 'uuid': d.uuid, 'mig_uuid': d.mig_uuid,
|
|
184
|
+
'product': d.product, 'memory_bytes': d.memory_bytes, 'mig_profile': d.mig_profile}
|
|
185
|
+
for d in self.devices],
|
|
186
|
+
'exclusive': self.exclusive,
|
|
187
|
+
'requested': self.requested,
|
|
188
|
+
'policy': self.policy,
|
|
189
|
+
'host': self.host,
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
@staticmethod
|
|
193
|
+
def from_dict(d : Mapping[str, Any]) -> 'DeliveredGrant':
|
|
194
|
+
return DeliveredGrant(
|
|
195
|
+
workload_id = str(d['workload_id']),
|
|
196
|
+
devices = tuple(DeviceIdentity(x.get('node'), x.get('ordinal'), x.get('uuid'), x.get('mig_uuid'),
|
|
197
|
+
str(x.get('product', '')), x.get('memory_bytes'), x.get('mig_profile'))
|
|
198
|
+
for x in d.get('devices', ())),
|
|
199
|
+
exclusive = bool(d.get('exclusive', False)),
|
|
200
|
+
requested = int(d.get('requested', 0)),
|
|
201
|
+
policy = str(d.get('policy', '')),
|
|
202
|
+
host = str(d.get('host', 'observed')),
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
class AcceleratorAllocationBackend(abc.ABC):
|
|
206
|
+
@abc.abstractmethod
|
|
207
|
+
def capabilities(self, environment : Mapping[str, Any]) -> AllocationCapabilities:
|
|
208
|
+
...
|
|
209
|
+
|
|
210
|
+
@abc.abstractmethod
|
|
211
|
+
def inventory(self, scope : Mapping[str, Any]) -> Observation[InventorySnapshot]:
|
|
212
|
+
...
|
|
213
|
+
|
|
214
|
+
@abc.abstractmethod
|
|
215
|
+
def plan(self, requests : Sequence[WorkloadRequest], snapshot : InventorySnapshot) -> PlanOutcome:
|
|
216
|
+
...
|
|
217
|
+
|
|
218
|
+
@abc.abstractmethod
|
|
219
|
+
def reserve(self, plan : FeasiblePlan, operation_id : str,
|
|
220
|
+
expected_generation : str | None) -> ClaimObservation:
|
|
221
|
+
...
|
|
222
|
+
|
|
223
|
+
@abc.abstractmethod
|
|
224
|
+
def bindings(self, claim_id : str, workload_id : str) -> WorkloadBindings:
|
|
225
|
+
...
|
|
226
|
+
|
|
227
|
+
@abc.abstractmethod
|
|
228
|
+
def observe(self, claim_id : str) -> Observation[ClaimObservation]:
|
|
229
|
+
...
|
|
230
|
+
|
|
231
|
+
@abc.abstractmethod
|
|
232
|
+
def reconcile(self, claim_id : str, desired : str, expected_generation : str) -> ClaimObservation:
|
|
233
|
+
...
|
|
234
|
+
|
|
235
|
+
@abc.abstractmethod
|
|
236
|
+
def release(self, claim_id : str, operation_id : str, expected_generation : str,
|
|
237
|
+
keep_workloads : bool = False) -> ReleaseObservation:
|
|
238
|
+
...
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
#: Feature vocabulary a request may name (``WorkloadRequest.features``); each is
|
|
242
|
+
#: admitted only when the backend's version matrix lists it as available — a
|
|
243
|
+
#: request for dynamic MIG on hardware or a driver that cannot partition is
|
|
244
|
+
#: rejected by name, never by a blanket "impossible" (ALLOC-023).
|
|
245
|
+
FEATURE_DYNAMIC_MIG = 'dynamic-mig'
|
|
246
|
+
FEATURE_MPS = 'mps'
|
|
247
|
+
FEATURE_CONSUMABLE_CAPACITY = 'consumable-capacity'
|
|
248
|
+
FEATURE_PEER_ACCESS = 'peer-access'
|
|
249
|
+
GATED_FEATURES = (FEATURE_DYNAMIC_MIG, FEATURE_MPS, FEATURE_CONSUMABLE_CAPACITY)
|
|
250
|
+
#: Combinations no backend can serve at once: MPS shares a device between clients
|
|
251
|
+
#: while dynamic MIG re-partitions it, and consumable shares already carve it.
|
|
252
|
+
EXCLUSIVE_FEATURE_PAIRS = ((FEATURE_MPS, FEATURE_DYNAMIC_MIG), (FEATURE_MPS, FEATURE_CONSUMABLE_CAPACITY))
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def allocation_rejections(requests : Sequence[WorkloadRequest],
|
|
256
|
+
capabilities : AllocationCapabilities) -> list[str]:
|
|
257
|
+
'''
|
|
258
|
+
Why the backend cannot serve these requests, before anything is planned or
|
|
259
|
+
written — the validation-before-write boundary. Empty means admitted.
|
|
260
|
+
Each reason names the request, the feature/policy/sharing kind it needs
|
|
261
|
+
and what the backend (``adapter``/``authority``) actually offers, so the
|
|
262
|
+
operator can tell "this driver version lacks the gate" from "this hardware
|
|
263
|
+
cannot do it".
|
|
264
|
+
'''
|
|
265
|
+
reasons : list[str] = []
|
|
266
|
+
who = f'{capabilities.adapter} ({capabilities.authority})'
|
|
267
|
+
available = set(capabilities.version_matrix.get('features', ()))
|
|
268
|
+
for request in requests:
|
|
269
|
+
w = request.workload_id
|
|
270
|
+
if request.sharing == SHARING_ISOLATED_MIG and not capabilities.isolated_mig:
|
|
271
|
+
reasons.append(f'{w}: needs hardware-isolated MIG slices, which {who} does not provide')
|
|
272
|
+
if request.sharing == SHARING_COOPERATIVE and not capabilities.cooperative_sharing:
|
|
273
|
+
reasons.append(f'{w}: needs cooperative device sharing, which {who} does not provide')
|
|
274
|
+
if request.sharing == SHARING_ACCOUNTING_ONLY and capabilities.memory_enforcement == ENFORCEMENT_NONE:
|
|
275
|
+
reasons.append(f'{w}: needs memory accounting, but {who} enforces nothing')
|
|
276
|
+
if request.device_count > 1 and not capabilities.multi_device:
|
|
277
|
+
reasons.append(f'{w}: needs {request.device_count} devices in one grant; {who} grants one')
|
|
278
|
+
if request.elasticity != ELASTICITY_FIXED and not capabilities.elastic:
|
|
279
|
+
reasons.append(f'{w}: elasticity {request.elasticity!r} is not supported by {who}')
|
|
280
|
+
if FEATURE_PEER_ACCESS in request.features and not capabilities.topology_verification:
|
|
281
|
+
reasons.append(f'{w}: needs verified peer access between devices, which {who} cannot verify')
|
|
282
|
+
for feature in sorted(request.features):
|
|
283
|
+
if feature in GATED_FEATURES and feature not in available:
|
|
284
|
+
matrix = capabilities.version_matrix.get('version') or 'unversioned'
|
|
285
|
+
reasons.append(f'{w}: feature {feature!r} is not available on {who} '
|
|
286
|
+
f'(capability snapshot {matrix}; available: {sorted(available) or "none"})')
|
|
287
|
+
for a, b in EXCLUSIVE_FEATURE_PAIRS:
|
|
288
|
+
if a in request.features and b in request.features:
|
|
289
|
+
reasons.append(f'{w}: {a!r} and {b!r} cannot be combined on one device')
|
|
290
|
+
return reasons
|