videoflow 1.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (123) hide show
  1. videoflow/__init__.py +1 -0
  2. videoflow/backends/__init__.py +28 -0
  3. videoflow/backends/allocation.py +290 -0
  4. videoflow/backends/capabilities.py +652 -0
  5. videoflow/backends/faults.py +304 -0
  6. videoflow/backends/identity.py +205 -0
  7. videoflow/backends/memory/__init__.py +11 -0
  8. videoflow/backends/memory/allocation.py +571 -0
  9. videoflow/backends/memory/clock.py +45 -0
  10. videoflow/backends/memory/messaging.py +729 -0
  11. videoflow/backends/memory/mig_geometry.py +124 -0
  12. videoflow/backends/memory/payload.py +408 -0
  13. videoflow/backends/memory/runtime_store.py +171 -0
  14. videoflow/backends/messaging.py +322 -0
  15. videoflow/backends/observation.py +97 -0
  16. videoflow/backends/outcomes.py +152 -0
  17. videoflow/backends/payload.py +209 -0
  18. videoflow/backends/payload_bridge.py +119 -0
  19. videoflow/backends/runtime.py +884 -0
  20. videoflow/cli.py +13 -0
  21. videoflow/compile.py +14 -0
  22. videoflow/components/__init__.py +0 -0
  23. videoflow/components/descriptor.py +306 -0
  24. videoflow/components/oci.py +147 -0
  25. videoflow/consumers/__init__.py +6 -0
  26. videoflow/consumers/basic.py +80 -0
  27. videoflow/consumers/video.py +65 -0
  28. videoflow/core/__init__.py +15 -0
  29. videoflow/core/compiler.py +399 -0
  30. videoflow/core/constants.py +11 -0
  31. videoflow/core/context.py +124 -0
  32. videoflow/core/engine.py +260 -0
  33. videoflow/core/errors.py +576 -0
  34. videoflow/core/flow.py +147 -0
  35. videoflow/core/graph.py +193 -0
  36. videoflow/core/node.py +836 -0
  37. videoflow/core/policies.py +542 -0
  38. videoflow/core/provenance.py +308 -0
  39. videoflow/core/remote.py +276 -0
  40. videoflow/core/supervision.py +400 -0
  41. videoflow/core/task.py +566 -0
  42. videoflow/deploy/__init__.py +0 -0
  43. videoflow/deploy/admission.py +672 -0
  44. videoflow/deploy/allocation_dra.py +349 -0
  45. videoflow/deploy/allocation_kubernetes.py +567 -0
  46. videoflow/deploy/allocation_local.py +391 -0
  47. videoflow/deploy/broker_profiles.py +281 -0
  48. videoflow/deploy/build.py +354 -0
  49. videoflow/deploy/cli.py +2132 -0
  50. videoflow/deploy/cluster.py +765 -0
  51. videoflow/deploy/compile.py +153 -0
  52. videoflow/deploy/gpu.py +2287 -0
  53. videoflow/deploy/images.py +84 -0
  54. videoflow/deploy/infra.py +528 -0
  55. videoflow/deploy/localinfra.py +160 -0
  56. videoflow/deploy/manifests.py +1333 -0
  57. videoflow/deploy/mig.py +470 -0
  58. videoflow/deploy/profiles.py +217 -0
  59. videoflow/deploy/solution.py +431 -0
  60. videoflow/engines/__init__.py +0 -0
  61. videoflow/engines/kubernetes.py +704 -0
  62. videoflow/engines/local.py +934 -0
  63. videoflow/messaging/__init__.py +0 -0
  64. videoflow/messaging/grouping.py +565 -0
  65. videoflow/messaging/jetstream_backend.py +1109 -0
  66. videoflow/messaging/nats_messenger.py +2161 -0
  67. videoflow/messaging/obligations.py +105 -0
  68. videoflow/messaging/topology.py +1060 -0
  69. videoflow/processors/__init__.py +2 -0
  70. videoflow/processors/aggregators.py +97 -0
  71. videoflow/processors/basic.py +55 -0
  72. videoflow/processors/vision/__init__.py +5 -0
  73. videoflow/processors/vision/annotators.py +221 -0
  74. videoflow/processors/vision/counters.py +1 -0
  75. videoflow/processors/vision/detectors.py +38 -0
  76. videoflow/processors/vision/pose.py +1 -0
  77. videoflow/processors/vision/segmentation.py +33 -0
  78. videoflow/processors/vision/trackers.py +31 -0
  79. videoflow/processors/vision/transformers.py +171 -0
  80. videoflow/producers/__init__.py +6 -0
  81. videoflow/producers/basic.py +40 -0
  82. videoflow/producers/video.py +260 -0
  83. videoflow/provision.py +13 -0
  84. videoflow/runtime/__init__.py +0 -0
  85. videoflow/runtime/assetcheck.py +89 -0
  86. videoflow/runtime/gpucheck.py +239 -0
  87. videoflow/runtime/health.py +531 -0
  88. videoflow/runtime/idempotency.py +63 -0
  89. videoflow/runtime/logging_config.py +40 -0
  90. videoflow/runtime/provision.py +183 -0
  91. videoflow/runtime/redis_runtime_store.py +195 -0
  92. videoflow/runtime/runtime_stores.py +71 -0
  93. videoflow/runtime/scaling.py +597 -0
  94. videoflow/runtime/watchdog.py +151 -0
  95. videoflow/runtime/worker.py +751 -0
  96. videoflow/serialization.py +9 -0
  97. videoflow/utils/__init__.py +0 -0
  98. videoflow/utils/downloader.py +304 -0
  99. videoflow/utils/generic_utils.py +208 -0
  100. videoflow/utils/graph.py +100 -0
  101. videoflow/utils/parsers.py +26 -0
  102. videoflow/utils/plugins.py +56 -0
  103. videoflow/utils/system.py +278 -0
  104. videoflow/utils/transforms.py +22 -0
  105. videoflow/v1/__init__.py +1 -0
  106. videoflow/v1/envelope_pb2.py +44 -0
  107. videoflow/v1/envelope_pb2.pyi +61 -0
  108. videoflow/v1/error_pb2.py +42 -0
  109. videoflow/v1/error_pb2.pyi +45 -0
  110. videoflow/v1/payloads_pb2.py +44 -0
  111. videoflow/v1/payloads_pb2.pyi +50 -0
  112. videoflow/v1/value_pb2.py +47 -0
  113. videoflow/v1/value_pb2.pyi +54 -0
  114. videoflow/version.py +1 -0
  115. videoflow/wire/__init__.py +0 -0
  116. videoflow/wire/redis_payload_store.py +846 -0
  117. videoflow/wire/serialization.py +775 -0
  118. videoflow/worker.py +16 -0
  119. videoflow-1.0.1.dist-info/METADATA +1048 -0
  120. videoflow-1.0.1.dist-info/RECORD +123 -0
  121. videoflow-1.0.1.dist-info/WHEEL +4 -0
  122. videoflow-1.0.1.dist-info/entry_points.txt +2 -0
  123. videoflow-1.0.1.dist-info/licenses/LICENSE +21 -0
videoflow/__init__.py ADDED
@@ -0,0 +1 @@
1
+ from .version import __version__
@@ -0,0 +1,28 @@
1
+ '''
2
+ Backend contracts: the responsibility boundaries behind a running flow.
3
+
4
+ Videoflow builds a graph on one machine and executes it on many. Between the graph
5
+ and the workers sit a handful of *backends* — a message transport, a payload store,
6
+ an accelerator allocator, an execution environment — and a runtime that decides what
7
+ correct processing means regardless of which concrete backend is underneath. This
8
+ package holds the contracts for those boundaries, the truthful outcome types they
9
+ share, and reference in-memory implementations that double as the executable
10
+ specification every real adapter is tested against.
11
+
12
+ - ``MessagingBackend`` (``messaging``): how does an envelope reach its required consumers?
13
+ - ``PayloadStore`` (``payload``): where are the image bytes, and how long must they survive?
14
+ - ``AcceleratorAllocationBackend`` (``allocation``): which accelerators may this workload use, \
15
+ under what guarantees? (Implemented by ``deploy.allocation_local`` for ``run-local``, \
16
+ ``deploy.allocation_kubernetes`` over the GPU strategies, and render-only \
17
+ ``deploy.allocation_dra``.)
18
+ - ``FlowRuntime`` (``runtime``): what constitutes correct processing and recovery?
19
+ - the composition planner (``capabilities``): can this graph meet its requested contract?
20
+
21
+ Two rules run through all of it. Policy lives above adapters: an adapter reports
22
+ what it can guarantee, and the planner rejects a request it cannot meet instead
23
+ of quietly weakening it. And observations are truthful: a read that failed is
24
+ ``Unknown``, never zero, never empty, never complete.
25
+
26
+ Nothing here imports a broker, a store or a cluster client at module scope, so the
27
+ contracts can be imported wherever the graph can.
28
+ '''
@@ -0,0 +1,290 @@
1
+ '''
2
+ The ``AcceleratorAllocationBackend`` contract: which accelerator resources may a
3
+ workload use, and under what guarantees.
4
+
5
+ Named for what it decides, not for one mechanism: a Kubernetes DRA driver, the
6
+ device plugin, videoflow's managed MIG geometry and a local ``nvidia-smi`` walk
7
+ are all implementations. What every implementation must share:
8
+
9
+ - **Inventory and plans are advisory.** A feasible plan is a snapshot, never a
10
+ reservation; authoritative allocation must survive concurrent demand.
11
+ - **Reservation is a compare-and-swap.** Ownership is written under a server
12
+ enforced precondition (a Kubernetes ``resourceVersion`` or a JSON-patch ``test``
13
+ op), and a stale rollback cannot remove a newer owner.
14
+ - **Readiness is observed, not inferred.** "Allocated", "prepared" and
15
+ "application-ready" are different states; a historical ``success`` label is
16
+ not evidence for a new operation.
17
+ - **Reads that failed are Unknown.** An occupancy listing the API refused does
18
+ not prove a device idle.
19
+ - **Release is idempotent and generation-fenced**, and a retained workload keeps
20
+ its allocation until an explicit later release.
21
+
22
+ Memory is expressed in bytes with separate meanings — usable minimum, reserved,
23
+ hard limit, declared peak — because one ``gpu_memory_gb`` with three meanings
24
+ was how a scheduler reservation got mistaken for isolation.
25
+ '''
26
+ from __future__ import absolute_import, division, print_function
27
+
28
+ import abc
29
+ from dataclasses import dataclass, field
30
+ from typing import Any, Mapping, Sequence, Union
31
+
32
+ from .capabilities import ENFORCEMENT_NONE, AllocationCapabilities
33
+ from .outcomes import Observation
34
+
35
+ SHARING_EXCLUSIVE = 'exclusive'
36
+ SHARING_ISOLATED_MIG = 'isolated_mig'
37
+ SHARING_COOPERATIVE = 'cooperative'
38
+ SHARING_ACCOUNTING_ONLY = 'accounting_only'
39
+ SHARING_MODES = (SHARING_EXCLUSIVE, SHARING_ISOLATED_MIG, SHARING_COOPERATIVE, SHARING_ACCOUNTING_ONLY)
40
+
41
+ ELASTICITY_FIXED = 'fixed'
42
+ ELASTICITY_ELASTIC = 'elastic'
43
+
44
+ CLAIM_PENDING = 'pending'
45
+ CLAIM_ALLOCATED = 'allocated'
46
+ CLAIM_PREPARED = 'prepared'
47
+ CLAIM_READY = 'ready'
48
+ CLAIM_FAILED = 'failed'
49
+ CLAIM_RELEASING = 'releasing'
50
+ CLAIM_STATES = (CLAIM_PENDING, CLAIM_ALLOCATED, CLAIM_PREPARED, CLAIM_READY, CLAIM_FAILED, CLAIM_RELEASING)
51
+
52
+ RELEASE_RELEASED = 'released'
53
+ RELEASE_PENDING_RECOVERY = 'pending_recovery'
54
+ RELEASE_STALE = 'stale'
55
+
56
+ COMPLETENESS_COMPLETE = 'complete'
57
+ COMPLETENESS_PARTIAL = 'partial'
58
+
59
+ PROVENANCE_EXPLICIT = 'explicit'
60
+ PROVENANCE_DESCRIPTOR = 'descriptor'
61
+ PROVENANCE_DEFAULT = 'default'
62
+
63
+ @dataclass(frozen = True)
64
+ class DeviceIdentity:
65
+ '''
66
+ One accelerator as three different identifier spaces: a host ordinal, a
67
+ physical GPU UUID, and (for a slice) a MIG UUID. They are not interchangeable;
68
+ a worker maps whatever it was granted onto CUDA-local ordinals itself.
69
+ '''
70
+ node : str | None
71
+ ordinal : int | None
72
+ uuid : str | None
73
+ mig_uuid : str | None
74
+ product : str
75
+ memory_bytes : int | None
76
+ mig_profile : str | None = None
77
+
78
+ @dataclass(frozen = True)
79
+ class InventorySnapshot:
80
+ '''
81
+ - Arguments:
82
+ - occupancy: device key -> units in use, as far as the reads could see.
83
+ - sharing: node -> classification (``physical``, ``mig``, ``time-sliced``, \
84
+ ``mps``, ``unknown``).
85
+ - owners: node -> owner recorded on it (videoflow's own stamp), if any.
86
+ - completeness: ``complete`` when every read succeeded, ``partial`` when \
87
+ some read failed — a partial snapshot may plan but may not admit.
88
+ '''
89
+ devices : tuple[DeviceIdentity, ...]
90
+ occupancy : Mapping[str, int]
91
+ sharing : Mapping[str, str]
92
+ owners : Mapping[str, str]
93
+ completeness : str
94
+ observed_at : float
95
+ generation : str
96
+
97
+ @dataclass(frozen = True)
98
+ class Constraint:
99
+ '''A hard requirement or a soft preference on device or node attributes.'''
100
+ key : str
101
+ operator : str
102
+ values : tuple[str, ...]
103
+ hard : bool = True
104
+
105
+ @dataclass(frozen = True)
106
+ class WorkloadRequest:
107
+ flow_id : str
108
+ run_id : str
109
+ workload_id : str
110
+ device_count : int
111
+ sharing : str
112
+ minimum_usable_memory_bytes : int | None = None
113
+ reserved_memory_bytes : int | None = None
114
+ hard_memory_limit_bytes : int | None = None
115
+ declared_peak_memory_bytes : int | None = None
116
+ features : frozenset[str] = frozenset()
117
+ constraints : tuple[Constraint, ...] = ()
118
+ elasticity : str = ELASTICITY_FIXED
119
+ host_cpu : str | None = None
120
+ host_memory : str | None = None
121
+ provenance : Mapping[str, str] = field(default_factory = dict)
122
+
123
+ @dataclass(frozen = True)
124
+ class FeasiblePlan:
125
+ '''Advisory. ``plan_id`` and ``snapshot_generation`` tie a later reservation to the inventory it was planned on.'''
126
+ assignments : Mapping[str, tuple[DeviceIdentity, ...]]
127
+ geometry : Mapping[str, str]
128
+ snapshot_generation : str
129
+ plan_id : str
130
+ notes : tuple[str, ...] = ()
131
+
132
+ @dataclass(frozen = True)
133
+ class Infeasible:
134
+ reasons : tuple[str, ...]
135
+
136
+ PlanOutcome = Union[FeasiblePlan, Infeasible]
137
+
138
+ @dataclass(frozen = True)
139
+ class ClaimObservation:
140
+ claim_id : str
141
+ owner : str
142
+ desired_generation : str
143
+ observed_generation : str | None
144
+ status : str
145
+ grant : tuple[DeviceIdentity, ...] | None
146
+ evidence : Mapping[str, Any] = field(default_factory = dict)
147
+
148
+ @dataclass(frozen = True)
149
+ class WorkloadBindings:
150
+ '''
151
+ How a granted claim reaches a workload: environment for a process, and
152
+ Kubernetes fragments (plain dicts — external API schema) for a pod.
153
+ '''
154
+ env : Mapping[str, str]
155
+ pod_fragment : dict
156
+ container_fragment : dict
157
+ claim_manifests : list[dict] = field(default_factory = list)
158
+ node_constraints : dict = field(default_factory = dict)
159
+
160
+ @dataclass(frozen = True)
161
+ class ReleaseObservation:
162
+ claim_id : str
163
+ status : str
164
+ remaining : tuple[str, ...] = ()
165
+ reason : str = ''
166
+
167
+ @dataclass(frozen = True)
168
+ class DeliveredGrant:
169
+ '''What a workload actually received, distinct from what it requested.'''
170
+ workload_id : str
171
+ devices : tuple[DeviceIdentity, ...]
172
+ exclusive : bool
173
+ requested : int
174
+ policy : str
175
+ #: ``observed`` when the grant reflects a successful host read; ``unobserved``
176
+ #: when discovery failed and the launcher went ahead without one — an empty
177
+ #: device list then means "could not tell", never "a zero-GPU machine".
178
+ host : str = 'observed'
179
+
180
+ def to_dict(self) -> dict[str, Any]:
181
+ return {
182
+ 'workload_id': self.workload_id,
183
+ 'devices': [{'node': d.node, 'ordinal': d.ordinal, 'uuid': d.uuid, 'mig_uuid': d.mig_uuid,
184
+ 'product': d.product, 'memory_bytes': d.memory_bytes, 'mig_profile': d.mig_profile}
185
+ for d in self.devices],
186
+ 'exclusive': self.exclusive,
187
+ 'requested': self.requested,
188
+ 'policy': self.policy,
189
+ 'host': self.host,
190
+ }
191
+
192
+ @staticmethod
193
+ def from_dict(d : Mapping[str, Any]) -> 'DeliveredGrant':
194
+ return DeliveredGrant(
195
+ workload_id = str(d['workload_id']),
196
+ devices = tuple(DeviceIdentity(x.get('node'), x.get('ordinal'), x.get('uuid'), x.get('mig_uuid'),
197
+ str(x.get('product', '')), x.get('memory_bytes'), x.get('mig_profile'))
198
+ for x in d.get('devices', ())),
199
+ exclusive = bool(d.get('exclusive', False)),
200
+ requested = int(d.get('requested', 0)),
201
+ policy = str(d.get('policy', '')),
202
+ host = str(d.get('host', 'observed')),
203
+ )
204
+
205
+ class AcceleratorAllocationBackend(abc.ABC):
206
+ @abc.abstractmethod
207
+ def capabilities(self, environment : Mapping[str, Any]) -> AllocationCapabilities:
208
+ ...
209
+
210
+ @abc.abstractmethod
211
+ def inventory(self, scope : Mapping[str, Any]) -> Observation[InventorySnapshot]:
212
+ ...
213
+
214
+ @abc.abstractmethod
215
+ def plan(self, requests : Sequence[WorkloadRequest], snapshot : InventorySnapshot) -> PlanOutcome:
216
+ ...
217
+
218
+ @abc.abstractmethod
219
+ def reserve(self, plan : FeasiblePlan, operation_id : str,
220
+ expected_generation : str | None) -> ClaimObservation:
221
+ ...
222
+
223
+ @abc.abstractmethod
224
+ def bindings(self, claim_id : str, workload_id : str) -> WorkloadBindings:
225
+ ...
226
+
227
+ @abc.abstractmethod
228
+ def observe(self, claim_id : str) -> Observation[ClaimObservation]:
229
+ ...
230
+
231
+ @abc.abstractmethod
232
+ def reconcile(self, claim_id : str, desired : str, expected_generation : str) -> ClaimObservation:
233
+ ...
234
+
235
+ @abc.abstractmethod
236
+ def release(self, claim_id : str, operation_id : str, expected_generation : str,
237
+ keep_workloads : bool = False) -> ReleaseObservation:
238
+ ...
239
+
240
+
241
+ #: Feature vocabulary a request may name (``WorkloadRequest.features``); each is
242
+ #: admitted only when the backend's version matrix lists it as available — a
243
+ #: request for dynamic MIG on hardware or a driver that cannot partition is
244
+ #: rejected by name, never by a blanket "impossible" (ALLOC-023).
245
+ FEATURE_DYNAMIC_MIG = 'dynamic-mig'
246
+ FEATURE_MPS = 'mps'
247
+ FEATURE_CONSUMABLE_CAPACITY = 'consumable-capacity'
248
+ FEATURE_PEER_ACCESS = 'peer-access'
249
+ GATED_FEATURES = (FEATURE_DYNAMIC_MIG, FEATURE_MPS, FEATURE_CONSUMABLE_CAPACITY)
250
+ #: Combinations no backend can serve at once: MPS shares a device between clients
251
+ #: while dynamic MIG re-partitions it, and consumable shares already carve it.
252
+ EXCLUSIVE_FEATURE_PAIRS = ((FEATURE_MPS, FEATURE_DYNAMIC_MIG), (FEATURE_MPS, FEATURE_CONSUMABLE_CAPACITY))
253
+
254
+
255
+ def allocation_rejections(requests : Sequence[WorkloadRequest],
256
+ capabilities : AllocationCapabilities) -> list[str]:
257
+ '''
258
+ Why the backend cannot serve these requests, before anything is planned or
259
+ written — the validation-before-write boundary. Empty means admitted.
260
+ Each reason names the request, the feature/policy/sharing kind it needs
261
+ and what the backend (``adapter``/``authority``) actually offers, so the
262
+ operator can tell "this driver version lacks the gate" from "this hardware
263
+ cannot do it".
264
+ '''
265
+ reasons : list[str] = []
266
+ who = f'{capabilities.adapter} ({capabilities.authority})'
267
+ available = set(capabilities.version_matrix.get('features', ()))
268
+ for request in requests:
269
+ w = request.workload_id
270
+ if request.sharing == SHARING_ISOLATED_MIG and not capabilities.isolated_mig:
271
+ reasons.append(f'{w}: needs hardware-isolated MIG slices, which {who} does not provide')
272
+ if request.sharing == SHARING_COOPERATIVE and not capabilities.cooperative_sharing:
273
+ reasons.append(f'{w}: needs cooperative device sharing, which {who} does not provide')
274
+ if request.sharing == SHARING_ACCOUNTING_ONLY and capabilities.memory_enforcement == ENFORCEMENT_NONE:
275
+ reasons.append(f'{w}: needs memory accounting, but {who} enforces nothing')
276
+ if request.device_count > 1 and not capabilities.multi_device:
277
+ reasons.append(f'{w}: needs {request.device_count} devices in one grant; {who} grants one')
278
+ if request.elasticity != ELASTICITY_FIXED and not capabilities.elastic:
279
+ reasons.append(f'{w}: elasticity {request.elasticity!r} is not supported by {who}')
280
+ if FEATURE_PEER_ACCESS in request.features and not capabilities.topology_verification:
281
+ reasons.append(f'{w}: needs verified peer access between devices, which {who} cannot verify')
282
+ for feature in sorted(request.features):
283
+ if feature in GATED_FEATURES and feature not in available:
284
+ matrix = capabilities.version_matrix.get('version') or 'unversioned'
285
+ reasons.append(f'{w}: feature {feature!r} is not available on {who} '
286
+ f'(capability snapshot {matrix}; available: {sorted(available) or "none"})')
287
+ for a, b in EXCLUSIVE_FEATURE_PAIRS:
288
+ if a in request.features and b in request.features:
289
+ reasons.append(f'{w}: {a!r} and {b!r} cannot be combined on one device')
290
+ return reasons