taskferry 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- taskferry/__init__.py +211 -0
- taskferry/aio.py +486 -0
- taskferry/backends/__init__.py +38 -0
- taskferry/backends/inline.py +235 -0
- taskferry/backends/process.py +292 -0
- taskferry/backends/subprocess.py +390 -0
- taskferry/backends/thread.py +351 -0
- taskferry/capabilities.py +90 -0
- taskferry/cli.py +445 -0
- taskferry/config.py +360 -0
- taskferry/contract/__init__.py +56 -0
- taskferry/contract/base.py +179 -0
- taskferry/contract/inline.py +89 -0
- taskferry/contract/job.py +91 -0
- taskferry/contract/task.py +91 -0
- taskferry/core/__init__.py +130 -0
- taskferry/core/capabilities.py +89 -0
- taskferry/core/config.py +167 -0
- taskferry/core/correlation.py +120 -0
- taskferry/core/delivery.py +36 -0
- taskferry/core/errors.py +55 -0
- taskferry/core/ids.py +37 -0
- taskferry/core/observability.py +136 -0
- taskferry/core/otel.py +83 -0
- taskferry/core/provider.py +50 -0
- taskferry/core/py.typed +0 -0
- taskferry/core/registry.py +92 -0
- taskferry/core/serialization.py +79 -0
- taskferry/core/typing.py +16 -0
- taskferry/envelope.py +197 -0
- taskferry/errors.py +144 -0
- taskferry/execution.py +239 -0
- taskferry/functions.py +290 -0
- taskferry/handle.py +186 -0
- taskferry/hooks.py +238 -0
- taskferry/plugins.py +183 -0
- taskferry/ports.py +356 -0
- taskferry/py.typed +0 -0
- taskferry/retry.py +205 -0
- taskferry/router.py +160 -0
- taskferry/runtime.py +609 -0
- taskferry/specs.py +353 -0
- taskferry/tracking.py +129 -0
- taskferry-0.2.0.dist-info/METADATA +109 -0
- taskferry-0.2.0.dist-info/RECORD +48 -0
- taskferry-0.2.0.dist-info/WHEEL +4 -0
- taskferry-0.2.0.dist-info/entry_points.txt +2 -0
- taskferry-0.2.0.dist-info/licenses/LICENSE +201 -0
taskferry/ports.py
ADDED
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
"""The ports — the whole surface an adapter has to implement.
|
|
2
|
+
|
|
3
|
+
```mermaid
|
|
4
|
+
flowchart BT
|
|
5
|
+
PRO["taskferry_procrastinate"]
|
|
6
|
+
CT["taskferry_cloudtasks"]
|
|
7
|
+
CR["taskferry_cloudrun"]
|
|
8
|
+
DJ["taskferry_django"]
|
|
9
|
+
|
|
10
|
+
TB["TaskBackend"]
|
|
11
|
+
JB["JobBackend"]
|
|
12
|
+
IB["InlineBackend"]
|
|
13
|
+
|
|
14
|
+
CORE["taskferry<br/>specs · execution · runtime"]
|
|
15
|
+
|
|
16
|
+
PRO --> TB
|
|
17
|
+
CT --> TB
|
|
18
|
+
CR --> JB
|
|
19
|
+
DJ --> TB
|
|
20
|
+
|
|
21
|
+
TB --> CORE
|
|
22
|
+
JB --> CORE
|
|
23
|
+
IB --> CORE
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
The surface is deliberately tiny: **submit** and **get**, plus **cancel** and
|
|
27
|
+
**result** whose availability is decided by the backend's advertised
|
|
28
|
+
capabilities rather than by whether a method happens to exist.
|
|
29
|
+
|
|
30
|
+
Why capabilities rather than method presence
|
|
31
|
+
--------------------------------------------
|
|
32
|
+
|
|
33
|
+
A duck-typed "does it have ``.cancel``?" check cannot express *"can cancel a
|
|
34
|
+
queued task but not a running one"*, cannot be inspected without instantiating
|
|
35
|
+
the backend, and cannot be reported by a CLI. So every backend implements the
|
|
36
|
+
full four-method surface, and :class:`~taskferry.capabilities.CapabilitySet` is
|
|
37
|
+
the single source of truth. Calling an unsupported operation raises
|
|
38
|
+
:class:`~taskferry.errors.UnsupportedCapability` — never a silent no-op and never
|
|
39
|
+
an emulation. See ADR-0005.
|
|
40
|
+
|
|
41
|
+
Adapters normally subclass :class:`BaseBackend`, which applies the capability
|
|
42
|
+
checks, the spec validation and the hook dispatch uniformly, leaving the adapter
|
|
43
|
+
with four underscore-prefixed methods that only talk to its engine.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
from __future__ import annotations
|
|
47
|
+
|
|
48
|
+
import asyncio
|
|
49
|
+
import time
|
|
50
|
+
from abc import ABC, abstractmethod
|
|
51
|
+
from typing import Protocol, runtime_checkable
|
|
52
|
+
|
|
53
|
+
from .capabilities import Capability, CapabilitySet
|
|
54
|
+
from .errors import ExecutionNotFound, TaskferryTimeoutError, UnsupportedCapability
|
|
55
|
+
from .execution import Execution, ExecutionId, ExecutionKind, ExecutionResult, ExecutionState
|
|
56
|
+
from .hooks import HookChain
|
|
57
|
+
from .specs import AnySpec, ExecutionSpec, InlineSpec, JobSpec, TaskSpec
|
|
58
|
+
|
|
59
|
+
DEFAULT_POLL_INTERVAL = 0.1
|
|
60
|
+
"""Seconds between polls in the generic :meth:`BaseBackend.wait` loop.
|
|
61
|
+
|
|
62
|
+
Backends with a native blocking wait should override ``_wait`` instead of
|
|
63
|
+
tuning this."""
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@runtime_checkable
|
|
67
|
+
class ExecutionBackend(Protocol):
|
|
68
|
+
"""What every Taskferry backend, of every kind, provides."""
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def name(self) -> str:
|
|
72
|
+
"""Stable backend name used in routing, errors and ``Execution.backend``."""
|
|
73
|
+
...
|
|
74
|
+
|
|
75
|
+
@property
|
|
76
|
+
def kind(self) -> ExecutionKind:
|
|
77
|
+
"""The single execution kind this backend accepts."""
|
|
78
|
+
...
|
|
79
|
+
|
|
80
|
+
@property
|
|
81
|
+
def capabilities(self) -> CapabilitySet:
|
|
82
|
+
"""What this backend can actually do. Authoritative."""
|
|
83
|
+
...
|
|
84
|
+
|
|
85
|
+
def submit(self, spec: ExecutionSpec) -> Execution:
|
|
86
|
+
"""Hand ``spec`` to the engine and return the resulting execution."""
|
|
87
|
+
...
|
|
88
|
+
|
|
89
|
+
def get(self, execution_id: ExecutionId | str) -> Execution:
|
|
90
|
+
"""Return the current snapshot of a previously submitted execution.
|
|
91
|
+
|
|
92
|
+
Accepts a Taskferry id or, where the adapter documents it, the engine's
|
|
93
|
+
own id — so an operator holding a job id from a dashboard can look one up
|
|
94
|
+
from a process that never submitted it.
|
|
95
|
+
|
|
96
|
+
Requires :attr:`~taskferry.capabilities.Capability.STATE`.
|
|
97
|
+
"""
|
|
98
|
+
...
|
|
99
|
+
|
|
100
|
+
def cancel(self, execution_id: ExecutionId | str) -> Execution:
|
|
101
|
+
"""Cancel an execution. Requires ``Capability.CANCEL``."""
|
|
102
|
+
...
|
|
103
|
+
|
|
104
|
+
def result(
|
|
105
|
+
self, execution_id: ExecutionId | str, *, timeout: float | None = None
|
|
106
|
+
) -> ExecutionResult:
|
|
107
|
+
"""Return the outcome, waiting up to ``timeout``. Requires ``Capability.RESULT``."""
|
|
108
|
+
...
|
|
109
|
+
|
|
110
|
+
def wait(self, execution_id: ExecutionId | str, *, timeout: float | None = None) -> Execution:
|
|
111
|
+
"""Block until the execution is terminal. Requires ``Capability.STATE``."""
|
|
112
|
+
...
|
|
113
|
+
|
|
114
|
+
# -- the async surface --------------------------------------------------- #
|
|
115
|
+
# Every backend has one. :class:`BaseBackend` derives it from the sync
|
|
116
|
+
# methods with ``asyncio.to_thread``, so an adapter gets a correct — never
|
|
117
|
+
# loop-blocking — async path for free, and overrides only where its client
|
|
118
|
+
# is genuinely async. See ADR-0015.
|
|
119
|
+
|
|
120
|
+
async def asubmit(self, spec: ExecutionSpec) -> Execution: ...
|
|
121
|
+
|
|
122
|
+
async def aget(self, execution_id: ExecutionId | str) -> Execution: ...
|
|
123
|
+
|
|
124
|
+
async def acancel(self, execution_id: ExecutionId | str) -> Execution: ...
|
|
125
|
+
|
|
126
|
+
async def aresult(
|
|
127
|
+
self, execution_id: ExecutionId | str, *, timeout: float | None = None
|
|
128
|
+
) -> ExecutionResult: ...
|
|
129
|
+
|
|
130
|
+
async def await_(
|
|
131
|
+
self, execution_id: ExecutionId | str, *, timeout: float | None = None
|
|
132
|
+
) -> Execution:
|
|
133
|
+
"""Async ``wait``. Trailing underscore because ``await`` is a keyword."""
|
|
134
|
+
...
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@runtime_checkable
|
|
138
|
+
class InlineBackend(ExecutionBackend, Protocol):
|
|
139
|
+
"""Runs an :class:`~taskferry.specs.InlineSpec` in the current process."""
|
|
140
|
+
|
|
141
|
+
def submit(self, spec: ExecutionSpec) -> Execution: ...
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
@runtime_checkable
|
|
145
|
+
class TaskBackend(ExecutionBackend, Protocol):
|
|
146
|
+
"""Hands a :class:`~taskferry.specs.TaskSpec` to a task engine."""
|
|
147
|
+
|
|
148
|
+
def submit(self, spec: ExecutionSpec) -> Execution: ...
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
@runtime_checkable
|
|
152
|
+
class JobBackend(ExecutionBackend, Protocol):
|
|
153
|
+
"""Hands a :class:`~taskferry.specs.JobSpec` to a batch runtime."""
|
|
154
|
+
|
|
155
|
+
def submit(self, spec: ExecutionSpec) -> Execution: ...
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
_SPEC_FOR_KIND: dict[ExecutionKind, type[ExecutionSpec]] = {
|
|
159
|
+
ExecutionKind.INLINE: InlineSpec,
|
|
160
|
+
ExecutionKind.TASK: TaskSpec,
|
|
161
|
+
ExecutionKind.JOB: JobSpec,
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class BaseBackend(ABC):
|
|
166
|
+
"""Template base applying the rules every backend must follow.
|
|
167
|
+
|
|
168
|
+
Subclasses declare :attr:`name`, :attr:`kind` and :attr:`capabilities`, then
|
|
169
|
+
implement ``_submit`` and as many of ``_get`` / ``_cancel`` / ``_result`` as
|
|
170
|
+
their capabilities promise. The public methods here guarantee, uniformly:
|
|
171
|
+
|
|
172
|
+
* the spec is of the right kind for this backend;
|
|
173
|
+
* every capability the spec requires is advertised — otherwise
|
|
174
|
+
:class:`~taskferry.errors.UnsupportedCapability` before anything is sent;
|
|
175
|
+
* operations the backend does not advertise raise rather than no-op;
|
|
176
|
+
* hooks fire around submission.
|
|
177
|
+
|
|
178
|
+
Instances are expected to be safe to share across threads once constructed;
|
|
179
|
+
a backend holding mutable state must say so in its own docstring.
|
|
180
|
+
"""
|
|
181
|
+
|
|
182
|
+
#: Hook chain invoked around submissions. The runtime injects its own.
|
|
183
|
+
hooks: HookChain = HookChain()
|
|
184
|
+
|
|
185
|
+
@property
|
|
186
|
+
@abstractmethod
|
|
187
|
+
def name(self) -> str: ...
|
|
188
|
+
|
|
189
|
+
@property
|
|
190
|
+
@abstractmethod
|
|
191
|
+
def kind(self) -> ExecutionKind: ...
|
|
192
|
+
|
|
193
|
+
@property
|
|
194
|
+
@abstractmethod
|
|
195
|
+
def capabilities(self) -> CapabilitySet: ...
|
|
196
|
+
|
|
197
|
+
# -- adapter hooks ------------------------------------------------------ #
|
|
198
|
+
@abstractmethod
|
|
199
|
+
def _submit(self, spec: ExecutionSpec) -> Execution:
|
|
200
|
+
"""Send the spec to the engine. Called only after validation."""
|
|
201
|
+
|
|
202
|
+
def _get(self, execution_id: ExecutionId) -> Execution:
|
|
203
|
+
raise ExecutionNotFound(f"{execution_id!r} is unknown to {self.name!r}", backend=self.name)
|
|
204
|
+
|
|
205
|
+
def _cancel(self, execution_id: ExecutionId) -> Execution: # pragma: no cover - overridden
|
|
206
|
+
raise UnsupportedCapability(Capability.CANCEL.value, provider=self.name)
|
|
207
|
+
|
|
208
|
+
def _result(
|
|
209
|
+
self, execution_id: ExecutionId, *, timeout: float | None
|
|
210
|
+
) -> ExecutionResult: # pragma: no cover - overridden
|
|
211
|
+
raise UnsupportedCapability(Capability.RESULT.value, provider=self.name)
|
|
212
|
+
|
|
213
|
+
def _wait(self, execution_id: ExecutionId, *, timeout: float | None) -> Execution:
|
|
214
|
+
"""Block until terminal. Default implementation polls ``_get``.
|
|
215
|
+
|
|
216
|
+
Override when the engine offers a native blocking wait — polling is a
|
|
217
|
+
fallback, not a design goal.
|
|
218
|
+
"""
|
|
219
|
+
deadline = None if timeout is None else time.monotonic() + timeout
|
|
220
|
+
while True:
|
|
221
|
+
execution = self.get(execution_id)
|
|
222
|
+
if execution.is_terminal:
|
|
223
|
+
return execution
|
|
224
|
+
if deadline is not None and time.monotonic() >= deadline:
|
|
225
|
+
raise TaskferryTimeoutError(
|
|
226
|
+
f"{execution_id!r} did not finish within {timeout}s "
|
|
227
|
+
f"(last state: {execution.state.value})"
|
|
228
|
+
)
|
|
229
|
+
time.sleep(DEFAULT_POLL_INTERVAL)
|
|
230
|
+
|
|
231
|
+
# -- public surface ----------------------------------------------------- #
|
|
232
|
+
def submit(self, spec: ExecutionSpec) -> Execution:
|
|
233
|
+
self.validate(spec)
|
|
234
|
+
self.hooks.before_submit(spec, self.name)
|
|
235
|
+
try:
|
|
236
|
+
execution = self._submit(spec)
|
|
237
|
+
except Exception as exc:
|
|
238
|
+
self.hooks.on_submit_error(spec, self.name, exc)
|
|
239
|
+
raise
|
|
240
|
+
self.hooks.after_submit(spec, execution)
|
|
241
|
+
return execution
|
|
242
|
+
|
|
243
|
+
def get(self, execution_id: ExecutionId | str) -> Execution:
|
|
244
|
+
self.capabilities.require(Capability.STATE)
|
|
245
|
+
return self._get(ExecutionId(str(execution_id)))
|
|
246
|
+
|
|
247
|
+
def cancel(self, execution_id: ExecutionId | str) -> Execution:
|
|
248
|
+
self.capabilities.require(Capability.CANCEL)
|
|
249
|
+
execution = self._cancel(ExecutionId(str(execution_id)))
|
|
250
|
+
self.hooks.on_cancel(execution)
|
|
251
|
+
return execution
|
|
252
|
+
|
|
253
|
+
def result(
|
|
254
|
+
self, execution_id: ExecutionId | str, *, timeout: float | None = None
|
|
255
|
+
) -> ExecutionResult:
|
|
256
|
+
self.capabilities.require(Capability.RESULT)
|
|
257
|
+
return self._result(ExecutionId(str(execution_id)), timeout=timeout)
|
|
258
|
+
|
|
259
|
+
def wait(self, execution_id: ExecutionId | str, *, timeout: float | None = None) -> Execution:
|
|
260
|
+
"""Block until the execution is terminal.
|
|
261
|
+
|
|
262
|
+
Requires ``STATE``: without it there is nothing to poll, and pretending
|
|
263
|
+
otherwise would mean returning a fabricated terminal state.
|
|
264
|
+
"""
|
|
265
|
+
self.capabilities.require(Capability.STATE)
|
|
266
|
+
return self._wait(ExecutionId(str(execution_id)), timeout=timeout)
|
|
267
|
+
|
|
268
|
+
# -- the async surface ---------------------------------------------------- #
|
|
269
|
+
# Derived from the sync methods by default. ``asyncio.to_thread`` runs the
|
|
270
|
+
# blocking call on a worker thread, so the caller's event loop keeps
|
|
271
|
+
# spinning — which is the whole point, and is why this is a real async path
|
|
272
|
+
# rather than a cosmetic ``async def`` around a blocking call.
|
|
273
|
+
#
|
|
274
|
+
# An adapter whose client is natively async (aiobotocore, an async
|
|
275
|
+
# Procrastinate connector) overrides these and skips the thread entirely.
|
|
276
|
+
# Correctness does not depend on it doing so.
|
|
277
|
+
|
|
278
|
+
async def asubmit(self, spec: ExecutionSpec) -> Execution:
|
|
279
|
+
return await asyncio.to_thread(self.submit, spec)
|
|
280
|
+
|
|
281
|
+
async def aget(self, execution_id: ExecutionId | str) -> Execution:
|
|
282
|
+
return await asyncio.to_thread(self.get, execution_id)
|
|
283
|
+
|
|
284
|
+
async def acancel(self, execution_id: ExecutionId | str) -> Execution:
|
|
285
|
+
return await asyncio.to_thread(self.cancel, execution_id)
|
|
286
|
+
|
|
287
|
+
async def aresult(
|
|
288
|
+
self, execution_id: ExecutionId | str, *, timeout: float | None = None
|
|
289
|
+
) -> ExecutionResult:
|
|
290
|
+
return await asyncio.to_thread(lambda: self.result(execution_id, timeout=timeout))
|
|
291
|
+
|
|
292
|
+
async def await_(
|
|
293
|
+
self, execution_id: ExecutionId | str, *, timeout: float | None = None
|
|
294
|
+
) -> Execution:
|
|
295
|
+
"""Async ``wait``. Trailing underscore because ``await`` is a keyword.
|
|
296
|
+
|
|
297
|
+
The default polls on a worker thread. A backend whose engine offers a
|
|
298
|
+
native async notification should override this — holding a thread for
|
|
299
|
+
the lifetime of a long job is wasteful, even if it is correct.
|
|
300
|
+
"""
|
|
301
|
+
return await asyncio.to_thread(lambda: self.wait(execution_id, timeout=timeout))
|
|
302
|
+
|
|
303
|
+
# -- validation --------------------------------------------------------- #
|
|
304
|
+
def validate(self, spec: ExecutionSpec) -> None:
|
|
305
|
+
"""Reject a spec this backend cannot honour, before anything is sent."""
|
|
306
|
+
expected = _SPEC_FOR_KIND[self.kind]
|
|
307
|
+
if not isinstance(spec, expected):
|
|
308
|
+
raise TypeError(
|
|
309
|
+
f"{self.name!r} is a {self.kind.value} backend and needs a "
|
|
310
|
+
f"{expected.__name__}, got {type(spec).__name__}"
|
|
311
|
+
)
|
|
312
|
+
missing = self.capabilities.missing(spec.required_capabilities())
|
|
313
|
+
if missing:
|
|
314
|
+
names = ", ".join(sorted(str(c) for c in missing))
|
|
315
|
+
raise UnsupportedCapability(names, provider=self.name)
|
|
316
|
+
|
|
317
|
+
# -- convenience for adapters ------------------------------------------- #
|
|
318
|
+
def close(self) -> None: # noqa: B027 - an optional hook, deliberately not abstract
|
|
319
|
+
"""Release resources (pools, clients, connections). Idempotent.
|
|
320
|
+
|
|
321
|
+
Not abstract: most adapters hold a stateless client and have nothing to
|
|
322
|
+
release, and forcing every one of them to write an empty override would
|
|
323
|
+
be noise that hides the few that genuinely need this.
|
|
324
|
+
"""
|
|
325
|
+
|
|
326
|
+
def __repr__(self) -> str:
|
|
327
|
+
return f"<{type(self).__name__} name={self.name!r} kind={self.kind.value!r}>"
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def supports(backend: ExecutionBackend, capability: Capability) -> bool:
|
|
331
|
+
"""Whether ``backend`` advertises ``capability``. Reads better than ``in``."""
|
|
332
|
+
return capability in backend.capabilities
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def accepts(backend: ExecutionBackend, spec: AnySpec) -> bool:
|
|
336
|
+
"""Whether ``backend`` could honour ``spec`` without raising.
|
|
337
|
+
|
|
338
|
+
Used by the router to skip a candidate backend rather than let submission
|
|
339
|
+
fail, and by ``taskferry capabilities`` to explain why a route was chosen.
|
|
340
|
+
"""
|
|
341
|
+
return backend.kind is spec.kind and backend.capabilities.supports_all(
|
|
342
|
+
spec.required_capabilities()
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
__all__ = [
|
|
347
|
+
"DEFAULT_POLL_INTERVAL",
|
|
348
|
+
"BaseBackend",
|
|
349
|
+
"ExecutionBackend",
|
|
350
|
+
"ExecutionState",
|
|
351
|
+
"InlineBackend",
|
|
352
|
+
"JobBackend",
|
|
353
|
+
"TaskBackend",
|
|
354
|
+
"accepts",
|
|
355
|
+
"supports",
|
|
356
|
+
]
|
taskferry/py.typed
ADDED
|
File without changes
|
taskferry/retry.py
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Retry intent — and, more importantly, who owns it.
|
|
2
|
+
|
|
3
|
+
Retries stack badly. If the application retries, and Taskferry retries, and the
|
|
4
|
+
engine retries, and the platform retries, five attempts become several hundred:
|
|
5
|
+
|
|
6
|
+
```mermaid
|
|
7
|
+
flowchart TD
|
|
8
|
+
APP["Application"]
|
|
9
|
+
TP["Taskferry"]
|
|
10
|
+
ENGINE["Engine"]
|
|
11
|
+
INFRA["Infrastructure"]
|
|
12
|
+
|
|
13
|
+
APP -. "retry?" .-> TP
|
|
14
|
+
TP -. "retry?" .-> ENGINE
|
|
15
|
+
ENGINE -. "retry?" .-> INFRA
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
Taskferry's rule: **Taskferry never retries.** A :class:`RetryPolicy` is a portable
|
|
19
|
+
*declaration of intent* that a backend translates into its engine's native retry
|
|
20
|
+
configuration. :class:`RetryOwner` records who is actually going to act on it, so
|
|
21
|
+
the answer is written down rather than assumed.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
from enum import StrEnum
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class Backoff(StrEnum):
|
|
31
|
+
"""How the delay between attempts grows."""
|
|
32
|
+
|
|
33
|
+
NONE = "none"
|
|
34
|
+
"""Retry immediately."""
|
|
35
|
+
|
|
36
|
+
FIXED = "fixed"
|
|
37
|
+
"""Always wait ``initial_delay``."""
|
|
38
|
+
|
|
39
|
+
LINEAR = "linear"
|
|
40
|
+
"""Wait ``initial_delay * attempt``."""
|
|
41
|
+
|
|
42
|
+
EXPONENTIAL = "exponential"
|
|
43
|
+
"""Wait ``initial_delay * 2 ** (attempt - 1)``."""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class RetryOwner(StrEnum):
|
|
47
|
+
"""Which layer performs the retry. Exactly one layer should."""
|
|
48
|
+
|
|
49
|
+
BACKEND = "backend"
|
|
50
|
+
"""The engine retries natively (Procrastinate, Cloud Tasks, Cloud Run,
|
|
51
|
+
Kubernetes). Taskferry translates the policy and then stays out of the way.
|
|
52
|
+
This is the default and the recommended value."""
|
|
53
|
+
|
|
54
|
+
APPLICATION = "application"
|
|
55
|
+
"""The caller retries. Taskferry passes ``max_attempts=1`` to the engine so
|
|
56
|
+
the engine does not retry as well."""
|
|
57
|
+
|
|
58
|
+
NONE = "none"
|
|
59
|
+
"""Nobody retries. A failure is final."""
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass(frozen=True, slots=True)
|
|
63
|
+
class RetryPolicy:
|
|
64
|
+
"""Portable retry intent.
|
|
65
|
+
|
|
66
|
+
Attributes:
|
|
67
|
+
max_attempts: Total attempts including the first. ``1`` disables retries.
|
|
68
|
+
backoff: Growth strategy for the delay between attempts.
|
|
69
|
+
initial_delay: Seconds before the second attempt.
|
|
70
|
+
max_delay: Ceiling for a computed delay, or ``None`` for no ceiling.
|
|
71
|
+
jitter: Ask the engine to randomise the delay. Engines that cannot do
|
|
72
|
+
this ignore the flag — it is a hint, not a capability requirement.
|
|
73
|
+
retry_on: Exception type names (``"ConnectionError"``,
|
|
74
|
+
``"myapp.errors.Transient"``) that *should* be retried. Empty means
|
|
75
|
+
"retry anything".
|
|
76
|
+
no_retry_on: Exception type names that must never be retried. Takes
|
|
77
|
+
precedence over :attr:`retry_on`.
|
|
78
|
+
owner: Who performs the retry. See :class:`RetryOwner`.
|
|
79
|
+
|
|
80
|
+
``retry_on``/``no_retry_on`` are advisory: a backend that cannot filter by
|
|
81
|
+
exception type advertises no such capability and ignores them. The contract
|
|
82
|
+
suite asserts the backend says so rather than pretending.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
max_attempts: int = 1
|
|
86
|
+
backoff: Backoff = Backoff.EXPONENTIAL
|
|
87
|
+
initial_delay: float = 1.0
|
|
88
|
+
max_delay: float | None = 300.0
|
|
89
|
+
jitter: bool = True
|
|
90
|
+
retry_on: tuple[str, ...] = ()
|
|
91
|
+
no_retry_on: tuple[str, ...] = ()
|
|
92
|
+
owner: RetryOwner = RetryOwner.BACKEND
|
|
93
|
+
|
|
94
|
+
def __post_init__(self) -> None:
|
|
95
|
+
if self.max_attempts < 1:
|
|
96
|
+
raise ValueError("max_attempts must be >= 1")
|
|
97
|
+
if self.initial_delay < 0:
|
|
98
|
+
raise ValueError("initial_delay must be >= 0")
|
|
99
|
+
if self.max_delay is not None and self.max_delay < 0:
|
|
100
|
+
raise ValueError("max_delay must be >= 0 when set")
|
|
101
|
+
|
|
102
|
+
@property
|
|
103
|
+
def enabled(self) -> bool:
|
|
104
|
+
"""Whether this policy asks for any retry at all."""
|
|
105
|
+
return self.max_attempts > 1 and self.owner is not RetryOwner.NONE
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
def engine_attempts(self) -> int:
|
|
109
|
+
"""Attempts to configure on the engine.
|
|
110
|
+
|
|
111
|
+
``1`` unless the engine owns the retry — this is what stops two layers
|
|
112
|
+
from multiplying their attempt counts.
|
|
113
|
+
"""
|
|
114
|
+
return self.max_attempts if self.owner is RetryOwner.BACKEND else 1
|
|
115
|
+
|
|
116
|
+
def delay_for(self, attempt: int) -> float:
|
|
117
|
+
"""Delay in seconds before ``attempt`` (1-based; ``delay_for(1)`` is 0).
|
|
118
|
+
|
|
119
|
+
Pure and deterministic — :attr:`jitter` is applied by the engine, not
|
|
120
|
+
here, so this stays testable.
|
|
121
|
+
"""
|
|
122
|
+
if attempt <= 1:
|
|
123
|
+
return 0.0
|
|
124
|
+
match self.backoff:
|
|
125
|
+
case Backoff.NONE:
|
|
126
|
+
delay = 0.0
|
|
127
|
+
case Backoff.FIXED:
|
|
128
|
+
delay = self.initial_delay
|
|
129
|
+
case Backoff.LINEAR:
|
|
130
|
+
delay = self.initial_delay * (attempt - 1)
|
|
131
|
+
case Backoff.EXPONENTIAL:
|
|
132
|
+
delay = self.initial_delay * (2 ** (attempt - 2))
|
|
133
|
+
if self.max_delay is not None:
|
|
134
|
+
delay = min(delay, self.max_delay)
|
|
135
|
+
return delay
|
|
136
|
+
|
|
137
|
+
def should_retry(self, exc: BaseException, attempt: int) -> bool:
|
|
138
|
+
"""Whether ``exc`` on ``attempt`` warrants another try.
|
|
139
|
+
|
|
140
|
+
Only backends that execute in-process (inline, local) can call this;
|
|
141
|
+
remote engines apply their own equivalent from the translated policy.
|
|
142
|
+
"""
|
|
143
|
+
if attempt >= self.max_attempts or self.owner is not RetryOwner.BACKEND:
|
|
144
|
+
return False
|
|
145
|
+
names = _type_names(exc)
|
|
146
|
+
if self.no_retry_on and names & set(self.no_retry_on):
|
|
147
|
+
return False
|
|
148
|
+
if self.retry_on:
|
|
149
|
+
return bool(names & set(self.retry_on))
|
|
150
|
+
return True
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _type_names(exc: BaseException) -> set[str]:
|
|
154
|
+
"""Every name ``exc`` can be matched by: bare and fully qualified, per MRO."""
|
|
155
|
+
names: set[str] = set()
|
|
156
|
+
for klass in type(exc).__mro__:
|
|
157
|
+
if klass is object:
|
|
158
|
+
break
|
|
159
|
+
names.add(klass.__name__)
|
|
160
|
+
names.add(f"{klass.__module__}.{klass.__qualname__}")
|
|
161
|
+
return names
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
NO_RETRY = RetryPolicy(max_attempts=1, owner=RetryOwner.NONE)
|
|
165
|
+
"""Shared instance for "this must not be retried"."""
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
@dataclass(frozen=True, slots=True)
|
|
169
|
+
class TimeoutPolicy:
|
|
170
|
+
"""Portable wall-clock timeout intent.
|
|
171
|
+
|
|
172
|
+
Attributes:
|
|
173
|
+
seconds: Wall-clock budget, or ``None`` for no limit.
|
|
174
|
+
cancel_on_timeout: Ask the engine to cancel the execution when the
|
|
175
|
+
budget is exhausted rather than letting it run on.
|
|
176
|
+
|
|
177
|
+
A backend that cannot enforce timeouts does not advertise
|
|
178
|
+
:attr:`~taskferry.capabilities.Capability.TIMEOUT`, and a spec carrying one is
|
|
179
|
+
rejected — an unenforced timeout is worse than no timeout.
|
|
180
|
+
"""
|
|
181
|
+
|
|
182
|
+
seconds: float | None = None
|
|
183
|
+
cancel_on_timeout: bool = True
|
|
184
|
+
|
|
185
|
+
def __post_init__(self) -> None:
|
|
186
|
+
if self.seconds is not None and self.seconds <= 0:
|
|
187
|
+
raise ValueError("timeout seconds must be positive when set")
|
|
188
|
+
|
|
189
|
+
@property
|
|
190
|
+
def enabled(self) -> bool:
|
|
191
|
+
return self.seconds is not None
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
NO_TIMEOUT = TimeoutPolicy()
|
|
195
|
+
"""Shared instance for "no wall-clock limit"."""
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
__all__ = [
|
|
199
|
+
"NO_RETRY",
|
|
200
|
+
"NO_TIMEOUT",
|
|
201
|
+
"Backoff",
|
|
202
|
+
"RetryOwner",
|
|
203
|
+
"RetryPolicy",
|
|
204
|
+
"TimeoutPolicy",
|
|
205
|
+
]
|