without-durability-postgres 0.0.7__tar.gz → 0.0.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {without_durability_postgres-0.0.7 → without_durability_postgres-0.0.9}/PKG-INFO +5 -4
- {without_durability_postgres-0.0.7 → without_durability_postgres-0.0.9}/README.md +3 -2
- {without_durability_postgres-0.0.7 → without_durability_postgres-0.0.9}/pyproject.toml +2 -2
- {without_durability_postgres-0.0.7 → without_durability_postgres-0.0.9}/pyproject.toml.orig +2 -2
- {without_durability_postgres-0.0.7 → without_durability_postgres-0.0.9}/src/without_durability_postgres/store.py +306 -36
- {without_durability_postgres-0.0.7 → without_durability_postgres-0.0.9}/src/without_durability_postgres/__init__.py +0 -0
- {without_durability_postgres-0.0.7 → without_durability_postgres-0.0.9}/src/without_durability_postgres/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: without-durability-postgres
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.9
|
|
4
4
|
Summary: A without-durability checkpoint store and queue backed by Postgres, where every guarantee is an ordinary transaction.
|
|
5
5
|
Author: Josh Karpel
|
|
6
6
|
Author-email: Josh Karpel <josh.karpel@gmail.com>
|
|
@@ -14,7 +14,7 @@ Classifier: Programming Language :: Python :: 3 :: Only
|
|
|
14
14
|
Classifier: Programming Language :: Python :: 3.14
|
|
15
15
|
Classifier: Topic :: Software Development :: Libraries
|
|
16
16
|
Classifier: Typing :: Typed
|
|
17
|
-
Requires-Dist: without-durability==0.0.
|
|
17
|
+
Requires-Dist: without-durability==0.0.9
|
|
18
18
|
Requires-Dist: psycopg[binary,pool]>=3.2
|
|
19
19
|
Requires-Python: >=3.14
|
|
20
20
|
Description-Content-Type: text/markdown
|
|
@@ -40,8 +40,9 @@ write the Redis store needs a Lua script for is one statement here, or one
|
|
|
40
40
|
transaction, and neither is something this package supplies. Redis needs scripts
|
|
41
41
|
because it has no way to say "check this, then write that, and let nobody in
|
|
42
42
|
between"; SQL says it by default. The claim is an upsert whose `DO UPDATE` carries
|
|
43
|
-
a `WHERE` on the
|
|
44
|
-
serializes it against a claim in flight
|
|
43
|
+
a `WHERE` on the liveness deadline; the fenced record is one statement whose
|
|
44
|
+
updating CTE serializes it against a claim in flight and notes the write as a sign
|
|
45
|
+
of life in the same breath; the queue takes with
|
|
45
46
|
`FOR UPDATE SKIP LOCKED`, so several workers polling one table fan out instead of
|
|
46
47
|
queueing on its head.
|
|
47
48
|
|
|
@@ -19,8 +19,9 @@ write the Redis store needs a Lua script for is one statement here, or one
|
|
|
19
19
|
transaction, and neither is something this package supplies. Redis needs scripts
|
|
20
20
|
because it has no way to say "check this, then write that, and let nobody in
|
|
21
21
|
between"; SQL says it by default. The claim is an upsert whose `DO UPDATE` carries
|
|
22
|
-
a `WHERE` on the
|
|
23
|
-
serializes it against a claim in flight
|
|
22
|
+
a `WHERE` on the liveness deadline; the fenced record is one statement whose
|
|
23
|
+
updating CTE serializes it against a claim in flight and notes the write as a sign
|
|
24
|
+
of life in the same breath; the queue takes with
|
|
24
25
|
`FOR UPDATE SKIP LOCKED`, so several workers polling one table fan out instead of
|
|
25
26
|
queueing on its head.
|
|
26
27
|
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "without-durability-postgres"
|
|
7
|
-
version = "0.0.
|
|
7
|
+
version = "0.0.9"
|
|
8
8
|
description = "A without-durability checkpoint store and queue backed by Postgres, where every guarantee is an ordinary transaction."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -21,7 +21,7 @@ classifiers = [
|
|
|
21
21
|
"Typing :: Typed",
|
|
22
22
|
]
|
|
23
23
|
dependencies = [
|
|
24
|
-
"without-durability==0.0.
|
|
24
|
+
"without-durability==0.0.9",
|
|
25
25
|
"psycopg[binary,pool]>=3.2",
|
|
26
26
|
]
|
|
27
27
|
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "without-durability-postgres"
|
|
7
|
-
version = "0.0.
|
|
7
|
+
version = "0.0.9"
|
|
8
8
|
description = "A without-durability checkpoint store and queue backed by Postgres, where every guarantee is an ordinary transaction."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -24,7 +24,7 @@ classifiers = [
|
|
|
24
24
|
"Typing :: Typed",
|
|
25
25
|
]
|
|
26
26
|
dependencies = [
|
|
27
|
-
"without-durability==0.0.
|
|
27
|
+
"without-durability==0.0.9",
|
|
28
28
|
"psycopg[binary,pool]>=3.2",
|
|
29
29
|
]
|
|
30
30
|
|
|
@@ -70,6 +70,7 @@ from without_durability.interfaces import Entry
|
|
|
70
70
|
from without_durability.interfaces import Fenced
|
|
71
71
|
from without_durability.interfaces import Pass
|
|
72
72
|
from without_durability.interfaces import Recorded
|
|
73
|
+
from without_durability.interfaces import Written
|
|
73
74
|
from without_durability.interfaces import check_duration
|
|
74
75
|
from without_durability.stepwise import now_utc
|
|
75
76
|
|
|
@@ -112,6 +113,18 @@ POLL = timedelta(milliseconds=50)
|
|
|
112
113
|
# looks like insertion order right up until it is not: the conflict update is a real MVCC
|
|
113
114
|
# update, so it writes a new tuple version and moves the row.
|
|
114
115
|
#
|
|
116
|
+
# `written_at` is what `history` reads, and its `DEFAULT` is doing the same work `seq`'s
|
|
117
|
+
# does: evaluated on insert, and left alone by every conflict update below, so a losing
|
|
118
|
+
# write moves the value, the position, and the time equally not at all. The clock is
|
|
119
|
+
# `clock_timestamp()` and not the `now()` every other statement here reads, which is the
|
|
120
|
+
# one place this file wants a time that is not the transaction's start: `transact` runs
|
|
121
|
+
# its effect *inside* the transaction, so a step that spent ten seconds at a gateway would
|
|
122
|
+
# be stamped ten seconds before it landed, and could carry an earlier time than a `supply`
|
|
123
|
+
# that committed while it ran and took a lower `seq`, which is `history` returning its
|
|
124
|
+
# records in one order and their times in another. Both are the *server's* clock, which is
|
|
125
|
+
# what makes two records' times comparable across the machines that wrote them, and is the
|
|
126
|
+
# same clock the claim's lease is measured by.
|
|
127
|
+
#
|
|
115
128
|
# `workflow_seq` is a named sequence with a `DEFAULT` rather than an identity column,
|
|
116
129
|
# because `append` has to mint a key from the *same* number that becomes the row's
|
|
117
130
|
# position, and an identity column is a number no statement is allowed to see. Two
|
|
@@ -136,13 +149,24 @@ CREATE TABLE IF NOT EXISTS workflow_checkpoint (
|
|
|
136
149
|
step text NOT NULL,
|
|
137
150
|
value jsonb NOT NULL,
|
|
138
151
|
seq bigint NOT NULL DEFAULT nextval('workflow_seq'),
|
|
152
|
+
written_at timestamptz NOT NULL DEFAULT clock_timestamp(),
|
|
139
153
|
PRIMARY KEY (workflow, step)
|
|
140
154
|
);
|
|
141
155
|
|
|
142
156
|
CREATE TABLE IF NOT EXISTS workflow_claim (
|
|
143
157
|
workflow text PRIMARY KEY,
|
|
144
158
|
token bigint NOT NULL,
|
|
145
|
-
|
|
159
|
+
-- The budget: the latest this claim can lapse at, whatever its holder does.
|
|
160
|
+
held_until timestamptz NOT NULL,
|
|
161
|
+
-- When it lapses if nothing more is heard, which is what `CLAIM` tests. Every statement
|
|
162
|
+
-- that writes it holds it at or below `held_until` with a `LEAST`, so a sign of life
|
|
163
|
+
-- cannot carry a pass past its budget or take a workflow back after a `RELEASE`.
|
|
164
|
+
alive_until timestamptz NOT NULL,
|
|
165
|
+
-- What one sign of life is worth, carried on the row because `RECORD` is one of them
|
|
166
|
+
-- and is not told: a write is the plainest word from a pass there is, and the statement
|
|
167
|
+
-- making it has only the workflow to go on. On the row rather than in a store-wide
|
|
168
|
+
-- setting so a workflow claimed with a short window cannot quietly renew on a long one.
|
|
169
|
+
alive_for interval NOT NULL
|
|
146
170
|
);
|
|
147
171
|
|
|
148
172
|
CREATE TABLE IF NOT EXISTS workflow_queue (
|
|
@@ -175,21 +199,75 @@ MIGRATION_LOCK = 0x77_0F_10_2026
|
|
|
175
199
|
# only as good as the agreement between the two, which is exactly what fails when a
|
|
176
200
|
# machine is unhealthy enough to stall mid-pass.
|
|
177
201
|
CLAIM = """
|
|
178
|
-
INSERT INTO workflow_claim AS held (workflow, token, held_until)
|
|
179
|
-
VALUES (
|
|
202
|
+
INSERT INTO workflow_claim AS held (workflow, token, held_until, alive_until, alive_for)
|
|
203
|
+
VALUES (
|
|
204
|
+
%(workflow)s, 1,
|
|
205
|
+
now() + %(budget)s, LEAST(now() + %(alive)s, now() + %(budget)s), %(alive)s
|
|
206
|
+
)
|
|
180
207
|
ON CONFLICT (workflow) DO UPDATE
|
|
181
|
-
SET token = held.token + 1,
|
|
182
|
-
|
|
208
|
+
SET token = held.token + 1,
|
|
209
|
+
held_until = now() + %(budget)s,
|
|
210
|
+
alive_until = LEAST(now() + %(alive)s, now() + %(budget)s),
|
|
211
|
+
alive_for = %(alive)s
|
|
212
|
+
WHERE held.alive_until <= now()
|
|
183
213
|
RETURNING token
|
|
184
214
|
"""
|
|
185
215
|
|
|
216
|
+
# A fresh budget, and a sign of life with it: what a step that declared a `within` spends
|
|
217
|
+
# before it runs its effect.
|
|
218
|
+
#
|
|
219
|
+
# Conditional on the token rather than on the deadline, because a claim that has *lapsed*
|
|
220
|
+
# but not been taken is still this pass's to stretch: nobody else has raised the fence, so
|
|
221
|
+
# nothing has gone wrong that refusing here would repair. The token is the only thing that
|
|
222
|
+
# says somebody else owns the workflow now.
|
|
223
|
+
#
|
|
224
|
+
# `GREATEST` against the budget already standing, so a step naming a window shorter than
|
|
225
|
+
# what is left takes nothing away from the unannotated steps behind it: a budget can only
|
|
226
|
+
# ever be too generous from here, which is the promise `extending` skips round trips on.
|
|
227
|
+
EXTEND = """
|
|
228
|
+
UPDATE workflow_claim
|
|
229
|
+
SET held_until = GREATEST(held_until, now() + %(budget)s),
|
|
230
|
+
alive_until = LEAST(now() + %(alive)s, GREATEST(held_until, now() + %(budget)s)),
|
|
231
|
+
alive_for = %(alive)s
|
|
232
|
+
WHERE workflow = %(workflow)s AND token <= %(token)s
|
|
233
|
+
"""
|
|
234
|
+
|
|
235
|
+
# A sign of life and nothing else: the worker's tick, which says this pass is still running
|
|
236
|
+
# without saying it may run any longer than it was already granted. `LEAST` against the
|
|
237
|
+
# budget is what makes that true rather than merely intended.
|
|
238
|
+
#
|
|
239
|
+
# Refused once the budget has run out as well as below the fence, because the answer is
|
|
240
|
+
# what the worker acts on. A renewal that reported success on a lapsed claim would keep a
|
|
241
|
+
# hung pass running, holding the only delivery for its workflow, for as long as nothing
|
|
242
|
+
# else happened to take it.
|
|
243
|
+
RENEW = """
|
|
244
|
+
UPDATE workflow_claim
|
|
245
|
+
SET alive_until = LEAST(now() + %(alive)s, held_until), alive_for = %(alive)s
|
|
246
|
+
WHERE workflow = %(workflow)s AND token <= %(token)s AND held_until > now()
|
|
247
|
+
"""
|
|
248
|
+
|
|
186
249
|
# The fenced, conditional write, and the whole of `record` in one statement.
|
|
187
250
|
#
|
|
188
|
-
# The `
|
|
189
|
-
#
|
|
190
|
-
#
|
|
191
|
-
#
|
|
192
|
-
#
|
|
251
|
+
# The fence CTE is an `UPDATE` because the write is also a sign of life, and the cheapest
|
|
252
|
+
# place to say so is the statement that was already taking a row lock on the claim. That is
|
|
253
|
+
# what makes a workflow of ordinary short steps renew itself for nothing, and leaves the
|
|
254
|
+
# worker's tick with the case it is really for: one step long enough that no write falls
|
|
255
|
+
# inside a whole lease.
|
|
256
|
+
#
|
|
257
|
+
# The lock is doing real work rather than being belt-and-braces. Without it the fence is
|
|
258
|
+
# read from the statement's snapshot, so a claim committing a microsecond after the
|
|
259
|
+
# statement began would go unseen and a superseded pass's write would land. Taking the row
|
|
260
|
+
# lock makes this statement queue behind any claim in flight and then re-read the row it
|
|
261
|
+
# locked, so the token compared against is the newest one. An `UPDATE` gives that as
|
|
262
|
+
# `SELECT ... FOR UPDATE` would, re-evaluating its `WHERE` against the committed row
|
|
263
|
+
# version and returning what it re-read.
|
|
264
|
+
#
|
|
265
|
+
# The token is in the CTE's `WHERE` rather than only in the insert's, so a refused write
|
|
266
|
+
# renews nothing: a superseded pass's stray writes would otherwise keep the *winner's* claim
|
|
267
|
+
# alive after the winner had died, delaying the takeover that its silence should have
|
|
268
|
+
# brought on. The `LEAST` covers the other stray write, after a `RELEASE`: the budget is
|
|
269
|
+
# already `now()` then, so a write still in flight renews to `now()` and takes nothing back
|
|
270
|
+
# from whoever has claimed the workflow since.
|
|
193
271
|
#
|
|
194
272
|
# The rest is `HSETNX` and its read-back, as one upsert. `DO UPDATE SET value = the value
|
|
195
273
|
# already there` is a write that changes nothing and therefore returns the row that was
|
|
@@ -206,10 +284,13 @@ RETURNING token
|
|
|
206
284
|
# all when the pass is fenced
|
|
207
285
|
RECORD = """
|
|
208
286
|
WITH fence AS (
|
|
209
|
-
|
|
287
|
+
UPDATE workflow_claim
|
|
288
|
+
SET alive_until = LEAST(now() + alive_for, held_until)
|
|
289
|
+
WHERE workflow = %(workflow)s AND token <= %(token)s
|
|
290
|
+
RETURNING token
|
|
210
291
|
)
|
|
211
292
|
INSERT INTO workflow_checkpoint AS recorded (workflow, step, value)
|
|
212
|
-
SELECT %(workflow)s, %(step)s, %(value)s::jsonb FROM fence
|
|
293
|
+
SELECT %(workflow)s, %(step)s, %(value)s::jsonb FROM fence
|
|
213
294
|
ON CONFLICT (workflow, step) DO UPDATE SET value = recorded.value
|
|
214
295
|
RETURNING recorded.value::text, recorded.value = %(value)s::jsonb
|
|
215
296
|
"""
|
|
@@ -254,27 +335,80 @@ RETURNING entry.step, entry.value::text
|
|
|
254
335
|
# work in the middle. They are separate strings rather than one because the effect is
|
|
255
336
|
# arbitrary application SQL that this store cannot see, which is precisely what makes the
|
|
256
337
|
# transaction worth having.
|
|
257
|
-
|
|
338
|
+
#
|
|
339
|
+
# The fence is read twice, and the split is where the row lock goes. `FENCE` is a plain
|
|
340
|
+
# read before the effect, so a pass already superseded performs nothing; it takes no lock,
|
|
341
|
+
# so the claim row stays free while the effect runs, and the worker's `RENEW` on another
|
|
342
|
+
# connection lands instead of queueing behind the transaction for the whole effect. `WRITE`
|
|
343
|
+
# is the locked re-read *after* the effect, in the statement that renews for the reason
|
|
344
|
+
# `RECORD`'s does: it queues behind any claim in flight and re-evaluates against the
|
|
345
|
+
# committed row, so a pass superseded while its effect ran is refused here and the effect
|
|
346
|
+
# rolls back with the transaction. The lock is then held for one statement's worth rather
|
|
347
|
+
# than for the effect, which is the same window `RECORD` holds it for.
|
|
348
|
+
#
|
|
349
|
+
# `clock_timestamp()` rather than `now()`, and this is the one claim statement that needs
|
|
350
|
+
# it: `now()` is the transaction's start, and a sign of life stamped from before a long
|
|
351
|
+
# effect began could already be in the past by the time it commits, which would say a
|
|
352
|
+
# pass had gone quiet at the moment it was speaking.
|
|
353
|
+
FENCE = "SELECT token FROM workflow_claim WHERE workflow = %s"
|
|
258
354
|
ALREADY = "SELECT value::text FROM workflow_checkpoint WHERE workflow = %s AND step = %s"
|
|
259
|
-
# `ON CONFLICT DO NOTHING` rather than
|
|
260
|
-
# gated on the claim and so is the one writer this transaction's fence does not
|
|
261
|
-
# approval landing between the `ALREADY` read and this write would otherwise
|
|
262
|
-
# into a duplicate-key error, which `transact` MUST not answer with: the step
|
|
263
|
-
# so the contract is to hand back what is recorded.
|
|
264
|
-
#
|
|
265
|
-
#
|
|
355
|
+
# `ON CONFLICT DO NOTHING` rather than `RECORD`'s upsert, because `supply` is deliberately
|
|
356
|
+
# not gated on the claim and so is the one writer this transaction's fence does not
|
|
357
|
+
# exclude. An approval landing between the `ALREADY` read and this write would otherwise
|
|
358
|
+
# turn a step into a duplicate-key error, which `transact` MUST not answer with: the step
|
|
359
|
+
# is recorded, so the contract is to hand back what is recorded. So the statement reports
|
|
360
|
+
# both halves rather than one row or none: whether the fence held, and what the insert
|
|
361
|
+
# wrote. No fence is `Fenced`; a fence with nothing written says that happened, and the
|
|
362
|
+
# caller rolls the effect back rather than committing work whose record belongs to somebody
|
|
363
|
+
# else.
|
|
266
364
|
WRITE = """
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
365
|
+
WITH fence AS (
|
|
366
|
+
UPDATE workflow_claim
|
|
367
|
+
SET alive_until = LEAST(clock_timestamp() + alive_for, held_until)
|
|
368
|
+
WHERE workflow = %(workflow)s AND token <= %(token)s
|
|
369
|
+
RETURNING token
|
|
370
|
+
),
|
|
371
|
+
written AS (
|
|
372
|
+
INSERT INTO workflow_checkpoint (workflow, step, value)
|
|
373
|
+
SELECT %(workflow)s, %(step)s, %(value)s::jsonb FROM fence
|
|
374
|
+
ON CONFLICT (workflow, step) DO NOTHING
|
|
375
|
+
RETURNING value::text
|
|
376
|
+
)
|
|
377
|
+
SELECT EXISTS (SELECT FROM fence), (SELECT value FROM written)
|
|
270
378
|
"""
|
|
271
379
|
|
|
272
380
|
LOAD = "SELECT step, value::text FROM workflow_checkpoint WHERE workflow = %s ORDER BY seq"
|
|
381
|
+
HISTORY = "SELECT step, value::text, written_at FROM workflow_checkpoint WHERE workflow = %s ORDER BY seq"
|
|
382
|
+
|
|
383
|
+
# Forget every record a workflow has. Paired with `SUPERSEDE` below and never run without
|
|
384
|
+
# it, which is what the transaction in `discard` is for.
|
|
385
|
+
DISCARD = "DELETE FROM workflow_checkpoint WHERE workflow = %s"
|
|
386
|
+
|
|
387
|
+
# Take the fencing token *up*, so a pass still holding one is refused at its next write.
|
|
388
|
+
#
|
|
389
|
+
# An `UPDATE` rather than the upsert `CLAIM` is, and the difference is what it declines to
|
|
390
|
+
# do: a workflow with no claim row has no `Pass` outstanding, since a `Pass` is only ever
|
|
391
|
+
# handed out by a `claim` that wrote one, so there is nothing to fence and a row minted
|
|
392
|
+
# here would be a tombstone for a workflow nobody ever claimed. `held_until = now()` hands
|
|
393
|
+
# the workflow back at the same time, so it is claimable again immediately: what is kept is
|
|
394
|
+
# the ordering, not the claim.
|
|
395
|
+
SUPERSEDE = """
|
|
396
|
+
UPDATE workflow_claim SET token = token + 1, held_until = now(), alive_until = now()
|
|
397
|
+
WHERE workflow = %s
|
|
398
|
+
"""
|
|
273
399
|
# Hand the workflow back early, but keep the token, so the next claim gets the next
|
|
274
400
|
# number up and a pass that comes back from the dead still loses. Conditional on the
|
|
275
401
|
# token for the same reason `release` is in the Redis store: a superseded pass letting go
|
|
276
402
|
# must not hand away a claim someone else is holding.
|
|
277
|
-
|
|
403
|
+
#
|
|
404
|
+
# Both deadlines, and the budget is the load-bearing one: a write this pass had already
|
|
405
|
+
# started is entitled to land (it keeps its token), and bringing `held_until` down to now
|
|
406
|
+
# is what stops that write's own renewal from claiming the workflow straight back, since
|
|
407
|
+
# every renewal is a `LEAST` against it.
|
|
408
|
+
RELEASE = """
|
|
409
|
+
UPDATE workflow_claim SET held_until = now(), alive_until = now()
|
|
410
|
+
WHERE workflow = %s AND token = %s
|
|
411
|
+
"""
|
|
278
412
|
|
|
279
413
|
# What an effect is for a store whose datastore is a Postgres database: an async callback
|
|
280
414
|
# handed a cursor that is already inside `transact`'s transaction. The Redis store's is a
|
|
@@ -390,14 +524,38 @@ class PostgresCheckpointer:
|
|
|
390
524
|
await cursor.execute(LOAD, (workflow,))
|
|
391
525
|
return {step: self.codec.decode(encoded) for step, encoded in await cursor.fetchall()}
|
|
392
526
|
|
|
393
|
-
async def
|
|
527
|
+
async def history(self, workflow: str) -> dict[str, Written]:
|
|
528
|
+
"""The same records `load` returns, each with the moment the server wrote it."""
|
|
394
529
|
async with self.pool.connection() as connection, connection.cursor() as cursor:
|
|
395
|
-
await cursor.execute(
|
|
530
|
+
await cursor.execute(HISTORY, (workflow,))
|
|
531
|
+
return {
|
|
532
|
+
step: Written(value=self.codec.decode(encoded), at=written_at)
|
|
533
|
+
for step, encoded, written_at in await cursor.fetchall()
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
async def claim(self, workflow: str, budget: timedelta, alive: timedelta) -> Pass | None:
|
|
537
|
+
async with self.pool.connection() as connection, connection.cursor() as cursor:
|
|
538
|
+
await cursor.execute(CLAIM, {"workflow": workflow, "budget": budget, "alive": alive})
|
|
396
539
|
taken = await cursor.fetchone()
|
|
397
540
|
if taken is None:
|
|
398
541
|
return None
|
|
399
542
|
return Pass(workflow=workflow, token=cast(int, taken[0]))
|
|
400
543
|
|
|
544
|
+
async def extend(self, holder: Pass, budget: timedelta, alive: timedelta) -> bool:
|
|
545
|
+
async with self.pool.connection() as connection, connection.cursor() as cursor:
|
|
546
|
+
await cursor.execute(
|
|
547
|
+
EXTEND,
|
|
548
|
+
{"workflow": holder.workflow, "token": holder.token, "budget": budget, "alive": alive},
|
|
549
|
+
)
|
|
550
|
+
# No row touched means the `WHERE` refused the token, which is the only way
|
|
551
|
+
# this misses: a `Pass` exists because a `claim` wrote the row it names.
|
|
552
|
+
return cursor.rowcount == 1
|
|
553
|
+
|
|
554
|
+
async def renew(self, holder: Pass, alive: timedelta) -> bool:
|
|
555
|
+
async with self.pool.connection() as connection, connection.cursor() as cursor:
|
|
556
|
+
await cursor.execute(RENEW, {"workflow": holder.workflow, "token": holder.token, "alive": alive})
|
|
557
|
+
return cursor.rowcount == 1
|
|
558
|
+
|
|
401
559
|
async def record(self, holder: Pass, key: str, value: object) -> Recorded:
|
|
402
560
|
async with self.pool.connection() as connection, connection.cursor() as cursor:
|
|
403
561
|
await cursor.execute(
|
|
@@ -437,13 +595,17 @@ class PostgresCheckpointer:
|
|
|
437
595
|
which comes back a list) then does so on the first pass rather than surprising the
|
|
438
596
|
second.
|
|
439
597
|
|
|
440
|
-
The
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
598
|
+
The claim row is *not* locked while the effect runs, and that is deliberate. The
|
|
599
|
+
fence is read plainly before the effect, so a superseded pass performs nothing, and
|
|
600
|
+
re-read under the row lock after it (`WRITE`), so a pass superseded meanwhile is
|
|
601
|
+
refused and the effect rolls back with the transaction. Holding the lock across the
|
|
602
|
+
effect instead would queue the worker's own renewal behind it, so a long effect
|
|
603
|
+
would run with nothing renewing the delivery and be taken over on commit for having
|
|
604
|
+
gone quiet; it would also make `claim` wait out the effect on a pinned connection
|
|
605
|
+
rather than being told the workflow is held. What keeps another pass out while the
|
|
606
|
+
effect runs is the claim's liveness, renewed by the worker's tick, and what keeps
|
|
607
|
+
the effect from landing twice if that lapses is the step row itself: a second
|
|
608
|
+
transaction's `WRITE` conflicts with the first's and rolls its effect back.
|
|
447
609
|
|
|
448
610
|
The fence excludes every other *pass*, and one writer is left over: `supply` is
|
|
449
611
|
ungated on purpose, so an approval can land under this key between the read and the
|
|
@@ -463,11 +625,17 @@ class PostgresCheckpointer:
|
|
|
463
625
|
recorded = await cursor.fetchone()
|
|
464
626
|
if recorded is not None:
|
|
465
627
|
return self.codec.decode(cast(str, recorded[0]))
|
|
466
|
-
|
|
467
|
-
|
|
628
|
+
encoded = self.codec.encode(await effect(cursor))
|
|
629
|
+
await cursor.execute(
|
|
630
|
+
WRITE,
|
|
631
|
+
{"workflow": holder.workflow, "step": key, "value": encoded, "token": holder.token},
|
|
632
|
+
)
|
|
633
|
+
held, written = cast(tuple[bool, str | None], await cursor.fetchone())
|
|
634
|
+
if not held:
|
|
635
|
+
raise Fenced(f"{holder.workflow!r} moved on while this pass held it")
|
|
468
636
|
if written is None:
|
|
469
637
|
raise Supplied
|
|
470
|
-
return self.codec.decode(
|
|
638
|
+
return self.codec.decode(written)
|
|
471
639
|
except Supplied:
|
|
472
640
|
pass
|
|
473
641
|
async with self.pool.connection() as connection, connection.cursor() as cursor:
|
|
@@ -498,6 +666,24 @@ class PostgresCheckpointer:
|
|
|
498
666
|
key, encoded = cast(tuple[str, str], await cursor.fetchone())
|
|
499
667
|
return Entry(key=key, value=self.codec.decode(encoded))
|
|
500
668
|
|
|
669
|
+
async def discard(self, workflow: str) -> int:
|
|
670
|
+
"""
|
|
671
|
+
Forget every record this workflow has, and raise its fence, in one transaction.
|
|
672
|
+
|
|
673
|
+
One commit rather than two statements, because the two are only right together: a
|
|
674
|
+
crash between them either leaves the records deleted with the fence unraised, so
|
|
675
|
+
the pass that was mid-flight writes them back one at a time, or the reverse, which
|
|
676
|
+
fences a live pass for a deletion that never happened.
|
|
677
|
+
|
|
678
|
+
What is left behind is one claim row carrying a number. Nothing here sweeps it, in
|
|
679
|
+
keeping with the rest of this store, where nothing expires and a control-plane
|
|
680
|
+
sweep is the deployment's homework.
|
|
681
|
+
"""
|
|
682
|
+
async with self.pool.connection() as connection, connection.cursor() as cursor:
|
|
683
|
+
await cursor.execute(SUPERSEDE, (workflow,))
|
|
684
|
+
await cursor.execute(DISCARD, (workflow,))
|
|
685
|
+
return cursor.rowcount
|
|
686
|
+
|
|
501
687
|
async def release(self, holder: Pass) -> None:
|
|
502
688
|
async with self.pool.connection() as connection:
|
|
503
689
|
await connection.execute(RELEASE, (holder.workflow, holder.token))
|
|
@@ -561,6 +747,11 @@ ON CONFLICT (namespace, workflow) DO UPDATE SET visible_at = EXCLUDED.visible_at
|
|
|
561
747
|
# wrote a different `visible_at`, so the equality is the whole check.
|
|
562
748
|
FINISH = "DELETE FROM workflow_queue WHERE namespace = %s AND workflow = %s AND visible_at = %s"
|
|
563
749
|
|
|
750
|
+
# Withdraw the workflow's right to run, whatever its row currently means. Unconditional
|
|
751
|
+
# where `FINISH` compares the receipt, which is the difference between finishing a pass
|
|
752
|
+
# (leave anything that asked for another) and cancelling the workflow (leave nothing).
|
|
753
|
+
CANCEL = "DELETE FROM workflow_queue WHERE namespace = %s AND workflow = %s"
|
|
754
|
+
|
|
564
755
|
# Suspend until a deadline, under the same comparison and for the same reason. A workflow
|
|
565
756
|
# holds one row here, so writing the deadline unconditionally would land on top of a
|
|
566
757
|
# `make_ready` that arrived while the pass was ending and push a confirmation out to a
|
|
@@ -572,6 +763,23 @@ UPDATE workflow_queue SET visible_at = %(when)s
|
|
|
572
763
|
WHERE namespace = %(namespace)s AND workflow = %(workflow)s AND visible_at = %(receipt)s
|
|
573
764
|
"""
|
|
574
765
|
|
|
766
|
+
# Keep a delivery this worker's for another `within`, under the same comparison as
|
|
767
|
+
# `SUSPEND` and for the same reason: a `make_ready` that landed since would have written a
|
|
768
|
+
# different `visible_at`, and pushing the visibility out on top of it would bury a wakeup
|
|
769
|
+
# that has already arrived.
|
|
770
|
+
#
|
|
771
|
+
# It returns the new visibility because that *is* the new receipt, which is the price of
|
|
772
|
+
# the trick that makes this table a queue. A worker that went on holding the old one would
|
|
773
|
+
# find its own `FINISH` refused by the equality above and the workflow redelivered for
|
|
774
|
+
# nothing, so the rename is reported rather than left to be discovered.
|
|
775
|
+
#
|
|
776
|
+
# returns the new receipt, or no row when this delivery is no longer this worker's
|
|
777
|
+
RENEW_DELIVERY = """
|
|
778
|
+
UPDATE workflow_queue SET visible_at = now() + %(within)s
|
|
779
|
+
WHERE namespace = %(namespace)s AND workflow = %(workflow)s AND visible_at = %(receipt)s
|
|
780
|
+
RETURNING visible_at
|
|
781
|
+
"""
|
|
782
|
+
|
|
575
783
|
|
|
576
784
|
@dataclass(frozen=True, slots=True)
|
|
577
785
|
class PostgresScheduler:
|
|
@@ -698,6 +906,52 @@ class PostgresScheduler:
|
|
|
698
906
|
"""Nothing to take over by hand: an abandoned workflow becomes visible on its own."""
|
|
699
907
|
return None
|
|
700
908
|
|
|
909
|
+
async def extend(self, delivery: Delivery, within: timedelta) -> Delivery:
|
|
910
|
+
"""
|
|
911
|
+
Push this delivery's visibility out, and say what it is called now.
|
|
912
|
+
|
|
913
|
+
The new visibility is the new receipt, since this table's receipt *is* its
|
|
914
|
+
visibility, so the caller is handed a delivery to use from here rather than left to
|
|
915
|
+
work out that the one it holds has been renamed.
|
|
916
|
+
|
|
917
|
+
No row means this delivery is no longer this worker's: taken over, cancelled, or
|
|
918
|
+
rescheduled by a wakeup that arrived mid-pass, all of which wrote a `visible_at`
|
|
919
|
+
that is not the one it took. The answer to every one of them is to hand back what
|
|
920
|
+
came in and let whoever now owns the row have it, exactly as `wake_at` and `done`
|
|
921
|
+
already do.
|
|
922
|
+
"""
|
|
923
|
+
async with self.pool.connection() as connection, connection.cursor() as cursor:
|
|
924
|
+
await cursor.execute(
|
|
925
|
+
RENEW_DELIVERY,
|
|
926
|
+
{
|
|
927
|
+
"namespace": self.namespace,
|
|
928
|
+
"workflow": delivery.workflow,
|
|
929
|
+
"receipt": datetime.fromisoformat(delivery.receipt),
|
|
930
|
+
"within": within,
|
|
931
|
+
},
|
|
932
|
+
)
|
|
933
|
+
renewed = await cursor.fetchone()
|
|
934
|
+
if renewed is None:
|
|
935
|
+
return delivery
|
|
936
|
+
return Delivery(workflow=delivery.workflow, receipt=cast(datetime, renewed[0]).isoformat())
|
|
937
|
+
|
|
938
|
+
async def cancel(self, workflow: str) -> None:
|
|
939
|
+
"""
|
|
940
|
+
Drop the workflow's row, whichever of the three things its `visible_at` means.
|
|
941
|
+
|
|
942
|
+
One `DELETE` covers queued, sleeping, and out with a worker, because this table
|
|
943
|
+
holds one row per workflow and the visibility is the only thing that differs
|
|
944
|
+
between them. That is the same collapse that leaves `wake_due` and `reclaim` with
|
|
945
|
+
nothing to do.
|
|
946
|
+
|
|
947
|
+
The half of `cancel` a queue sweep cannot reach comes free with it: a pass still in
|
|
948
|
+
flight answers with `wake_at`, which is an `UPDATE` conditional on the visibility
|
|
949
|
+
still being the one it took, and a deleted row has none. So the deadline it was
|
|
950
|
+
about to write updates nothing and a deleted workflow is not put back to sleep.
|
|
951
|
+
"""
|
|
952
|
+
async with self.pool.connection() as connection:
|
|
953
|
+
await connection.execute(CANCEL, (self.namespace, workflow))
|
|
954
|
+
|
|
701
955
|
async def done(self, delivery: Delivery) -> None:
|
|
702
956
|
"""
|
|
703
957
|
Drop the workflow, unless something asked for another pass while this one ran.
|
|
@@ -774,3 +1028,19 @@ class PostgresDurable:
|
|
|
774
1028
|
{"namespace": self.scheduler.namespace, "workflow": workflow, "visible_at": self.scheduler.now()},
|
|
775
1029
|
)
|
|
776
1030
|
return Entry(key=key, value=codec.decode(encoded))
|
|
1031
|
+
|
|
1032
|
+
async def delete(self, workflow: str) -> int:
|
|
1033
|
+
"""
|
|
1034
|
+
Cancel the workflow's wakeups and forget its records, together or not at all.
|
|
1035
|
+
|
|
1036
|
+
Three statements in one commit, so the ordering `SplitDurable` has to reason about
|
|
1037
|
+
does not arise: there is no window in which the records are gone and a wakeup is
|
|
1038
|
+
not, and none in which the fence has been raised for a deletion that did not
|
|
1039
|
+
happen. Which is the same thing `arrive` gets from this store and for the same
|
|
1040
|
+
reason, one datastore.
|
|
1041
|
+
"""
|
|
1042
|
+
async with self.checkpointer.pool.connection() as connection, connection.cursor() as cursor:
|
|
1043
|
+
await cursor.execute(CANCEL, (self.scheduler.namespace, workflow))
|
|
1044
|
+
await cursor.execute(SUPERSEDE, (workflow,))
|
|
1045
|
+
await cursor.execute(DISCARD, (workflow,))
|
|
1046
|
+
return cursor.rowcount
|
|
File without changes
|
|
File without changes
|