without-durability-postgres 0.0.8__tar.gz → 0.0.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: without-durability-postgres
3
- Version: 0.0.8
3
+ Version: 0.0.9
4
4
  Summary: A without-durability checkpoint store and queue backed by Postgres, where every guarantee is an ordinary transaction.
5
5
  Author: Josh Karpel
6
6
  Author-email: Josh Karpel <josh.karpel@gmail.com>
@@ -14,7 +14,7 @@ Classifier: Programming Language :: Python :: 3 :: Only
14
14
  Classifier: Programming Language :: Python :: 3.14
15
15
  Classifier: Topic :: Software Development :: Libraries
16
16
  Classifier: Typing :: Typed
17
- Requires-Dist: without-durability==0.0.8
17
+ Requires-Dist: without-durability==0.0.9
18
18
  Requires-Dist: psycopg[binary,pool]>=3.2
19
19
  Requires-Python: >=3.14
20
20
  Description-Content-Type: text/markdown
@@ -40,8 +40,9 @@ write the Redis store needs a Lua script for is one statement here, or one
40
40
  transaction, and neither is something this package supplies. Redis needs scripts
41
41
  because it has no way to say "check this, then write that, and let nobody in
42
42
  between"; SQL says it by default. The claim is an upsert whose `DO UPDATE` carries
43
- a `WHERE` on the lease; the fenced record is one statement whose `FOR UPDATE` CTE
44
- serializes it against a claim in flight; the queue takes with
43
+ a `WHERE` on the liveness deadline; the fenced record is one statement whose
44
+ updating CTE serializes it against a claim in flight and notes the write as a sign
45
+ of life in the same breath; the queue takes with
45
46
  `FOR UPDATE SKIP LOCKED`, so several workers polling one table fan out instead of
46
47
  queueing on its head.
47
48
 
@@ -19,8 +19,9 @@ write the Redis store needs a Lua script for is one statement here, or one
19
19
  transaction, and neither is something this package supplies. Redis needs scripts
20
20
  because it has no way to say "check this, then write that, and let nobody in
21
21
  between"; SQL says it by default. The claim is an upsert whose `DO UPDATE` carries
22
- a `WHERE` on the lease; the fenced record is one statement whose `FOR UPDATE` CTE
23
- serializes it against a claim in flight; the queue takes with
22
+ a `WHERE` on the liveness deadline; the fenced record is one statement whose
23
+ updating CTE serializes it against a claim in flight and notes the write as a sign
24
+ of life in the same breath; the queue takes with
24
25
  `FOR UPDATE SKIP LOCKED`, so several workers polling one table fan out instead of
25
26
  queueing on its head.
26
27
 
@@ -4,7 +4,7 @@ build-backend = "uv_build"
4
4
 
5
5
  [project]
6
6
  name = "without-durability-postgres"
7
- version = "0.0.8"
7
+ version = "0.0.9"
8
8
  description = "A without-durability checkpoint store and queue backed by Postgres, where every guarantee is an ordinary transaction."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -21,7 +21,7 @@ classifiers = [
21
21
  "Typing :: Typed",
22
22
  ]
23
23
  dependencies = [
24
- "without-durability==0.0.8",
24
+ "without-durability==0.0.9",
25
25
  "psycopg[binary,pool]>=3.2",
26
26
  ]
27
27
 
@@ -4,7 +4,7 @@ build-backend = "uv_build"
4
4
 
5
5
  [project]
6
6
  name = "without-durability-postgres"
7
- version = "0.0.8"
7
+ version = "0.0.9"
8
8
  description = "A without-durability checkpoint store and queue backed by Postgres, where every guarantee is an ordinary transaction."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -24,7 +24,7 @@ classifiers = [
24
24
  "Typing :: Typed",
25
25
  ]
26
26
  dependencies = [
27
- "without-durability==0.0.8",
27
+ "without-durability==0.0.9",
28
28
  "psycopg[binary,pool]>=3.2",
29
29
  ]
30
30
 
@@ -156,7 +156,17 @@ CREATE TABLE IF NOT EXISTS workflow_checkpoint (
156
156
  CREATE TABLE IF NOT EXISTS workflow_claim (
157
157
  workflow text PRIMARY KEY,
158
158
  token bigint NOT NULL,
159
- held_until timestamptz NOT NULL
159
+ -- The budget: the latest this claim can lapse at, whatever its holder does.
160
+ held_until timestamptz NOT NULL,
161
+ -- When it lapses if nothing more is heard, which is what `CLAIM` tests. Every statement
162
+ -- that writes it holds it at or below `held_until` with a `LEAST`, so a sign of life
163
+ -- cannot carry a pass past its budget or take a workflow back after a `RELEASE`.
164
+ alive_until timestamptz NOT NULL,
165
+ -- What one sign of life is worth, carried on the row because `RECORD` is one of them
166
+ -- and is not told: a write is the plainest word from a pass there is, and the statement
167
+ -- making it has only the workflow to go on. On the row rather than in a store-wide
168
+ -- setting so a workflow claimed with a short window cannot quietly renew on a long one.
169
+ alive_for interval NOT NULL
160
170
  );
161
171
 
162
172
  CREATE TABLE IF NOT EXISTS workflow_queue (
@@ -189,21 +199,75 @@ MIGRATION_LOCK = 0x77_0F_10_2026
189
199
  # only as good as the agreement between the two, which is exactly what fails when a
190
200
  # machine is unhealthy enough to stall mid-pass.
191
201
  CLAIM = """
192
- INSERT INTO workflow_claim AS held (workflow, token, held_until)
193
- VALUES (%(workflow)s, 1, now() + %(lease)s)
202
+ INSERT INTO workflow_claim AS held (workflow, token, held_until, alive_until, alive_for)
203
+ VALUES (
204
+ %(workflow)s, 1,
205
+ now() + %(budget)s, LEAST(now() + %(alive)s, now() + %(budget)s), %(alive)s
206
+ )
194
207
  ON CONFLICT (workflow) DO UPDATE
195
- SET token = held.token + 1, held_until = now() + %(lease)s
196
- WHERE held.held_until <= now()
208
+ SET token = held.token + 1,
209
+ held_until = now() + %(budget)s,
210
+ alive_until = LEAST(now() + %(alive)s, now() + %(budget)s),
211
+ alive_for = %(alive)s
212
+ WHERE held.alive_until <= now()
197
213
  RETURNING token
198
214
  """
199
215
 
216
+ # A fresh budget, and a sign of life with it: what a step that declared a `within` spends
217
+ # before it runs its effect.
218
+ #
219
+ # Conditional on the token rather than on the deadline, because a claim that has *lapsed*
220
+ # but not been taken is still this pass's to stretch: nobody else has raised the fence, so
221
+ # nothing has gone wrong that refusing here would repair. The token is the only thing that
222
+ # says somebody else owns the workflow now.
223
+ #
224
+ # `GREATEST` against the budget already standing, so a step naming a window shorter than
225
+ # what is left takes nothing away from the unannotated steps behind it: a budget can only
226
+ # ever be too generous from here, which is the promise `extending` skips round trips on.
227
+ EXTEND = """
228
+ UPDATE workflow_claim
229
+ SET held_until = GREATEST(held_until, now() + %(budget)s),
230
+ alive_until = LEAST(now() + %(alive)s, GREATEST(held_until, now() + %(budget)s)),
231
+ alive_for = %(alive)s
232
+ WHERE workflow = %(workflow)s AND token <= %(token)s
233
+ """
234
+
235
+ # A sign of life and nothing else: the worker's tick, which says this pass is still running
236
+ # without saying it may run any longer than it was already granted. `LEAST` against the
237
+ # budget is what makes that true rather than merely intended.
238
+ #
239
+ # Refused once the budget has run out as well as below the fence, because the answer is
240
+ # what the worker acts on. A renewal that reported success on a lapsed claim would keep a
241
+ # hung pass running, holding the only delivery for its workflow, for as long as nothing
242
+ # else happened to take it.
243
+ RENEW = """
244
+ UPDATE workflow_claim
245
+ SET alive_until = LEAST(now() + %(alive)s, held_until), alive_for = %(alive)s
246
+ WHERE workflow = %(workflow)s AND token <= %(token)s AND held_until > now()
247
+ """
248
+
200
249
  # The fenced, conditional write, and the whole of `record` in one statement.
201
250
  #
202
- # The `FOR UPDATE` is doing real work rather than being belt-and-braces. Without it the
203
- # fence is read from the statement's snapshot, so a claim committing a microsecond after
204
- # the statement began would go unseen and a superseded pass's write would land. Taking
205
- # the row lock makes this statement queue behind any claim in flight and then re-read the
206
- # row it locked, so the token compared against is the newest one.
251
+ # The fence CTE is an `UPDATE` because the write is also a sign of life, and the cheapest
252
+ # place to say so is the statement that was already taking a row lock on the claim. That is
253
+ # what makes a workflow of ordinary short steps renew itself for nothing, and leaves the
254
+ # worker's tick with the case it is really for: one step long enough that no write falls
255
+ # inside a whole lease.
256
+ #
257
+ # The lock is doing real work rather than being belt-and-braces. Without it the fence is
258
+ # read from the statement's snapshot, so a claim committing a microsecond after the
259
+ # statement began would go unseen and a superseded pass's write would land. Taking the row
260
+ # lock makes this statement queue behind any claim in flight and then re-read the row it
261
+ # locked, so the token compared against is the newest one. An `UPDATE` gives that as
262
+ # `SELECT ... FOR UPDATE` would, re-evaluating its `WHERE` against the committed row
263
+ # version and returning what it re-read.
264
+ #
265
+ # The token is in the CTE's `WHERE` rather than only in the insert's, so a refused write
266
+ # renews nothing: a superseded pass's stray writes would otherwise keep the *winner's* claim
267
+ # alive after the winner had died, delaying the takeover that its silence should have
268
+ # brought on. The `LEAST` covers the other stray write, after a `RELEASE`: the budget is
269
+ # already `now()` then, so a write still in flight renews to `now()` and takes nothing back
270
+ # from whoever has claimed the workflow since.
207
271
  #
208
272
  # The rest is `HSETNX` and its read-back, as one upsert. `DO UPDATE SET value = the value
209
273
  # already there` is a write that changes nothing and therefore returns the row that was
@@ -220,10 +284,13 @@ RETURNING token
220
284
  # all when the pass is fenced
221
285
  RECORD = """
222
286
  WITH fence AS (
223
- SELECT token FROM workflow_claim WHERE workflow = %(workflow)s FOR UPDATE
287
+ UPDATE workflow_claim
288
+ SET alive_until = LEAST(now() + alive_for, held_until)
289
+ WHERE workflow = %(workflow)s AND token <= %(token)s
290
+ RETURNING token
224
291
  )
225
292
  INSERT INTO workflow_checkpoint AS recorded (workflow, step, value)
226
- SELECT %(workflow)s, %(step)s, %(value)s::jsonb FROM fence WHERE fence.token <= %(token)s
293
+ SELECT %(workflow)s, %(step)s, %(value)s::jsonb FROM fence
227
294
  ON CONFLICT (workflow, step) DO UPDATE SET value = recorded.value
228
295
  RETURNING recorded.value::text, recorded.value = %(value)s::jsonb
229
296
  """
@@ -268,19 +335,46 @@ RETURNING entry.step, entry.value::text
268
335
  # work in the middle. They are separate strings rather than one because the effect is
269
336
  # arbitrary application SQL that this store cannot see, which is precisely what makes the
270
337
  # transaction worth having.
271
- FENCE = "SELECT token FROM workflow_claim WHERE workflow = %s FOR UPDATE"
338
+ #
339
+ # The fence is read twice, and the split is where the row lock goes. `FENCE` is a plain
340
+ # read before the effect, so a pass already superseded performs nothing; it takes no lock,
341
+ # so the claim row stays free while the effect runs, and the worker's `RENEW` on another
342
+ # connection lands instead of queueing behind the transaction for the whole effect. `WRITE`
343
+ # is the locked re-read *after* the effect, in the statement that renews for the reason
344
+ # `RECORD`'s does: it queues behind any claim in flight and re-evaluates against the
345
+ # committed row, so a pass superseded while its effect ran is refused here and the effect
346
+ # rolls back with the transaction. The lock is then held for one statement's worth rather
347
+ # than for the effect, which is the same window `RECORD` holds it for.
348
+ #
349
+ # `clock_timestamp()` rather than `now()`, and this is the one claim statement that needs
350
+ # it: `now()` is the transaction's start, and a sign of life stamped from before a long
351
+ # effect began could already be in the past by the time it commits, which would say a
352
+ # pass had gone quiet at the moment it was speaking.
353
+ FENCE = "SELECT token FROM workflow_claim WHERE workflow = %s"
272
354
  ALREADY = "SELECT value::text FROM workflow_checkpoint WHERE workflow = %s AND step = %s"
273
- # `ON CONFLICT DO NOTHING` rather than a plain insert, because `supply` is deliberately not
274
- # gated on the claim and so is the one writer this transaction's fence does not exclude. An
275
- # approval landing between the `ALREADY` read and this write would otherwise turn a step
276
- # into a duplicate-key error, which `transact` MUST not answer with: the step is recorded,
277
- # so the contract is to hand back what is recorded. Returning no row says that happened,
278
- # and the caller rolls the effect back rather than committing work whose record belongs to
279
- # somebody else.
355
+ # `ON CONFLICT DO NOTHING` rather than `RECORD`'s upsert, because `supply` is deliberately
356
+ # not gated on the claim and so is the one writer this transaction's fence does not
357
+ # exclude. An approval landing between the `ALREADY` read and this write would otherwise
358
+ # turn a step into a duplicate-key error, which `transact` MUST not answer with: the step
359
+ # is recorded, so the contract is to hand back what is recorded. So the statement reports
360
+ # both halves rather than one row or none: whether the fence held, and what the insert
361
+ # wrote. No fence is `Fenced`; a fence with nothing written says that happened, and the
362
+ # caller rolls the effect back rather than committing work whose record belongs to somebody
363
+ # else.
280
364
  WRITE = """
281
- INSERT INTO workflow_checkpoint (workflow, step, value) VALUES (%s, %s, %s::jsonb)
282
- ON CONFLICT (workflow, step) DO NOTHING
283
- RETURNING value::text
365
+ WITH fence AS (
366
+ UPDATE workflow_claim
367
+ SET alive_until = LEAST(clock_timestamp() + alive_for, held_until)
368
+ WHERE workflow = %(workflow)s AND token <= %(token)s
369
+ RETURNING token
370
+ ),
371
+ written AS (
372
+ INSERT INTO workflow_checkpoint (workflow, step, value)
373
+ SELECT %(workflow)s, %(step)s, %(value)s::jsonb FROM fence
374
+ ON CONFLICT (workflow, step) DO NOTHING
375
+ RETURNING value::text
376
+ )
377
+ SELECT EXISTS (SELECT FROM fence), (SELECT value FROM written)
284
378
  """
285
379
 
286
380
  LOAD = "SELECT step, value::text FROM workflow_checkpoint WHERE workflow = %s ORDER BY seq"
@@ -298,12 +392,23 @@ DISCARD = "DELETE FROM workflow_checkpoint WHERE workflow = %s"
298
392
  # here would be a tombstone for a workflow nobody ever claimed. `held_until = now()` hands
299
393
  # the workflow back at the same time, so it is claimable again immediately: what is kept is
300
394
  # the ordering, not the claim.
301
- SUPERSEDE = "UPDATE workflow_claim SET token = token + 1, held_until = now() WHERE workflow = %s"
395
+ SUPERSEDE = """
396
+ UPDATE workflow_claim SET token = token + 1, held_until = now(), alive_until = now()
397
+ WHERE workflow = %s
398
+ """
302
399
  # Hand the workflow back early, but keep the token, so the next claim gets the next
303
400
  # number up and a pass that comes back from the dead still loses. Conditional on the
304
401
  # token for the same reason `release` is in the Redis store: a superseded pass letting go
305
402
  # must not hand away a claim someone else is holding.
306
- RELEASE = "UPDATE workflow_claim SET held_until = now() WHERE workflow = %s AND token = %s"
403
+ #
404
+ # Both deadlines, and the budget is the load-bearing one: a write this pass had already
405
+ # started is entitled to land (it keeps its token), and bringing `held_until` down to now
406
+ # is what stops that write's own renewal from claiming the workflow straight back, since
407
+ # every renewal is a `LEAST` against it.
408
+ RELEASE = """
409
+ UPDATE workflow_claim SET held_until = now(), alive_until = now()
410
+ WHERE workflow = %s AND token = %s
411
+ """
307
412
 
308
413
  # What an effect is for a store whose datastore is a Postgres database: an async callback
309
414
  # handed a cursor that is already inside `transact`'s transaction. The Redis store's is a
@@ -428,14 +533,29 @@ class PostgresCheckpointer:
428
533
  for step, encoded, written_at in await cursor.fetchall()
429
534
  }
430
535
 
431
- async def claim(self, workflow: str, lease: timedelta) -> Pass | None:
536
+ async def claim(self, workflow: str, budget: timedelta, alive: timedelta) -> Pass | None:
432
537
  async with self.pool.connection() as connection, connection.cursor() as cursor:
433
- await cursor.execute(CLAIM, {"workflow": workflow, "lease": lease})
538
+ await cursor.execute(CLAIM, {"workflow": workflow, "budget": budget, "alive": alive})
434
539
  taken = await cursor.fetchone()
435
540
  if taken is None:
436
541
  return None
437
542
  return Pass(workflow=workflow, token=cast(int, taken[0]))
438
543
 
544
+ async def extend(self, holder: Pass, budget: timedelta, alive: timedelta) -> bool:
545
+ async with self.pool.connection() as connection, connection.cursor() as cursor:
546
+ await cursor.execute(
547
+ EXTEND,
548
+ {"workflow": holder.workflow, "token": holder.token, "budget": budget, "alive": alive},
549
+ )
550
+ # No row touched means the `WHERE` refused the token, which is the only way
551
+ # this misses: a `Pass` exists because a `claim` wrote the row it names.
552
+ return cursor.rowcount == 1
553
+
554
+ async def renew(self, holder: Pass, alive: timedelta) -> bool:
555
+ async with self.pool.connection() as connection, connection.cursor() as cursor:
556
+ await cursor.execute(RENEW, {"workflow": holder.workflow, "token": holder.token, "alive": alive})
557
+ return cursor.rowcount == 1
558
+
439
559
  async def record(self, holder: Pass, key: str, value: object) -> Recorded:
440
560
  async with self.pool.connection() as connection, connection.cursor() as cursor:
441
561
  await cursor.execute(
@@ -475,13 +595,17 @@ class PostgresCheckpointer:
475
595
  which comes back a list) then does so on the first pass rather than surprising the
476
596
  second.
477
597
 
478
- The fence is held for as long as the effect runs, which is what makes it a fence
479
- and is worth stating because of what it costs elsewhere: `claim` takes the same
480
- row, so a worker trying to take this workflow over *waits* for the effect rather
481
- than being told the workflow is held, and each one that waits holds a pool
482
- connection while it does. A long effect and a small pool is a worker that stops
483
- pulling work for unrelated namespaces. Size the pool for the passes a deployment
484
- runs concurrently plus the takeovers it expects, or keep the effects short.
598
+ The claim row is *not* locked while the effect runs, and that is deliberate. The
599
+ fence is read plainly before the effect, so a superseded pass performs nothing, and
600
+ re-read under the row lock after it (`WRITE`), so a pass superseded meanwhile is
601
+ refused and the effect rolls back with the transaction. Holding the lock across the
602
+ effect instead would queue the worker's own renewal behind it, so a long effect
603
+ would run with nothing renewing the delivery and be taken over on commit for having
604
+ gone quiet; it would also make `claim` wait out the effect on a pinned connection
605
+ rather than being told the workflow is held. What keeps another pass out while the
606
+ effect runs is the claim's liveness, renewed by the worker's tick, and what keeps
607
+ the effect from landing twice if that lapses is the step row itself: a second
608
+ transaction's `WRITE` conflicts with the first's and rolls its effect back.
485
609
 
486
610
  The fence excludes every other *pass*, and one writer is left over: `supply` is
487
611
  ungated on purpose, so an approval can land under this key between the read and the
@@ -501,11 +625,17 @@ class PostgresCheckpointer:
501
625
  recorded = await cursor.fetchone()
502
626
  if recorded is not None:
503
627
  return self.codec.decode(cast(str, recorded[0]))
504
- await cursor.execute(WRITE, (holder.workflow, key, self.codec.encode(await effect(cursor))))
505
- written = await cursor.fetchone()
628
+ encoded = self.codec.encode(await effect(cursor))
629
+ await cursor.execute(
630
+ WRITE,
631
+ {"workflow": holder.workflow, "step": key, "value": encoded, "token": holder.token},
632
+ )
633
+ held, written = cast(tuple[bool, str | None], await cursor.fetchone())
634
+ if not held:
635
+ raise Fenced(f"{holder.workflow!r} moved on while this pass held it")
506
636
  if written is None:
507
637
  raise Supplied
508
- return self.codec.decode(cast(str, written[0]))
638
+ return self.codec.decode(written)
509
639
  except Supplied:
510
640
  pass
511
641
  async with self.pool.connection() as connection, connection.cursor() as cursor:
@@ -633,6 +763,23 @@ UPDATE workflow_queue SET visible_at = %(when)s
633
763
  WHERE namespace = %(namespace)s AND workflow = %(workflow)s AND visible_at = %(receipt)s
634
764
  """
635
765
 
766
+ # Keep a delivery this worker's for another `within`, under the same comparison as
767
+ # `SUSPEND` and for the same reason: a `make_ready` that landed since would have written a
768
+ # different `visible_at`, and pushing the visibility out on top of it would bury a wakeup
769
+ # that has already arrived.
770
+ #
771
+ # It returns the new visibility because that *is* the new receipt, which is the price of
772
+ # the trick that makes this table a queue. A worker that went on holding the old one would
773
+ # find its own `FINISH` refused by the equality above and the workflow redelivered for
774
+ # nothing, so the rename is reported rather than left to be discovered.
775
+ #
776
+ # returns the new receipt, or no row when this delivery is no longer this worker's
777
+ RENEW_DELIVERY = """
778
+ UPDATE workflow_queue SET visible_at = now() + %(within)s
779
+ WHERE namespace = %(namespace)s AND workflow = %(workflow)s AND visible_at = %(receipt)s
780
+ RETURNING visible_at
781
+ """
782
+
636
783
 
637
784
  @dataclass(frozen=True, slots=True)
638
785
  class PostgresScheduler:
@@ -759,6 +906,35 @@ class PostgresScheduler:
759
906
  """Nothing to take over by hand: an abandoned workflow becomes visible on its own."""
760
907
  return None
761
908
 
909
+ async def extend(self, delivery: Delivery, within: timedelta) -> Delivery:
910
+ """
911
+ Push this delivery's visibility out, and say what it is called now.
912
+
913
+ The new visibility is the new receipt, since this table's receipt *is* its
914
+ visibility, so the caller is handed a delivery to use from here rather than left to
915
+ work out that the one it holds has been renamed.
916
+
917
+ No row means this delivery is no longer this worker's: taken over, cancelled, or
918
+ rescheduled by a wakeup that arrived mid-pass, all of which wrote a `visible_at`
919
+ that is not the one it took. The answer to every one of them is to hand back what
920
+ came in and let whoever now owns the row have it, exactly as `wake_at` and `done`
921
+ already do.
922
+ """
923
+ async with self.pool.connection() as connection, connection.cursor() as cursor:
924
+ await cursor.execute(
925
+ RENEW_DELIVERY,
926
+ {
927
+ "namespace": self.namespace,
928
+ "workflow": delivery.workflow,
929
+ "receipt": datetime.fromisoformat(delivery.receipt),
930
+ "within": within,
931
+ },
932
+ )
933
+ renewed = await cursor.fetchone()
934
+ if renewed is None:
935
+ return delivery
936
+ return Delivery(workflow=delivery.workflow, receipt=cast(datetime, renewed[0]).isoformat())
937
+
762
938
  async def cancel(self, workflow: str) -> None:
763
939
  """
764
940
  Drop the workflow's row, whichever of the three things its `visible_at` means.