cairnq 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +11 -5
  2. package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
  3. package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
  4. package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
  5. package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
  6. package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
  7. package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
  8. package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
  9. package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
  10. package/dist/backoff.d.ts +31 -0
  11. package/dist/backoff.js +40 -0
  12. package/dist/client.d.ts +28 -2
  13. package/dist/client.js +30 -3
  14. package/dist/context.d.ts +48 -2
  15. package/dist/context.js +101 -10
  16. package/dist/errors.d.ts +60 -4
  17. package/dist/errors.js +94 -9
  18. package/dist/index.d.ts +6 -2
  19. package/dist/index.js +2 -1
  20. package/dist/retention.d.ts +60 -0
  21. package/dist/retention.js +115 -0
  22. package/dist/store/base.d.ts +50 -1
  23. package/dist/store/base.js +114 -19
  24. package/dist/store/sqlite.js +4 -1
  25. package/dist/wait.d.ts +20 -5
  26. package/dist/wait.js +34 -9
  27. package/dist/worker.d.ts +214 -16
  28. package/dist/worker.js +500 -131
  29. package/package.json +1 -1
  30. package/src/backoff.ts +53 -0
  31. package/src/client.ts +43 -5
  32. package/src/context.ts +116 -9
  33. package/src/errors.ts +101 -9
  34. package/src/index.ts +6 -1
  35. package/src/retention.ts +136 -0
  36. package/src/store/base.ts +121 -17
  37. package/src/store/sqlite.ts +4 -1
  38. package/src/wait.ts +60 -16
  39. package/src/worker.ts +640 -146
package/src/store/base.ts CHANGED
@@ -6,7 +6,7 @@ import {
6
6
  ProtocolVersionMismatch,
7
7
  SerializationError,
8
8
  } from "../errors.js";
9
- import { rowToTask, STATUSES, type Task, type TaskStatus } from "../models.js";
9
+ import { rowToTask, STATUSES, TERMINAL, type Task, type TaskStatus } from "../models.js";
10
10
  import { type BackpressureOptions, QueueDepthGate } from "../backpressure.js";
11
11
 
12
12
  const rejectMangled = function (this: unknown, _key: string, v: unknown): unknown {
@@ -62,9 +62,25 @@ export function checkProtocolVersion(version: number): void {
62
62
  // CONFLICTS is the canonical declaration; the type derives from it so the
63
63
  // runtime guard in submit() and the type can't drift apart (same pattern as
64
64
  // STATUSES/TaskStatus in models.ts).
65
- const CONFLICTS = ["reuse", "reject", "replace"] as const;
65
+ const CONFLICTS = ["reuse", "reuse-succeeded", "reject", "replace"] as const;
66
66
  export type Conflict = (typeof CONFLICTS)[number];
67
67
 
68
+ /**
69
+ * Whether a keyed submit's strategy accepts the task the key already points at.
70
+ *
71
+ * Both reuse strategies deduplicate work that is still in play — that is what a
72
+ * key is for, and the answer cannot depend on the outcome of a task that has no
73
+ * outcome yet. They differ only on what a *finished* task means: `reuse` treats
74
+ * the key as free again, while `reuse-succeeded` reads a succeeded task as a
75
+ * cached result. Neither ever hands back a failed or canceled one, which would
76
+ * poison the key for every later submit (see PROTOCOL.md "Key conflict").
77
+ */
78
+ function reusable(conflict: Conflict, status: TaskStatus): boolean {
79
+ if (conflict === "replace") return false;
80
+ if (!TERMINAL.includes(status)) return true;
81
+ return conflict === "reuse-succeeded" && status === "succeeded";
82
+ }
83
+
68
84
  /** The queue a submit lands on when it names none. Owned here, where the
69
85
  * default is applied, so nothing above has to re-derive it. */
70
86
  export const DEFAULT_QUEUE = "default";
@@ -283,10 +299,15 @@ export abstract class TaskStore {
283
299
  // free after all, whatever the strategy.
284
300
  const current = (await fetch("get", { id: existing[0].task_id }))[0];
285
301
  if (current) {
286
- if (conflict === "reuse") return rowToTask(current);
287
302
  if (conflict === "reject") throw new AlreadyExists(key);
288
- // "replace": cancel the recorded task, then repoint the key below.
289
- await fetch("cancel", { id: existing[0].task_id });
303
+ if (reusable(conflict, current.status as TaskStatus)) return rowToTask(current);
304
+ // The strategy declined the recorded task, so the key repoints to the
305
+ // fresh one inserted below. Cancel only what is still live: a terminal
306
+ // task has nothing to stop, and cancelling it would rewrite a settled
307
+ // row (and hand a `canceled` back to whoever is waiting on it).
308
+ if (!TERMINAL.includes(current.status as TaskStatus)) {
309
+ await fetch("cancel", { id: existing[0].task_id });
310
+ }
290
311
  }
291
312
  }
292
313
  const row = (await fetch("insert_task", ins))[0];
@@ -425,26 +446,82 @@ export abstract class TaskStore {
425
446
  limit?: number;
426
447
  names?: string[];
427
448
  }): Promise<Task[]> {
428
- // One queue is the common case and gets its own statement: a list-valued queue
429
- // filter cannot be read in claim order, so the planner sorts every claimable
430
- // row to take LIMIT of them, and claim's cost grows with the queued backlog
431
- // while it holds the claim transaction. See claim_one_queue.sql.
449
+ const names = input.names ?? null;
450
+ const claimed = await this.claimSession(
451
+ { queues: input.queues, workerId: input.workerId, leaseMs: input.leaseMs, names },
452
+ (claim) => claim(names, input.limit ?? 1),
453
+ );
454
+ return claimed ?? [];
455
+ }
456
+
457
+ /**
458
+ * Open one claim transaction and let the caller draw from it repeatedly.
459
+ *
460
+ * The transaction is what has to live here: the read-only probe that keeps an
461
+ * idle worker off SQLite's single write lock, the `recover_leases` whose
462
+ * reclaimed leases must be visible to the claims that follow and to nobody in
463
+ * between, and the write lock itself. *What* gets claimed under it is the
464
+ * caller's business — a worker drawing a separate quota per task name is
465
+ * scheduling policy, and this layer has no vocabulary for the "handler call"
466
+ * that policy is denominated in. It knows queues, names, limits and rows.
467
+ *
468
+ * `plan` is handed a `claim(names, limit)` it may call any number of times,
469
+ * each a separate statement under the same lock and the same recovery, and
470
+ * each free to size itself from what the previous one returned. That feedback
471
+ * is the reason this is a callback rather than a list of quotas: a caller
472
+ * dividing a budget up front has to guess, and every share handed to a name
473
+ * with nothing queued is a slot left idle until the next poll.
474
+ *
475
+ * `plan` runs with the write lock held, so it must await nothing but that
476
+ * callback.
477
+ *
478
+ * `names` is the union `plan` might ask for — the probe and the recovery are
479
+ * filtered by it. Returns undefined when the probe finds nothing claimable, in
480
+ * which case `plan` never runs and no transaction is opened.
481
+ */
482
+ async claimSession<T>(
483
+ input: { queues: string[]; workerId: string; leaseMs?: number; names: string[] | null },
484
+ plan: (claim: (names: string[] | null, limit: number) => Promise<Task[]>) => Promise<T>,
485
+ ): Promise<T | undefined> {
486
+ // A list-valued filter cannot be read in claim order, so the planner sorts
487
+ // every claimable row to take LIMIT of them and the claim's cost grows with
488
+ // the backlog while it holds the transaction. Both filters therefore have an
489
+ // equality form, picked per draw: one queue is the common deployment, and one
490
+ // name is every per-name quota. See claim_one_queue.sql and claim_one_name.sql.
432
491
  const oneQueue = input.queues.length === 1;
433
- const params: Params = {
492
+ const base: Params = {
434
493
  queues: input.queues,
435
494
  queue: oneQueue ? input.queues[0] : null,
436
- names: input.names ?? null,
495
+ names: input.names,
496
+ name: null,
437
497
  worker_id: input.workerId,
438
498
  lease_ms: input.leaseMs ?? 30_000,
439
- limit: input.limit ?? 1,
499
+ limit: 1,
440
500
  lease_expired_error: LEASE_EXPIRED_ERROR_JSON,
441
501
  };
442
- if (!(await this.hasClaimableWork(params))) return [];
443
- // Recovery must share the claim's transaction: a lease reclaimed here has to
444
- // be visible to the claim that follows, and to nobody in between.
502
+ if (!(await this.hasClaimableWork(base))) return undefined;
445
503
  return this.tx(async (fetch) => {
446
- await fetch("recover_leases", params);
447
- return (await fetch(oneQueue ? "claim_one_queue" : "claim", params)).map(rowToTask);
504
+ await fetch("recover_leases", base);
505
+ return plan(async (names, limit) => {
506
+ // A draw asking for nothing, or filtered to no names, claims nothing —
507
+ // answer it here rather than spending a statement to learn that.
508
+ if (limit <= 0 || names?.length === 0) return [];
509
+ const oneName = names?.length === 1;
510
+ const statement = oneName
511
+ ? oneQueue
512
+ ? "claim_one_queue_one_name"
513
+ : "claim_one_name"
514
+ : oneQueue
515
+ ? "claim_one_queue"
516
+ : "claim";
517
+ const rows = await fetch(statement, {
518
+ ...base,
519
+ names,
520
+ name: oneName ? names![0] : null,
521
+ limit,
522
+ });
523
+ return rows.map(rowToTask);
524
+ });
448
525
  });
449
526
  }
450
527
 
@@ -456,6 +533,33 @@ export abstract class TaskStore {
456
533
  });
457
534
  }
458
535
 
536
+ /**
537
+ * Renew several leases in one statement. Returns `taskId -> cancel requested`
538
+ * for the tasks this worker still holds.
539
+ *
540
+ * Deliberately not an ownedWrite: ownership is per task here, so there is no
541
+ * single answer to "did it work". A task **absent** from the result lost its
542
+ * lease, and the caller decides what that means for that one task rather than
543
+ * failing the whole beat.
544
+ *
545
+ * It returns flags rather than Tasks because nothing downstream needs a task:
546
+ * the caller renews leases and observes cancellation, and whole rows would drag
547
+ * every payload back on every beat for the life of the call.
548
+ */
549
+ async heartbeatBatch(input: {
550
+ taskIds: string[];
551
+ workerId: string;
552
+ leaseMs?: number;
553
+ }): Promise<Map<string, boolean>> {
554
+ if (!input.taskIds.length) return new Map();
555
+ const rows = await this.fetch("heartbeat_batch", {
556
+ ids: input.taskIds,
557
+ worker_id: input.workerId,
558
+ lease_ms: input.leaseMs ?? 30_000,
559
+ });
560
+ return new Map(rows.map((r) => [r.id as string, r.cancel_requested_at_ms != null]));
561
+ }
562
+
459
563
  async progress(input: {
460
564
  taskId: string;
461
565
  workerId: string;
@@ -357,7 +357,10 @@ export class SQLiteStore extends TaskStore {
357
357
  bound[name] = now - (params.older_than_ms as number);
358
358
  break;
359
359
  case "queues":
360
- bound[name] = JSON.stringify(params.queues);
360
+ case "ids":
361
+ // json_each needs a JSON array. Postgres binds the array itself as
362
+ // text[], so only this dialect encodes.
363
+ bound[name] = JSON.stringify(params[name]);
361
364
  break;
362
365
  case "names":
363
366
  // json_each needs a JSON array; null stays null so the SQL's
package/src/wait.ts CHANGED
@@ -7,6 +7,14 @@ export const DEFAULT_POLL_MS = 100;
7
7
  export const MAX_POLL_MS = 500;
8
8
  const GROWTH = 1.5;
9
9
 
10
+ const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
11
+
12
+ export interface PollOptions {
13
+ timeoutMs: number;
14
+ pollMs?: number;
15
+ maxPollMs?: number;
16
+ }
17
+
10
18
  /**
11
19
  * Grow the polling interval towards the ceiling.
12
20
  *
@@ -19,28 +27,64 @@ export function nextPollMs(current: number, maxMs: number): number {
19
27
  return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
20
28
  }
21
29
 
22
- /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
23
- * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
24
- * it backs off towards `maxPollMs`. */
25
- export async function pollWait(
26
- store: TaskStore,
27
- taskId: string,
28
- {
29
- timeoutMs,
30
- pollMs = DEFAULT_POLL_MS,
31
- maxPollMs = MAX_POLL_MS,
32
- }: { timeoutMs: number; pollMs?: number; maxPollMs?: number },
30
+ /**
31
+ * Poll `read` until it yields a terminal task, or the timeout elapses.
32
+ *
33
+ * `wake` is what the loop sleeps on between reads: a store with a push channel
34
+ * (Postgres) cuts it short when the task goes terminal, but the re-read is the
35
+ * source of truth either way, so a plain sleep is always a correct answer.
36
+ */
37
+ async function poll(
38
+ read: () => Promise<Task | null>,
39
+ wake: (task: Task | null, ms: number) => Promise<void>,
40
+ subject: string,
41
+ key: string | null,
42
+ { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS }: PollOptions,
33
43
  ): Promise<Task> {
34
44
  const deadline = nowMs() + timeoutMs;
35
45
  let interval = pollMs;
36
46
  for (;;) {
37
- const task = await store.get(taskId);
47
+ const task = await read();
38
48
  if (task && isTerminal(task)) return task;
39
49
  const remaining = deadline - nowMs();
40
- if (remaining <= 0) throw new TaskTimeout(taskId, { timeoutMs, task });
41
- // A store with a push channel (Postgres) cuts the sleep short when the task
42
- // goes terminal; the re-get above stays the source of truth either way.
43
- await store.taskDoneWake(taskId, Math.min(interval, remaining));
50
+ if (remaining <= 0) throw new TaskTimeout(task?.id ?? subject, { timeoutMs, task, key });
51
+ await wake(task, Math.min(interval, remaining));
44
52
  interval = nextPollMs(interval, maxPollMs);
45
53
  }
46
54
  }
55
+
56
+ /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
57
+ * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
58
+ * it backs off towards `maxPollMs`. */
59
+ export function pollWait(store: TaskStore, taskId: string, opts: PollOptions): Promise<Task> {
60
+ return poll(
61
+ () => store.get(taskId),
62
+ (_task, ms) => store.taskDoneWake(taskId, ms),
63
+ taskId,
64
+ null,
65
+ opts,
66
+ );
67
+ }
68
+
69
+ /**
70
+ * The same wait, following a key instead of an id.
71
+ *
72
+ * The key is re-resolved on every read, because that is what a key means: a
73
+ * pointer to the task that is *current* under it. A `replace` landing mid-wait
74
+ * moves the wait onto the new task rather than reporting the cancellation of the
75
+ * old one, and a key that points at nothing yet is simply not finished — it
76
+ * polls until something appears, the same way waiting on an id that does not
77
+ * exist yet does.
78
+ *
79
+ * There is nothing to subscribe to before the key resolves, so those naps are
80
+ * plain sleeps; once it resolves, the store's push channel applies as usual.
81
+ */
82
+ export function pollWaitByKey(store: TaskStore, key: string, opts: PollOptions): Promise<Task> {
83
+ return poll(
84
+ () => store.getByKey(key),
85
+ (task, ms) => (task ? store.taskDoneWake(task.id, ms) : sleep(ms)),
86
+ key,
87
+ key,
88
+ opts,
89
+ );
90
+ }