cairnq 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "cairnq",
3
- "version": "0.5.0",
3
+ "version": "0.6.0",
4
4
  "description": "SQLite-first, cross-language, storage-centered durable task runtime",
5
5
  "license": "MIT",
6
6
  "author": "Jannchie <jannchie@gmail.com>",
package/src/backoff.ts ADDED
@@ -0,0 +1,53 @@
1
+ /**
2
+ * Retry backoff, in its own module because two callers need it.
3
+ *
4
+ * The worker computes it when a handler's failure ends an attempt; TaskContext
5
+ * computes it when a handler fails one task of a batch itself. Keeping it in
6
+ * worker.ts would make context.ts import the module that imports it.
7
+ */
8
+
9
+ export const DEFAULT_RETRY_BACKOFF_MS = 1_000;
10
+ export const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
11
+
12
+ /**
13
+ * Exponential backoff with equal jitter: the window doubles per attempt up to
14
+ * `maxMs`, and the delay lands uniformly in its upper half, `[w/2, w)`.
15
+ *
16
+ * The jitter is what keeps a fleet from retrying in lockstep. Failures align
17
+ * when the downstream fails fast enough that a whole concurrency batch raises
18
+ * at once (connection refused, DNS gone), and capped exponential backoff then
19
+ * *preserves* that alignment — once every task sits at `maxMs`, they all retry
20
+ * on the same beat forever. Spreading over half the window breaks it; keeping
21
+ * the lower half as a floor means jitter never shortens the wait to less than
22
+ * half of what plain exponential backoff would have asked for.
23
+ *
24
+ * `rand` is injected so tests can pin an exact delay.
25
+ */
26
+ export function retryDelayMs(
27
+ attempt: number,
28
+ baseMs: number,
29
+ maxMs: number,
30
+ rand: () => number = Math.random,
31
+ ): number {
32
+ if (baseMs <= 0) return 0;
33
+ const exponent = Math.max(0, attempt - 1);
34
+ const window = Math.min(maxMs, baseMs * 2 ** exponent);
35
+ const floor = Math.floor(window / 2);
36
+ return floor + Math.floor(rand() * (window - floor));
37
+ }
38
+
39
+ /**
40
+ * The delay a `fail` write should carry. Not just the backoff: a permanent
41
+ * failure is never re-run, so it always delays 0. Both settlement paths — the
42
+ * worker's and a handler's `ctx.fail` — go through this, so they cannot end up
43
+ * backing off differently.
44
+ */
45
+ export function failDelayMs(
46
+ attempt: number,
47
+ retryable: boolean,
48
+ baseMs: number,
49
+ maxMs: number,
50
+ rand: () => number = Math.random,
51
+ ): number {
52
+ return retryable ? retryDelayMs(attempt, baseMs, maxMs, rand) : 0;
53
+ }
package/src/context.ts CHANGED
@@ -1,24 +1,48 @@
1
- import { LostLease } from "./errors.js";
1
+ import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
2
+ import { asEnvelope, type FailReason, LostLease } from "./errors.js";
2
3
  import { cancelRequested, type Task } from "./models.js";
3
4
  import type { SubmitOptions } from "./client.js";
4
5
  import type { TaskStore } from "./store/base.js";
5
6
  import { type TaskDef, taskName } from "./task.js";
6
7
  import { pollWait } from "./wait.js";
7
8
 
8
- /** Handed to a task handler. Worker-side capabilities mirror the Python SDK. */
9
+ export interface TaskContextOptions {
10
+ retryBackoffMs?: number;
11
+ retryBackoffMaxMs?: number;
12
+ }
13
+
14
+ /**
15
+ * Handed to a task handler. Worker-side capabilities mirror the Python SDK.
16
+ *
17
+ * One of these per task, whether a handler is delivered one task or a batch: a
18
+ * batch handler receives a `TaskContext[]`, so a single-task handler's `ctx` is
19
+ * literally the batch-of-one element. Lease, cancellation and settlement are per
20
+ * task, which is why they live here rather than on anything batch-shaped.
21
+ */
9
22
  export class TaskContext {
10
23
  private readonly abort = new AbortController();
11
24
  private leaseLost = false;
12
25
  // Cancellation is monotonic: once the DB has told us a cancel was requested it
13
26
  // can't be taken back, so canceled() can answer from this without a re-read.
14
27
  private cancelSeen = false;
28
+ // Set once this task reached a terminal state through succeed()/fail(). The
29
+ // worker reads it to know which tasks a batch handler already decided, so it
30
+ // neither settles them twice nor keeps renewing their leases — the bookkeeping
31
+ // every ack/nack-style handler otherwise has to carry itself.
32
+ private isSettled = false;
33
+ private readonly backoffMs: number;
34
+ private readonly backoffMaxMs: number;
15
35
 
16
36
  constructor(
17
37
  private readonly store: TaskStore,
18
38
  private readonly task: Task,
19
39
  public readonly workerId: string,
20
40
  private readonly leaseMs: number,
21
- ) {}
41
+ opts: TaskContextOptions = {},
42
+ ) {
43
+ this.backoffMs = opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
44
+ this.backoffMaxMs = opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS;
45
+ }
22
46
 
23
47
  get taskId(): string {
24
48
  return this.task.id;
@@ -44,6 +68,19 @@ export class TaskContext {
44
68
  get payload(): any {
45
69
  return this.task.payload;
46
70
  }
71
+ /**
72
+ * True once this task reached a terminal state — whether the handler settled
73
+ * it with succeed()/fail() or the worker settled it on the handler's behalf.
74
+ * The heartbeat and the settlement paths both read it.
75
+ */
76
+ get settled(): boolean {
77
+ return this.isSettled;
78
+ }
79
+
80
+ /** @internal Called by the worker when it finalizes this task itself. */
81
+ markSettled(): void {
82
+ this.isSettled = true;
83
+ }
47
84
 
48
85
  /**
49
86
  * True once this worker has lost the task's lease — it expired and another
@@ -70,17 +107,36 @@ export class TaskContext {
70
107
  // Every owned write returns the current row, so cancellation and lease loss
71
108
  // ride along on writes the handler was making anyway.
72
109
  private observe(task: Task): Task {
73
- if (cancelRequested(task)) this.cancelSeen = true;
110
+ this.observeCancel(cancelRequested(task));
74
111
  return task;
75
112
  }
76
113
 
114
+ /**
115
+ * @internal The same observation from just the flag, for a caller that read it
116
+ * without materializing a Task — the shared heartbeat, whose statement returns
117
+ * only the id and the cancel column precisely so it does not have to drag
118
+ * every payload back on every beat.
119
+ */
120
+ observeCancel(cancelRequested: boolean): void {
121
+ if (cancelRequested) this.cancelSeen = true;
122
+ }
123
+
77
124
  private async owned(write: () => Promise<Task>): Promise<Task> {
78
- // Short-circuit once the lease is known lost: nothing this context writes
79
- // may be recorded any more. Locally, not just via the store's ownership
80
- // check — after an abandoned (timed-out) attempt the same worker may
81
- // re-claim this task under the same workerId, and a zombie handler's write
82
- // would then pass ownership against the NEW attempt.
125
+ // One gate for every write through this context, so "may I still write?" is
126
+ // answered in one place rather than at each call site.
127
+ //
128
+ // Lease lost: nothing this context writes may be recorded any more. Checked
129
+ // locally, not just via the store's ownership check — after an abandoned
130
+ // (timed-out) attempt the same worker may re-claim this task under the same
131
+ // workerId, and a zombie handler's write would then pass ownership against
132
+ // the NEW attempt.
83
133
  if (this.leaseLost) throw new LostLease(this.task.id);
134
+ // Settled: the task is terminal, so the statement would match no row and come
135
+ // back as a lost lease — telling the handler "another worker took this" when
136
+ // the truth is "you already finished it", and flipping lostLease on the way.
137
+ // Refuse here instead, without the round trip and without corrupting the
138
+ // lease state.
139
+ if (this.isSettled) throw new LostLease(this.task.id);
84
140
  try {
85
141
  return this.observe(await write());
86
142
  } catch (err) {
@@ -119,6 +175,57 @@ export class TaskContext {
119
175
  return this.cancelSeen || t.status === "canceled";
120
176
  }
121
177
 
178
+ // ------------------------------------------------------------- settlement
179
+ // Finalizing a task is normally the worker's job, decided by whether the
180
+ // handler returned or threw. These two let a handler decide one task itself,
181
+ // which is what a batch needs: four of 256 tasks failing for four different
182
+ // reasons is the ordinary case, not the edge one, and it cannot be expressed
183
+ // by a single return value or a single throw.
184
+ //
185
+ // Settling twice is a no-op rather than an error. Handlers built on ack/nack
186
+ // queues all end up carrying a `finalizedIds` set to guarantee exactly that;
187
+ // holding it here instead is the point.
188
+
189
+ /**
190
+ * Finalize this task as succeeded, now, without waiting for the handler to
191
+ * return. `complete` semantics: a cancel requested while it ran wins and the
192
+ * task finalizes as canceled instead, its result discarded. Returns null if
193
+ * this task was already settled.
194
+ */
195
+ async succeed(result: unknown = null): Promise<Task | null> {
196
+ if (this.isSettled) return null;
197
+ const task = await this.owned(() =>
198
+ this.store.complete({ taskId: this.task.id, workerId: this.workerId, result }),
199
+ );
200
+ this.markSettled();
201
+ return task;
202
+ }
203
+
204
+ /**
205
+ * Finalize this task as failed, now. `error` may be a string reason, an Error,
206
+ * a TaskError (which carries its own retryability), or a ready envelope.
207
+ * Retryable failures get the worker's backoff and are re-queued while attempts
208
+ * remain, exactly as a thrown error would be. Returns null if already settled.
209
+ */
210
+ async fail(
211
+ error: FailReason = "task failed",
212
+ opts: { retryable?: boolean } = {},
213
+ ): Promise<Task | null> {
214
+ if (this.isSettled) return null;
215
+ const [envelope, retryable] = asEnvelope(error, opts.retryable ?? true);
216
+ const task = await this.owned(() =>
217
+ this.store.fail({
218
+ taskId: this.task.id,
219
+ workerId: this.workerId,
220
+ error: envelope,
221
+ retryable,
222
+ delayMs: failDelayMs(this.task.attempt, retryable, this.backoffMs, this.backoffMaxMs),
223
+ }),
224
+ );
225
+ this.markSettled();
226
+ return task;
227
+ }
228
+
122
229
  /** Submit a child task; parent/root/correlation are wired automatically. */
123
230
  submit(name: string, payload?: unknown, opts?: SubmitOptions): Promise<Task>;
124
231
  submit<P, R>(task: TaskDef<P, R>, payload?: P, opts?: SubmitOptions): Promise<Task>;
package/src/errors.ts CHANGED
@@ -20,6 +20,24 @@ export function errorEnvelope(e: {
20
20
  };
21
21
  }
22
22
 
23
+ /**
24
+ * How an arbitrary thrown value becomes an envelope. Split out from `asEnvelope`
25
+ * below because the worker also reaches it directly, for a thrown plain object —
26
+ * which `asEnvelope` reads as a ready envelope, the right call for `ctx.fail` and
27
+ * the wrong one for something that was thrown. Both must agree on `code` and on
28
+ * deriving `type` from the error's name, or the same error reads differently
29
+ * depending on which way it was recorded.
30
+ */
31
+ export function exceptionEnvelope(err: unknown, retryable = true): Record<string, unknown> {
32
+ const e = err as { name?: string; message?: string };
33
+ return errorEnvelope({
34
+ type: e?.name ?? "Error",
35
+ code: "handler_error",
36
+ message: String(e?.message ?? err),
37
+ retryable,
38
+ });
39
+ }
40
+
23
41
  export class CairnQError extends Error {
24
42
  constructor(message?: string) {
25
43
  super(message);
@@ -183,3 +201,33 @@ export class TaskError extends CairnQError {
183
201
  });
184
202
  }
185
203
  }
204
+
205
+ /** What a handler may pass to `ctx.fail`. */
206
+ export type FailReason = string | Error | TaskError | Record<string, unknown>;
207
+
208
+ /**
209
+ * Normalize anything that can end a task into [envelope, retryable].
210
+ *
211
+ * Shared by both ways a failure is recorded — a handler passing a reason to
212
+ * `ctx.fail`, and the worker classifying an error that ended an attempt — so the
213
+ * two cannot disagree about what a given error means. It lives here, beside the
214
+ * envelope constructors it dispatches to, rather than in the module that happens
215
+ * to expose it to handlers.
216
+ *
217
+ * A handler failing one task of a batch has a reason, not an exception object:
218
+ * `item.fail("no source records", { retryable: false })` is the shape the real
219
+ * code wants. A TaskError carries its own retryability and wins over the option;
220
+ * everything else takes the caller's. A ready envelope passes through, which is
221
+ * how the worker hands in the ones it composes itself.
222
+ */
223
+ export function asEnvelope(
224
+ error: FailReason,
225
+ retryable: boolean,
226
+ ): [Record<string, unknown>, boolean] {
227
+ if (error instanceof TaskError) return [error.envelope(), error.retryable];
228
+ if (error instanceof Error) return [exceptionEnvelope(error, retryable), retryable];
229
+ if (typeof error === "object" && error !== null) return [error, retryable];
230
+ // A bare reason is a TaskError in everything but the throwing, so let
231
+ // TaskError own its own type/code defaults rather than restating them.
232
+ return [new TaskError(String(error), { retryable }).envelope(), retryable];
233
+ }
package/src/index.ts CHANGED
@@ -3,8 +3,9 @@ export type { CallOptions, ClientOptions, SubmitOptions } from "./client.js";
3
3
  export { QueueDepthGate } from "./backpressure.js";
4
4
  export type { BackpressureOptions, QueueDepthLimit } from "./backpressure.js";
5
5
  export { Worker } from "./worker.js";
6
- export type { Handler, TypedHandler, WorkerOptions } from "./worker.js";
6
+ export type { BatchHandler, Handler, TypedHandler, WorkerOptions } from "./worker.js";
7
7
  export { TaskContext } from "./context.js";
8
+ export type { TaskContextOptions } from "./context.js";
8
9
  export { defineTask } from "./task.js";
9
10
  export type { TaskDef } from "./task.js";
10
11
  export { SQLiteStore } from "./store/sqlite.js";
@@ -34,3 +35,4 @@ export {
34
35
  ProtocolVersionMismatch,
35
36
  SerializationError,
36
37
  } from "./errors.js";
38
+ export type { FailReason } from "./errors.js";
package/src/store/base.ts CHANGED
@@ -425,26 +425,82 @@ export abstract class TaskStore {
425
425
  limit?: number;
426
426
  names?: string[];
427
427
  }): Promise<Task[]> {
428
- // One queue is the common case and gets its own statement: a list-valued queue
429
- // filter cannot be read in claim order, so the planner sorts every claimable
430
- // row to take LIMIT of them, and claim's cost grows with the queued backlog
431
- // while it holds the claim transaction. See claim_one_queue.sql.
428
+ const names = input.names ?? null;
429
+ const claimed = await this.claimSession(
430
+ { queues: input.queues, workerId: input.workerId, leaseMs: input.leaseMs, names },
431
+ (claim) => claim(names, input.limit ?? 1),
432
+ );
433
+ return claimed ?? [];
434
+ }
435
+
436
+ /**
437
+ * Open one claim transaction and let the caller draw from it repeatedly.
438
+ *
439
+ * The transaction is what has to live here: the read-only probe that keeps an
440
+ * idle worker off SQLite's single write lock, the `recover_leases` whose
441
+ * reclaimed leases must be visible to the claims that follow and to nobody in
442
+ * between, and the write lock itself. *What* gets claimed under it is the
443
+ * caller's business — a worker drawing a separate quota per task name is
444
+ * scheduling policy, and this layer has no vocabulary for the "handler call"
445
+ * that policy is denominated in. It knows queues, names, limits and rows.
446
+ *
447
+ * `plan` is handed a `claim(names, limit)` it may call any number of times,
448
+ * each a separate statement under the same lock and the same recovery, and
449
+ * each free to size itself from what the previous one returned. That feedback
450
+ * is the reason this is a callback rather than a list of quotas: a caller
451
+ * dividing a budget up front has to guess, and every share handed to a name
452
+ * with nothing queued is a slot left idle until the next poll.
453
+ *
454
+ * `plan` runs with the write lock held, so it must await nothing but that
455
+ * callback.
456
+ *
457
+ * `names` is the union `plan` might ask for — the probe and the recovery are
458
+ * filtered by it. Returns undefined when the probe finds nothing claimable, in
459
+ * which case `plan` never runs and no transaction is opened.
460
+ */
461
+ async claimSession<T>(
462
+ input: { queues: string[]; workerId: string; leaseMs?: number; names: string[] | null },
463
+ plan: (claim: (names: string[] | null, limit: number) => Promise<Task[]>) => Promise<T>,
464
+ ): Promise<T | undefined> {
465
+ // A list-valued filter cannot be read in claim order, so the planner sorts
466
+ // every claimable row to take LIMIT of them and the claim's cost grows with
467
+ // the backlog while it holds the transaction. Both filters therefore have an
468
+ // equality form, picked per draw: one queue is the common deployment, and one
469
+ // name is every per-name quota. See claim_one_queue.sql and claim_one_name.sql.
432
470
  const oneQueue = input.queues.length === 1;
433
- const params: Params = {
471
+ const base: Params = {
434
472
  queues: input.queues,
435
473
  queue: oneQueue ? input.queues[0] : null,
436
- names: input.names ?? null,
474
+ names: input.names,
475
+ name: null,
437
476
  worker_id: input.workerId,
438
477
  lease_ms: input.leaseMs ?? 30_000,
439
- limit: input.limit ?? 1,
478
+ limit: 1,
440
479
  lease_expired_error: LEASE_EXPIRED_ERROR_JSON,
441
480
  };
442
- if (!(await this.hasClaimableWork(params))) return [];
443
- // Recovery must share the claim's transaction: a lease reclaimed here has to
444
- // be visible to the claim that follows, and to nobody in between.
481
+ if (!(await this.hasClaimableWork(base))) return undefined;
445
482
  return this.tx(async (fetch) => {
446
- await fetch("recover_leases", params);
447
- return (await fetch(oneQueue ? "claim_one_queue" : "claim", params)).map(rowToTask);
483
+ await fetch("recover_leases", base);
484
+ return plan(async (names, limit) => {
485
+ // A draw asking for nothing, or filtered to no names, claims nothing —
486
+ // answer it here rather than spending a statement to learn that.
487
+ if (limit <= 0 || names?.length === 0) return [];
488
+ const oneName = names?.length === 1;
489
+ const statement = oneName
490
+ ? oneQueue
491
+ ? "claim_one_queue_one_name"
492
+ : "claim_one_name"
493
+ : oneQueue
494
+ ? "claim_one_queue"
495
+ : "claim";
496
+ const rows = await fetch(statement, {
497
+ ...base,
498
+ names,
499
+ name: oneName ? names![0] : null,
500
+ limit,
501
+ });
502
+ return rows.map(rowToTask);
503
+ });
448
504
  });
449
505
  }
450
506
 
@@ -456,6 +512,33 @@ export abstract class TaskStore {
456
512
  });
457
513
  }
458
514
 
515
+ /**
516
+ * Renew several leases in one statement. Returns `taskId -> cancel requested`
517
+ * for the tasks this worker still holds.
518
+ *
519
+ * Deliberately not an ownedWrite: ownership is per task here, so there is no
520
+ * single answer to "did it work". A task **absent** from the result lost its
521
+ * lease, and the caller decides what that means for that one task rather than
522
+ * failing the whole beat.
523
+ *
524
+ * It returns flags rather than Tasks because nothing downstream needs a task:
525
+ * the caller renews leases and observes cancellation, and whole rows would drag
526
+ * every payload back on every beat for the life of the call.
527
+ */
528
+ async heartbeatBatch(input: {
529
+ taskIds: string[];
530
+ workerId: string;
531
+ leaseMs?: number;
532
+ }): Promise<Map<string, boolean>> {
533
+ if (!input.taskIds.length) return new Map();
534
+ const rows = await this.fetch("heartbeat_batch", {
535
+ ids: input.taskIds,
536
+ worker_id: input.workerId,
537
+ lease_ms: input.leaseMs ?? 30_000,
538
+ });
539
+ return new Map(rows.map((r) => [r.id as string, r.cancel_requested_at_ms != null]));
540
+ }
541
+
459
542
  async progress(input: {
460
543
  taskId: string;
461
544
  workerId: string;
@@ -357,7 +357,10 @@ export class SQLiteStore extends TaskStore {
357
357
  bound[name] = now - (params.older_than_ms as number);
358
358
  break;
359
359
  case "queues":
360
- bound[name] = JSON.stringify(params.queues);
360
+ case "ids":
361
+ // json_each needs a JSON array. Postgres binds the array itself as
362
+ // text[], so only this dialect encodes.
363
+ bound[name] = JSON.stringify(params[name]);
361
364
  break;
362
365
  case "names":
363
366
  // json_each needs a JSON array; null stays null so the SQL's