cairnq 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +11 -5
  2. package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
  3. package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
  4. package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
  5. package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
  6. package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
  7. package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
  8. package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
  9. package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
  10. package/dist/backoff.d.ts +31 -0
  11. package/dist/backoff.js +40 -0
  12. package/dist/client.d.ts +28 -2
  13. package/dist/client.js +30 -3
  14. package/dist/context.d.ts +48 -2
  15. package/dist/context.js +101 -10
  16. package/dist/errors.d.ts +60 -4
  17. package/dist/errors.js +94 -9
  18. package/dist/index.d.ts +6 -2
  19. package/dist/index.js +2 -1
  20. package/dist/retention.d.ts +60 -0
  21. package/dist/retention.js +115 -0
  22. package/dist/store/base.d.ts +50 -1
  23. package/dist/store/base.js +114 -19
  24. package/dist/store/sqlite.js +4 -1
  25. package/dist/wait.d.ts +20 -5
  26. package/dist/wait.js +34 -9
  27. package/dist/worker.d.ts +214 -16
  28. package/dist/worker.js +500 -131
  29. package/package.json +1 -1
  30. package/src/backoff.ts +53 -0
  31. package/src/client.ts +43 -5
  32. package/src/context.ts +116 -9
  33. package/src/errors.ts +101 -9
  34. package/src/index.ts +6 -1
  35. package/src/retention.ts +136 -0
  36. package/src/store/base.ts +121 -17
  37. package/src/store/sqlite.ts +4 -1
  38. package/src/wait.ts +60 -16
  39. package/src/worker.ts +640 -146
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "cairnq",
3
- "version": "0.5.0",
3
+ "version": "0.7.0",
4
4
  "description": "SQLite-first, cross-language, storage-centered durable task runtime",
5
5
  "license": "MIT",
6
6
  "author": "Jannchie <jannchie@gmail.com>",
package/src/backoff.ts ADDED
@@ -0,0 +1,53 @@
1
+ /**
2
+ * Retry backoff, in its own module because two callers need it.
3
+ *
4
+ * The worker computes it when a handler's failure ends an attempt; TaskContext
5
+ * computes it when a handler fails one task of a batch itself. Keeping it in
6
+ * worker.ts would make context.ts import the module that imports it.
7
+ */
8
+
9
+ export const DEFAULT_RETRY_BACKOFF_MS = 1_000;
10
+ export const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
11
+
12
+ /**
13
+ * Exponential backoff with equal jitter: the window doubles per attempt up to
14
+ * `maxMs`, and the delay lands uniformly in its upper half, `[w/2, w)`.
15
+ *
16
+ * The jitter is what keeps a fleet from retrying in lockstep. Failures align
17
+ * when the downstream fails fast enough that a whole concurrency batch raises
18
+ * at once (connection refused, DNS gone), and capped exponential backoff then
19
+ * *preserves* that alignment — once every task sits at `maxMs`, they all retry
20
+ * on the same beat forever. Spreading over half the window breaks it; keeping
21
+ * the lower half as a floor means jitter never shortens the wait to less than
22
+ * half of what plain exponential backoff would have asked for.
23
+ *
24
+ * `rand` is injected so tests can pin an exact delay.
25
+ */
26
+ export function retryDelayMs(
27
+ attempt: number,
28
+ baseMs: number,
29
+ maxMs: number,
30
+ rand: () => number = Math.random,
31
+ ): number {
32
+ if (baseMs <= 0) return 0;
33
+ const exponent = Math.max(0, attempt - 1);
34
+ const window = Math.min(maxMs, baseMs * 2 ** exponent);
35
+ const floor = Math.floor(window / 2);
36
+ return floor + Math.floor(rand() * (window - floor));
37
+ }
38
+
39
+ /**
40
+ * The delay a `fail` write should carry. Not just the backoff: a permanent
41
+ * failure is never re-run, so it always delays 0. Both settlement paths — the
42
+ * worker's and a handler's `ctx.fail` — go through this, so they cannot end up
43
+ * backing off differently.
44
+ */
45
+ export function failDelayMs(
46
+ attempt: number,
47
+ retryable: boolean,
48
+ baseMs: number,
49
+ maxMs: number,
50
+ rand: () => number = Math.random,
51
+ ): number {
52
+ return retryable ? retryDelayMs(attempt, baseMs, maxMs, rand) : 0;
53
+ }
package/src/client.ts CHANGED
@@ -1,11 +1,12 @@
1
1
  import type { BackpressureOptions } from "./backpressure.js";
2
+ import { type RetentionOptions, RetentionSweeper } from "./retention.js";
2
3
  import { TaskCanceled, TaskFailed } from "./errors.js";
3
4
  import { isFailed, isSucceeded, type Task, type TaskStatus } from "./models.js";
4
5
  import { SQLiteStore } from "./store/sqlite.js";
5
6
  import { PostgresStore } from "./store/postgres.js";
6
7
  import type { ListInput, PurgeInput, SubmitInput, TaskStore } from "./store/base.js";
7
8
  import { type TaskDef, taskName } from "./task.js";
8
- import { pollWait } from "./wait.js";
9
+ import { pollWait, pollWaitByKey } from "./wait.js";
9
10
 
10
11
  export type SubmitOptions = Omit<SubmitInput, "name" | "payload">;
11
12
  export interface CallOptions extends SubmitOptions {
@@ -15,10 +16,18 @@ export interface CallOptions extends SubmitOptions {
15
16
 
16
17
  /** Options this handle configures on the store it wraps, rather than the
17
18
  * store's own constructor arguments. */
18
- export type ClientOptions = Partial<BackpressureOptions>;
19
+ export type ClientOptions = Partial<BackpressureOptions> & {
20
+ /** Delete terminal tasks older than a cutoff, on a schedule, for as long as
21
+ * this handle is open. Off unless set — and off means rows accumulate forever,
22
+ * because nothing else in CairnQ removes them. */
23
+ retention?: RetentionOptions;
24
+ };
19
25
 
20
26
  /** API-side handle. Thin wrapper over a TaskStore + SDK-orchestrated wait/call. */
21
27
  export class CairnQ {
28
+ /** null unless `retention` was configured. */
29
+ private readonly sweeper: RetentionSweeper | null;
30
+
22
31
  constructor(
23
32
  private readonly _store: TaskStore,
24
33
  opts: ClientOptions = {},
@@ -28,6 +37,13 @@ export class CairnQ {
28
37
  if (opts.maxQueueDepth != null) {
29
38
  _store.useBackpressure(opts as BackpressureOptions);
30
39
  }
40
+ // Retention is the opposite case: it belongs to the handle, because a worker
41
+ // sharing the store must not also be deleting rows behind the API's back.
42
+ // Started here rather than in connect(), which is optional — every other
43
+ // path connects lazily, and retention that silently depends on an optional
44
+ // call is retention that silently does not happen.
45
+ this.sweeper = opts.retention ? new RetentionSweeper(_store, opts.retention) : null;
46
+ this.sweeper?.start();
31
47
  }
32
48
 
33
49
  static sqlite(path: string, opts: { busyTimeoutMs?: number } & ClientOptions = {}): CairnQ {
@@ -50,8 +66,11 @@ export class CairnQ {
50
66
  return this._store.connect();
51
67
  }
52
68
 
53
- close(): Promise<void> {
54
- return this._store.close();
69
+ /** Stop retention (waiting for a sweep in flight, so no purge outlives the
70
+ * store) and close the store. */
71
+ async close(): Promise<void> {
72
+ await this.sweeper?.stop();
73
+ await this._store.close();
55
74
  }
56
75
 
57
76
  /** Enqueue a task. With `maxQueueDepth` configured this blocks while the
@@ -114,6 +133,9 @@ export class CairnQ {
114
133
  return this._store.stats();
115
134
  }
116
135
 
136
+ /** Wait for a task to finish. Resolves with the terminal Task (any status);
137
+ * throws TaskTimeout without stopping the task, so `wait(err.taskId)` picks the
138
+ * same wait back up — from another process, or after a longer deadline. */
117
139
  wait(
118
140
  taskId: string,
119
141
  opts: { timeoutMs?: number; pollMs?: number } = {},
@@ -124,9 +146,25 @@ export class CairnQ {
124
146
  });
125
147
  }
126
148
 
149
+ /** Wait for whatever task the `key` currently points at — the cross-process
150
+ * form of picking a wait back up, when the id was never in hand or the process
151
+ * that held it is gone. Re-resolves the key on each poll, so a `replace`
152
+ * landing mid-wait moves the wait onto the new task, and a key with no task
153
+ * yet is waited for rather than rejected. */
154
+ waitByKey(key: string, opts: { timeoutMs?: number; pollMs?: number } = {}): Promise<Task> {
155
+ return pollWaitByKey(this._store, key, {
156
+ timeoutMs: opts.timeoutMs ?? 30_000,
157
+ pollMs: opts.pollMs,
158
+ });
159
+ }
160
+
127
161
  /** submit + wait. Resolves with the result on success; rejects with
128
162
  * TaskFailed / TaskCanceled / TaskTimeout otherwise. Pass a TaskDef and the
129
- * resolved value is typed as its Result. */
163
+ * resolved value is typed as its Result.
164
+ *
165
+ * `waitTimeoutMs` bounds the wait, not the task: on timeout the task runs on,
166
+ * and `wait(err.taskId)` — or `waitByKey`, from a process that only has the
167
+ * key — resumes the wait rather than starting the work over. */
130
168
  async call(name: string, payload?: unknown, opts?: CallOptions): Promise<unknown>;
131
169
  async call<P, R>(task: TaskDef<P, R>, payload?: P, opts?: CallOptions): Promise<R>;
132
170
  async call(task: string | TaskDef, payload?: unknown, opts: CallOptions = {}): Promise<unknown> {
package/src/context.ts CHANGED
@@ -1,24 +1,48 @@
1
- import { LostLease } from "./errors.js";
1
+ import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
2
+ import { asEnvelope, type FailReason, LostLease } from "./errors.js";
2
3
  import { cancelRequested, type Task } from "./models.js";
3
4
  import type { SubmitOptions } from "./client.js";
4
5
  import type { TaskStore } from "./store/base.js";
5
6
  import { type TaskDef, taskName } from "./task.js";
6
7
  import { pollWait } from "./wait.js";
7
8
 
8
- /** Handed to a task handler. Worker-side capabilities mirror the Python SDK. */
9
+ export interface TaskContextOptions {
10
+ retryBackoffMs?: number;
11
+ retryBackoffMaxMs?: number;
12
+ }
13
+
14
+ /**
15
+ * Handed to a task handler. Worker-side capabilities mirror the Python SDK.
16
+ *
17
+ * One of these per task, whether a handler is delivered one task or a batch: a
18
+ * batch handler receives a `TaskContext[]`, so a single-task handler's `ctx` is
19
+ * literally the batch-of-one element. Lease, cancellation and settlement are per
20
+ * task, which is why they live here rather than on anything batch-shaped.
21
+ */
9
22
  export class TaskContext {
10
23
  private readonly abort = new AbortController();
11
24
  private leaseLost = false;
12
25
  // Cancellation is monotonic: once the DB has told us a cancel was requested it
13
26
  // can't be taken back, so canceled() can answer from this without a re-read.
14
27
  private cancelSeen = false;
28
+ // Set once this task reached a terminal state through succeed()/fail(). The
29
+ // worker reads it to know which tasks a batch handler already decided, so it
30
+ // neither settles them twice nor keeps renewing their leases — the bookkeeping
31
+ // every ack/nack-style handler otherwise has to carry itself.
32
+ private isSettled = false;
33
+ private readonly backoffMs: number;
34
+ private readonly backoffMaxMs: number;
15
35
 
16
36
  constructor(
17
37
  private readonly store: TaskStore,
18
38
  private readonly task: Task,
19
39
  public readonly workerId: string,
20
40
  private readonly leaseMs: number,
21
- ) {}
41
+ opts: TaskContextOptions = {},
42
+ ) {
43
+ this.backoffMs = opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
44
+ this.backoffMaxMs = opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS;
45
+ }
22
46
 
23
47
  get taskId(): string {
24
48
  return this.task.id;
@@ -44,6 +68,19 @@ export class TaskContext {
44
68
  get payload(): any {
45
69
  return this.task.payload;
46
70
  }
71
+ /**
72
+ * True once this task reached a terminal state — whether the handler settled
73
+ * it with succeed()/fail() or the worker settled it on the handler's behalf.
74
+ * The heartbeat and the settlement paths both read it.
75
+ */
76
+ get settled(): boolean {
77
+ return this.isSettled;
78
+ }
79
+
80
+ /** @internal Called by the worker when it finalizes this task itself. */
81
+ markSettled(): void {
82
+ this.isSettled = true;
83
+ }
47
84
 
48
85
  /**
49
86
  * True once this worker has lost the task's lease — it expired and another
@@ -70,17 +107,36 @@ export class TaskContext {
70
107
  // Every owned write returns the current row, so cancellation and lease loss
71
108
  // ride along on writes the handler was making anyway.
72
109
  private observe(task: Task): Task {
73
- if (cancelRequested(task)) this.cancelSeen = true;
110
+ this.observeCancel(cancelRequested(task));
74
111
  return task;
75
112
  }
76
113
 
114
+ /**
115
+ * @internal The same observation from just the flag, for a caller that read it
116
+ * without materializing a Task — the shared heartbeat, whose statement returns
117
+ * only the id and the cancel column precisely so it does not have to drag
118
+ * every payload back on every beat.
119
+ */
120
+ observeCancel(cancelRequested: boolean): void {
121
+ if (cancelRequested) this.cancelSeen = true;
122
+ }
123
+
77
124
  private async owned(write: () => Promise<Task>): Promise<Task> {
78
- // Short-circuit once the lease is known lost: nothing this context writes
79
- // may be recorded any more. Locally, not just via the store's ownership
80
- // check — after an abandoned (timed-out) attempt the same worker may
81
- // re-claim this task under the same workerId, and a zombie handler's write
82
- // would then pass ownership against the NEW attempt.
125
+ // One gate for every write through this context, so "may I still write?" is
126
+ // answered in one place rather than at each call site.
127
+ //
128
+ // Lease lost: nothing this context writes may be recorded any more. Checked
129
+ // locally, not just via the store's ownership check — after an abandoned
130
+ // (timed-out) attempt the same worker may re-claim this task under the same
131
+ // workerId, and a zombie handler's write would then pass ownership against
132
+ // the NEW attempt.
83
133
  if (this.leaseLost) throw new LostLease(this.task.id);
134
+ // Settled: the task is terminal, so the statement would match no row and come
135
+ // back as a lost lease — telling the handler "another worker took this" when
136
+ // the truth is "you already finished it", and flipping lostLease on the way.
137
+ // Refuse here instead, without the round trip and without corrupting the
138
+ // lease state.
139
+ if (this.isSettled) throw new LostLease(this.task.id);
84
140
  try {
85
141
  return this.observe(await write());
86
142
  } catch (err) {
@@ -119,6 +175,57 @@ export class TaskContext {
119
175
  return this.cancelSeen || t.status === "canceled";
120
176
  }
121
177
 
178
+ // ------------------------------------------------------------- settlement
179
+ // Finalizing a task is normally the worker's job, decided by whether the
180
+ // handler returned or threw. These two let a handler decide one task itself,
181
+ // which is what a batch needs: four of 256 tasks failing for four different
182
+ // reasons is the ordinary case, not the edge one, and it cannot be expressed
183
+ // by a single return value or a single throw.
184
+ //
185
+ // Settling twice is a no-op rather than an error. Handlers built on ack/nack
186
+ // queues all end up carrying a `finalizedIds` set to guarantee exactly that;
187
+ // holding it here instead is the point.
188
+
189
+ /**
190
+ * Finalize this task as succeeded, now, without waiting for the handler to
191
+ * return. `complete` semantics: a cancel requested while it ran wins and the
192
+ * task finalizes as canceled instead, its result discarded. Returns null if
193
+ * this task was already settled.
194
+ */
195
+ async succeed(result: unknown = null): Promise<Task | null> {
196
+ if (this.isSettled) return null;
197
+ const task = await this.owned(() =>
198
+ this.store.complete({ taskId: this.task.id, workerId: this.workerId, result }),
199
+ );
200
+ this.markSettled();
201
+ return task;
202
+ }
203
+
204
+ /**
205
+ * Finalize this task as failed, now. `error` may be a string reason, an Error,
206
+ * a TaskError (which carries its own retryability), or a ready envelope.
207
+ * Retryable failures get the worker's backoff and are re-queued while attempts
208
+ * remain, exactly as a thrown error would be. Returns null if already settled.
209
+ */
210
+ async fail(
211
+ error: FailReason = "task failed",
212
+ opts: { retryable?: boolean } = {},
213
+ ): Promise<Task | null> {
214
+ if (this.isSettled) return null;
215
+ const [envelope, retryable] = asEnvelope(error, opts.retryable ?? true);
216
+ const task = await this.owned(() =>
217
+ this.store.fail({
218
+ taskId: this.task.id,
219
+ workerId: this.workerId,
220
+ error: envelope,
221
+ retryable,
222
+ delayMs: failDelayMs(this.task.attempt, retryable, this.backoffMs, this.backoffMaxMs),
223
+ }),
224
+ );
225
+ this.markSettled();
226
+ return task;
227
+ }
228
+
122
229
  /** Submit a child task; parent/root/correlation are wired automatically. */
123
230
  submit(name: string, payload?: unknown, opts?: SubmitOptions): Promise<Task>;
124
231
  submit<P, R>(task: TaskDef<P, R>, payload?: P, opts?: SubmitOptions): Promise<Task>;
package/src/errors.ts CHANGED
@@ -20,6 +20,24 @@ export function errorEnvelope(e: {
20
20
  };
21
21
  }
22
22
 
23
+ /**
24
+ * How an arbitrary thrown value becomes an envelope. Split out from `asEnvelope`
25
+ * below because the worker also reaches it directly, for a thrown plain object —
26
+ * which `asEnvelope` reads as a ready envelope, the right call for `ctx.fail` and
27
+ * the wrong one for something that was thrown. Both must agree on `code` and on
28
+ * deriving `type` from the error's name, or the same error reads differently
29
+ * depending on which way it was recorded.
30
+ */
31
+ export function exceptionEnvelope(err: unknown, retryable = true): Record<string, unknown> {
32
+ const e = err as { name?: string; message?: string };
33
+ return errorEnvelope({
34
+ type: e?.name ?? "Error",
35
+ code: "handler_error",
36
+ message: String(e?.message ?? err),
37
+ retryable,
38
+ });
39
+ }
40
+
23
41
  export class CairnQError extends Error {
24
42
  constructor(message?: string) {
25
43
  super(message);
@@ -59,8 +77,12 @@ export class QueueFull extends CairnQError {
59
77
  * observed. No worker running, no handler for the name, wrong queue, and two
60
78
  * processes on different database files all look identical from the API side —
61
79
  * queued, never claimed — so that case names the likely causes. */
62
- function timeoutDetail(task: Task | null): string {
63
- if (!task) return "task not found — wrong database file, or already purged?";
80
+ function timeoutDetail(task: Task | null, key: string | null): string {
81
+ if (!task) {
82
+ return key === null
83
+ ? "task not found — wrong database file, or already purged?"
84
+ : "no task under this key — never submitted, or already purged?";
85
+ }
64
86
  if (isQueued(task)) {
65
87
  const delayMs = task.run_at_ms - nowMs();
66
88
  if (task.attempt === 0 && delayMs <= 0) {
@@ -76,23 +98,33 @@ function timeoutDetail(task: Task | null): string {
76
98
  return `still running (attempt ${task.attempt}/${task.max_attempts})`;
77
99
  }
78
100
 
79
- /** wait/call did not reach a terminal status in time. The task keeps running.
80
- * `task` is the last snapshot wait() observed (null if get() found nothing), and
81
- * the message says what state it was stuck in — a queued-never-claimed task is
82
- * the classic first-run failure (no worker, no handler, wrong queue or file). */
101
+ /** wait/call did not reach a terminal status in time. The task keeps running, so
102
+ * `taskId` is the handle for picking the wait back up — `wait(err.taskId)`
103
+ * re-attaches to the same task from anywhere that can reach the store. `task` is
104
+ * the last snapshot wait() observed (null if the lookup found nothing), and the
105
+ * message says what state it was stuck in — a queued-never-claimed task is the
106
+ * classic first-run failure (no worker, no handler, wrong queue or file).
107
+ *
108
+ * `key` is set when the wait watched a key rather than an id; `taskId` is then
109
+ * the task the key pointed at, or the key itself when it pointed at nothing —
110
+ * there was no id to report. */
83
111
  export class TaskTimeout extends CairnQError {
84
112
  readonly task: Task | null;
113
+ readonly key: string | null;
85
114
  constructor(
86
115
  public taskId: string,
87
- opts: { timeoutMs?: number; task?: Task | null } = {},
116
+ opts: { timeoutMs?: number; task?: Task | null; key?: string | null } = {},
88
117
  ) {
118
+ const key = opts.key ?? null;
119
+ const subject = key === null ? `task ${taskId}` : `key ${key}`;
89
120
  super(
90
121
  opts.timeoutMs == null
91
- ? `task ${taskId} did not finish in time`
92
- : `task ${taskId} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null)}`,
122
+ ? `${subject} did not finish in time`
123
+ : `${subject} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null, key)}`,
93
124
  );
94
125
  this.name = "TaskTimeout";
95
126
  this.task = opts.task ?? null;
127
+ this.key = key;
96
128
  }
97
129
  }
98
130
 
@@ -128,6 +160,36 @@ export class TaskCanceled extends CairnQError {
128
160
  }
129
161
  }
130
162
 
163
+ /**
164
+ * A heartbeat beat came back later than its own interval allowed.
165
+ *
166
+ * The heartbeat shares the event loop with the handlers whose leases it renews,
167
+ * so a handler that blocks the loop stops the renewal with it: the lease expires,
168
+ * the task is recovered and redelivered, and a second worker starts computing
169
+ * what the first is still computing — one task, billed twice, with no error
170
+ * anywhere. Nothing inside the blocked handler can observe that, which is why it
171
+ * is reported through `onError` alongside the other things the run loop survived.
172
+ *
173
+ * The cause is always synchronous work in a handler: a tight loop, a large
174
+ * JSON.parse, a `*Sync` filesystem or crypto call. Node has one loop and no way
175
+ * to preempt it — move the work to a worker thread, a child process, or an async
176
+ * API that yields.
177
+ */
178
+ export class EventLoopBlocked extends CairnQError {
179
+ constructor(
180
+ readonly lateMs: number,
181
+ readonly intervalMs: number,
182
+ readonly leaseMs: number,
183
+ ) {
184
+ super(
185
+ `heartbeat beat was ${lateMs}ms late (interval ${intervalMs}ms, lease ${leaseMs}ms): ` +
186
+ `the event loop was blocked long enough to miss a beat. Synchronous work in a ` +
187
+ `handler starves lease renewal — move it off the loop.`,
188
+ );
189
+ this.name = "EventLoopBlocked";
190
+ }
191
+ }
192
+
131
193
  /** A worker write affected 0 rows: the lease expired and was reclaimed. */
132
194
  export class LostLease extends CairnQError {
133
195
  constructor(public taskId: string) {
@@ -183,3 +245,33 @@ export class TaskError extends CairnQError {
183
245
  });
184
246
  }
185
247
  }
248
+
249
+ /** What a handler may pass to `ctx.fail`. */
250
+ export type FailReason = string | Error | TaskError | Record<string, unknown>;
251
+
252
+ /**
253
+ * Normalize anything that can end a task into [envelope, retryable].
254
+ *
255
+ * Shared by both ways a failure is recorded — a handler passing a reason to
256
+ * `ctx.fail`, and the worker classifying an error that ended an attempt — so the
257
+ * two cannot disagree about what a given error means. It lives here, beside the
258
+ * envelope constructors it dispatches to, rather than in the module that happens
259
+ * to expose it to handlers.
260
+ *
261
+ * A handler failing one task of a batch has a reason, not an exception object:
262
+ * `item.fail("no source records", { retryable: false })` is the shape the real
263
+ * code wants. A TaskError carries its own retryability and wins over the option;
264
+ * everything else takes the caller's. A ready envelope passes through, which is
265
+ * how the worker hands in the ones it composes itself.
266
+ */
267
+ export function asEnvelope(
268
+ error: FailReason,
269
+ retryable: boolean,
270
+ ): [Record<string, unknown>, boolean] {
271
+ if (error instanceof TaskError) return [error.envelope(), error.retryable];
272
+ if (error instanceof Error) return [exceptionEnvelope(error, retryable), retryable];
273
+ if (typeof error === "object" && error !== null) return [error, retryable];
274
+ // A bare reason is a TaskError in everything but the throwing, so let
275
+ // TaskError own its own type/code defaults rather than restating them.
276
+ return [new TaskError(String(error), { retryable }).envelope(), retryable];
277
+ }
package/src/index.ts CHANGED
@@ -2,9 +2,12 @@ export { CairnQ } from "./client.js";
2
2
  export type { CallOptions, ClientOptions, SubmitOptions } from "./client.js";
3
3
  export { QueueDepthGate } from "./backpressure.js";
4
4
  export type { BackpressureOptions, QueueDepthLimit } from "./backpressure.js";
5
+ export { RetentionSweeper } from "./retention.js";
6
+ export type { RetentionOptions } from "./retention.js";
5
7
  export { Worker } from "./worker.js";
6
- export type { Handler, TypedHandler, WorkerOptions } from "./worker.js";
8
+ export type { BatchHandler, Handler, TypedHandler, WorkerOptions } from "./worker.js";
7
9
  export { TaskContext } from "./context.js";
10
+ export type { TaskContextOptions } from "./context.js";
8
11
  export { defineTask } from "./task.js";
9
12
  export type { TaskDef } from "./task.js";
10
13
  export { SQLiteStore } from "./store/sqlite.js";
@@ -31,6 +34,8 @@ export {
31
34
  TaskCanceled,
32
35
  TaskError,
33
36
  LostLease,
37
+ EventLoopBlocked,
34
38
  ProtocolVersionMismatch,
35
39
  SerializationError,
36
40
  } from "./errors.js";
41
+ export type { FailReason } from "./errors.js";
@@ -0,0 +1,136 @@
1
+ import type { PurgeInput, TaskStore } from "./store/base.js";
2
+
3
+ /** Sweep every hour unless asked otherwise — often enough that a queue with a
4
+ * day of retention never carries more than an hour of extra rows, rare enough
5
+ * that the sweep is invisible next to the task traffic. */
6
+ const DEFAULT_INTERVAL_MS = 3_600_000;
7
+ /** Rows per purge statement. The same bound `purge` defaults to: big enough that
8
+ * a backlog drains in few statements, small enough that each is a short write. */
9
+ const DEFAULT_LIMIT = 1_000;
10
+
11
+ export interface RetentionOptions {
12
+ /**
13
+ * How long a terminal task is kept after it finished. Required: there is no
14
+ * safe default for how long someone else's results stay readable.
15
+ */
16
+ olderThanMs: number;
17
+ /** Time between sweeps. Default 3_600_000 (one hour). */
18
+ intervalMs?: number;
19
+ /** Rows deleted per statement while draining. Default 1_000. */
20
+ limit?: number;
21
+ /**
22
+ * Called for a sweep that threw. The next sweep runs on schedule regardless —
23
+ * a purge that failed because the database was busy is not a reason to stop
24
+ * retaining — so without this a store quietly stops being swept. Must not throw.
25
+ */
26
+ onError?: (err: unknown) => void;
27
+ }
28
+
29
+ /**
30
+ * Deletes terminal tasks on a schedule, for as long as the handle is open.
31
+ *
32
+ * `purge` exists because nothing else in CairnQ removes rows, and a queue whose
33
+ * payloads carry real data — an image, a document, a batch of embeddings — turns
34
+ * that into a disk leak measured in gigabytes per backfill. Every deployment
35
+ * that runs longer than a demo needs the sweep; leaving it to an external
36
+ * scheduler means the leak is the default and remembering is the opt-in.
37
+ *
38
+ * It sweeps in bounded batches with a yield between them, so draining a backlog
39
+ * that accumulated while nothing was sweeping stays a sequence of short writes
40
+ * rather than one long one — on SQLite that matters, since a long write holds
41
+ * the single write lock against every producer and worker on the file.
42
+ */
43
+ export class RetentionSweeper {
44
+ /** Whether the scheduled loop is running. */
45
+ private active = false;
46
+ /** Set by stop(), so a drain in progress can cut itself short too. */
47
+ private stopping = false;
48
+ /** Resolves the current sleep early, so stop() need not wait out an interval. */
49
+ private wake: (() => void) | null = null;
50
+ /** The loop itself, awaited by stop() so no purge outlives the store. */
51
+ private loop: Promise<void> | null = null;
52
+ private readonly intervalMs: number;
53
+ private readonly purgeInput: PurgeInput;
54
+
55
+ constructor(
56
+ private readonly store: TaskStore,
57
+ private readonly opts: RetentionOptions,
58
+ ) {
59
+ if (!Number.isFinite(opts.olderThanMs) || opts.olderThanMs < 0) {
60
+ throw new Error(`retention.olderThanMs must be >= 0, got ${opts.olderThanMs}`);
61
+ }
62
+ this.intervalMs = opts.intervalMs ?? DEFAULT_INTERVAL_MS;
63
+ if (!Number.isFinite(this.intervalMs) || this.intervalMs < 1) {
64
+ throw new Error(`retention.intervalMs must be >= 1, got ${this.intervalMs}`);
65
+ }
66
+ this.purgeInput = { olderThanMs: opts.olderThanMs, limit: opts.limit ?? DEFAULT_LIMIT };
67
+ }
68
+
69
+ start(): void {
70
+ if (this.active) return;
71
+ this.active = true;
72
+ this.stopping = false;
73
+ this.loop = this.run();
74
+ }
75
+
76
+ /** Stop sweeping and wait for the sweep in flight, if any. */
77
+ async stop(): Promise<void> {
78
+ this.stopping = true;
79
+ this.active = false;
80
+ this.wake?.();
81
+ await this.loop;
82
+ this.loop = null;
83
+ }
84
+
85
+ private async run(): Promise<void> {
86
+ // Sleep first: a process that restarts often would otherwise purge on every
87
+ // boot, which is a write burst exactly when the store is busiest.
88
+ while (!this.stopping) {
89
+ await this.sleep(this.intervalMs);
90
+ if (this.stopping) return;
91
+ try {
92
+ await this.sweep();
93
+ } catch (err) {
94
+ try {
95
+ this.opts.onError?.(err);
96
+ } catch {
97
+ // A reporting hook must never take the sweep down with it — the same
98
+ // rule the worker's onError follows.
99
+ }
100
+ }
101
+ }
102
+ }
103
+
104
+ /**
105
+ * Delete everything past the cutoff now, in bounded batches, and return how
106
+ * many rows went. The scheduled loop calls this; call it directly to drain on
107
+ * demand — after a backfill, or from a maintenance command.
108
+ */
109
+ async sweep(): Promise<number> {
110
+ const limit = this.purgeInput.limit as number;
111
+ let deleted = 0;
112
+ for (;;) {
113
+ const ids = await this.store.purge(this.purgeInput);
114
+ deleted += ids.length;
115
+ if (ids.length < limit || this.stopping) return deleted;
116
+ // Hand the loop back between batches: a large drain must not starve the
117
+ // submits and claims sharing this process.
118
+ await this.sleep(0);
119
+ }
120
+ }
121
+
122
+ /** Sleep, interruptible by stop(). Unref'd: retention is housekeeping, and a
123
+ * pending sweep must never be the reason a process refuses to exit. */
124
+ private sleep(ms: number): Promise<void> {
125
+ return new Promise<void>((resolve) => {
126
+ const timer = setTimeout(resolve, ms);
127
+ timer.unref?.();
128
+ this.wake = () => {
129
+ clearTimeout(timer);
130
+ resolve();
131
+ };
132
+ }).finally(() => {
133
+ this.wake = null;
134
+ });
135
+ }
136
+ }