cairnq 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +11 -5
  2. package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
  3. package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
  4. package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
  5. package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
  6. package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
  7. package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
  8. package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
  9. package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
  10. package/dist/backoff.d.ts +31 -0
  11. package/dist/backoff.js +40 -0
  12. package/dist/client.d.ts +28 -2
  13. package/dist/client.js +30 -3
  14. package/dist/context.d.ts +48 -2
  15. package/dist/context.js +101 -10
  16. package/dist/errors.d.ts +60 -4
  17. package/dist/errors.js +94 -9
  18. package/dist/index.d.ts +6 -2
  19. package/dist/index.js +2 -1
  20. package/dist/retention.d.ts +60 -0
  21. package/dist/retention.js +115 -0
  22. package/dist/store/base.d.ts +50 -1
  23. package/dist/store/base.js +114 -19
  24. package/dist/store/sqlite.js +4 -1
  25. package/dist/wait.d.ts +20 -5
  26. package/dist/wait.js +34 -9
  27. package/dist/worker.d.ts +214 -16
  28. package/dist/worker.js +500 -131
  29. package/package.json +1 -1
  30. package/src/backoff.ts +53 -0
  31. package/src/client.ts +43 -5
  32. package/src/context.ts +116 -9
  33. package/src/errors.ts +101 -9
  34. package/src/index.ts +6 -1
  35. package/src/retention.ts +136 -0
  36. package/src/store/base.ts +121 -17
  37. package/src/store/sqlite.ts +4 -1
  38. package/src/wait.ts +60 -16
  39. package/src/worker.ts +640 -146
package/dist/context.js CHANGED
@@ -1,8 +1,16 @@
1
- import { LostLease } from "./errors.js";
1
+ import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
2
+ import { asEnvelope, LostLease } from "./errors.js";
2
3
  import { cancelRequested } from "./models.js";
3
4
  import { taskName } from "./task.js";
4
5
  import { pollWait } from "./wait.js";
5
- /** Handed to a task handler. Worker-side capabilities mirror the Python SDK. */
6
+ /**
7
+ * Handed to a task handler. Worker-side capabilities mirror the Python SDK.
8
+ *
9
+ * One of these per task, whether a handler is delivered one task or a batch: a
10
+ * batch handler receives a `TaskContext[]`, so a single-task handler's `ctx` is
11
+ * literally the batch-of-one element. Lease, cancellation and settlement are per
12
+ * task, which is why they live here rather than on anything batch-shaped.
13
+ */
6
14
  export class TaskContext {
7
15
  store;
8
16
  task;
@@ -13,11 +21,20 @@ export class TaskContext {
13
21
  // Cancellation is monotonic: once the DB has told us a cancel was requested it
14
22
  // can't be taken back, so canceled() can answer from this without a re-read.
15
23
  cancelSeen = false;
16
- constructor(store, task, workerId, leaseMs) {
24
+ // Set once this task reached a terminal state through succeed()/fail(). The
25
+ // worker reads it to know which tasks a batch handler already decided, so it
26
+ // neither settles them twice nor keeps renewing their leases — the bookkeeping
27
+ // every ack/nack-style handler otherwise has to carry itself.
28
+ isSettled = false;
29
+ backoffMs;
30
+ backoffMaxMs;
31
+ constructor(store, task, workerId, leaseMs, opts = {}) {
17
32
  this.store = store;
18
33
  this.task = task;
19
34
  this.workerId = workerId;
20
35
  this.leaseMs = leaseMs;
36
+ this.backoffMs = opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
37
+ this.backoffMaxMs = opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS;
21
38
  }
22
39
  get taskId() {
23
40
  return this.task.id;
@@ -43,6 +60,18 @@ export class TaskContext {
43
60
  get payload() {
44
61
  return this.task.payload;
45
62
  }
63
+ /**
64
+ * True once this task reached a terminal state — whether the handler settled
65
+ * it with succeed()/fail() or the worker settled it on the handler's behalf.
66
+ * The heartbeat and the settlement paths both read it.
67
+ */
68
+ get settled() {
69
+ return this.isSettled;
70
+ }
71
+ /** @internal Called by the worker when it finalizes this task itself. */
72
+ markSettled() {
73
+ this.isSettled = true;
74
+ }
46
75
  /**
47
76
  * True once this worker has lost the task's lease — it expired and another
48
77
  * worker reclaimed it. Nothing this handler writes will be recorded any more
@@ -66,18 +95,37 @@ export class TaskContext {
66
95
  // Every owned write returns the current row, so cancellation and lease loss
67
96
  // ride along on writes the handler was making anyway.
68
97
  observe(task) {
69
- if (cancelRequested(task))
70
- this.cancelSeen = true;
98
+ this.observeCancel(cancelRequested(task));
71
99
  return task;
72
100
  }
101
+ /**
102
+ * @internal The same observation from just the flag, for a caller that read it
103
+ * without materializing a Task — the shared heartbeat, whose statement returns
104
+ * only the id and the cancel column precisely so it does not have to drag
105
+ * every payload back on every beat.
106
+ */
107
+ observeCancel(cancelRequested) {
108
+ if (cancelRequested)
109
+ this.cancelSeen = true;
110
+ }
73
111
  async owned(write) {
74
- // Short-circuit once the lease is known lost: nothing this context writes
75
- // may be recorded any more. Locally, not just via the store's ownership
76
- // check — after an abandoned (timed-out) attempt the same worker may
77
- // re-claim this task under the same workerId, and a zombie handler's write
78
- // would then pass ownership against the NEW attempt.
112
+ // One gate for every write through this context, so "may I still write?" is
113
+ // answered in one place rather than at each call site.
114
+ //
115
+ // Lease lost: nothing this context writes may be recorded any more. Checked
116
+ // locally, not just via the store's ownership check — after an abandoned
117
+ // (timed-out) attempt the same worker may re-claim this task under the same
118
+ // workerId, and a zombie handler's write would then pass ownership against
119
+ // the NEW attempt.
79
120
  if (this.leaseLost)
80
121
  throw new LostLease(this.task.id);
122
+ // Settled: the task is terminal, so the statement would match no row and come
123
+ // back as a lost lease — telling the handler "another worker took this" when
124
+ // the truth is "you already finished it", and flipping lostLease on the way.
125
+ // Refuse here instead, without the round trip and without corrupting the
126
+ // lease state.
127
+ if (this.isSettled)
128
+ throw new LostLease(this.task.id);
81
129
  try {
82
130
  return this.observe(await write());
83
131
  }
@@ -113,6 +161,49 @@ export class TaskContext {
113
161
  this.cancelSeen = true;
114
162
  return this.cancelSeen || t.status === "canceled";
115
163
  }
164
+ // ------------------------------------------------------------- settlement
165
+ // Finalizing a task is normally the worker's job, decided by whether the
166
+ // handler returned or threw. These two let a handler decide one task itself,
167
+ // which is what a batch needs: four of 256 tasks failing for four different
168
+ // reasons is the ordinary case, not the edge one, and it cannot be expressed
169
+ // by a single return value or a single throw.
170
+ //
171
+ // Settling twice is a no-op rather than an error. Handlers built on ack/nack
172
+ // queues all end up carrying a `finalizedIds` set to guarantee exactly that;
173
+ // holding it here instead is the point.
174
+ /**
175
+ * Finalize this task as succeeded, now, without waiting for the handler to
176
+ * return. `complete` semantics: a cancel requested while it ran wins and the
177
+ * task finalizes as canceled instead, its result discarded. Returns null if
178
+ * this task was already settled.
179
+ */
180
+ async succeed(result = null) {
181
+ if (this.isSettled)
182
+ return null;
183
+ const task = await this.owned(() => this.store.complete({ taskId: this.task.id, workerId: this.workerId, result }));
184
+ this.markSettled();
185
+ return task;
186
+ }
187
+ /**
188
+ * Finalize this task as failed, now. `error` may be a string reason, an Error,
189
+ * a TaskError (which carries its own retryability), or a ready envelope.
190
+ * Retryable failures get the worker's backoff and are re-queued while attempts
191
+ * remain, exactly as a thrown error would be. Returns null if already settled.
192
+ */
193
+ async fail(error = "task failed", opts = {}) {
194
+ if (this.isSettled)
195
+ return null;
196
+ const [envelope, retryable] = asEnvelope(error, opts.retryable ?? true);
197
+ const task = await this.owned(() => this.store.fail({
198
+ taskId: this.task.id,
199
+ workerId: this.workerId,
200
+ error: envelope,
201
+ retryable,
202
+ delayMs: failDelayMs(this.task.attempt, retryable, this.backoffMs, this.backoffMaxMs),
203
+ }));
204
+ this.markSettled();
205
+ return task;
206
+ }
116
207
  async submit(task, payload, opts = {}) {
117
208
  return this.store.submit({
118
209
  name: taskName(task),
package/dist/errors.d.ts CHANGED
@@ -9,6 +9,15 @@ export declare function errorEnvelope(e: {
9
9
  retryable: boolean;
10
10
  details?: Record<string, unknown>;
11
11
  }): Record<string, unknown>;
12
+ /**
13
+ * How an arbitrary thrown value becomes an envelope. Split out from `asEnvelope`
14
+ * below because the worker also reaches it directly, for a thrown plain object —
15
+ * which `asEnvelope` reads as a ready envelope, the right call for `ctx.fail` and
16
+ * the wrong one for something that was thrown. Both must agree on `code` and on
17
+ * deriving `type` from the error's name, or the same error reads differently
18
+ * depending on which way it was recorded.
19
+ */
20
+ export declare function exceptionEnvelope(err: unknown, retryable?: boolean): Record<string, unknown>;
12
21
  export declare class CairnQError extends Error {
13
22
  constructor(message?: string);
14
23
  }
@@ -26,16 +35,24 @@ export declare class QueueFull extends CairnQError {
26
35
  waitedMs: number;
27
36
  constructor(queue: string, maxDepth: number, waitedMs: number);
28
37
  }
29
- /** wait/call did not reach a terminal status in time. The task keeps running.
30
- * `task` is the last snapshot wait() observed (null if get() found nothing), and
31
- * the message says what state it was stuck in — a queued-never-claimed task is
32
- * the classic first-run failure (no worker, no handler, wrong queue or file). */
38
+ /** wait/call did not reach a terminal status in time. The task keeps running, so
39
+ * `taskId` is the handle for picking the wait back up — `wait(err.taskId)`
40
+ * re-attaches to the same task from anywhere that can reach the store. `task` is
41
+ * the last snapshot wait() observed (null if the lookup found nothing), and the
42
+ * message says what state it was stuck in — a queued-never-claimed task is the
43
+ * classic first-run failure (no worker, no handler, wrong queue or file).
44
+ *
45
+ * `key` is set when the wait watched a key rather than an id; `taskId` is then
46
+ * the task the key pointed at, or the key itself when it pointed at nothing —
47
+ * there was no id to report. */
33
48
  export declare class TaskTimeout extends CairnQError {
34
49
  taskId: string;
35
50
  readonly task: Task | null;
51
+ readonly key: string | null;
36
52
  constructor(taskId: string, opts?: {
37
53
  timeoutMs?: number;
38
54
  task?: Task | null;
55
+ key?: string | null;
39
56
  });
40
57
  }
41
58
  /** A waited-on task ended in `failed`. The envelope's fields are unpacked onto the
@@ -53,6 +70,27 @@ export declare class TaskCanceled extends CairnQError {
53
70
  taskId: string;
54
71
  constructor(taskId: string);
55
72
  }
73
+ /**
74
+ * A heartbeat beat came back later than its own interval allowed.
75
+ *
76
+ * The heartbeat shares the event loop with the handlers whose leases it renews,
77
+ * so a handler that blocks the loop stops the renewal with it: the lease expires,
78
+ * the task is recovered and redelivered, and a second worker starts computing
79
+ * what the first is still computing — one task, billed twice, with no error
80
+ * anywhere. Nothing inside the blocked handler can observe that, which is why it
81
+ * is reported through `onError` alongside the other things the run loop survived.
82
+ *
83
+ * The cause is always synchronous work in a handler: a tight loop, a large
84
+ * JSON.parse, a `*Sync` filesystem or crypto call. Node has one loop and no way
85
+ * to preempt it — move the work to a worker thread, a child process, or an async
86
+ * API that yields.
87
+ */
88
+ export declare class EventLoopBlocked extends CairnQError {
89
+ readonly lateMs: number;
90
+ readonly intervalMs: number;
91
+ readonly leaseMs: number;
92
+ constructor(lateMs: number, intervalMs: number, leaseMs: number);
93
+ }
56
94
  /** A worker write affected 0 rows: the lease expired and was reclaimed. */
57
95
  export declare class LostLease extends CairnQError {
58
96
  taskId: string;
@@ -84,3 +122,21 @@ export declare class TaskError extends CairnQError {
84
122
  });
85
123
  envelope(): Record<string, unknown>;
86
124
  }
125
+ /** What a handler may pass to `ctx.fail`. */
126
+ export type FailReason = string | Error | TaskError | Record<string, unknown>;
127
+ /**
128
+ * Normalize anything that can end a task into [envelope, retryable].
129
+ *
130
+ * Shared by both ways a failure is recorded — a handler passing a reason to
131
+ * `ctx.fail`, and the worker classifying an error that ended an attempt — so the
132
+ * two cannot disagree about what a given error means. It lives here, beside the
133
+ * envelope constructors it dispatches to, rather than in the module that happens
134
+ * to expose it to handlers.
135
+ *
136
+ * A handler failing one task of a batch has a reason, not an exception object:
137
+ * `item.fail("no source records", { retryable: false })` is the shape the real
138
+ * code wants. A TaskError carries its own retryability and wins over the option;
139
+ * everything else takes the caller's. A ready envelope passes through, which is
140
+ * how the worker hands in the ones it composes itself.
141
+ */
142
+ export declare function asEnvelope(error: FailReason, retryable: boolean): [Record<string, unknown>, boolean];
package/dist/errors.js CHANGED
@@ -12,6 +12,23 @@ export function errorEnvelope(e) {
12
12
  details: e.details ?? {},
13
13
  };
14
14
  }
15
+ /**
16
+ * How an arbitrary thrown value becomes an envelope. Split out from `asEnvelope`
17
+ * below because the worker also reaches it directly, for a thrown plain object —
18
+ * which `asEnvelope` reads as a ready envelope, the right call for `ctx.fail` and
19
+ * the wrong one for something that was thrown. Both must agree on `code` and on
20
+ * deriving `type` from the error's name, or the same error reads differently
21
+ * depending on which way it was recorded.
22
+ */
23
+ export function exceptionEnvelope(err, retryable = true) {
24
+ const e = err;
25
+ return errorEnvelope({
26
+ type: e?.name ?? "Error",
27
+ code: "handler_error",
28
+ message: String(e?.message ?? err),
29
+ retryable,
30
+ });
31
+ }
15
32
  export class CairnQError extends Error {
16
33
  constructor(message) {
17
34
  super(message);
@@ -50,9 +67,12 @@ export class QueueFull extends CairnQError {
50
67
  * observed. No worker running, no handler for the name, wrong queue, and two
51
68
  * processes on different database files all look identical from the API side —
52
69
  * queued, never claimed — so that case names the likely causes. */
53
- function timeoutDetail(task) {
54
- if (!task)
55
- return "task not found — wrong database file, or already purged?";
70
+ function timeoutDetail(task, key) {
71
+ if (!task) {
72
+ return key === null
73
+ ? "task not found — wrong database file, or already purged?"
74
+ : "no task under this key — never submitted, or already purged?";
75
+ }
56
76
  if (isQueued(task)) {
57
77
  const delayMs = task.run_at_ms - nowMs();
58
78
  if (task.attempt === 0 && delayMs <= 0) {
@@ -66,20 +86,30 @@ function timeoutDetail(task) {
66
86
  return "cancel requested, waiting for the handler to observe it";
67
87
  return `still running (attempt ${task.attempt}/${task.max_attempts})`;
68
88
  }
69
- /** wait/call did not reach a terminal status in time. The task keeps running.
70
- * `task` is the last snapshot wait() observed (null if get() found nothing), and
71
- * the message says what state it was stuck in — a queued-never-claimed task is
72
- * the classic first-run failure (no worker, no handler, wrong queue or file). */
89
+ /** wait/call did not reach a terminal status in time. The task keeps running, so
90
+ * `taskId` is the handle for picking the wait back up — `wait(err.taskId)`
91
+ * re-attaches to the same task from anywhere that can reach the store. `task` is
92
+ * the last snapshot wait() observed (null if the lookup found nothing), and the
93
+ * message says what state it was stuck in — a queued-never-claimed task is the
94
+ * classic first-run failure (no worker, no handler, wrong queue or file).
95
+ *
96
+ * `key` is set when the wait watched a key rather than an id; `taskId` is then
97
+ * the task the key pointed at, or the key itself when it pointed at nothing —
98
+ * there was no id to report. */
73
99
  export class TaskTimeout extends CairnQError {
74
100
  taskId;
75
101
  task;
102
+ key;
76
103
  constructor(taskId, opts = {}) {
104
+ const key = opts.key ?? null;
105
+ const subject = key === null ? `task ${taskId}` : `key ${key}`;
77
106
  super(opts.timeoutMs == null
78
- ? `task ${taskId} did not finish in time`
79
- : `task ${taskId} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null)}`);
107
+ ? `${subject} did not finish in time`
108
+ : `${subject} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null, key)}`);
80
109
  this.taskId = taskId;
81
110
  this.name = "TaskTimeout";
82
111
  this.task = opts.task ?? null;
112
+ this.key = key;
83
113
  }
84
114
  }
85
115
  /** A waited-on task ended in `failed`. The envelope's fields are unpacked onto the
@@ -110,6 +140,35 @@ export class TaskCanceled extends CairnQError {
110
140
  this.name = "TaskCanceled";
111
141
  }
112
142
  }
143
+ /**
144
+ * A heartbeat beat came back later than its own interval allowed.
145
+ *
146
+ * The heartbeat shares the event loop with the handlers whose leases it renews,
147
+ * so a handler that blocks the loop stops the renewal with it: the lease expires,
148
+ * the task is recovered and redelivered, and a second worker starts computing
149
+ * what the first is still computing — one task, billed twice, with no error
150
+ * anywhere. Nothing inside the blocked handler can observe that, which is why it
151
+ * is reported through `onError` alongside the other things the run loop survived.
152
+ *
153
+ * The cause is always synchronous work in a handler: a tight loop, a large
154
+ * JSON.parse, a `*Sync` filesystem or crypto call. Node has one loop and no way
155
+ * to preempt it — move the work to a worker thread, a child process, or an async
156
+ * API that yields.
157
+ */
158
+ export class EventLoopBlocked extends CairnQError {
159
+ lateMs;
160
+ intervalMs;
161
+ leaseMs;
162
+ constructor(lateMs, intervalMs, leaseMs) {
163
+ super(`heartbeat beat was ${lateMs}ms late (interval ${intervalMs}ms, lease ${leaseMs}ms): ` +
164
+ `the event loop was blocked long enough to miss a beat. Synchronous work in a ` +
165
+ `handler starves lease renewal — move it off the loop.`);
166
+ this.lateMs = lateMs;
167
+ this.intervalMs = intervalMs;
168
+ this.leaseMs = leaseMs;
169
+ this.name = "EventLoopBlocked";
170
+ }
171
+ }
113
172
  /** A worker write affected 0 rows: the lease expired and was reclaimed. */
114
173
  export class LostLease extends CairnQError {
115
174
  taskId;
@@ -161,3 +220,29 @@ export class TaskError extends CairnQError {
161
220
  });
162
221
  }
163
222
  }
223
+ /**
224
+ * Normalize anything that can end a task into [envelope, retryable].
225
+ *
226
+ * Shared by both ways a failure is recorded — a handler passing a reason to
227
+ * `ctx.fail`, and the worker classifying an error that ended an attempt — so the
228
+ * two cannot disagree about what a given error means. It lives here, beside the
229
+ * envelope constructors it dispatches to, rather than in the module that happens
230
+ * to expose it to handlers.
231
+ *
232
+ * A handler failing one task of a batch has a reason, not an exception object:
233
+ * `item.fail("no source records", { retryable: false })` is the shape the real
234
+ * code wants. A TaskError carries its own retryability and wins over the option;
235
+ * everything else takes the caller's. A ready envelope passes through, which is
236
+ * how the worker hands in the ones it composes itself.
237
+ */
238
+ export function asEnvelope(error, retryable) {
239
+ if (error instanceof TaskError)
240
+ return [error.envelope(), error.retryable];
241
+ if (error instanceof Error)
242
+ return [exceptionEnvelope(error, retryable), retryable];
243
+ if (typeof error === "object" && error !== null)
244
+ return [error, retryable];
245
+ // A bare reason is a TaskError in everything but the throwing, so let
246
+ // TaskError own its own type/code defaults rather than restating them.
247
+ return [new TaskError(String(error), { retryable }).envelope(), retryable];
248
+ }
package/dist/index.d.ts CHANGED
@@ -2,9 +2,12 @@ export { CairnQ } from "./client.js";
2
2
  export type { CallOptions, ClientOptions, SubmitOptions } from "./client.js";
3
3
  export { QueueDepthGate } from "./backpressure.js";
4
4
  export type { BackpressureOptions, QueueDepthLimit } from "./backpressure.js";
5
+ export { RetentionSweeper } from "./retention.js";
6
+ export type { RetentionOptions } from "./retention.js";
5
7
  export { Worker } from "./worker.js";
6
- export type { Handler, TypedHandler, WorkerOptions } from "./worker.js";
8
+ export type { BatchHandler, Handler, TypedHandler, WorkerOptions } from "./worker.js";
7
9
  export { TaskContext } from "./context.js";
10
+ export type { TaskContextOptions } from "./context.js";
8
11
  export { defineTask } from "./task.js";
9
12
  export type { TaskDef } from "./task.js";
10
13
  export { SQLiteStore } from "./store/sqlite.js";
@@ -13,4 +16,5 @@ export { TaskStore } from "./store/base.js";
13
16
  export type { ListInput, PurgeInput, SubmitInput, Conflict } from "./store/base.js";
14
17
  export type { Task, TaskStatus } from "./models.js";
15
18
  export { STATUSES, isTerminal, cancelRequested, isQueued, isRunning, isSucceeded, isFailed, isCanceled, } from "./models.js";
16
- export { CairnQError, AlreadyExists, QueueFull, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, SerializationError, } from "./errors.js";
19
+ export { CairnQError, AlreadyExists, QueueFull, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, EventLoopBlocked, ProtocolVersionMismatch, SerializationError, } from "./errors.js";
20
+ export type { FailReason } from "./errors.js";
package/dist/index.js CHANGED
@@ -1,5 +1,6 @@
1
1
  export { CairnQ } from "./client.js";
2
2
  export { QueueDepthGate } from "./backpressure.js";
3
+ export { RetentionSweeper } from "./retention.js";
3
4
  export { Worker } from "./worker.js";
4
5
  export { TaskContext } from "./context.js";
5
6
  export { defineTask } from "./task.js";
@@ -7,4 +8,4 @@ export { SQLiteStore } from "./store/sqlite.js";
7
8
  export { PostgresStore } from "./store/postgres.js";
8
9
  export { TaskStore } from "./store/base.js";
9
10
  export { STATUSES, isTerminal, cancelRequested, isQueued, isRunning, isSucceeded, isFailed, isCanceled, } from "./models.js";
10
- export { CairnQError, AlreadyExists, QueueFull, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, SerializationError, } from "./errors.js";
11
+ export { CairnQError, AlreadyExists, QueueFull, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, EventLoopBlocked, ProtocolVersionMismatch, SerializationError, } from "./errors.js";
@@ -0,0 +1,60 @@
1
+ import type { TaskStore } from "./store/base.js";
2
+ export interface RetentionOptions {
3
+ /**
4
+ * How long a terminal task is kept after it finished. Required: there is no
5
+ * safe default for how long someone else's results stay readable.
6
+ */
7
+ olderThanMs: number;
8
+ /** Time between sweeps. Default 3_600_000 (one hour). */
9
+ intervalMs?: number;
10
+ /** Rows deleted per statement while draining. Default 1_000. */
11
+ limit?: number;
12
+ /**
13
+ * Called for a sweep that threw. The next sweep runs on schedule regardless —
14
+ * a purge that failed because the database was busy is not a reason to stop
15
+ * retaining — so without this a store quietly stops being swept. Must not throw.
16
+ */
17
+ onError?: (err: unknown) => void;
18
+ }
19
+ /**
20
+ * Deletes terminal tasks on a schedule, for as long as the handle is open.
21
+ *
22
+ * `purge` exists because nothing else in CairnQ removes rows, and a queue whose
23
+ * payloads carry real data — an image, a document, a batch of embeddings — turns
24
+ * that into a disk leak measured in gigabytes per backfill. Every deployment
25
+ * that runs longer than a demo needs the sweep; leaving it to an external
26
+ * scheduler means the leak is the default and remembering is the opt-in.
27
+ *
28
+ * It sweeps in bounded batches with a yield between them, so draining a backlog
29
+ * that accumulated while nothing was sweeping stays a sequence of short writes
30
+ * rather than one long one — on SQLite that matters, since a long write holds
31
+ * the single write lock against every producer and worker on the file.
32
+ */
33
+ export declare class RetentionSweeper {
34
+ private readonly store;
35
+ private readonly opts;
36
+ /** Whether the scheduled loop is running. */
37
+ private active;
38
+ /** Set by stop(), so a drain in progress can cut itself short too. */
39
+ private stopping;
40
+ /** Resolves the current sleep early, so stop() need not wait out an interval. */
41
+ private wake;
42
+ /** The loop itself, awaited by stop() so no purge outlives the store. */
43
+ private loop;
44
+ private readonly intervalMs;
45
+ private readonly purgeInput;
46
+ constructor(store: TaskStore, opts: RetentionOptions);
47
+ start(): void;
48
+ /** Stop sweeping and wait for the sweep in flight, if any. */
49
+ stop(): Promise<void>;
50
+ private run;
51
+ /**
52
+ * Delete everything past the cutoff now, in bounded batches, and return how
53
+ * many rows went. The scheduled loop calls this; call it directly to drain on
54
+ * demand — after a backfill, or from a maintenance command.
55
+ */
56
+ sweep(): Promise<number>;
57
+ /** Sleep, interruptible by stop(). Unref'd: retention is housekeeping, and a
58
+ * pending sweep must never be the reason a process refuses to exit. */
59
+ private sleep;
60
+ }
@@ -0,0 +1,115 @@
1
+ /** Sweep every hour unless asked otherwise — often enough that a queue with a
2
+ * day of retention never carries more than an hour of extra rows, rare enough
3
+ * that the sweep is invisible next to the task traffic. */
4
+ const DEFAULT_INTERVAL_MS = 3_600_000;
5
+ /** Rows per purge statement. The same bound `purge` defaults to: big enough that
6
+ * a backlog drains in few statements, small enough that each is a short write. */
7
+ const DEFAULT_LIMIT = 1_000;
8
+ /**
9
+ * Deletes terminal tasks on a schedule, for as long as the handle is open.
10
+ *
11
+ * `purge` exists because nothing else in CairnQ removes rows, and a queue whose
12
+ * payloads carry real data — an image, a document, a batch of embeddings — turns
13
+ * that into a disk leak measured in gigabytes per backfill. Every deployment
14
+ * that runs longer than a demo needs the sweep; leaving it to an external
15
+ * scheduler means the leak is the default and remembering is the opt-in.
16
+ *
17
+ * It sweeps in bounded batches with a yield between them, so draining a backlog
18
+ * that accumulated while nothing was sweeping stays a sequence of short writes
19
+ * rather than one long one — on SQLite that matters, since a long write holds
20
+ * the single write lock against every producer and worker on the file.
21
+ */
22
+ export class RetentionSweeper {
23
+ store;
24
+ opts;
25
+ /** Whether the scheduled loop is running. */
26
+ active = false;
27
+ /** Set by stop(), so a drain in progress can cut itself short too. */
28
+ stopping = false;
29
+ /** Resolves the current sleep early, so stop() need not wait out an interval. */
30
+ wake = null;
31
+ /** The loop itself, awaited by stop() so no purge outlives the store. */
32
+ loop = null;
33
+ intervalMs;
34
+ purgeInput;
35
+ constructor(store, opts) {
36
+ this.store = store;
37
+ this.opts = opts;
38
+ if (!Number.isFinite(opts.olderThanMs) || opts.olderThanMs < 0) {
39
+ throw new Error(`retention.olderThanMs must be >= 0, got ${opts.olderThanMs}`);
40
+ }
41
+ this.intervalMs = opts.intervalMs ?? DEFAULT_INTERVAL_MS;
42
+ if (!Number.isFinite(this.intervalMs) || this.intervalMs < 1) {
43
+ throw new Error(`retention.intervalMs must be >= 1, got ${this.intervalMs}`);
44
+ }
45
+ this.purgeInput = { olderThanMs: opts.olderThanMs, limit: opts.limit ?? DEFAULT_LIMIT };
46
+ }
47
+ start() {
48
+ if (this.active)
49
+ return;
50
+ this.active = true;
51
+ this.stopping = false;
52
+ this.loop = this.run();
53
+ }
54
+ /** Stop sweeping and wait for the sweep in flight, if any. */
55
+ async stop() {
56
+ this.stopping = true;
57
+ this.active = false;
58
+ this.wake?.();
59
+ await this.loop;
60
+ this.loop = null;
61
+ }
62
+ async run() {
63
+ // Sleep first: a process that restarts often would otherwise purge on every
64
+ // boot, which is a write burst exactly when the store is busiest.
65
+ while (!this.stopping) {
66
+ await this.sleep(this.intervalMs);
67
+ if (this.stopping)
68
+ return;
69
+ try {
70
+ await this.sweep();
71
+ }
72
+ catch (err) {
73
+ try {
74
+ this.opts.onError?.(err);
75
+ }
76
+ catch {
77
+ // A reporting hook must never take the sweep down with it — the same
78
+ // rule the worker's onError follows.
79
+ }
80
+ }
81
+ }
82
+ }
83
+ /**
84
+ * Delete everything past the cutoff now, in bounded batches, and return how
85
+ * many rows went. The scheduled loop calls this; call it directly to drain on
86
+ * demand — after a backfill, or from a maintenance command.
87
+ */
88
+ async sweep() {
89
+ const limit = this.purgeInput.limit;
90
+ let deleted = 0;
91
+ for (;;) {
92
+ const ids = await this.store.purge(this.purgeInput);
93
+ deleted += ids.length;
94
+ if (ids.length < limit || this.stopping)
95
+ return deleted;
96
+ // Hand the loop back between batches: a large drain must not starve the
97
+ // submits and claims sharing this process.
98
+ await this.sleep(0);
99
+ }
100
+ }
101
+ /** Sleep, interruptible by stop(). Unref'd: retention is housekeeping, and a
102
+ * pending sweep must never be the reason a process refuses to exit. */
103
+ sleep(ms) {
104
+ return new Promise((resolve) => {
105
+ const timer = setTimeout(resolve, ms);
106
+ timer.unref?.();
107
+ this.wake = () => {
108
+ clearTimeout(timer);
109
+ resolve();
110
+ };
111
+ }).finally(() => {
112
+ this.wake = null;
113
+ });
114
+ }
115
+ }