cairnq 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +69 -4
  2. package/dist/_protocol/migrations/postgres/0007_purge_status_index.sql +11 -0
  3. package/dist/_protocol/migrations/sqlite/0007_purge_status_index.sql +11 -0
  4. package/dist/_protocol/sql/postgres/get_status.sql +7 -0
  5. package/dist/_protocol/sql/postgres/get_status_by_key.sql +7 -0
  6. package/dist/_protocol/sql/postgres/purge.sql +8 -1
  7. package/dist/_protocol/sql/sqlite/get_status.sql +6 -0
  8. package/dist/_protocol/sql/sqlite/get_status_by_key.sql +7 -0
  9. package/dist/_protocol/sql/sqlite/purge.sql +7 -1
  10. package/dist/client.d.ts +32 -15
  11. package/dist/client.js +38 -12
  12. package/dist/context.d.ts +25 -0
  13. package/dist/context.js +33 -0
  14. package/dist/errors.d.ts +4 -2
  15. package/dist/errors.js +7 -4
  16. package/dist/index.d.ts +9 -5
  17. package/dist/index.js +4 -1
  18. package/dist/models.d.ts +14 -2
  19. package/dist/models.js +12 -1
  20. package/dist/retention.d.ts +15 -3
  21. package/dist/retention.js +36 -14
  22. package/dist/store/base.d.ts +118 -1
  23. package/dist/store/base.js +122 -20
  24. package/dist/store/pg-executor.d.ts +77 -0
  25. package/dist/store/pg-executor.js +26 -0
  26. package/dist/store/pg-pool.d.ts +16 -0
  27. package/dist/store/pg-pool.js +147 -0
  28. package/dist/store/postgres.d.ts +41 -13
  29. package/dist/store/postgres.js +135 -147
  30. package/dist/store/sqlite.js +20 -2
  31. package/dist/wait.d.ts +9 -4
  32. package/dist/wait.js +27 -13
  33. package/dist/worker.d.ts +6 -3
  34. package/dist/worker.js +6 -5
  35. package/package.json +6 -4
  36. package/src/client.ts +60 -21
  37. package/src/context.ts +37 -0
  38. package/src/errors.ts +7 -4
  39. package/src/index.ts +16 -4
  40. package/src/models.ts +24 -3
  41. package/src/retention.ts +45 -15
  42. package/src/store/base.ts +204 -9
  43. package/src/store/pg-executor.ts +90 -0
  44. package/src/store/pg-pool.ts +156 -0
  45. package/src/store/postgres.ts +144 -141
  46. package/src/store/sqlite.ts +23 -2
  47. package/src/wait.ts +35 -14
  48. package/src/worker.ts +8 -6
@@ -1,9 +1,27 @@
1
1
  import { mkdirSync } from "node:fs";
2
2
  import { dirname, resolve } from "node:path";
3
- import Database from "better-sqlite3";
3
+ import { createRequire } from "node:module";
4
4
  import { nowMs } from "../ids.js";
5
5
  import { loadMigrations, loadStatements } from "../sql.js";
6
6
  import { checkProtocolVersion, COMMENT, statementParams, TaskStore, } from "./base.js";
7
+ // `better-sqlite3` is an optional dependency, matching `pg` on the Postgres side:
8
+ // a Postgres-only deployment should not have to build a native module it never
9
+ // loads, and importing this file must not pull one in. Required (not imported)
10
+ // because ensure() is synchronous — the open path applies migrations and cannot
11
+ // await — and createRequire gives a synchronous load from ESM.
12
+ const require = createRequire(import.meta.url);
13
+ let sqliteModule = null;
14
+ function loadSqlite() {
15
+ if (sqliteModule)
16
+ return sqliteModule;
17
+ try {
18
+ sqliteModule = require("better-sqlite3");
19
+ }
20
+ catch {
21
+ throw new Error("SQLiteStore requires the 'better-sqlite3' package — install it (e.g. `npm i better-sqlite3`)");
22
+ }
23
+ return sqliteModule;
24
+ }
7
25
  const WAL_RETRY_DELAY_MS = 50;
8
26
  const WAL_RETRY_BUDGET_MS = 5_000;
9
27
  const BUSY_RETRY_BASE_MS = 1;
@@ -221,7 +239,7 @@ export class SQLiteStore extends TaskStore {
221
239
  const memory = isMemory(this.path);
222
240
  if (!memory)
223
241
  mkdirSync(dirname(this.path), { recursive: true });
224
- const db = new Database(this.path);
242
+ const db = new (loadSqlite())(this.path);
225
243
  // Only the synchronous part of the open path gets a real busy_timeout: the WAL
226
244
  // switch and the migrations cannot await a retry. See the class comment.
227
245
  db.pragma(`busy_timeout = ${this.busyBudgetMs}`);
package/dist/wait.d.ts CHANGED
@@ -1,10 +1,15 @@
1
1
  import { type Task } from "./models.js";
2
2
  import type { TaskStore } from "./store/base.js";
3
+ export declare const DEFAULT_WAIT_TIMEOUT_MS = 30000;
3
4
  export declare const DEFAULT_POLL_MS = 100;
4
5
  export declare const MAX_POLL_MS = 500;
5
6
  export interface PollOptions {
6
7
  timeoutMs: number;
8
+ /** The first poll interval (default 100). */
7
9
  pollMs?: number;
10
+ /** Ceiling the poll interval backs off to (default 500). Worth raising for a
11
+ * task known to take minutes — fewer reads — or lowering when shaving the
12
+ * average half-interval of completion-detection latency matters. */
8
13
  maxPollMs?: number;
9
14
  }
10
15
  /**
@@ -16,14 +21,14 @@ export interface PollOptions {
16
21
  * Math.floor(1 * 1.5) === 1 would otherwise never grow past 1.
17
22
  */
18
23
  export declare function nextPollMs(current: number, maxMs: number): number;
19
- /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
20
- * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
21
- * it backs off towards `maxPollMs`. */
24
+ /** Poll the task's status until terminal or timeout. Returns the terminal Task
25
+ * (any status). Throws TaskTimeout, leaving the task running. `pollMs` is the
26
+ * *first* interval; it backs off towards `maxPollMs`. */
22
27
  export declare function pollWait(store: TaskStore, taskId: string, opts: PollOptions): Promise<Task>;
23
28
  /**
24
29
  * The same wait, following a key instead of an id.
25
30
  *
26
- * The key is re-resolved on every read, because that is what a key means: a
31
+ * The key is re-resolved on every probe, because that is what a key means: a
27
32
  * pointer to the task that is *current* under it. A `replace` landing mid-wait
28
33
  * moves the wait onto the new task rather than reporting the cancellation of the
29
34
  * old one, and a key that points at nothing yet is simply not finished — it
package/dist/wait.js CHANGED
@@ -1,6 +1,7 @@
1
1
  import { TaskTimeout } from "./errors.js";
2
2
  import { nowMs } from "./ids.js";
3
3
  import { isTerminal } from "./models.js";
4
+ export const DEFAULT_WAIT_TIMEOUT_MS = 30_000;
4
5
  export const DEFAULT_POLL_MS = 100;
5
6
  export const MAX_POLL_MS = 500;
6
7
  const GROWTH = 1.5;
@@ -17,36 +18,49 @@ export function nextPollMs(current, maxMs) {
17
18
  return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
18
19
  }
19
20
  /**
20
- * Poll `read` until it yields a terminal task, or the timeout elapses.
21
+ * Poll `probe` until it reports a terminal status, then return the full task
22
+ * via `read`; or throw once the timeout elapses.
23
+ *
24
+ * The loop's repeated read is the status-only `probe` (see get_status.sql): a
25
+ * waiting caller asks nothing but "is it finished yet", and re-reading the whole
26
+ * row would drag the payload back — and re-parse it — on every beat for the life
27
+ * of the wait. The full row is read once, when the probe turns terminal or, on
28
+ * the timeout beat, for the error's snapshot. Between the probe and that read
29
+ * the row can vanish (purge) or the key repoint (`replace`); a read that comes
30
+ * back empty or non-terminal is simply not finished, and the loop keeps polling.
21
31
  *
22
32
  * `wake` is what the loop sleeps on between reads: a store with a push channel
23
- * (Postgres) cuts it short when the task goes terminal, but the re-read is the
33
+ * (Postgres) cuts it short when the task goes terminal, but the re-probe is the
24
34
  * source of truth either way, so a plain sleep is always a correct answer.
25
35
  */
26
- async function poll(read, wake, subject, key, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS }) {
36
+ async function poll(probe, read, wake, subject, key, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS }) {
27
37
  const deadline = nowMs() + timeoutMs;
28
38
  let interval = pollMs;
29
39
  for (;;) {
30
- const task = await read();
40
+ const ref = await probe();
41
+ const remaining = deadline - nowMs();
42
+ // The one full-read site: when the probe says finished, or on the timeout
43
+ // beat for the error's stuck-in-what-state snapshot. No ref means no row,
44
+ // so there is nothing for a read to add to either case.
45
+ const task = ref && (isTerminal(ref) || remaining <= 0) ? await read() : null;
31
46
  if (task && isTerminal(task))
32
47
  return task;
33
- const remaining = deadline - nowMs();
34
48
  if (remaining <= 0)
35
- throw new TaskTimeout(task?.id ?? subject, { timeoutMs, task, key });
36
- await wake(task, Math.min(interval, remaining));
49
+ throw new TaskTimeout(ref?.id ?? subject, { timeoutMs, task, key });
50
+ await wake(ref, Math.min(interval, remaining));
37
51
  interval = nextPollMs(interval, maxPollMs);
38
52
  }
39
53
  }
40
- /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
41
- * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
42
- * it backs off towards `maxPollMs`. */
54
+ /** Poll the task's status until terminal or timeout. Returns the terminal Task
55
+ * (any status). Throws TaskTimeout, leaving the task running. `pollMs` is the
56
+ * *first* interval; it backs off towards `maxPollMs`. */
43
57
  export function pollWait(store, taskId, opts) {
44
- return poll(() => store.get(taskId), (_task, ms) => store.taskDoneWake(taskId, ms), taskId, null, opts);
58
+ return poll(() => store.getStatus(taskId), () => store.get(taskId), (_ref, ms) => store.taskDoneWake(taskId, ms), taskId, null, opts);
45
59
  }
46
60
  /**
47
61
  * The same wait, following a key instead of an id.
48
62
  *
49
- * The key is re-resolved on every read, because that is what a key means: a
63
+ * The key is re-resolved on every probe, because that is what a key means: a
50
64
  * pointer to the task that is *current* under it. A `replace` landing mid-wait
51
65
  * moves the wait onto the new task rather than reporting the cancellation of the
52
66
  * old one, and a key that points at nothing yet is simply not finished — it
@@ -57,5 +71,5 @@ export function pollWait(store, taskId, opts) {
57
71
  * plain sleeps; once it resolves, the store's push channel applies as usual.
58
72
  */
59
73
  export function pollWaitByKey(store, key, opts) {
60
- return poll(() => store.getByKey(key), (task, ms) => (task ? store.taskDoneWake(task.id, ms) : sleep(ms)), key, key, opts);
74
+ return poll(() => store.getStatusByKey(key), () => store.getByKey(key), (ref, ms) => (ref ? store.taskDoneWake(ref.id, ms) : sleep(ms)), key, key, opts);
61
75
  }
package/dist/worker.d.ts CHANGED
@@ -1,5 +1,6 @@
1
1
  import type { BackpressureOptions } from "./backpressure.js";
2
2
  import { TaskContext } from "./context.js";
3
+ import type { PgExecutor } from "./store/pg-executor.js";
3
4
  import type { TaskStore } from "./store/base.js";
4
5
  import { type TaskDef } from "./task.js";
5
6
  export { retryDelayMs, DEFAULT_RETRY_BACKOFF_MS, DEFAULT_RETRY_BACKOFF_MAX_MS } from "./backoff.js";
@@ -130,11 +131,13 @@ export declare class Worker {
130
131
  queues?: string[];
131
132
  busyTimeoutMs?: number;
132
133
  }): Worker;
133
- /** Multi-host backend. `dsn` is a libpq connection string; requires the
134
- * optional `pg` package. */
135
- static postgres(dsn: string, opts?: WorkerOptions & {
134
+ /** Multi-host backend. `source` is a libpq connection string — which requires
135
+ * the optional `pg` package — or a PgExecutor over a driver the application
136
+ * already runs, which cairnq then shares instead of opening a second pool. */
137
+ static postgres(source: string | PgExecutor, opts?: WorkerOptions & {
136
138
  queues?: string[];
137
139
  max?: number;
140
+ schema?: string;
138
141
  }): Worker;
139
142
  get id(): string;
140
143
  task(handler: Handler): this;
package/dist/worker.js CHANGED
@@ -133,11 +133,12 @@ export class Worker {
133
133
  worker.ownsStore = true;
134
134
  return worker;
135
135
  }
136
- /** Multi-host backend. `dsn` is a libpq connection string; requires the
137
- * optional `pg` package. */
138
- static postgres(dsn, opts = {}) {
139
- const { queues = ["default"], max, ...rest } = opts;
140
- const worker = new Worker(new PostgresStore(dsn, { max }), queues, rest);
136
+ /** Multi-host backend. `source` is a libpq connection string — which requires
137
+ * the optional `pg` package — or a PgExecutor over a driver the application
138
+ * already runs, which cairnq then shares instead of opening a second pool. */
139
+ static postgres(source, opts = {}) {
140
+ const { queues = ["default"], max, schema, ...rest } = opts;
141
+ const worker = new Worker(new PostgresStore(source, { max, schema }), queues, rest);
141
142
  worker.ownsStore = true;
142
143
  return worker;
143
144
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "cairnq",
3
- "version": "0.7.0",
3
+ "version": "0.9.0",
4
4
  "description": "SQLite-first, cross-language, storage-centered durable task runtime",
5
5
  "license": "MIT",
6
6
  "author": "Jannchie <jannchie@gmail.com>",
@@ -39,19 +39,21 @@
39
39
  "test": "vitest run",
40
40
  "typecheck": "tsc -p tsconfig.json --noEmit"
41
41
  },
42
- "dependencies": {
43
- "better-sqlite3": "^13.0.2"
44
- },
45
42
  "peerDependencies": {
43
+ "better-sqlite3": "^13.0.2",
46
44
  "pg": "^8.13.0"
47
45
  },
48
46
  "peerDependenciesMeta": {
47
+ "better-sqlite3": {
48
+ "optional": true
49
+ },
49
50
  "pg": {
50
51
  "optional": true
51
52
  }
52
53
  },
53
54
  "devDependencies": {
54
55
  "@types/better-sqlite3": "^7.6.11",
56
+ "better-sqlite3": "^13.0.2",
55
57
  "@types/node": "^22.7.0",
56
58
  "@types/pg": "^8.11.10",
57
59
  "pg": "^8.13.0",
package/src/client.ts CHANGED
@@ -1,17 +1,27 @@
1
1
  import type { BackpressureOptions } from "./backpressure.js";
2
2
  import { type RetentionOptions, RetentionSweeper } from "./retention.js";
3
3
  import { TaskCanceled, TaskFailed } from "./errors.js";
4
- import { isFailed, isSucceeded, type Task, type TaskStatus } from "./models.js";
4
+ import { isFailed, isSucceeded, type Task, type TaskRef, type TaskStatus } from "./models.js";
5
5
  import { SQLiteStore } from "./store/sqlite.js";
6
6
  import { PostgresStore } from "./store/postgres.js";
7
- import type { ListInput, PurgeInput, SubmitInput, TaskStore } from "./store/base.js";
7
+ import type { PgExecutor } from "./store/pg-executor.js";
8
+ import type {
9
+ ListInput,
10
+ PurgeInput,
11
+ SubmitInput,
12
+ TaskStore,
13
+ WatchOptions,
14
+ WatchSignal,
15
+ } from "./store/base.js";
8
16
  import { type TaskDef, taskName } from "./task.js";
9
- import { pollWait, pollWaitByKey } from "./wait.js";
17
+ import { DEFAULT_WAIT_TIMEOUT_MS, type PollOptions, pollWait, pollWaitByKey } from "./wait.js";
10
18
 
11
19
  export type SubmitOptions = Omit<SubmitInput, "name" | "payload">;
12
- export interface CallOptions extends SubmitOptions {
20
+ /** The wait loop's knobs with the timeout optional (default 30s) — the public
21
+ * face of PollOptions, whose comments document each knob. */
22
+ export type WaitOptions = Partial<PollOptions>;
23
+ export interface CallOptions extends SubmitOptions, Omit<WaitOptions, "timeoutMs"> {
13
24
  waitTimeoutMs?: number;
14
- pollMs?: number;
15
25
  }
16
26
 
17
27
  /** Options this handle configures on the store it wraps, rather than the
@@ -51,11 +61,16 @@ export class CairnQ {
51
61
  return new CairnQ(new SQLiteStore(path, { busyTimeoutMs }), client);
52
62
  }
53
63
 
54
- /** Multi-host backend. `dsn` is a libpq connection string; requires the
55
- * optional `pg` package. */
56
- static postgres(dsn: string, opts: { max?: number } & ClientOptions = {}): CairnQ {
57
- const { max, ...client } = opts;
58
- return new CairnQ(new PostgresStore(dsn, { max }), client);
64
+ /** Multi-host backend. `source` is a libpq connection string — which requires
65
+ * the optional `pg` package — or a PgExecutor over a driver the application
66
+ * already runs (an ORM's pool, say), which cairnq then shares instead of
67
+ * opening a second one. */
68
+ static postgres(
69
+ source: string | PgExecutor,
70
+ opts: { max?: number; schema?: string } & ClientOptions = {},
71
+ ): CairnQ {
72
+ const { max, schema, ...client } = opts;
73
+ return new CairnQ(new PostgresStore(source, { max, schema }), client);
59
74
  }
60
75
 
61
76
  get store(): TaskStore {
@@ -99,6 +114,17 @@ export class CairnQ {
99
114
  return this._store.getByKey(key);
100
115
  }
101
116
 
117
+ /** The status-only probe wait polls on: id + status, no payload. Public for
118
+ * the same reason it exists — a dashboard or poller that only asks "is it
119
+ * finished yet" should not drag the payload back per ask. */
120
+ getStatus(taskId: string): Promise<TaskRef | null> {
121
+ return this._store.getStatus(taskId);
122
+ }
123
+
124
+ getStatusByKey(key: string): Promise<TaskRef | null> {
125
+ return this._store.getStatusByKey(key);
126
+ }
127
+
102
128
  list(input?: ListInput): Promise<Task[]> {
103
129
  return this._store.list(input);
104
130
  }
@@ -133,16 +159,29 @@ export class CairnQ {
133
159
  return this._store.stats();
134
160
  }
135
161
 
162
+ /**
163
+ * Call `onSignal` when the tasks on `queues` may have changed. Returns an
164
+ * unsubscribe.
165
+ *
166
+ * Notify-accelerated polling, not an event log: a signal means "re-read now",
167
+ * and `stats()` / `list()` / `get()` are where the truth is. On Postgres an
168
+ * idle watch costs nothing and signals land in milliseconds; everywhere else
169
+ * the timer alone still delivers, so the same consumer code is correct either
170
+ * way. See TaskStore.watch for the full contract.
171
+ */
172
+ watch(opts: WatchOptions, onSignal: (signal: WatchSignal) => void): () => void {
173
+ return this._store.watch(opts, onSignal);
174
+ }
175
+
136
176
  /** Wait for a task to finish. Resolves with the terminal Task (any status);
137
177
  * throws TaskTimeout without stopping the task, so `wait(err.taskId)` picks the
138
178
  * same wait back up — from another process, or after a longer deadline. */
139
- wait(
140
- taskId: string,
141
- opts: { timeoutMs?: number; pollMs?: number } = {},
142
- ): Promise<Task> {
179
+ wait(taskId: string, opts: WaitOptions = {}): Promise<Task> {
180
+ // `??`, not a spread default: a caller forwarding `timeoutMs: undefined`
181
+ // (call() does) must still get the default, and a spread would override it.
143
182
  return pollWait(this._store, taskId, {
144
- timeoutMs: opts.timeoutMs ?? 30_000,
145
- pollMs: opts.pollMs,
183
+ ...opts,
184
+ timeoutMs: opts.timeoutMs ?? DEFAULT_WAIT_TIMEOUT_MS,
146
185
  });
147
186
  }
148
187
 
@@ -151,10 +190,10 @@ export class CairnQ {
151
190
  * that held it is gone. Re-resolves the key on each poll, so a `replace`
152
191
  * landing mid-wait moves the wait onto the new task, and a key with no task
153
192
  * yet is waited for rather than rejected. */
154
- waitByKey(key: string, opts: { timeoutMs?: number; pollMs?: number } = {}): Promise<Task> {
193
+ waitByKey(key: string, opts: WaitOptions = {}): Promise<Task> {
155
194
  return pollWaitByKey(this._store, key, {
156
- timeoutMs: opts.timeoutMs ?? 30_000,
157
- pollMs: opts.pollMs,
195
+ ...opts,
196
+ timeoutMs: opts.timeoutMs ?? DEFAULT_WAIT_TIMEOUT_MS,
158
197
  });
159
198
  }
160
199
 
@@ -168,9 +207,9 @@ export class CairnQ {
168
207
  async call(name: string, payload?: unknown, opts?: CallOptions): Promise<unknown>;
169
208
  async call<P, R>(task: TaskDef<P, R>, payload?: P, opts?: CallOptions): Promise<R>;
170
209
  async call(task: string | TaskDef, payload?: unknown, opts: CallOptions = {}): Promise<unknown> {
171
- const { waitTimeoutMs = 30_000, pollMs, ...submit } = opts;
210
+ const { waitTimeoutMs, pollMs, maxPollMs, ...submit } = opts;
172
211
  const created = await this.submit(taskName(task), payload, submit);
173
- const final = await pollWait(this._store, created.id, { timeoutMs: waitTimeoutMs, pollMs });
212
+ const final = await this.wait(created.id, { timeoutMs: waitTimeoutMs, pollMs, maxPollMs });
174
213
  if (isSucceeded(final)) return final.result;
175
214
  if (isFailed(final)) throw new TaskFailed(final.error);
176
215
  throw new TaskCanceled(final.id);
package/src/context.ts CHANGED
@@ -3,6 +3,7 @@ import { asEnvelope, type FailReason, LostLease } from "./errors.js";
3
3
  import { cancelRequested, type Task } from "./models.js";
4
4
  import type { SubmitOptions } from "./client.js";
5
5
  import type { TaskStore } from "./store/base.js";
6
+ import type { PgSession } from "./store/pg-executor.js";
6
7
  import { type TaskDef, taskName } from "./task.js";
7
8
  import { pollWait } from "./wait.js";
8
9
 
@@ -201,6 +202,42 @@ export class TaskContext {
201
202
  return task;
202
203
  }
203
204
 
205
+ /**
206
+ * Finalize this task as succeeded, committing the caller's own writes in the
207
+ * SAME transaction as the settlement. Whatever `write` returns becomes the
208
+ * task's result.
209
+ *
210
+ * await ctx.succeedIn(async (session) => {
211
+ * await db.withSession(session).insert(pages).values(rendered)
212
+ * return { pages: rendered.length }
213
+ * })
214
+ *
215
+ * The alternative — write the rows, then settle — has a window between the two
216
+ * commits where the work is durable but the task still reads as running. A
217
+ * crash there re-runs the whole task, which for a render or an ingest means
218
+ * recomputing it, and for non-idempotent work means doing it twice.
219
+ *
220
+ * `session` is the driver's, so this needs a Postgres store built on a
221
+ * PgExecutor the application shares with its own driver; anything else throws.
222
+ * If the settlement finds the lease gone, `write`'s work is rolled back with
223
+ * it and LostLease is raised. Returns null if this task was already settled.
224
+ *
225
+ * `write` may be replayed if the backend retries the transaction — derive
226
+ * nothing inside it that cannot be derived twice.
227
+ */
228
+ async succeedIn<T>(write: (session: PgSession) => Promise<T>): Promise<Task | null> {
229
+ if (this.isSettled) return null;
230
+ const task = await this.owned(async () => {
231
+ const { task } = await this.store.completeIn<PgSession, T>(
232
+ { taskId: this.task.id, workerId: this.workerId },
233
+ write,
234
+ );
235
+ return task;
236
+ });
237
+ this.markSettled();
238
+ return task;
239
+ }
240
+
204
241
  /**
205
242
  * Finalize this task as failed, now. `error` may be a string reason, an Error,
206
243
  * a TaskError (which carries its own retryability), or a ready envelope.
package/src/errors.ts CHANGED
@@ -170,10 +170,12 @@ export class TaskCanceled extends CairnQError {
170
170
  * anywhere. Nothing inside the blocked handler can observe that, which is why it
171
171
  * is reported through `onError` alongside the other things the run loop survived.
172
172
  *
173
- * The cause is always synchronous work in a handler: a tight loop, a large
173
+ * The usual cause is synchronous work in a handler: a tight loop, a large
174
174
  * JSON.parse, a `*Sync` filesystem or crypto call. Node has one loop and no way
175
175
  * to preempt it — move the work to a worker thread, a child process, or an async
176
- * API that yields.
176
+ * API that yields. The other cause is a worker simply oversubscribed for its
177
+ * `leaseMs` — nothing is blocking, there is just more work than turns — which the
178
+ * same report covers, because the lease is at equal risk either way.
177
179
  */
178
180
  export class EventLoopBlocked extends CairnQError {
179
181
  constructor(
@@ -183,8 +185,9 @@ export class EventLoopBlocked extends CairnQError {
183
185
  ) {
184
186
  super(
185
187
  `heartbeat beat was ${lateMs}ms late (interval ${intervalMs}ms, lease ${leaseMs}ms): ` +
186
- `the event loop was blocked long enough to miss a beat. Synchronous work in a ` +
187
- `handler starves lease renewal — move it off the loop.`,
188
+ `the event loop was blocked long enough to miss a beat, so this worker's leases ` +
189
+ `are at risk. Usually synchronous work in a handler (move it off the loop); ` +
190
+ `otherwise the worker is oversubscribed for its leaseMs.`,
188
191
  );
189
192
  this.name = "EventLoopBlocked";
190
193
  }
package/src/index.ts CHANGED
@@ -1,9 +1,9 @@
1
1
  export { CairnQ } from "./client.js";
2
- export type { CallOptions, ClientOptions, SubmitOptions } from "./client.js";
2
+ export type { CallOptions, ClientOptions, SubmitOptions, WaitOptions } from "./client.js";
3
3
  export { QueueDepthGate } from "./backpressure.js";
4
4
  export type { BackpressureOptions, QueueDepthLimit } from "./backpressure.js";
5
5
  export { RetentionSweeper } from "./retention.js";
6
- export type { RetentionOptions } from "./retention.js";
6
+ export type { RetentionCutoffs, RetentionOptions } from "./retention.js";
7
7
  export { Worker } from "./worker.js";
8
8
  export type { BatchHandler, Handler, TypedHandler, WorkerOptions } from "./worker.js";
9
9
  export { TaskContext } from "./context.js";
@@ -12,12 +12,24 @@ export { defineTask } from "./task.js";
12
12
  export type { TaskDef } from "./task.js";
13
13
  export { SQLiteStore } from "./store/sqlite.js";
14
14
  export { PostgresStore } from "./store/postgres.js";
15
+ export { ListenUnavailable } from "./store/pg-executor.js";
16
+ export type { PgExecutor, PgSession, Row } from "./store/pg-executor.js";
17
+ export { createPoolExecutor } from "./store/pg-pool.js";
15
18
  export { TaskStore } from "./store/base.js";
16
- export type { ListInput, PurgeInput, SubmitInput, Conflict } from "./store/base.js";
17
- export type { Task, TaskStatus } from "./models.js";
19
+ export type {
20
+ ListInput,
21
+ PurgeInput,
22
+ SubmitInput,
23
+ Conflict,
24
+ WatchOptions,
25
+ WatchSignal,
26
+ } from "./store/base.js";
27
+ export { DEFAULT_WATCH_POLL_MS } from "./store/base.js";
28
+ export type { Task, TaskRef, TaskStatus, TerminalStatus } from "./models.js";
18
29
  export {
19
30
  STATUSES,
20
31
  isTerminal,
32
+ isTerminalStatus,
21
33
  cancelRequested,
22
34
  isQueued,
23
35
  isRunning,
package/src/models.ts CHANGED
@@ -31,8 +31,28 @@ export interface Task {
31
31
  completed_at_ms: number | null;
32
32
  }
33
33
 
34
+ /** The id + status pair the wait loop polls on (see get_status.sql) — a probe,
35
+ * not a snapshot: everything else about the task is deliberately not read. */
36
+ export interface TaskRef {
37
+ id: string;
38
+ status: TaskStatus;
39
+ }
40
+
34
41
  const JSON_COLUMNS = ["payload", "result", "error", "metadata"] as const;
35
- export const TERMINAL: TaskStatus[] = ["succeeded", "failed", "canceled"];
42
+ // As a const tuple so TerminalStatus derives from it — the same declare-once
43
+ // pattern as STATUSES/TaskStatus above.
44
+ export const TERMINAL = ["succeeded", "failed", "canceled"] as const;
45
+ export type TerminalStatus = (typeof TERMINAL)[number];
46
+
47
+ export function isTerminalStatus(status: TaskStatus): status is TerminalStatus {
48
+ return (TERMINAL as readonly TaskStatus[]).includes(status);
49
+ }
50
+
51
+ /** Map a probe row (see get_status.sql) to a TaskRef — the ref twin of
52
+ * rowToTask, so the row shape stays models' knowledge alone. */
53
+ export function rowToRef(row: Record<string, unknown>): TaskRef {
54
+ return { id: row.id as string, status: row.status as TaskStatus };
55
+ }
36
56
 
37
57
  export function rowToTask(row: Record<string, unknown>): Task {
38
58
  const t: Record<string, unknown> = { ...row };
@@ -46,8 +66,9 @@ export function rowToTask(row: Record<string, unknown>): Task {
46
66
  return t as unknown as Task;
47
67
  }
48
68
 
49
- export function isTerminal(task: Task): boolean {
50
- return TERMINAL.includes(task.status);
69
+ /** Accepts anything carrying a status — a Task or a TaskRef probe. */
70
+ export function isTerminal(task: Pick<Task, "status">): boolean {
71
+ return isTerminalStatus(task.status);
51
72
  }
52
73
 
53
74
  export function cancelRequested(task: Task): boolean {
package/src/retention.ts CHANGED
@@ -1,4 +1,5 @@
1
- import type { PurgeInput, TaskStore } from "./store/base.js";
1
+ import type { TaskStatus, TerminalStatus } from "./models.js";
2
+ import { validatePurgeInput, type PurgeInput, type TaskStore } from "./store/base.js";
2
3
 
3
4
  /** Sweep every hour unless asked otherwise — often enough that a queue with a
4
5
  * day of retention never carries more than an hour of extra rows, rare enough
@@ -8,12 +9,21 @@ const DEFAULT_INTERVAL_MS = 3_600_000;
8
9
  * a backlog drains in few statements, small enough that each is a short write. */
9
10
  const DEFAULT_LIMIT = 1_000;
10
11
 
12
+ /** Per-status cutoffs. A status left out is never swept — granular retention is
13
+ * an explicit statement of what may go, not a default for what wasn't named. */
14
+ export type RetentionCutoffs = Partial<Record<TerminalStatus, number>>;
15
+
11
16
  export interface RetentionOptions {
12
17
  /**
13
18
  * How long a terminal task is kept after it finished. Required: there is no
14
19
  * safe default for how long someone else's results stay readable.
20
+ *
21
+ * A number keeps every terminal status the same time. Retention needs are
22
+ * often tiered — a succeeded row is spent once its result is consumed, while
23
+ * a failed one is worth keeping for diagnosis — so a per-status map sets a
24
+ * cutoff per status instead: `{ succeeded: 300_000, failed: 86_400_000 }`.
15
25
  */
16
- olderThanMs: number;
26
+ olderThanMs: number | RetentionCutoffs;
17
27
  /** Time between sweeps. Default 3_600_000 (one hour). */
18
28
  intervalMs?: number;
19
29
  /** Rows deleted per statement while draining. Default 1_000. */
@@ -50,20 +60,37 @@ export class RetentionSweeper {
50
60
  /** The loop itself, awaited by stop() so no purge outlives the store. */
51
61
  private loop: Promise<void> | null = null;
52
62
  private readonly intervalMs: number;
53
- private readonly purgeInput: PurgeInput;
63
+ /** Rows per purge statement while draining — see DEFAULT_LIMIT. */
64
+ private readonly limit: number;
65
+ /** One purge per cutoff: a lone entry for a number, one per status for a map. */
66
+ private readonly purgeInputs: PurgeInput[];
54
67
 
55
68
  constructor(
56
69
  private readonly store: TaskStore,
57
70
  private readonly opts: RetentionOptions,
58
71
  ) {
59
- if (!Number.isFinite(opts.olderThanMs) || opts.olderThanMs < 0) {
60
- throw new Error(`retention.olderThanMs must be >= 0, got ${opts.olderThanMs}`);
61
- }
62
72
  this.intervalMs = opts.intervalMs ?? DEFAULT_INTERVAL_MS;
63
73
  if (!Number.isFinite(this.intervalMs) || this.intervalMs < 1) {
64
74
  throw new Error(`retention.intervalMs must be >= 1, got ${this.intervalMs}`);
65
75
  }
66
- this.purgeInput = { olderThanMs: opts.olderThanMs, limit: opts.limit ?? DEFAULT_LIMIT };
76
+ this.limit = opts.limit ?? DEFAULT_LIMIT;
77
+ const cutoffs: [TaskStatus | undefined, number][] =
78
+ typeof opts.olderThanMs === "number"
79
+ ? [[undefined, opts.olderThanMs]]
80
+ : (Object.entries(opts.olderThanMs) as [TaskStatus, number][]);
81
+ // An empty map retains nothing and sweeps nothing — almost certainly a bug
82
+ // upstream of this call, so refuse it rather than silently never purging.
83
+ if (!cutoffs.length) {
84
+ throw new Error("retention.olderThanMs must name at least one status");
85
+ }
86
+ this.purgeInputs = cutoffs.map(([status, ms]) => ({
87
+ olderThanMs: ms,
88
+ status,
89
+ limit: this.limit,
90
+ }));
91
+ // Fail fast on the store's own purge rules (terminal status, cutoff >= 0):
92
+ // the sweep runs an hour from now, and its errors only surface via onError.
93
+ for (const input of this.purgeInputs) validatePurgeInput(input);
67
94
  }
68
95
 
69
96
  start(): void {
@@ -107,16 +134,19 @@ export class RetentionSweeper {
107
134
  * demand — after a backfill, or from a maintenance command.
108
135
  */
109
136
  async sweep(): Promise<number> {
110
- const limit = this.purgeInput.limit as number;
111
137
  let deleted = 0;
112
- for (;;) {
113
- const ids = await this.store.purge(this.purgeInput);
114
- deleted += ids.length;
115
- if (ids.length < limit || this.stopping) return deleted;
116
- // Hand the loop back between batches: a large drain must not starve the
117
- // submits and claims sharing this process.
118
- await this.sleep(0);
138
+ for (const input of this.purgeInputs) {
139
+ for (;;) {
140
+ const ids = await this.store.purge(input);
141
+ deleted += ids.length;
142
+ if (this.stopping) return deleted;
143
+ if (ids.length < this.limit) break;
144
+ // Hand the loop back between batches: a large drain must not starve the
145
+ // submits and claims sharing this process.
146
+ await this.sleep(0);
147
+ }
119
148
  }
149
+ return deleted;
120
150
  }
121
151
 
122
152
  /** Sleep, interruptible by stop(). Unref'd: retention is housekeeping, and a