cairnq 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/README.md +28 -0
  2. package/dist/_protocol/migrations/postgres/0002_purge_index.sql +6 -0
  3. package/dist/_protocol/migrations/postgres/0003_notify.sql +38 -0
  4. package/dist/_protocol/migrations/sqlite/0002_purge_index.sql +6 -0
  5. package/dist/_protocol/sql/postgres/claim.sql +18 -5
  6. package/dist/_protocol/sql/postgres/fail.sql +28 -8
  7. package/dist/_protocol/sql/postgres/insert_task.sql +6 -3
  8. package/dist/_protocol/sql/postgres/list.sql +3 -1
  9. package/dist/_protocol/sql/postgres/lock_key.sql +9 -0
  10. package/dist/_protocol/sql/postgres/progress.sql +4 -3
  11. package/dist/_protocol/sql/postgres/protocol_version.sql +4 -0
  12. package/dist/_protocol/sql/postgres/purge.sql +25 -0
  13. package/dist/_protocol/sql/postgres/recover_leases.sql +38 -14
  14. package/dist/_protocol/sql/postgres/retry.sql +3 -0
  15. package/dist/_protocol/sql/postgres/stats.sql +8 -0
  16. package/dist/_protocol/sql/sqlite/claim.sql +13 -2
  17. package/dist/_protocol/sql/sqlite/claimable_probe.sql +6 -2
  18. package/dist/_protocol/sql/sqlite/fail.sql +30 -8
  19. package/dist/_protocol/sql/sqlite/list.sql +3 -1
  20. package/dist/_protocol/sql/sqlite/lock_key.sql +5 -0
  21. package/dist/_protocol/sql/sqlite/progress.sql +6 -2
  22. package/dist/_protocol/sql/sqlite/protocol_version.sql +4 -0
  23. package/dist/_protocol/sql/sqlite/purge.sql +18 -0
  24. package/dist/_protocol/sql/sqlite/recover_leases.sql +25 -7
  25. package/dist/_protocol/sql/sqlite/retry.sql +3 -0
  26. package/dist/_protocol/sql/sqlite/stats.sql +8 -0
  27. package/dist/client.d.ts +10 -2
  28. package/dist/client.js +12 -0
  29. package/dist/context.d.ts +17 -1
  30. package/dist/context.js +60 -6
  31. package/dist/errors.d.ts +18 -2
  32. package/dist/errors.js +49 -3
  33. package/dist/index.d.ts +3 -2
  34. package/dist/index.js +2 -1
  35. package/dist/sql.js +16 -9
  36. package/dist/store/base.d.ts +107 -9
  37. package/dist/store/base.js +370 -1
  38. package/dist/store/postgres.d.ts +62 -63
  39. package/dist/store/postgres.js +245 -222
  40. package/dist/store/sqlite.d.ts +34 -59
  41. package/dist/store/sqlite.js +200 -232
  42. package/dist/wait.d.ts +15 -2
  43. package/dist/wait.js +23 -5
  44. package/dist/worker.d.ts +53 -1
  45. package/dist/worker.js +202 -42
  46. package/package.json +9 -2
  47. package/src/client.ts +16 -2
  48. package/src/context.ts +70 -13
  49. package/src/errors.ts +59 -4
  50. package/src/index.ts +3 -1
  51. package/src/sql.ts +15 -8
  52. package/src/store/base.ts +430 -27
  53. package/src/store/postgres.ts +243 -267
  54. package/src/store/sqlite.ts +211 -265
  55. package/src/wait.ts +28 -5
  56. package/src/worker.ts +242 -42
package/dist/worker.js CHANGED
@@ -1,9 +1,20 @@
1
1
  import { TaskContext } from "./context.js";
2
- import { errorEnvelope, LostLease, TaskError } from "./errors.js";
2
+ import { errorEnvelope, LostLease, SerializationError, TaskError } from "./errors.js";
3
3
  import { newId } from "./ids.js";
4
4
  import { SQLiteStore } from "./store/sqlite.js";
5
5
  import { PostgresStore } from "./store/postgres.js";
6
6
  import { taskName } from "./task.js";
7
+ const DEFAULT_RETRY_BACKOFF_MS = 1_000;
8
+ const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
9
+ /** Wait after a failed claim, so a broken database is not polled in a tight loop. */
10
+ const CLAIM_ERROR_BACKOFF_MS = 250;
11
+ /** Exponential backoff for the next attempt of a task that just failed. */
12
+ export function retryDelayMs(attempt, baseMs, maxMs) {
13
+ if (baseMs <= 0)
14
+ return 0;
15
+ const exponent = Math.max(0, attempt - 1);
16
+ return Math.min(maxMs, baseMs * 2 ** exponent);
17
+ }
7
18
  function exceptionEnvelope(err) {
8
19
  const e = err;
9
20
  return errorEnvelope({
@@ -13,6 +24,23 @@ function exceptionEnvelope(err) {
13
24
  retryable: true,
14
25
  });
15
26
  }
27
+ /** Internal: an attempt outran maxRunMs and was abandoned. */
28
+ class AttemptTimeout extends Error {
29
+ maxRunMs;
30
+ constructor(maxRunMs) {
31
+ super(`attempt exceeded ${maxRunMs}ms`);
32
+ this.maxRunMs = maxRunMs;
33
+ }
34
+ }
35
+ function timeoutEnvelope(name, maxRunMs) {
36
+ return errorEnvelope({
37
+ type: "HandlerTimeout",
38
+ code: "handler_timeout",
39
+ message: `handler for ${name} exceeded maxRunMs=${maxRunMs}ms; the attempt was abandoned`,
40
+ retryable: true,
41
+ });
42
+ }
43
+ const TIMED_OUT = Symbol("cairnq.timedOut");
16
44
  export class Worker {
17
45
  store;
18
46
  queues;
@@ -20,7 +48,10 @@ export class Worker {
20
48
  handlers = new Map();
21
49
  workerId = newId("worker");
22
50
  stopped = false;
23
- stopResolvers = [];
51
+ stopWake;
52
+ // Resolved once by stop(); every sleep races against it. A stopped worker
53
+ // never restarts, so one promise serves the instance's lifetime.
54
+ stopped$ = new Promise((r) => (this.stopWake = r));
24
55
  // True only when this worker created its own store (via Worker.sqlite); an
25
56
  // injected store may be shared, so serve()/background() must not close it.
26
57
  ownsStore = false;
@@ -28,6 +59,9 @@ export class Worker {
28
59
  this.store = store;
29
60
  this.queues = queues;
30
61
  this.opts = opts;
62
+ if (opts.maxRunMs != null && opts.maxRunMs <= 0) {
63
+ throw new Error(`maxRunMs must be > 0, got ${opts.maxRunMs}`);
64
+ }
31
65
  }
32
66
  static sqlite(path, opts = {}) {
33
67
  const { queues = ["default"], busyTimeoutMs, ...rest } = opts;
@@ -51,8 +85,10 @@ export class Worker {
51
85
  let fn;
52
86
  if (typeof arg === "function") {
53
87
  // Bare form: worker.task(fn) — registered under the function's name.
88
+ // Strip the "bound " prefix .bind() stamps on it: otherwise a bound
89
+ // method registers under "bound process", a name no submit ever uses.
54
90
  fn = arg;
55
- name = fn.name;
91
+ name = fn.name.replace(/^(bound )+/, "");
56
92
  if (!name) {
57
93
  throw new Error("worker.task(fn): the handler is anonymous; pass a name explicitly, " +
58
94
  "e.g. worker.task('summary.create', fn)");
@@ -68,10 +104,7 @@ export class Worker {
68
104
  }
69
105
  stop() {
70
106
  this.stopped = true;
71
- const resolvers = this.stopResolvers;
72
- this.stopResolvers = [];
73
- for (const r of resolvers)
74
- r();
107
+ this.stopWake();
75
108
  }
76
109
  /** Close the underlying store connection. Call after run() returns. */
77
110
  async close() {
@@ -84,28 +117,66 @@ export class Worker {
84
117
  if (this.ownsStore)
85
118
  await this.close();
86
119
  }
120
+ report(err, info) {
121
+ try {
122
+ this.opts.onError?.(err, info);
123
+ }
124
+ catch {
125
+ // A reporting hook must never take the worker down with it.
126
+ }
127
+ }
87
128
  async run(opts = {}) {
88
- const concurrency = opts.concurrency ?? this.opts.concurrency ?? 1;
129
+ // Clamped: at 0 the loop would await Promise.race([]) — pending forever,
130
+ // beyond even stop()'s reach.
131
+ const concurrency = Math.max(1, opts.concurrency ?? this.opts.concurrency ?? 1);
89
132
  const leaseMs = this.opts.leaseMs ?? 30_000;
90
- const pollMs = this.opts.pollIntervalMs ?? 500;
91
133
  const batch = this.opts.claimBatch ?? concurrency;
92
134
  await this.store.connect();
93
- this.installSignals();
94
135
  const running = new Set();
136
+ try {
137
+ await this.loop(concurrency, batch, leaseMs, running);
138
+ }
139
+ finally {
140
+ // Whatever ends the loop — stop(), or something unexpected out of the body
141
+ // — nothing this worker started may outlive run(). serve() closes the store
142
+ // as soon as run() settles, and a handler still holding the connection
143
+ // would fault on it.
144
+ await Promise.all([...running]);
145
+ }
146
+ }
147
+ async loop(concurrency, batch, leaseMs, running) {
148
+ const pollMs = this.opts.pollIntervalMs ?? 500;
95
149
  while (!this.stopped) {
96
150
  const free = concurrency - running.size;
97
151
  if (free <= 0) {
98
- await this.sleepOrStop(5);
152
+ // Wait for a slot rather than spinning. execute() never rejects, so
153
+ // racing these is safe.
154
+ await Promise.race([...running]);
155
+ continue;
156
+ }
157
+ let claimed;
158
+ try {
159
+ claimed = await this.store.claim({
160
+ queues: this.queues,
161
+ // Only what this worker can run. Queues do not partition work by task
162
+ // name, so another worker's tasks would otherwise be claimed here and
163
+ // failed for want of a handler. Read each poll: handlers may be
164
+ // registered after run() started.
165
+ names: [...this.handlers.keys()],
166
+ workerId: this.workerId,
167
+ leaseMs,
168
+ limit: Math.min(batch, free),
169
+ });
170
+ }
171
+ catch (err) {
172
+ // A claim can fail transiently (lock contention, a dropped connection).
173
+ // Report it and keep polling — one bad poll must not end the worker.
174
+ this.report(err, { phase: "claim" });
175
+ await this.sleepOrStop(CLAIM_ERROR_BACKOFF_MS);
99
176
  continue;
100
177
  }
101
- const claimed = await this.store.claim({
102
- queues: this.queues,
103
- workerId: this.workerId,
104
- leaseMs,
105
- limit: Math.min(batch, free),
106
- });
107
178
  if (claimed.length === 0) {
108
- await this.sleepOrStop(pollMs);
179
+ await this.idle(pollMs);
109
180
  continue;
110
181
  }
111
182
  for (const task of claimed) {
@@ -113,16 +184,21 @@ export class Worker {
113
184
  running.add(p);
114
185
  }
115
186
  }
116
- await Promise.all([...running]);
117
187
  }
118
188
  /** Blocking-style entry point for a standalone worker process: run until
119
189
  * SIGINT/SIGTERM, then close the store. Use this at a script's top level;
120
190
  * use run() / background() when you manage the event loop yourself. */
121
191
  async serve(opts = {}) {
192
+ // Signals are installed here rather than in run(): serve() is the entry point
193
+ // that owns the process. run()/background() embed the worker in someone
194
+ // else's process, where a leftover listener suppresses Node's default Ctrl-C
195
+ // handling for the host long after the worker is done.
196
+ const removeSignalHandlers = this.installSignals();
122
197
  try {
123
198
  await this.run(opts);
124
199
  }
125
200
  finally {
201
+ removeSignalHandlers();
126
202
  await this.closeIfOwned();
127
203
  }
128
204
  }
@@ -138,13 +214,18 @@ export class Worker {
138
214
  await this.closeIfOwned();
139
215
  }
140
216
  }
217
+ /**
218
+ * Run one task to completion. Never rejects: a task-level failure is reported
219
+ * through onError and the loop moves on. (It used to reject into a promise
220
+ * nobody awaited — an unhandled rejection that took the process down.)
221
+ */
141
222
  async execute(task, leaseMs) {
142
223
  const ctx = new TaskContext(this.store, task, this.workerId, leaseMs);
143
224
  const hb = this.startHeartbeat(ctx, leaseMs);
144
225
  try {
145
226
  const handler = this.handlers.get(task.name);
146
227
  if (!handler) {
147
- await this.safeFail(task.id, errorEnvelope({
228
+ await this.safeFail(task, errorEnvelope({
148
229
  type: "NoHandler",
149
230
  code: "no_handler",
150
231
  message: `no handler registered for ${task.name}`,
@@ -154,16 +235,22 @@ export class Worker {
154
235
  }
155
236
  let result;
156
237
  try {
157
- result = await handler(ctx, task.payload);
238
+ result = await this.attempt(handler, ctx, task);
158
239
  }
159
240
  catch (err) {
160
241
  if (err instanceof LostLease)
161
242
  return;
243
+ if (err instanceof AttemptTimeout) {
244
+ // Recorded as a retryable failure, so backoff / maxAttempts /
245
+ // cancel-wins all apply exactly as for a thrown error.
246
+ await this.safeFail(task, timeoutEnvelope(task.name, err.maxRunMs), true);
247
+ return;
248
+ }
162
249
  if (err instanceof TaskError) {
163
- await this.safeFail(task.id, err.envelope(), err.retryable);
250
+ await this.safeFail(task, err.envelope(), err.retryable);
164
251
  }
165
252
  else {
166
- await this.safeFail(task.id, exceptionEnvelope(err), true);
253
+ await this.safeFail(task, exceptionEnvelope(err), true);
167
254
  }
168
255
  return;
169
256
  }
@@ -173,20 +260,68 @@ export class Worker {
173
260
  await this.store.complete({ taskId: task.id, workerId: this.workerId, result });
174
261
  }
175
262
  catch (err) {
176
- if (err instanceof LostLease)
263
+ if (err instanceof LostLease) {
264
+ ctx.markLeaseLost();
177
265
  return;
266
+ }
267
+ if (err instanceof SerializationError) {
268
+ // The handler succeeded but its return value can't cross the JSON
269
+ // protocol (BigInt, non-finite number, circular). Deterministic, so
270
+ // fail fast and permanently — the alternative is sitting `running`
271
+ // until lease expiry redelivers a task that fails the same way every
272
+ // attempt.
273
+ await this.safeFail(task, errorEnvelope({
274
+ type: "SerializationError",
275
+ code: "unserializable_result",
276
+ message: `handler result is not JSON-serializable: ${err.message}`,
277
+ retryable: false,
278
+ }), false);
279
+ return;
280
+ }
178
281
  throw err;
179
282
  }
180
283
  }
284
+ catch (err) {
285
+ this.report(err, { phase: "execute", taskId: task.id });
286
+ }
181
287
  finally {
182
288
  hb.cancel();
183
289
  await hb.done;
184
290
  }
185
291
  }
292
+ /**
293
+ * Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
294
+ * abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
295
+ * a handler that keeps running can never write again — see
296
+ * TaskContext.owned), then the still-pending promise is left to settle on
297
+ * its own, its outcome discarded. The caller records the handler_timeout
298
+ * failure; lease recovery is NOT involved, so redelivery is immediate.
299
+ */
300
+ async attempt(handler, ctx, task) {
301
+ const maxRunMs = this.opts.maxRunMs;
302
+ if (maxRunMs == null)
303
+ return handler(ctx, task.payload);
304
+ // As a real promise: the race needs one (a handler may return a plain
305
+ // value), and the timeout path .catch()es it.
306
+ const run = (async () => handler(ctx, task.payload))();
307
+ let timer;
308
+ const winner = await Promise.race([
309
+ run,
310
+ new Promise((r) => (timer = setTimeout(() => r(TIMED_OUT), maxRunMs))),
311
+ ]).finally(() => clearTimeout(timer));
312
+ if (winner !== TIMED_OUT)
313
+ return winner;
314
+ ctx.markLeaseLost();
315
+ // The zombie may still reject later; that must not become an unhandled
316
+ // rejection — its outcome was already decided to be handler_timeout.
317
+ run.catch(() => { });
318
+ throw new AttemptTimeout(maxRunMs);
319
+ }
186
320
  startHeartbeat(ctx, leaseMs) {
187
321
  let active = true;
188
322
  let wake = null;
189
- const interval = this.opts.heartbeatIntervalMs ?? Math.max(1_000, Math.floor(leaseMs / 3));
323
+ // lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
324
+ const interval = this.opts.heartbeatIntervalMs ?? Math.max(50, Math.floor(leaseMs / 3));
190
325
  const done = (async () => {
191
326
  while (active) {
192
327
  // Cancellable sleep: cancel() resolves this immediately and clears the
@@ -205,8 +340,10 @@ export class Worker {
205
340
  await ctx.heartbeat();
206
341
  }
207
342
  catch (err) {
343
+ // ctx.heartbeat() already flagged the lease as lost for the handler.
208
344
  if (err instanceof LostLease)
209
345
  break;
346
+ this.report(err, { phase: "execute", taskId: ctx.taskId });
210
347
  }
211
348
  }
212
349
  })();
@@ -219,34 +356,57 @@ export class Worker {
219
356
  done,
220
357
  };
221
358
  }
222
- async safeFail(taskId, envelope, retryable) {
359
+ async safeFail(task, envelope, retryable) {
360
+ const delayMs = retryable
361
+ ? retryDelayMs(task.attempt, this.opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS, this.opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS)
362
+ : 0;
223
363
  try {
224
- await this.store.fail({ taskId, workerId: this.workerId, error: envelope, retryable });
364
+ await this.store.fail({
365
+ taskId: task.id,
366
+ workerId: this.workerId,
367
+ error: envelope,
368
+ retryable,
369
+ delayMs,
370
+ });
225
371
  }
226
372
  catch (err) {
227
373
  if (!(err instanceof LostLease))
228
374
  throw err;
229
375
  }
230
376
  }
377
+ /**
378
+ * The empty-poll sleep. A store with a push channel (Postgres LISTEN/NOTIFY)
379
+ * cuts it short when a task on this worker's queues becomes claimable;
380
+ * stop() interrupts it either way, and sleepOrStop bounds it at `ms` so the
381
+ * poll fallback — which also drives lease recovery — never stretches.
382
+ */
383
+ idle(ms) {
384
+ return Promise.race([this.sleepOrStop(ms), this.store.claimWake(this.queues, ms)]);
385
+ }
231
386
  sleepOrStop(ms) {
232
387
  if (this.stopped)
233
388
  return Promise.resolve();
234
- return new Promise((resolve) => {
235
- let done = false;
236
- const finish = () => {
237
- if (done)
238
- return;
239
- done = true;
240
- clearTimeout(timer);
241
- resolve();
242
- };
243
- const timer = setTimeout(finish, ms);
244
- this.stopResolvers.push(finish);
245
- });
389
+ let timer;
390
+ const nap = new Promise((r) => (timer = setTimeout(r, ms)));
391
+ // Clear the timer whichever side wins, so a stop is never followed by a
392
+ // leftover poll timer holding the process open.
393
+ return Promise.race([nap, this.stopped$]).finally(() => clearTimeout(timer));
246
394
  }
395
+ /** Take SIGINT/SIGTERM for the duration of serve(). Returns the undo. */
247
396
  installSignals() {
248
- const handler = () => this.stop();
249
- process.once("SIGINT", handler);
250
- process.once("SIGTERM", handler);
397
+ const remove = () => {
398
+ process.off("SIGINT", handler);
399
+ process.off("SIGTERM", handler);
400
+ };
401
+ const handler = () => {
402
+ // Stand down after the first signal, so a second Ctrl-C reaches Node's
403
+ // default and kills a worker that will not drain. `once` would only drop
404
+ // whichever signal fired and leave the other suppressing the default.
405
+ remove();
406
+ this.stop();
407
+ };
408
+ process.on("SIGINT", handler);
409
+ process.on("SIGTERM", handler);
410
+ return remove;
251
411
  }
252
412
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "cairnq",
3
- "version": "0.1.0",
3
+ "version": "0.2.0",
4
4
  "description": "SQLite-first, cross-language, storage-centered durable task runtime",
5
5
  "license": "MIT",
6
6
  "author": "Jannchie <jannchie@gmail.com>",
@@ -33,6 +33,7 @@
33
33
  "files": ["dist", "src"],
34
34
  "engines": { "node": ">=20" },
35
35
  "scripts": {
36
+ "bench": "tsx bench/run.ts",
36
37
  "build": "tsc -p tsconfig.json",
37
38
  "test": "vitest run",
38
39
  "typecheck": "tsc -p tsconfig.json --noEmit"
@@ -40,13 +41,19 @@
40
41
  "dependencies": {
41
42
  "better-sqlite3": "^11.3.0"
42
43
  },
43
- "optionalDependencies": {
44
+ "peerDependencies": {
44
45
  "pg": "^8.13.0"
45
46
  },
47
+ "peerDependenciesMeta": {
48
+ "pg": {
49
+ "optional": true
50
+ }
51
+ },
46
52
  "devDependencies": {
47
53
  "@types/better-sqlite3": "^7.6.11",
48
54
  "@types/node": "^22.7.0",
49
55
  "@types/pg": "^8.11.10",
56
+ "pg": "^8.13.0",
50
57
  "tsx": "^4.19.0",
51
58
  "typescript": "^5.6.0",
52
59
  "vitest": "^2.1.0"
package/src/client.ts CHANGED
@@ -1,8 +1,8 @@
1
1
  import { TaskCanceled, TaskFailed } from "./errors.js";
2
- import { isFailed, isSucceeded, type Task } from "./models.js";
2
+ import { isFailed, isSucceeded, type Task, type TaskStatus } from "./models.js";
3
3
  import { SQLiteStore } from "./store/sqlite.js";
4
4
  import { PostgresStore } from "./store/postgres.js";
5
- import type { ListInput, SubmitInput, TaskStore } from "./store/base.js";
5
+ import type { ListInput, PurgeInput, SubmitInput, TaskStore } from "./store/base.js";
6
6
  import { type TaskDef, taskName } from "./task.js";
7
7
  import { pollWait } from "./wait.js";
8
8
 
@@ -72,6 +72,20 @@ export class CairnQ {
72
72
  return this._store.retryByKey(key, opts);
73
73
  }
74
74
 
75
+ /** Delete terminal tasks that finished more than `olderThanMs` ago and return
76
+ * their ids. Nothing else in CairnQ removes rows, so a long-lived database
77
+ * needs this on a schedule. Each call is bounded by `limit` to keep the write
78
+ * short; loop until it returns fewer than `limit`. */
79
+ purge(input?: PurgeInput): Promise<string[]> {
80
+ return this._store.purge(input);
81
+ }
82
+
83
+ /** Task counts per queue, keyed by status and zero-filled across all statuses
84
+ * — `(await stats()).default.queued` is the backlog of a queue. */
85
+ stats(): Promise<Record<string, Record<TaskStatus, number>>> {
86
+ return this._store.stats();
87
+ }
88
+
75
89
  wait(
76
90
  taskId: string,
77
91
  opts: { timeoutMs?: number; pollMs?: number } = {},
package/src/context.ts CHANGED
@@ -1,3 +1,4 @@
1
+ import { LostLease } from "./errors.js";
1
2
  import { cancelRequested, type Task } from "./models.js";
2
3
  import type { SubmitOptions } from "./client.js";
3
4
  import type { TaskStore } from "./store/base.js";
@@ -6,6 +7,12 @@ import { pollWait } from "./wait.js";
6
7
 
7
8
  /** Handed to a task handler. Worker-side capabilities mirror the Python SDK. */
8
9
  export class TaskContext {
10
+ private readonly abort = new AbortController();
11
+ private leaseLost = false;
12
+ // Cancellation is monotonic: once the DB has told us a cancel was requested it
13
+ // can't be taken back, so canceled() can answer from this without a re-read.
14
+ private cancelSeen = false;
15
+
9
16
  constructor(
10
17
  private readonly store: TaskStore,
11
18
  private readonly task: Task,
@@ -38,28 +45,78 @@ export class TaskContext {
38
45
  return this.task.payload;
39
46
  }
40
47
 
48
+ /**
49
+ * True once this worker has lost the task's lease — it expired and another
50
+ * worker reclaimed it. Nothing this handler writes will be recorded any more
51
+ * and the task is already running elsewhere, so a long handler should check
52
+ * this (or `signal`) and bail out instead of continuing to do side effects.
53
+ */
54
+ get lostLease(): boolean {
55
+ return this.leaseLost;
56
+ }
57
+
58
+ /** Aborts when the lease is lost. Pass it to fetch / any AbortSignal-aware API. */
59
+ get signal(): AbortSignal {
60
+ return this.abort.signal;
61
+ }
62
+
63
+ /** @internal Called by the worker when an owned write reports a lost lease. */
64
+ markLeaseLost(): void {
65
+ if (this.leaseLost) return;
66
+ this.leaseLost = true;
67
+ this.abort.abort(new LostLease(this.task.id));
68
+ }
69
+
70
+ // Every owned write returns the current row, so cancellation and lease loss
71
+ // ride along on writes the handler was making anyway.
72
+ private observe(task: Task): Task {
73
+ if (cancelRequested(task)) this.cancelSeen = true;
74
+ return task;
75
+ }
76
+
77
+ private async owned(write: () => Promise<Task>): Promise<Task> {
78
+ // Short-circuit once the lease is known lost: nothing this context writes
79
+ // may be recorded any more. Locally, not just via the store's ownership
80
+ // check — after an abandoned (timed-out) attempt the same worker may
81
+ // re-claim this task under the same workerId, and a zombie handler's write
82
+ // would then pass ownership against the NEW attempt.
83
+ if (this.leaseLost) throw new LostLease(this.task.id);
84
+ try {
85
+ return this.observe(await write());
86
+ } catch (err) {
87
+ if (err instanceof LostLease) this.markLeaseLost();
88
+ throw err;
89
+ }
90
+ }
91
+
41
92
  async progress(value: number | null, message: string | null = null): Promise<Task> {
42
- return this.store.progress({
43
- taskId: this.task.id,
44
- workerId: this.workerId,
45
- progress: value,
46
- message,
47
- });
93
+ return this.owned(() =>
94
+ this.store.progress({
95
+ taskId: this.task.id,
96
+ workerId: this.workerId,
97
+ progress: value,
98
+ message,
99
+ }),
100
+ );
48
101
  }
49
102
 
50
103
  async heartbeat(): Promise<Task> {
51
- return this.store.heartbeat({
52
- taskId: this.task.id,
53
- workerId: this.workerId,
54
- leaseMs: this.leaseMs,
55
- });
104
+ return this.owned(() =>
105
+ this.store.heartbeat({
106
+ taskId: this.task.id,
107
+ workerId: this.workerId,
108
+ leaseMs: this.leaseMs,
109
+ }),
110
+ );
56
111
  }
57
112
 
58
- /** Cooperative cancel check. */
113
+ /** Cooperative cancel check. Free once a heartbeat has already seen the flag. */
59
114
  async canceled(): Promise<boolean> {
115
+ if (this.cancelSeen) return true;
60
116
  const t = await this.store.get(this.task.id);
61
117
  if (!t) return true;
62
- return cancelRequested(t) || t.status === "canceled";
118
+ if (cancelRequested(t)) this.cancelSeen = true;
119
+ return this.cancelSeen || t.status === "canceled";
63
120
  }
64
121
 
65
122
  /** Submit a child task; parent/root/correlation are wired automatically. */
package/src/errors.ts CHANGED
@@ -1,3 +1,6 @@
1
+ import { nowMs } from "./ids.js";
2
+ import { cancelRequested, isQueued, type Task } from "./models.js";
3
+
1
4
  /** The single shape of the JSON error envelope (see PROTOCOL.md). Everything that
2
5
  * records an error — a handler exception, a missing handler, lease expiry, a thrown
3
6
  * TaskError — builds it here, so the contract's fields live in one place. */
@@ -17,7 +20,15 @@ export function errorEnvelope(e: {
17
20
  };
18
21
  }
19
22
 
20
- export class CairnQError extends Error {}
23
+ export class CairnQError extends Error {
24
+ constructor(message?: string) {
25
+ super(message);
26
+ // Subclasses each set their own; without this a bare CairnQError reports
27
+ // "Error", and `err.name` is how callers (and the conformance runner) tell
28
+ // one apart from another.
29
+ this.name = "CairnQError";
30
+ }
31
+ }
21
32
 
22
33
  export class AlreadyExists extends CairnQError {
23
34
  constructor(public key: string) {
@@ -26,11 +37,44 @@ export class AlreadyExists extends CairnQError {
26
37
  }
27
38
  }
28
39
 
29
- /** wait/call did not reach a terminal status in time. The task keeps running. */
40
+ /** One line of "why hasn't this finished" from the last snapshot wait()
41
+ * observed. No worker running, no handler for the name, wrong queue, and two
42
+ * processes on different database files all look identical from the API side —
43
+ * queued, never claimed — so that case names the likely causes. */
44
+ function timeoutDetail(task: Task | null): string {
45
+ if (!task) return "task not found — wrong database file, or already purged?";
46
+ if (isQueued(task)) {
47
+ const delayMs = task.run_at_ms - nowMs();
48
+ if (task.attempt === 0 && delayMs <= 0) {
49
+ return (
50
+ `never claimed by a worker — is a worker running with a handler for ` +
51
+ `'${task.name}' on queue '${task.queue}', against this same database?`
52
+ );
53
+ }
54
+ const next = delayMs > 0 ? `, next run in ~${delayMs}ms` : "";
55
+ return `still queued (attempt ${task.attempt}/${task.max_attempts})${next}`;
56
+ }
57
+ if (cancelRequested(task)) return "cancel requested, waiting for the handler to observe it";
58
+ return `still running (attempt ${task.attempt}/${task.max_attempts})`;
59
+ }
60
+
61
+ /** wait/call did not reach a terminal status in time. The task keeps running.
62
+ * `task` is the last snapshot wait() observed (null if get() found nothing), and
63
+ * the message says what state it was stuck in — a queued-never-claimed task is
64
+ * the classic first-run failure (no worker, no handler, wrong queue or file). */
30
65
  export class TaskTimeout extends CairnQError {
31
- constructor(public taskId: string) {
32
- super(`task ${taskId} did not finish in time`);
66
+ readonly task: Task | null;
67
+ constructor(
68
+ public taskId: string,
69
+ opts: { timeoutMs?: number; task?: Task | null } = {},
70
+ ) {
71
+ super(
72
+ opts.timeoutMs == null
73
+ ? `task ${taskId} did not finish in time`
74
+ : `task ${taskId} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null)}`,
75
+ );
33
76
  this.name = "TaskTimeout";
77
+ this.task = opts.task ?? null;
34
78
  }
35
79
  }
36
80
 
@@ -81,6 +125,17 @@ export class ProtocolVersionMismatch extends CairnQError {
81
125
  }
82
126
  }
83
127
 
128
+ /** A value could not be encoded for a protocol JSON column (non-finite number,
129
+ * BigInt, circular structure, …). Raised at the boundary — submit rejects with
130
+ * it, and a worker records a handler result that triggers it as a permanent
131
+ * `unserializable_result` failure. The Python SDK raises the same named error. */
132
+ export class SerializationError extends CairnQError {
133
+ constructor(message: string) {
134
+ super(message);
135
+ this.name = "SerializationError";
136
+ }
137
+ }
138
+
84
139
  /** Throw inside a handler to control how the failure is recorded. Defaults to
85
140
  * non-retryable so deterministic errors fail fast instead of burning retries.
86
141
  * Any other thrown value is treated as retryable. */
package/src/index.ts CHANGED
@@ -7,7 +7,8 @@ export { defineTask } from "./task.js";
7
7
  export type { TaskDef } from "./task.js";
8
8
  export { SQLiteStore } from "./store/sqlite.js";
9
9
  export { PostgresStore } from "./store/postgres.js";
10
- export type { ListInput, SubmitInput, TaskStore, Conflict } from "./store/base.js";
10
+ export { TaskStore } from "./store/base.js";
11
+ export type { ListInput, PurgeInput, SubmitInput, Conflict } from "./store/base.js";
11
12
  export type { Task, TaskStatus } from "./models.js";
12
13
  export {
13
14
  STATUSES,
@@ -28,4 +29,5 @@ export {
28
29
  TaskError,
29
30
  LostLease,
30
31
  ProtocolVersionMismatch,
32
+ SerializationError,
31
33
  } from "./errors.js";