cairnq 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -0
- package/dist/_protocol/migrations/postgres/0002_purge_index.sql +6 -0
- package/dist/_protocol/migrations/postgres/0003_notify.sql +38 -0
- package/dist/_protocol/migrations/sqlite/0002_purge_index.sql +6 -0
- package/dist/_protocol/sql/postgres/claim.sql +18 -5
- package/dist/_protocol/sql/postgres/fail.sql +28 -8
- package/dist/_protocol/sql/postgres/insert_task.sql +6 -3
- package/dist/_protocol/sql/postgres/list.sql +3 -1
- package/dist/_protocol/sql/postgres/lock_key.sql +9 -0
- package/dist/_protocol/sql/postgres/progress.sql +4 -3
- package/dist/_protocol/sql/postgres/protocol_version.sql +4 -0
- package/dist/_protocol/sql/postgres/purge.sql +25 -0
- package/dist/_protocol/sql/postgres/recover_leases.sql +38 -14
- package/dist/_protocol/sql/postgres/retry.sql +3 -0
- package/dist/_protocol/sql/postgres/stats.sql +8 -0
- package/dist/_protocol/sql/sqlite/claim.sql +13 -2
- package/dist/_protocol/sql/sqlite/claimable_probe.sql +6 -2
- package/dist/_protocol/sql/sqlite/fail.sql +30 -8
- package/dist/_protocol/sql/sqlite/list.sql +3 -1
- package/dist/_protocol/sql/sqlite/lock_key.sql +5 -0
- package/dist/_protocol/sql/sqlite/progress.sql +6 -2
- package/dist/_protocol/sql/sqlite/protocol_version.sql +4 -0
- package/dist/_protocol/sql/sqlite/purge.sql +18 -0
- package/dist/_protocol/sql/sqlite/recover_leases.sql +25 -7
- package/dist/_protocol/sql/sqlite/retry.sql +3 -0
- package/dist/_protocol/sql/sqlite/stats.sql +8 -0
- package/dist/client.d.ts +10 -2
- package/dist/client.js +12 -0
- package/dist/context.d.ts +17 -1
- package/dist/context.js +60 -6
- package/dist/errors.d.ts +18 -2
- package/dist/errors.js +49 -3
- package/dist/index.d.ts +3 -2
- package/dist/index.js +2 -1
- package/dist/sql.js +16 -9
- package/dist/store/base.d.ts +107 -9
- package/dist/store/base.js +370 -1
- package/dist/store/postgres.d.ts +62 -63
- package/dist/store/postgres.js +245 -222
- package/dist/store/sqlite.d.ts +34 -59
- package/dist/store/sqlite.js +200 -232
- package/dist/wait.d.ts +15 -2
- package/dist/wait.js +23 -5
- package/dist/worker.d.ts +53 -1
- package/dist/worker.js +202 -42
- package/package.json +9 -2
- package/src/client.ts +16 -2
- package/src/context.ts +70 -13
- package/src/errors.ts +59 -4
- package/src/index.ts +3 -1
- package/src/sql.ts +15 -8
- package/src/store/base.ts +430 -27
- package/src/store/postgres.ts +243 -267
- package/src/store/sqlite.ts +211 -265
- package/src/wait.ts +28 -5
- package/src/worker.ts +242 -42
package/dist/worker.js
CHANGED
|
@@ -1,9 +1,20 @@
|
|
|
1
1
|
import { TaskContext } from "./context.js";
|
|
2
|
-
import { errorEnvelope, LostLease, TaskError } from "./errors.js";
|
|
2
|
+
import { errorEnvelope, LostLease, SerializationError, TaskError } from "./errors.js";
|
|
3
3
|
import { newId } from "./ids.js";
|
|
4
4
|
import { SQLiteStore } from "./store/sqlite.js";
|
|
5
5
|
import { PostgresStore } from "./store/postgres.js";
|
|
6
6
|
import { taskName } from "./task.js";
|
|
7
|
+
const DEFAULT_RETRY_BACKOFF_MS = 1_000;
|
|
8
|
+
const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
|
|
9
|
+
/** Wait after a failed claim, so a broken database is not polled in a tight loop. */
|
|
10
|
+
const CLAIM_ERROR_BACKOFF_MS = 250;
|
|
11
|
+
/** Exponential backoff for the next attempt of a task that just failed. */
|
|
12
|
+
export function retryDelayMs(attempt, baseMs, maxMs) {
|
|
13
|
+
if (baseMs <= 0)
|
|
14
|
+
return 0;
|
|
15
|
+
const exponent = Math.max(0, attempt - 1);
|
|
16
|
+
return Math.min(maxMs, baseMs * 2 ** exponent);
|
|
17
|
+
}
|
|
7
18
|
function exceptionEnvelope(err) {
|
|
8
19
|
const e = err;
|
|
9
20
|
return errorEnvelope({
|
|
@@ -13,6 +24,23 @@ function exceptionEnvelope(err) {
|
|
|
13
24
|
retryable: true,
|
|
14
25
|
});
|
|
15
26
|
}
|
|
27
|
+
/** Internal: an attempt outran maxRunMs and was abandoned. */
|
|
28
|
+
class AttemptTimeout extends Error {
|
|
29
|
+
maxRunMs;
|
|
30
|
+
constructor(maxRunMs) {
|
|
31
|
+
super(`attempt exceeded ${maxRunMs}ms`);
|
|
32
|
+
this.maxRunMs = maxRunMs;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
function timeoutEnvelope(name, maxRunMs) {
|
|
36
|
+
return errorEnvelope({
|
|
37
|
+
type: "HandlerTimeout",
|
|
38
|
+
code: "handler_timeout",
|
|
39
|
+
message: `handler for ${name} exceeded maxRunMs=${maxRunMs}ms; the attempt was abandoned`,
|
|
40
|
+
retryable: true,
|
|
41
|
+
});
|
|
42
|
+
}
|
|
43
|
+
const TIMED_OUT = Symbol("cairnq.timedOut");
|
|
16
44
|
export class Worker {
|
|
17
45
|
store;
|
|
18
46
|
queues;
|
|
@@ -20,7 +48,10 @@ export class Worker {
|
|
|
20
48
|
handlers = new Map();
|
|
21
49
|
workerId = newId("worker");
|
|
22
50
|
stopped = false;
|
|
23
|
-
|
|
51
|
+
stopWake;
|
|
52
|
+
// Resolved once by stop(); every sleep races against it. A stopped worker
|
|
53
|
+
// never restarts, so one promise serves the instance's lifetime.
|
|
54
|
+
stopped$ = new Promise((r) => (this.stopWake = r));
|
|
24
55
|
// True only when this worker created its own store (via Worker.sqlite); an
|
|
25
56
|
// injected store may be shared, so serve()/background() must not close it.
|
|
26
57
|
ownsStore = false;
|
|
@@ -28,6 +59,9 @@ export class Worker {
|
|
|
28
59
|
this.store = store;
|
|
29
60
|
this.queues = queues;
|
|
30
61
|
this.opts = opts;
|
|
62
|
+
if (opts.maxRunMs != null && opts.maxRunMs <= 0) {
|
|
63
|
+
throw new Error(`maxRunMs must be > 0, got ${opts.maxRunMs}`);
|
|
64
|
+
}
|
|
31
65
|
}
|
|
32
66
|
static sqlite(path, opts = {}) {
|
|
33
67
|
const { queues = ["default"], busyTimeoutMs, ...rest } = opts;
|
|
@@ -51,8 +85,10 @@ export class Worker {
|
|
|
51
85
|
let fn;
|
|
52
86
|
if (typeof arg === "function") {
|
|
53
87
|
// Bare form: worker.task(fn) — registered under the function's name.
|
|
88
|
+
// Strip the "bound " prefix .bind() stamps on it: otherwise a bound
|
|
89
|
+
// method registers under "bound process", a name no submit ever uses.
|
|
54
90
|
fn = arg;
|
|
55
|
-
name = fn.name;
|
|
91
|
+
name = fn.name.replace(/^(bound )+/, "");
|
|
56
92
|
if (!name) {
|
|
57
93
|
throw new Error("worker.task(fn): the handler is anonymous; pass a name explicitly, " +
|
|
58
94
|
"e.g. worker.task('summary.create', fn)");
|
|
@@ -68,10 +104,7 @@ export class Worker {
|
|
|
68
104
|
}
|
|
69
105
|
stop() {
|
|
70
106
|
this.stopped = true;
|
|
71
|
-
|
|
72
|
-
this.stopResolvers = [];
|
|
73
|
-
for (const r of resolvers)
|
|
74
|
-
r();
|
|
107
|
+
this.stopWake();
|
|
75
108
|
}
|
|
76
109
|
/** Close the underlying store connection. Call after run() returns. */
|
|
77
110
|
async close() {
|
|
@@ -84,28 +117,66 @@ export class Worker {
|
|
|
84
117
|
if (this.ownsStore)
|
|
85
118
|
await this.close();
|
|
86
119
|
}
|
|
120
|
+
report(err, info) {
|
|
121
|
+
try {
|
|
122
|
+
this.opts.onError?.(err, info);
|
|
123
|
+
}
|
|
124
|
+
catch {
|
|
125
|
+
// A reporting hook must never take the worker down with it.
|
|
126
|
+
}
|
|
127
|
+
}
|
|
87
128
|
async run(opts = {}) {
|
|
88
|
-
|
|
129
|
+
// Clamped: at 0 the loop would await Promise.race([]) — pending forever,
|
|
130
|
+
// beyond even stop()'s reach.
|
|
131
|
+
const concurrency = Math.max(1, opts.concurrency ?? this.opts.concurrency ?? 1);
|
|
89
132
|
const leaseMs = this.opts.leaseMs ?? 30_000;
|
|
90
|
-
const pollMs = this.opts.pollIntervalMs ?? 500;
|
|
91
133
|
const batch = this.opts.claimBatch ?? concurrency;
|
|
92
134
|
await this.store.connect();
|
|
93
|
-
this.installSignals();
|
|
94
135
|
const running = new Set();
|
|
136
|
+
try {
|
|
137
|
+
await this.loop(concurrency, batch, leaseMs, running);
|
|
138
|
+
}
|
|
139
|
+
finally {
|
|
140
|
+
// Whatever ends the loop — stop(), or something unexpected out of the body
|
|
141
|
+
// — nothing this worker started may outlive run(). serve() closes the store
|
|
142
|
+
// as soon as run() settles, and a handler still holding the connection
|
|
143
|
+
// would fault on it.
|
|
144
|
+
await Promise.all([...running]);
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
async loop(concurrency, batch, leaseMs, running) {
|
|
148
|
+
const pollMs = this.opts.pollIntervalMs ?? 500;
|
|
95
149
|
while (!this.stopped) {
|
|
96
150
|
const free = concurrency - running.size;
|
|
97
151
|
if (free <= 0) {
|
|
98
|
-
|
|
152
|
+
// Wait for a slot rather than spinning. execute() never rejects, so
|
|
153
|
+
// racing these is safe.
|
|
154
|
+
await Promise.race([...running]);
|
|
155
|
+
continue;
|
|
156
|
+
}
|
|
157
|
+
let claimed;
|
|
158
|
+
try {
|
|
159
|
+
claimed = await this.store.claim({
|
|
160
|
+
queues: this.queues,
|
|
161
|
+
// Only what this worker can run. Queues do not partition work by task
|
|
162
|
+
// name, so another worker's tasks would otherwise be claimed here and
|
|
163
|
+
// failed for want of a handler. Read each poll: handlers may be
|
|
164
|
+
// registered after run() started.
|
|
165
|
+
names: [...this.handlers.keys()],
|
|
166
|
+
workerId: this.workerId,
|
|
167
|
+
leaseMs,
|
|
168
|
+
limit: Math.min(batch, free),
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
catch (err) {
|
|
172
|
+
// A claim can fail transiently (lock contention, a dropped connection).
|
|
173
|
+
// Report it and keep polling — one bad poll must not end the worker.
|
|
174
|
+
this.report(err, { phase: "claim" });
|
|
175
|
+
await this.sleepOrStop(CLAIM_ERROR_BACKOFF_MS);
|
|
99
176
|
continue;
|
|
100
177
|
}
|
|
101
|
-
const claimed = await this.store.claim({
|
|
102
|
-
queues: this.queues,
|
|
103
|
-
workerId: this.workerId,
|
|
104
|
-
leaseMs,
|
|
105
|
-
limit: Math.min(batch, free),
|
|
106
|
-
});
|
|
107
178
|
if (claimed.length === 0) {
|
|
108
|
-
await this.
|
|
179
|
+
await this.idle(pollMs);
|
|
109
180
|
continue;
|
|
110
181
|
}
|
|
111
182
|
for (const task of claimed) {
|
|
@@ -113,16 +184,21 @@ export class Worker {
|
|
|
113
184
|
running.add(p);
|
|
114
185
|
}
|
|
115
186
|
}
|
|
116
|
-
await Promise.all([...running]);
|
|
117
187
|
}
|
|
118
188
|
/** Blocking-style entry point for a standalone worker process: run until
|
|
119
189
|
* SIGINT/SIGTERM, then close the store. Use this at a script's top level;
|
|
120
190
|
* use run() / background() when you manage the event loop yourself. */
|
|
121
191
|
async serve(opts = {}) {
|
|
192
|
+
// Signals are installed here rather than in run(): serve() is the entry point
|
|
193
|
+
// that owns the process. run()/background() embed the worker in someone
|
|
194
|
+
// else's process, where a leftover listener suppresses Node's default Ctrl-C
|
|
195
|
+
// handling for the host long after the worker is done.
|
|
196
|
+
const removeSignalHandlers = this.installSignals();
|
|
122
197
|
try {
|
|
123
198
|
await this.run(opts);
|
|
124
199
|
}
|
|
125
200
|
finally {
|
|
201
|
+
removeSignalHandlers();
|
|
126
202
|
await this.closeIfOwned();
|
|
127
203
|
}
|
|
128
204
|
}
|
|
@@ -138,13 +214,18 @@ export class Worker {
|
|
|
138
214
|
await this.closeIfOwned();
|
|
139
215
|
}
|
|
140
216
|
}
|
|
217
|
+
/**
|
|
218
|
+
* Run one task to completion. Never rejects: a task-level failure is reported
|
|
219
|
+
* through onError and the loop moves on. (It used to reject into a promise
|
|
220
|
+
* nobody awaited — an unhandled rejection that took the process down.)
|
|
221
|
+
*/
|
|
141
222
|
async execute(task, leaseMs) {
|
|
142
223
|
const ctx = new TaskContext(this.store, task, this.workerId, leaseMs);
|
|
143
224
|
const hb = this.startHeartbeat(ctx, leaseMs);
|
|
144
225
|
try {
|
|
145
226
|
const handler = this.handlers.get(task.name);
|
|
146
227
|
if (!handler) {
|
|
147
|
-
await this.safeFail(task
|
|
228
|
+
await this.safeFail(task, errorEnvelope({
|
|
148
229
|
type: "NoHandler",
|
|
149
230
|
code: "no_handler",
|
|
150
231
|
message: `no handler registered for ${task.name}`,
|
|
@@ -154,16 +235,22 @@ export class Worker {
|
|
|
154
235
|
}
|
|
155
236
|
let result;
|
|
156
237
|
try {
|
|
157
|
-
result = await handler
|
|
238
|
+
result = await this.attempt(handler, ctx, task);
|
|
158
239
|
}
|
|
159
240
|
catch (err) {
|
|
160
241
|
if (err instanceof LostLease)
|
|
161
242
|
return;
|
|
243
|
+
if (err instanceof AttemptTimeout) {
|
|
244
|
+
// Recorded as a retryable failure, so backoff / maxAttempts /
|
|
245
|
+
// cancel-wins all apply exactly as for a thrown error.
|
|
246
|
+
await this.safeFail(task, timeoutEnvelope(task.name, err.maxRunMs), true);
|
|
247
|
+
return;
|
|
248
|
+
}
|
|
162
249
|
if (err instanceof TaskError) {
|
|
163
|
-
await this.safeFail(task
|
|
250
|
+
await this.safeFail(task, err.envelope(), err.retryable);
|
|
164
251
|
}
|
|
165
252
|
else {
|
|
166
|
-
await this.safeFail(task
|
|
253
|
+
await this.safeFail(task, exceptionEnvelope(err), true);
|
|
167
254
|
}
|
|
168
255
|
return;
|
|
169
256
|
}
|
|
@@ -173,20 +260,68 @@ export class Worker {
|
|
|
173
260
|
await this.store.complete({ taskId: task.id, workerId: this.workerId, result });
|
|
174
261
|
}
|
|
175
262
|
catch (err) {
|
|
176
|
-
if (err instanceof LostLease)
|
|
263
|
+
if (err instanceof LostLease) {
|
|
264
|
+
ctx.markLeaseLost();
|
|
177
265
|
return;
|
|
266
|
+
}
|
|
267
|
+
if (err instanceof SerializationError) {
|
|
268
|
+
// The handler succeeded but its return value can't cross the JSON
|
|
269
|
+
// protocol (BigInt, non-finite number, circular). Deterministic, so
|
|
270
|
+
// fail fast and permanently — the alternative is sitting `running`
|
|
271
|
+
// until lease expiry redelivers a task that fails the same way every
|
|
272
|
+
// attempt.
|
|
273
|
+
await this.safeFail(task, errorEnvelope({
|
|
274
|
+
type: "SerializationError",
|
|
275
|
+
code: "unserializable_result",
|
|
276
|
+
message: `handler result is not JSON-serializable: ${err.message}`,
|
|
277
|
+
retryable: false,
|
|
278
|
+
}), false);
|
|
279
|
+
return;
|
|
280
|
+
}
|
|
178
281
|
throw err;
|
|
179
282
|
}
|
|
180
283
|
}
|
|
284
|
+
catch (err) {
|
|
285
|
+
this.report(err, { phase: "execute", taskId: task.id });
|
|
286
|
+
}
|
|
181
287
|
finally {
|
|
182
288
|
hb.cancel();
|
|
183
289
|
await hb.done;
|
|
184
290
|
}
|
|
185
291
|
}
|
|
292
|
+
/**
|
|
293
|
+
* Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
|
|
294
|
+
* abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
|
|
295
|
+
* a handler that keeps running can never write again — see
|
|
296
|
+
* TaskContext.owned), then the still-pending promise is left to settle on
|
|
297
|
+
* its own, its outcome discarded. The caller records the handler_timeout
|
|
298
|
+
* failure; lease recovery is NOT involved, so redelivery is immediate.
|
|
299
|
+
*/
|
|
300
|
+
async attempt(handler, ctx, task) {
|
|
301
|
+
const maxRunMs = this.opts.maxRunMs;
|
|
302
|
+
if (maxRunMs == null)
|
|
303
|
+
return handler(ctx, task.payload);
|
|
304
|
+
// As a real promise: the race needs one (a handler may return a plain
|
|
305
|
+
// value), and the timeout path .catch()es it.
|
|
306
|
+
const run = (async () => handler(ctx, task.payload))();
|
|
307
|
+
let timer;
|
|
308
|
+
const winner = await Promise.race([
|
|
309
|
+
run,
|
|
310
|
+
new Promise((r) => (timer = setTimeout(() => r(TIMED_OUT), maxRunMs))),
|
|
311
|
+
]).finally(() => clearTimeout(timer));
|
|
312
|
+
if (winner !== TIMED_OUT)
|
|
313
|
+
return winner;
|
|
314
|
+
ctx.markLeaseLost();
|
|
315
|
+
// The zombie may still reject later; that must not become an unhandled
|
|
316
|
+
// rejection — its outcome was already decided to be handler_timeout.
|
|
317
|
+
run.catch(() => { });
|
|
318
|
+
throw new AttemptTimeout(maxRunMs);
|
|
319
|
+
}
|
|
186
320
|
startHeartbeat(ctx, leaseMs) {
|
|
187
321
|
let active = true;
|
|
188
322
|
let wake = null;
|
|
189
|
-
|
|
323
|
+
// lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
|
|
324
|
+
const interval = this.opts.heartbeatIntervalMs ?? Math.max(50, Math.floor(leaseMs / 3));
|
|
190
325
|
const done = (async () => {
|
|
191
326
|
while (active) {
|
|
192
327
|
// Cancellable sleep: cancel() resolves this immediately and clears the
|
|
@@ -205,8 +340,10 @@ export class Worker {
|
|
|
205
340
|
await ctx.heartbeat();
|
|
206
341
|
}
|
|
207
342
|
catch (err) {
|
|
343
|
+
// ctx.heartbeat() already flagged the lease as lost for the handler.
|
|
208
344
|
if (err instanceof LostLease)
|
|
209
345
|
break;
|
|
346
|
+
this.report(err, { phase: "execute", taskId: ctx.taskId });
|
|
210
347
|
}
|
|
211
348
|
}
|
|
212
349
|
})();
|
|
@@ -219,34 +356,57 @@ export class Worker {
|
|
|
219
356
|
done,
|
|
220
357
|
};
|
|
221
358
|
}
|
|
222
|
-
async safeFail(
|
|
359
|
+
async safeFail(task, envelope, retryable) {
|
|
360
|
+
const delayMs = retryable
|
|
361
|
+
? retryDelayMs(task.attempt, this.opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS, this.opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS)
|
|
362
|
+
: 0;
|
|
223
363
|
try {
|
|
224
|
-
await this.store.fail({
|
|
364
|
+
await this.store.fail({
|
|
365
|
+
taskId: task.id,
|
|
366
|
+
workerId: this.workerId,
|
|
367
|
+
error: envelope,
|
|
368
|
+
retryable,
|
|
369
|
+
delayMs,
|
|
370
|
+
});
|
|
225
371
|
}
|
|
226
372
|
catch (err) {
|
|
227
373
|
if (!(err instanceof LostLease))
|
|
228
374
|
throw err;
|
|
229
375
|
}
|
|
230
376
|
}
|
|
377
|
+
/**
|
|
378
|
+
* The empty-poll sleep. A store with a push channel (Postgres LISTEN/NOTIFY)
|
|
379
|
+
* cuts it short when a task on this worker's queues becomes claimable;
|
|
380
|
+
* stop() interrupts it either way, and sleepOrStop bounds it at `ms` so the
|
|
381
|
+
* poll fallback — which also drives lease recovery — never stretches.
|
|
382
|
+
*/
|
|
383
|
+
idle(ms) {
|
|
384
|
+
return Promise.race([this.sleepOrStop(ms), this.store.claimWake(this.queues, ms)]);
|
|
385
|
+
}
|
|
231
386
|
sleepOrStop(ms) {
|
|
232
387
|
if (this.stopped)
|
|
233
388
|
return Promise.resolve();
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
done = true;
|
|
240
|
-
clearTimeout(timer);
|
|
241
|
-
resolve();
|
|
242
|
-
};
|
|
243
|
-
const timer = setTimeout(finish, ms);
|
|
244
|
-
this.stopResolvers.push(finish);
|
|
245
|
-
});
|
|
389
|
+
let timer;
|
|
390
|
+
const nap = new Promise((r) => (timer = setTimeout(r, ms)));
|
|
391
|
+
// Clear the timer whichever side wins, so a stop is never followed by a
|
|
392
|
+
// leftover poll timer holding the process open.
|
|
393
|
+
return Promise.race([nap, this.stopped$]).finally(() => clearTimeout(timer));
|
|
246
394
|
}
|
|
395
|
+
/** Take SIGINT/SIGTERM for the duration of serve(). Returns the undo. */
|
|
247
396
|
installSignals() {
|
|
248
|
-
const
|
|
249
|
-
|
|
250
|
-
|
|
397
|
+
const remove = () => {
|
|
398
|
+
process.off("SIGINT", handler);
|
|
399
|
+
process.off("SIGTERM", handler);
|
|
400
|
+
};
|
|
401
|
+
const handler = () => {
|
|
402
|
+
// Stand down after the first signal, so a second Ctrl-C reaches Node's
|
|
403
|
+
// default and kills a worker that will not drain. `once` would only drop
|
|
404
|
+
// whichever signal fired and leave the other suppressing the default.
|
|
405
|
+
remove();
|
|
406
|
+
this.stop();
|
|
407
|
+
};
|
|
408
|
+
process.on("SIGINT", handler);
|
|
409
|
+
process.on("SIGTERM", handler);
|
|
410
|
+
return remove;
|
|
251
411
|
}
|
|
252
412
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "cairnq",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.2.0",
|
|
4
4
|
"description": "SQLite-first, cross-language, storage-centered durable task runtime",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Jannchie <jannchie@gmail.com>",
|
|
@@ -33,6 +33,7 @@
|
|
|
33
33
|
"files": ["dist", "src"],
|
|
34
34
|
"engines": { "node": ">=20" },
|
|
35
35
|
"scripts": {
|
|
36
|
+
"bench": "tsx bench/run.ts",
|
|
36
37
|
"build": "tsc -p tsconfig.json",
|
|
37
38
|
"test": "vitest run",
|
|
38
39
|
"typecheck": "tsc -p tsconfig.json --noEmit"
|
|
@@ -40,13 +41,19 @@
|
|
|
40
41
|
"dependencies": {
|
|
41
42
|
"better-sqlite3": "^11.3.0"
|
|
42
43
|
},
|
|
43
|
-
"
|
|
44
|
+
"peerDependencies": {
|
|
44
45
|
"pg": "^8.13.0"
|
|
45
46
|
},
|
|
47
|
+
"peerDependenciesMeta": {
|
|
48
|
+
"pg": {
|
|
49
|
+
"optional": true
|
|
50
|
+
}
|
|
51
|
+
},
|
|
46
52
|
"devDependencies": {
|
|
47
53
|
"@types/better-sqlite3": "^7.6.11",
|
|
48
54
|
"@types/node": "^22.7.0",
|
|
49
55
|
"@types/pg": "^8.11.10",
|
|
56
|
+
"pg": "^8.13.0",
|
|
50
57
|
"tsx": "^4.19.0",
|
|
51
58
|
"typescript": "^5.6.0",
|
|
52
59
|
"vitest": "^2.1.0"
|
package/src/client.ts
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { TaskCanceled, TaskFailed } from "./errors.js";
|
|
2
|
-
import { isFailed, isSucceeded, type Task } from "./models.js";
|
|
2
|
+
import { isFailed, isSucceeded, type Task, type TaskStatus } from "./models.js";
|
|
3
3
|
import { SQLiteStore } from "./store/sqlite.js";
|
|
4
4
|
import { PostgresStore } from "./store/postgres.js";
|
|
5
|
-
import type { ListInput, SubmitInput, TaskStore } from "./store/base.js";
|
|
5
|
+
import type { ListInput, PurgeInput, SubmitInput, TaskStore } from "./store/base.js";
|
|
6
6
|
import { type TaskDef, taskName } from "./task.js";
|
|
7
7
|
import { pollWait } from "./wait.js";
|
|
8
8
|
|
|
@@ -72,6 +72,20 @@ export class CairnQ {
|
|
|
72
72
|
return this._store.retryByKey(key, opts);
|
|
73
73
|
}
|
|
74
74
|
|
|
75
|
+
/** Delete terminal tasks that finished more than `olderThanMs` ago and return
|
|
76
|
+
* their ids. Nothing else in CairnQ removes rows, so a long-lived database
|
|
77
|
+
* needs this on a schedule. Each call is bounded by `limit` to keep the write
|
|
78
|
+
* short; loop until it returns fewer than `limit`. */
|
|
79
|
+
purge(input?: PurgeInput): Promise<string[]> {
|
|
80
|
+
return this._store.purge(input);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/** Task counts per queue, keyed by status and zero-filled across all statuses
|
|
84
|
+
* — `(await stats()).default.queued` is the backlog of a queue. */
|
|
85
|
+
stats(): Promise<Record<string, Record<TaskStatus, number>>> {
|
|
86
|
+
return this._store.stats();
|
|
87
|
+
}
|
|
88
|
+
|
|
75
89
|
wait(
|
|
76
90
|
taskId: string,
|
|
77
91
|
opts: { timeoutMs?: number; pollMs?: number } = {},
|
package/src/context.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { LostLease } from "./errors.js";
|
|
1
2
|
import { cancelRequested, type Task } from "./models.js";
|
|
2
3
|
import type { SubmitOptions } from "./client.js";
|
|
3
4
|
import type { TaskStore } from "./store/base.js";
|
|
@@ -6,6 +7,12 @@ import { pollWait } from "./wait.js";
|
|
|
6
7
|
|
|
7
8
|
/** Handed to a task handler. Worker-side capabilities mirror the Python SDK. */
|
|
8
9
|
export class TaskContext {
|
|
10
|
+
private readonly abort = new AbortController();
|
|
11
|
+
private leaseLost = false;
|
|
12
|
+
// Cancellation is monotonic: once the DB has told us a cancel was requested it
|
|
13
|
+
// can't be taken back, so canceled() can answer from this without a re-read.
|
|
14
|
+
private cancelSeen = false;
|
|
15
|
+
|
|
9
16
|
constructor(
|
|
10
17
|
private readonly store: TaskStore,
|
|
11
18
|
private readonly task: Task,
|
|
@@ -38,28 +45,78 @@ export class TaskContext {
|
|
|
38
45
|
return this.task.payload;
|
|
39
46
|
}
|
|
40
47
|
|
|
48
|
+
/**
|
|
49
|
+
* True once this worker has lost the task's lease — it expired and another
|
|
50
|
+
* worker reclaimed it. Nothing this handler writes will be recorded any more
|
|
51
|
+
* and the task is already running elsewhere, so a long handler should check
|
|
52
|
+
* this (or `signal`) and bail out instead of continuing to do side effects.
|
|
53
|
+
*/
|
|
54
|
+
get lostLease(): boolean {
|
|
55
|
+
return this.leaseLost;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** Aborts when the lease is lost. Pass it to fetch / any AbortSignal-aware API. */
|
|
59
|
+
get signal(): AbortSignal {
|
|
60
|
+
return this.abort.signal;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** @internal Called by the worker when an owned write reports a lost lease. */
|
|
64
|
+
markLeaseLost(): void {
|
|
65
|
+
if (this.leaseLost) return;
|
|
66
|
+
this.leaseLost = true;
|
|
67
|
+
this.abort.abort(new LostLease(this.task.id));
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// Every owned write returns the current row, so cancellation and lease loss
|
|
71
|
+
// ride along on writes the handler was making anyway.
|
|
72
|
+
private observe(task: Task): Task {
|
|
73
|
+
if (cancelRequested(task)) this.cancelSeen = true;
|
|
74
|
+
return task;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
private async owned(write: () => Promise<Task>): Promise<Task> {
|
|
78
|
+
// Short-circuit once the lease is known lost: nothing this context writes
|
|
79
|
+
// may be recorded any more. Locally, not just via the store's ownership
|
|
80
|
+
// check — after an abandoned (timed-out) attempt the same worker may
|
|
81
|
+
// re-claim this task under the same workerId, and a zombie handler's write
|
|
82
|
+
// would then pass ownership against the NEW attempt.
|
|
83
|
+
if (this.leaseLost) throw new LostLease(this.task.id);
|
|
84
|
+
try {
|
|
85
|
+
return this.observe(await write());
|
|
86
|
+
} catch (err) {
|
|
87
|
+
if (err instanceof LostLease) this.markLeaseLost();
|
|
88
|
+
throw err;
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
41
92
|
async progress(value: number | null, message: string | null = null): Promise<Task> {
|
|
42
|
-
return this.
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
93
|
+
return this.owned(() =>
|
|
94
|
+
this.store.progress({
|
|
95
|
+
taskId: this.task.id,
|
|
96
|
+
workerId: this.workerId,
|
|
97
|
+
progress: value,
|
|
98
|
+
message,
|
|
99
|
+
}),
|
|
100
|
+
);
|
|
48
101
|
}
|
|
49
102
|
|
|
50
103
|
async heartbeat(): Promise<Task> {
|
|
51
|
-
return this.
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
104
|
+
return this.owned(() =>
|
|
105
|
+
this.store.heartbeat({
|
|
106
|
+
taskId: this.task.id,
|
|
107
|
+
workerId: this.workerId,
|
|
108
|
+
leaseMs: this.leaseMs,
|
|
109
|
+
}),
|
|
110
|
+
);
|
|
56
111
|
}
|
|
57
112
|
|
|
58
|
-
/** Cooperative cancel check. */
|
|
113
|
+
/** Cooperative cancel check. Free once a heartbeat has already seen the flag. */
|
|
59
114
|
async canceled(): Promise<boolean> {
|
|
115
|
+
if (this.cancelSeen) return true;
|
|
60
116
|
const t = await this.store.get(this.task.id);
|
|
61
117
|
if (!t) return true;
|
|
62
|
-
|
|
118
|
+
if (cancelRequested(t)) this.cancelSeen = true;
|
|
119
|
+
return this.cancelSeen || t.status === "canceled";
|
|
63
120
|
}
|
|
64
121
|
|
|
65
122
|
/** Submit a child task; parent/root/correlation are wired automatically. */
|
package/src/errors.ts
CHANGED
|
@@ -1,3 +1,6 @@
|
|
|
1
|
+
import { nowMs } from "./ids.js";
|
|
2
|
+
import { cancelRequested, isQueued, type Task } from "./models.js";
|
|
3
|
+
|
|
1
4
|
/** The single shape of the JSON error envelope (see PROTOCOL.md). Everything that
|
|
2
5
|
* records an error — a handler exception, a missing handler, lease expiry, a thrown
|
|
3
6
|
* TaskError — builds it here, so the contract's fields live in one place. */
|
|
@@ -17,7 +20,15 @@ export function errorEnvelope(e: {
|
|
|
17
20
|
};
|
|
18
21
|
}
|
|
19
22
|
|
|
20
|
-
export class CairnQError extends Error {
|
|
23
|
+
export class CairnQError extends Error {
|
|
24
|
+
constructor(message?: string) {
|
|
25
|
+
super(message);
|
|
26
|
+
// Subclasses each set their own; without this a bare CairnQError reports
|
|
27
|
+
// "Error", and `err.name` is how callers (and the conformance runner) tell
|
|
28
|
+
// one apart from another.
|
|
29
|
+
this.name = "CairnQError";
|
|
30
|
+
}
|
|
31
|
+
}
|
|
21
32
|
|
|
22
33
|
export class AlreadyExists extends CairnQError {
|
|
23
34
|
constructor(public key: string) {
|
|
@@ -26,11 +37,44 @@ export class AlreadyExists extends CairnQError {
|
|
|
26
37
|
}
|
|
27
38
|
}
|
|
28
39
|
|
|
29
|
-
/**
|
|
40
|
+
/** One line of "why hasn't this finished" from the last snapshot wait()
|
|
41
|
+
* observed. No worker running, no handler for the name, wrong queue, and two
|
|
42
|
+
* processes on different database files all look identical from the API side —
|
|
43
|
+
* queued, never claimed — so that case names the likely causes. */
|
|
44
|
+
function timeoutDetail(task: Task | null): string {
|
|
45
|
+
if (!task) return "task not found — wrong database file, or already purged?";
|
|
46
|
+
if (isQueued(task)) {
|
|
47
|
+
const delayMs = task.run_at_ms - nowMs();
|
|
48
|
+
if (task.attempt === 0 && delayMs <= 0) {
|
|
49
|
+
return (
|
|
50
|
+
`never claimed by a worker — is a worker running with a handler for ` +
|
|
51
|
+
`'${task.name}' on queue '${task.queue}', against this same database?`
|
|
52
|
+
);
|
|
53
|
+
}
|
|
54
|
+
const next = delayMs > 0 ? `, next run in ~${delayMs}ms` : "";
|
|
55
|
+
return `still queued (attempt ${task.attempt}/${task.max_attempts})${next}`;
|
|
56
|
+
}
|
|
57
|
+
if (cancelRequested(task)) return "cancel requested, waiting for the handler to observe it";
|
|
58
|
+
return `still running (attempt ${task.attempt}/${task.max_attempts})`;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** wait/call did not reach a terminal status in time. The task keeps running.
|
|
62
|
+
* `task` is the last snapshot wait() observed (null if get() found nothing), and
|
|
63
|
+
* the message says what state it was stuck in — a queued-never-claimed task is
|
|
64
|
+
* the classic first-run failure (no worker, no handler, wrong queue or file). */
|
|
30
65
|
export class TaskTimeout extends CairnQError {
|
|
31
|
-
|
|
32
|
-
|
|
66
|
+
readonly task: Task | null;
|
|
67
|
+
constructor(
|
|
68
|
+
public taskId: string,
|
|
69
|
+
opts: { timeoutMs?: number; task?: Task | null } = {},
|
|
70
|
+
) {
|
|
71
|
+
super(
|
|
72
|
+
opts.timeoutMs == null
|
|
73
|
+
? `task ${taskId} did not finish in time`
|
|
74
|
+
: `task ${taskId} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null)}`,
|
|
75
|
+
);
|
|
33
76
|
this.name = "TaskTimeout";
|
|
77
|
+
this.task = opts.task ?? null;
|
|
34
78
|
}
|
|
35
79
|
}
|
|
36
80
|
|
|
@@ -81,6 +125,17 @@ export class ProtocolVersionMismatch extends CairnQError {
|
|
|
81
125
|
}
|
|
82
126
|
}
|
|
83
127
|
|
|
128
|
+
/** A value could not be encoded for a protocol JSON column (non-finite number,
|
|
129
|
+
* BigInt, circular structure, …). Raised at the boundary — submit rejects with
|
|
130
|
+
* it, and a worker records a handler result that triggers it as a permanent
|
|
131
|
+
* `unserializable_result` failure. The Python SDK raises the same named error. */
|
|
132
|
+
export class SerializationError extends CairnQError {
|
|
133
|
+
constructor(message: string) {
|
|
134
|
+
super(message);
|
|
135
|
+
this.name = "SerializationError";
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
|
|
84
139
|
/** Throw inside a handler to control how the failure is recorded. Defaults to
|
|
85
140
|
* non-retryable so deterministic errors fail fast instead of burning retries.
|
|
86
141
|
* Any other thrown value is treated as retryable. */
|
package/src/index.ts
CHANGED
|
@@ -7,7 +7,8 @@ export { defineTask } from "./task.js";
|
|
|
7
7
|
export type { TaskDef } from "./task.js";
|
|
8
8
|
export { SQLiteStore } from "./store/sqlite.js";
|
|
9
9
|
export { PostgresStore } from "./store/postgres.js";
|
|
10
|
-
export
|
|
10
|
+
export { TaskStore } from "./store/base.js";
|
|
11
|
+
export type { ListInput, PurgeInput, SubmitInput, Conflict } from "./store/base.js";
|
|
11
12
|
export type { Task, TaskStatus } from "./models.js";
|
|
12
13
|
export {
|
|
13
14
|
STATUSES,
|
|
@@ -28,4 +29,5 @@ export {
|
|
|
28
29
|
TaskError,
|
|
29
30
|
LostLease,
|
|
30
31
|
ProtocolVersionMismatch,
|
|
32
|
+
SerializationError,
|
|
31
33
|
} from "./errors.js";
|