cairnq 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -0
- package/dist/_protocol/migrations/postgres/0001_init.sql +3 -1
- package/dist/_protocol/migrations/postgres/0002_purge_index.sql +6 -0
- package/dist/_protocol/migrations/postgres/0003_notify.sql +38 -0
- package/dist/_protocol/migrations/postgres/0004_lease_index.sql +16 -0
- package/dist/_protocol/migrations/postgres/0005_clear_terminal_lease.sql +17 -0
- package/dist/_protocol/migrations/sqlite/0001_init.sql +3 -1
- package/dist/_protocol/migrations/sqlite/0002_purge_index.sql +6 -0
- package/dist/_protocol/migrations/sqlite/0004_lease_index.sql +22 -0
- package/dist/_protocol/migrations/sqlite/0005_clear_terminal_lease.sql +17 -0
- package/dist/_protocol/sql/postgres/claim.sql +18 -5
- package/dist/_protocol/sql/postgres/claim_one_queue.sql +35 -0
- package/dist/_protocol/sql/postgres/complete.sql +3 -0
- package/dist/_protocol/sql/postgres/fail.sql +30 -8
- package/dist/_protocol/sql/postgres/insert_task.sql +6 -3
- package/dist/_protocol/sql/postgres/list.sql +3 -1
- package/dist/_protocol/sql/postgres/lock_key.sql +9 -0
- package/dist/_protocol/sql/postgres/progress.sql +4 -3
- package/dist/_protocol/sql/postgres/protocol_version.sql +4 -0
- package/dist/_protocol/sql/postgres/purge.sql +25 -0
- package/dist/_protocol/sql/postgres/recover_leases.sql +49 -14
- package/dist/_protocol/sql/postgres/retry.sql +3 -0
- package/dist/_protocol/sql/postgres/stats.sql +8 -0
- package/dist/_protocol/sql/postgres/succeed.sql +4 -0
- package/dist/_protocol/sql/sqlite/claim.sql +13 -2
- package/dist/_protocol/sql/sqlite/claim_one_queue.sql +36 -0
- package/dist/_protocol/sql/sqlite/claimable_probe.sql +6 -2
- package/dist/_protocol/sql/sqlite/complete.sql +3 -0
- package/dist/_protocol/sql/sqlite/fail.sql +32 -8
- package/dist/_protocol/sql/sqlite/list.sql +3 -1
- package/dist/_protocol/sql/sqlite/lock_key.sql +5 -0
- package/dist/_protocol/sql/sqlite/progress.sql +6 -2
- package/dist/_protocol/sql/sqlite/protocol_version.sql +4 -0
- package/dist/_protocol/sql/sqlite/purge.sql +18 -0
- package/dist/_protocol/sql/sqlite/recover_leases.sql +25 -7
- package/dist/_protocol/sql/sqlite/retry.sql +3 -0
- package/dist/_protocol/sql/sqlite/stats.sql +8 -0
- package/dist/_protocol/sql/sqlite/succeed.sql +4 -0
- package/dist/client.d.ts +10 -2
- package/dist/client.js +12 -0
- package/dist/context.d.ts +17 -1
- package/dist/context.js +60 -6
- package/dist/errors.d.ts +18 -2
- package/dist/errors.js +49 -3
- package/dist/index.d.ts +3 -2
- package/dist/index.js +2 -1
- package/dist/sql.js +16 -9
- package/dist/store/base.d.ts +114 -9
- package/dist/store/base.js +376 -1
- package/dist/store/postgres.d.ts +62 -63
- package/dist/store/postgres.js +245 -222
- package/dist/store/sqlite.d.ts +83 -60
- package/dist/store/sqlite.js +370 -234
- package/dist/wait.d.ts +15 -2
- package/dist/wait.js +23 -5
- package/dist/worker.d.ts +53 -1
- package/dist/worker.js +202 -42
- package/package.json +9 -2
- package/src/client.ts +16 -2
- package/src/context.ts +70 -13
- package/src/errors.ts +59 -4
- package/src/index.ts +3 -1
- package/src/sql.ts +15 -8
- package/src/store/base.ts +443 -27
- package/src/store/postgres.ts +243 -267
- package/src/store/sqlite.ts +378 -263
- package/src/wait.ts +28 -5
- package/src/worker.ts +242 -42
package/src/wait.ts
CHANGED
|
@@ -3,21 +3,44 @@ import { nowMs } from "./ids.js";
|
|
|
3
3
|
import { isTerminal, type Task } from "./models.js";
|
|
4
4
|
import type { TaskStore } from "./store/base.js";
|
|
5
5
|
|
|
6
|
-
const
|
|
6
|
+
export const DEFAULT_POLL_MS = 100;
|
|
7
|
+
export const MAX_POLL_MS = 500;
|
|
8
|
+
const GROWTH = 1.5;
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Grow the polling interval towards the ceiling.
|
|
12
|
+
*
|
|
13
|
+
* wait() has no idea whether the task takes 50ms or an hour. Starting tight keeps
|
|
14
|
+
* short tasks snappy; growing keeps a long wait from costing a read every 100ms
|
|
15
|
+
* for its whole duration. The +1 keeps truncation from pinning tiny intervals:
|
|
16
|
+
* Math.floor(1 * 1.5) === 1 would otherwise never grow past 1.
|
|
17
|
+
*/
|
|
18
|
+
export function nextPollMs(current: number, maxMs: number): number {
|
|
19
|
+
return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
|
|
20
|
+
}
|
|
7
21
|
|
|
8
22
|
/** Poll get() until terminal or timeout. Returns the terminal Task (any status).
|
|
9
|
-
* Throws TaskTimeout, leaving the task running.
|
|
23
|
+
* Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
|
|
24
|
+
* it backs off towards `maxPollMs`. */
|
|
10
25
|
export async function pollWait(
|
|
11
26
|
store: TaskStore,
|
|
12
27
|
taskId: string,
|
|
13
|
-
{
|
|
28
|
+
{
|
|
29
|
+
timeoutMs,
|
|
30
|
+
pollMs = DEFAULT_POLL_MS,
|
|
31
|
+
maxPollMs = MAX_POLL_MS,
|
|
32
|
+
}: { timeoutMs: number; pollMs?: number; maxPollMs?: number },
|
|
14
33
|
): Promise<Task> {
|
|
15
34
|
const deadline = nowMs() + timeoutMs;
|
|
35
|
+
let interval = pollMs;
|
|
16
36
|
for (;;) {
|
|
17
37
|
const task = await store.get(taskId);
|
|
18
38
|
if (task && isTerminal(task)) return task;
|
|
19
39
|
const remaining = deadline - nowMs();
|
|
20
|
-
if (remaining <= 0) throw new TaskTimeout(taskId);
|
|
21
|
-
|
|
40
|
+
if (remaining <= 0) throw new TaskTimeout(taskId, { timeoutMs, task });
|
|
41
|
+
// A store with a push channel (Postgres) cuts the sleep short when the task
|
|
42
|
+
// goes terminal; the re-get above stays the source of truth either way.
|
|
43
|
+
await store.taskDoneWake(taskId, Math.min(interval, remaining));
|
|
44
|
+
interval = nextPollMs(interval, maxPollMs);
|
|
22
45
|
}
|
|
23
46
|
}
|
package/src/worker.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { TaskContext } from "./context.js";
|
|
2
|
-
import { errorEnvelope, LostLease, TaskError } from "./errors.js";
|
|
2
|
+
import { errorEnvelope, LostLease, SerializationError, TaskError } from "./errors.js";
|
|
3
3
|
import { newId } from "./ids.js";
|
|
4
4
|
import { type Task } from "./models.js";
|
|
5
5
|
import { SQLiteStore } from "./store/sqlite.js";
|
|
@@ -11,12 +11,47 @@ export type Handler = (ctx: TaskContext, payload: any) => unknown | Promise<unkn
|
|
|
11
11
|
/** Handler typed against a TaskDef<P, R>: payload is P, the return is R. */
|
|
12
12
|
export type TypedHandler<P, R> = (ctx: TaskContext, payload: P) => R | Promise<R>;
|
|
13
13
|
|
|
14
|
+
/** Where an error the worker recovered from came from. */
|
|
15
|
+
export type ErrorPhase = "claim" | "execute";
|
|
16
|
+
|
|
14
17
|
export interface WorkerOptions {
|
|
15
18
|
concurrency?: number;
|
|
16
19
|
leaseMs?: number;
|
|
17
20
|
heartbeatIntervalMs?: number;
|
|
18
21
|
pollIntervalMs?: number;
|
|
19
22
|
claimBatch?: number;
|
|
23
|
+
/** Base delay before re-running a failed attempt; doubles per attempt. 0 disables. */
|
|
24
|
+
retryBackoffMs?: number;
|
|
25
|
+
/** Ceiling for the doubling. */
|
|
26
|
+
retryBackoffMaxMs?: number;
|
|
27
|
+
/**
|
|
28
|
+
* Wall-clock ceiling for one attempt. The heartbeat renews the lease for as
|
|
29
|
+
* long as a handler runs, so a hung handler would otherwise hold its task
|
|
30
|
+
* `running` (and its concurrency slot) forever — cancel can't help,
|
|
31
|
+
* cooperative checks need a live handler. On expiry the worker abandons the
|
|
32
|
+
* attempt (ctx.signal aborts, further ctx writes throw LostLease) and records
|
|
33
|
+
* a retryable `handler_timeout` failure. Unset disables the ceiling.
|
|
34
|
+
*/
|
|
35
|
+
maxRunMs?: number;
|
|
36
|
+
/**
|
|
37
|
+
* Called for errors the worker survived — a claim that threw, a store write
|
|
38
|
+
* that failed while finalizing a task. Without it these are silent: the run
|
|
39
|
+
* loop carries on either way, so this is the only place an operator learns a
|
|
40
|
+
* worker is limping. Must not throw.
|
|
41
|
+
*/
|
|
42
|
+
onError?: (err: unknown, info: { phase: ErrorPhase; taskId?: string }) => void;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
const DEFAULT_RETRY_BACKOFF_MS = 1_000;
|
|
46
|
+
const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
|
|
47
|
+
/** Wait after a failed claim, so a broken database is not polled in a tight loop. */
|
|
48
|
+
const CLAIM_ERROR_BACKOFF_MS = 250;
|
|
49
|
+
|
|
50
|
+
/** Exponential backoff for the next attempt of a task that just failed. */
|
|
51
|
+
export function retryDelayMs(attempt: number, baseMs: number, maxMs: number): number {
|
|
52
|
+
if (baseMs <= 0) return 0;
|
|
53
|
+
const exponent = Math.max(0, attempt - 1);
|
|
54
|
+
return Math.min(maxMs, baseMs * 2 ** exponent);
|
|
20
55
|
}
|
|
21
56
|
|
|
22
57
|
function exceptionEnvelope(err: unknown): Record<string, unknown> {
|
|
@@ -29,11 +64,32 @@ function exceptionEnvelope(err: unknown): Record<string, unknown> {
|
|
|
29
64
|
});
|
|
30
65
|
}
|
|
31
66
|
|
|
67
|
+
/** Internal: an attempt outran maxRunMs and was abandoned. */
|
|
68
|
+
class AttemptTimeout extends Error {
|
|
69
|
+
constructor(readonly maxRunMs: number) {
|
|
70
|
+
super(`attempt exceeded ${maxRunMs}ms`);
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function timeoutEnvelope(name: string, maxRunMs: number): Record<string, unknown> {
|
|
75
|
+
return errorEnvelope({
|
|
76
|
+
type: "HandlerTimeout",
|
|
77
|
+
code: "handler_timeout",
|
|
78
|
+
message: `handler for ${name} exceeded maxRunMs=${maxRunMs}ms; the attempt was abandoned`,
|
|
79
|
+
retryable: true,
|
|
80
|
+
});
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
const TIMED_OUT = Symbol("cairnq.timedOut");
|
|
84
|
+
|
|
32
85
|
export class Worker {
|
|
33
86
|
private readonly handlers = new Map<string, Handler>();
|
|
34
87
|
private readonly workerId = newId("worker");
|
|
35
88
|
private stopped = false;
|
|
36
|
-
private
|
|
89
|
+
private stopWake!: () => void;
|
|
90
|
+
// Resolved once by stop(); every sleep races against it. A stopped worker
|
|
91
|
+
// never restarts, so one promise serves the instance's lifetime.
|
|
92
|
+
private readonly stopped$ = new Promise<void>((r) => (this.stopWake = r));
|
|
37
93
|
// True only when this worker created its own store (via Worker.sqlite); an
|
|
38
94
|
// injected store may be shared, so serve()/background() must not close it.
|
|
39
95
|
private ownsStore = false;
|
|
@@ -42,7 +98,11 @@ export class Worker {
|
|
|
42
98
|
private readonly store: TaskStore,
|
|
43
99
|
private readonly queues: string[],
|
|
44
100
|
private readonly opts: WorkerOptions = {},
|
|
45
|
-
) {
|
|
101
|
+
) {
|
|
102
|
+
if (opts.maxRunMs != null && opts.maxRunMs <= 0) {
|
|
103
|
+
throw new Error(`maxRunMs must be > 0, got ${opts.maxRunMs}`);
|
|
104
|
+
}
|
|
105
|
+
}
|
|
46
106
|
|
|
47
107
|
static sqlite(
|
|
48
108
|
path: string,
|
|
@@ -78,8 +138,10 @@ export class Worker {
|
|
|
78
138
|
let fn: Handler;
|
|
79
139
|
if (typeof arg === "function") {
|
|
80
140
|
// Bare form: worker.task(fn) — registered under the function's name.
|
|
141
|
+
// Strip the "bound " prefix .bind() stamps on it: otherwise a bound
|
|
142
|
+
// method registers under "bound process", a name no submit ever uses.
|
|
81
143
|
fn = arg;
|
|
82
|
-
name = fn.name;
|
|
144
|
+
name = fn.name.replace(/^(bound )+/, "");
|
|
83
145
|
if (!name) {
|
|
84
146
|
throw new Error(
|
|
85
147
|
"worker.task(fn): the handler is anonymous; pass a name explicitly, " +
|
|
@@ -97,9 +159,7 @@ export class Worker {
|
|
|
97
159
|
|
|
98
160
|
stop(): void {
|
|
99
161
|
this.stopped = true;
|
|
100
|
-
|
|
101
|
-
this.stopResolvers = [];
|
|
102
|
-
for (const r of resolvers) r();
|
|
162
|
+
this.stopWake();
|
|
103
163
|
}
|
|
104
164
|
|
|
105
165
|
/** Close the underlying store connection. Call after run() returns. */
|
|
@@ -114,28 +174,70 @@ export class Worker {
|
|
|
114
174
|
if (this.ownsStore) await this.close();
|
|
115
175
|
}
|
|
116
176
|
|
|
177
|
+
private report(err: unknown, info: { phase: ErrorPhase; taskId?: string }): void {
|
|
178
|
+
try {
|
|
179
|
+
this.opts.onError?.(err, info);
|
|
180
|
+
} catch {
|
|
181
|
+
// A reporting hook must never take the worker down with it.
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
117
185
|
async run(opts: { concurrency?: number } = {}): Promise<void> {
|
|
118
|
-
|
|
186
|
+
// Clamped: at 0 the loop would await Promise.race([]) — pending forever,
|
|
187
|
+
// beyond even stop()'s reach.
|
|
188
|
+
const concurrency = Math.max(1, opts.concurrency ?? this.opts.concurrency ?? 1);
|
|
119
189
|
const leaseMs = this.opts.leaseMs ?? 30_000;
|
|
120
|
-
const pollMs = this.opts.pollIntervalMs ?? 500;
|
|
121
190
|
const batch = this.opts.claimBatch ?? concurrency;
|
|
122
191
|
await this.store.connect();
|
|
123
|
-
this.installSignals();
|
|
124
192
|
const running = new Set<Promise<void>>();
|
|
193
|
+
try {
|
|
194
|
+
await this.loop(concurrency, batch, leaseMs, running);
|
|
195
|
+
} finally {
|
|
196
|
+
// Whatever ends the loop — stop(), or something unexpected out of the body
|
|
197
|
+
// — nothing this worker started may outlive run(). serve() closes the store
|
|
198
|
+
// as soon as run() settles, and a handler still holding the connection
|
|
199
|
+
// would fault on it.
|
|
200
|
+
await Promise.all([...running]);
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
private async loop(
|
|
205
|
+
concurrency: number,
|
|
206
|
+
batch: number,
|
|
207
|
+
leaseMs: number,
|
|
208
|
+
running: Set<Promise<void>>,
|
|
209
|
+
): Promise<void> {
|
|
210
|
+
const pollMs = this.opts.pollIntervalMs ?? 500;
|
|
125
211
|
while (!this.stopped) {
|
|
126
212
|
const free = concurrency - running.size;
|
|
127
213
|
if (free <= 0) {
|
|
128
|
-
|
|
214
|
+
// Wait for a slot rather than spinning. execute() never rejects, so
|
|
215
|
+
// racing these is safe.
|
|
216
|
+
await Promise.race([...running]);
|
|
217
|
+
continue;
|
|
218
|
+
}
|
|
219
|
+
let claimed: Task[];
|
|
220
|
+
try {
|
|
221
|
+
claimed = await this.store.claim({
|
|
222
|
+
queues: this.queues,
|
|
223
|
+
// Only what this worker can run. Queues do not partition work by task
|
|
224
|
+
// name, so another worker's tasks would otherwise be claimed here and
|
|
225
|
+
// failed for want of a handler. Read each poll: handlers may be
|
|
226
|
+
// registered after run() started.
|
|
227
|
+
names: [...this.handlers.keys()],
|
|
228
|
+
workerId: this.workerId,
|
|
229
|
+
leaseMs,
|
|
230
|
+
limit: Math.min(batch, free),
|
|
231
|
+
});
|
|
232
|
+
} catch (err) {
|
|
233
|
+
// A claim can fail transiently (lock contention, a dropped connection).
|
|
234
|
+
// Report it and keep polling — one bad poll must not end the worker.
|
|
235
|
+
this.report(err, { phase: "claim" });
|
|
236
|
+
await this.sleepOrStop(CLAIM_ERROR_BACKOFF_MS);
|
|
129
237
|
continue;
|
|
130
238
|
}
|
|
131
|
-
const claimed = await this.store.claim({
|
|
132
|
-
queues: this.queues,
|
|
133
|
-
workerId: this.workerId,
|
|
134
|
-
leaseMs,
|
|
135
|
-
limit: Math.min(batch, free),
|
|
136
|
-
});
|
|
137
239
|
if (claimed.length === 0) {
|
|
138
|
-
await this.
|
|
240
|
+
await this.idle(pollMs);
|
|
139
241
|
continue;
|
|
140
242
|
}
|
|
141
243
|
for (const task of claimed) {
|
|
@@ -143,16 +245,21 @@ export class Worker {
|
|
|
143
245
|
running.add(p);
|
|
144
246
|
}
|
|
145
247
|
}
|
|
146
|
-
await Promise.all([...running]);
|
|
147
248
|
}
|
|
148
249
|
|
|
149
250
|
/** Blocking-style entry point for a standalone worker process: run until
|
|
150
251
|
* SIGINT/SIGTERM, then close the store. Use this at a script's top level;
|
|
151
252
|
* use run() / background() when you manage the event loop yourself. */
|
|
152
253
|
async serve(opts: { concurrency?: number } = {}): Promise<void> {
|
|
254
|
+
// Signals are installed here rather than in run(): serve() is the entry point
|
|
255
|
+
// that owns the process. run()/background() embed the worker in someone
|
|
256
|
+
// else's process, where a leftover listener suppresses Node's default Ctrl-C
|
|
257
|
+
// handling for the host long after the worker is done.
|
|
258
|
+
const removeSignalHandlers = this.installSignals();
|
|
153
259
|
try {
|
|
154
260
|
await this.run(opts);
|
|
155
261
|
} finally {
|
|
262
|
+
removeSignalHandlers();
|
|
156
263
|
await this.closeIfOwned();
|
|
157
264
|
}
|
|
158
265
|
}
|
|
@@ -169,6 +276,11 @@ export class Worker {
|
|
|
169
276
|
}
|
|
170
277
|
}
|
|
171
278
|
|
|
279
|
+
/**
|
|
280
|
+
* Run one task to completion. Never rejects: a task-level failure is reported
|
|
281
|
+
* through onError and the loop moves on. (It used to reject into a promise
|
|
282
|
+
* nobody awaited — an unhandled rejection that took the process down.)
|
|
283
|
+
*/
|
|
172
284
|
private async execute(task: Task, leaseMs: number): Promise<void> {
|
|
173
285
|
const ctx = new TaskContext(this.store, task, this.workerId, leaseMs);
|
|
174
286
|
const hb = this.startHeartbeat(ctx, leaseMs);
|
|
@@ -176,7 +288,7 @@ export class Worker {
|
|
|
176
288
|
const handler = this.handlers.get(task.name);
|
|
177
289
|
if (!handler) {
|
|
178
290
|
await this.safeFail(
|
|
179
|
-
task
|
|
291
|
+
task,
|
|
180
292
|
errorEnvelope({
|
|
181
293
|
type: "NoHandler",
|
|
182
294
|
code: "no_handler",
|
|
@@ -189,13 +301,19 @@ export class Worker {
|
|
|
189
301
|
}
|
|
190
302
|
let result: unknown;
|
|
191
303
|
try {
|
|
192
|
-
result = await handler
|
|
304
|
+
result = await this.attempt(handler, ctx, task);
|
|
193
305
|
} catch (err) {
|
|
194
306
|
if (err instanceof LostLease) return;
|
|
307
|
+
if (err instanceof AttemptTimeout) {
|
|
308
|
+
// Recorded as a retryable failure, so backoff / maxAttempts /
|
|
309
|
+
// cancel-wins all apply exactly as for a thrown error.
|
|
310
|
+
await this.safeFail(task, timeoutEnvelope(task.name, err.maxRunMs), true);
|
|
311
|
+
return;
|
|
312
|
+
}
|
|
195
313
|
if (err instanceof TaskError) {
|
|
196
|
-
await this.safeFail(task
|
|
314
|
+
await this.safeFail(task, err.envelope(), err.retryable);
|
|
197
315
|
} else {
|
|
198
|
-
await this.safeFail(task
|
|
316
|
+
await this.safeFail(task, exceptionEnvelope(err), true);
|
|
199
317
|
}
|
|
200
318
|
return;
|
|
201
319
|
}
|
|
@@ -204,22 +322,73 @@ export class Worker {
|
|
|
204
322
|
// requested while the handler ran, else succeeded.
|
|
205
323
|
await this.store.complete({ taskId: task.id, workerId: this.workerId, result });
|
|
206
324
|
} catch (err) {
|
|
207
|
-
if (err instanceof LostLease)
|
|
325
|
+
if (err instanceof LostLease) {
|
|
326
|
+
ctx.markLeaseLost();
|
|
327
|
+
return;
|
|
328
|
+
}
|
|
329
|
+
if (err instanceof SerializationError) {
|
|
330
|
+
// The handler succeeded but its return value can't cross the JSON
|
|
331
|
+
// protocol (BigInt, non-finite number, circular). Deterministic, so
|
|
332
|
+
// fail fast and permanently — the alternative is sitting `running`
|
|
333
|
+
// until lease expiry redelivers a task that fails the same way every
|
|
334
|
+
// attempt.
|
|
335
|
+
await this.safeFail(
|
|
336
|
+
task,
|
|
337
|
+
errorEnvelope({
|
|
338
|
+
type: "SerializationError",
|
|
339
|
+
code: "unserializable_result",
|
|
340
|
+
message: `handler result is not JSON-serializable: ${err.message}`,
|
|
341
|
+
retryable: false,
|
|
342
|
+
}),
|
|
343
|
+
false,
|
|
344
|
+
);
|
|
345
|
+
return;
|
|
346
|
+
}
|
|
208
347
|
throw err;
|
|
209
348
|
}
|
|
349
|
+
} catch (err) {
|
|
350
|
+
this.report(err, { phase: "execute", taskId: task.id });
|
|
210
351
|
} finally {
|
|
211
352
|
hb.cancel();
|
|
212
353
|
await hb.done;
|
|
213
354
|
}
|
|
214
355
|
}
|
|
215
356
|
|
|
357
|
+
/**
|
|
358
|
+
* Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
|
|
359
|
+
* abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
|
|
360
|
+
* a handler that keeps running can never write again — see
|
|
361
|
+
* TaskContext.owned), then the still-pending promise is left to settle on
|
|
362
|
+
* its own, its outcome discarded. The caller records the handler_timeout
|
|
363
|
+
* failure; lease recovery is NOT involved, so redelivery is immediate.
|
|
364
|
+
*/
|
|
365
|
+
private async attempt(handler: Handler, ctx: TaskContext, task: Task): Promise<unknown> {
|
|
366
|
+
const maxRunMs = this.opts.maxRunMs;
|
|
367
|
+
if (maxRunMs == null) return handler(ctx, task.payload);
|
|
368
|
+
// As a real promise: the race needs one (a handler may return a plain
|
|
369
|
+
// value), and the timeout path .catch()es it.
|
|
370
|
+
const run = (async () => handler(ctx, task.payload))();
|
|
371
|
+
let timer!: NodeJS.Timeout;
|
|
372
|
+
const winner = await Promise.race([
|
|
373
|
+
run,
|
|
374
|
+
new Promise<typeof TIMED_OUT>((r) => (timer = setTimeout(() => r(TIMED_OUT), maxRunMs))),
|
|
375
|
+
]).finally(() => clearTimeout(timer));
|
|
376
|
+
if (winner !== TIMED_OUT) return winner;
|
|
377
|
+
ctx.markLeaseLost();
|
|
378
|
+
// The zombie may still reject later; that must not become an unhandled
|
|
379
|
+
// rejection — its outcome was already decided to be handler_timeout.
|
|
380
|
+
run.catch(() => {});
|
|
381
|
+
throw new AttemptTimeout(maxRunMs);
|
|
382
|
+
}
|
|
383
|
+
|
|
216
384
|
private startHeartbeat(
|
|
217
385
|
ctx: TaskContext,
|
|
218
386
|
leaseMs: number,
|
|
219
387
|
): { cancel: () => void; done: Promise<void> } {
|
|
220
388
|
let active = true;
|
|
221
389
|
let wake: (() => void) | null = null;
|
|
222
|
-
|
|
390
|
+
// lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
|
|
391
|
+
const interval = this.opts.heartbeatIntervalMs ?? Math.max(50, Math.floor(leaseMs / 3));
|
|
223
392
|
const done = (async () => {
|
|
224
393
|
while (active) {
|
|
225
394
|
// Cancellable sleep: cancel() resolves this immediately and clears the
|
|
@@ -236,7 +405,9 @@ export class Worker {
|
|
|
236
405
|
try {
|
|
237
406
|
await ctx.heartbeat();
|
|
238
407
|
} catch (err) {
|
|
408
|
+
// ctx.heartbeat() already flagged the lease as lost for the handler.
|
|
239
409
|
if (err instanceof LostLease) break;
|
|
410
|
+
this.report(err, { phase: "execute", taskId: ctx.taskId });
|
|
240
411
|
}
|
|
241
412
|
}
|
|
242
413
|
})();
|
|
@@ -250,35 +421,64 @@ export class Worker {
|
|
|
250
421
|
}
|
|
251
422
|
|
|
252
423
|
private async safeFail(
|
|
253
|
-
|
|
424
|
+
task: Task,
|
|
254
425
|
envelope: Record<string, unknown>,
|
|
255
426
|
retryable: boolean,
|
|
256
427
|
): Promise<void> {
|
|
428
|
+
const delayMs = retryable
|
|
429
|
+
? retryDelayMs(
|
|
430
|
+
task.attempt,
|
|
431
|
+
this.opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS,
|
|
432
|
+
this.opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS,
|
|
433
|
+
)
|
|
434
|
+
: 0;
|
|
257
435
|
try {
|
|
258
|
-
await this.store.fail({
|
|
436
|
+
await this.store.fail({
|
|
437
|
+
taskId: task.id,
|
|
438
|
+
workerId: this.workerId,
|
|
439
|
+
error: envelope,
|
|
440
|
+
retryable,
|
|
441
|
+
delayMs,
|
|
442
|
+
});
|
|
259
443
|
} catch (err) {
|
|
260
444
|
if (!(err instanceof LostLease)) throw err;
|
|
261
445
|
}
|
|
262
446
|
}
|
|
263
447
|
|
|
448
|
+
/**
|
|
449
|
+
* The empty-poll sleep. A store with a push channel (Postgres LISTEN/NOTIFY)
|
|
450
|
+
* cuts it short when a task on this worker's queues becomes claimable;
|
|
451
|
+
* stop() interrupts it either way, and sleepOrStop bounds it at `ms` so the
|
|
452
|
+
* poll fallback — which also drives lease recovery — never stretches.
|
|
453
|
+
*/
|
|
454
|
+
private idle(ms: number): Promise<void> {
|
|
455
|
+
return Promise.race([this.sleepOrStop(ms), this.store.claimWake(this.queues, ms)]);
|
|
456
|
+
}
|
|
457
|
+
|
|
264
458
|
private sleepOrStop(ms: number): Promise<void> {
|
|
265
459
|
if (this.stopped) return Promise.resolve();
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
clearTimeout(timer);
|
|
272
|
-
resolve();
|
|
273
|
-
};
|
|
274
|
-
const timer = setTimeout(finish, ms);
|
|
275
|
-
this.stopResolvers.push(finish);
|
|
276
|
-
});
|
|
460
|
+
let timer!: NodeJS.Timeout;
|
|
461
|
+
const nap = new Promise<void>((r) => (timer = setTimeout(r, ms)));
|
|
462
|
+
// Clear the timer whichever side wins, so a stop is never followed by a
|
|
463
|
+
// leftover poll timer holding the process open.
|
|
464
|
+
return Promise.race([nap, this.stopped$]).finally(() => clearTimeout(timer));
|
|
277
465
|
}
|
|
278
466
|
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
467
|
+
/** Take SIGINT/SIGTERM for the duration of serve(). Returns the undo. */
|
|
468
|
+
private installSignals(): () => void {
|
|
469
|
+
const remove = () => {
|
|
470
|
+
process.off("SIGINT", handler);
|
|
471
|
+
process.off("SIGTERM", handler);
|
|
472
|
+
};
|
|
473
|
+
const handler = () => {
|
|
474
|
+
// Stand down after the first signal, so a second Ctrl-C reaches Node's
|
|
475
|
+
// default and kills a worker that will not drain. `once` would only drop
|
|
476
|
+
// whichever signal fired and leave the other suppressing the default.
|
|
477
|
+
remove();
|
|
478
|
+
this.stop();
|
|
479
|
+
};
|
|
480
|
+
process.on("SIGINT", handler);
|
|
481
|
+
process.on("SIGTERM", handler);
|
|
482
|
+
return remove;
|
|
283
483
|
}
|
|
284
484
|
}
|