cairnq 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -2
- package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
- package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
- package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
- package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
- package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
- package/dist/_protocol/sql/postgres/queue_depth.sql +22 -0
- package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
- package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
- package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
- package/dist/_protocol/sql/sqlite/queue_depth.sql +26 -0
- package/dist/backoff.d.ts +31 -0
- package/dist/backoff.js +40 -0
- package/dist/backpressure.d.ts +59 -0
- package/dist/backpressure.js +122 -0
- package/dist/client.d.ts +16 -3
- package/dist/client.js +19 -5
- package/dist/context.d.ts +48 -2
- package/dist/context.js +101 -10
- package/dist/errors.d.ts +37 -0
- package/dist/errors.js +60 -0
- package/dist/index.d.ts +7 -3
- package/dist/index.js +2 -1
- package/dist/store/base.d.ts +73 -0
- package/dist/store/base.js +124 -14
- package/dist/store/sqlite.d.ts +35 -0
- package/dist/store/sqlite.js +163 -3
- package/dist/worker.d.ts +238 -13
- package/dist/worker.js +512 -120
- package/package.json +2 -1
- package/src/backoff.ts +53 -0
- package/src/backpressure.ts +140 -0
- package/src/client.ts +33 -5
- package/src/context.ts +116 -9
- package/src/errors.ts +66 -0
- package/src/index.ts +7 -2
- package/src/store/base.ts +136 -13
- package/src/store/sqlite.ts +168 -2
- package/src/worker.ts +671 -132
package/dist/worker.js
CHANGED
|
@@ -1,29 +1,16 @@
|
|
|
1
|
+
import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
|
|
1
2
|
import { TaskContext } from "./context.js";
|
|
2
|
-
import { errorEnvelope, LostLease, SerializationError,
|
|
3
|
+
import { asEnvelope, errorEnvelope, exceptionEnvelope, LostLease, SerializationError, } from "./errors.js";
|
|
3
4
|
import { newId } from "./ids.js";
|
|
4
5
|
import { SQLiteStore } from "./store/sqlite.js";
|
|
5
6
|
import { PostgresStore } from "./store/postgres.js";
|
|
6
7
|
import { taskName } from "./task.js";
|
|
7
|
-
|
|
8
|
-
|
|
8
|
+
// Re-exported: they moved to backoff.ts so TaskContext.fail could share them
|
|
9
|
+
// without context.ts importing the module that imports it. They stay importable
|
|
10
|
+
// from this module, which is where they used to live.
|
|
11
|
+
export { retryDelayMs, DEFAULT_RETRY_BACKOFF_MS, DEFAULT_RETRY_BACKOFF_MAX_MS } from "./backoff.js";
|
|
9
12
|
/** Wait after a failed claim, so a broken database is not polled in a tight loop. */
|
|
10
13
|
const CLAIM_ERROR_BACKOFF_MS = 250;
|
|
11
|
-
/** Exponential backoff for the next attempt of a task that just failed. */
|
|
12
|
-
export function retryDelayMs(attempt, baseMs, maxMs) {
|
|
13
|
-
if (baseMs <= 0)
|
|
14
|
-
return 0;
|
|
15
|
-
const exponent = Math.max(0, attempt - 1);
|
|
16
|
-
return Math.min(maxMs, baseMs * 2 ** exponent);
|
|
17
|
-
}
|
|
18
|
-
function exceptionEnvelope(err) {
|
|
19
|
-
const e = err;
|
|
20
|
-
return errorEnvelope({
|
|
21
|
-
type: e?.name ?? "Error",
|
|
22
|
-
code: "handler_error",
|
|
23
|
-
message: String(e?.message ?? err),
|
|
24
|
-
retryable: true,
|
|
25
|
-
});
|
|
26
|
-
}
|
|
27
14
|
/** Internal: an attempt outran maxRunMs and was abandoned. */
|
|
28
15
|
class AttemptTimeout extends Error {
|
|
29
16
|
maxRunMs;
|
|
@@ -41,12 +28,74 @@ function timeoutEnvelope(name, maxRunMs) {
|
|
|
41
28
|
});
|
|
42
29
|
}
|
|
43
30
|
const TIMED_OUT = Symbol("cairnq.timedOut");
|
|
31
|
+
/**
|
|
32
|
+
* Resident size of a task's payload, for the maxInFlightBytes budget.
|
|
33
|
+
*
|
|
34
|
+
* Re-serializes because by this point the wire form is gone: `pg` parses a jsonb
|
|
35
|
+
* column with JSON.parse and discards the text, so on Postgres there is nothing
|
|
36
|
+
* cheaper to read. On SQLite the column does arrive as a string that rowToTask
|
|
37
|
+
* sees before parsing — capturing its length there would make this free, at the
|
|
38
|
+
* cost of carrying a non-protocol field on Task in both SDKs. Left for when the
|
|
39
|
+
* measurement shows up in a profile.
|
|
40
|
+
*
|
|
41
|
+
* What the budget is really after is the memory a payload pins while its handler
|
|
42
|
+
* runs, and its JSON length tracks that closely enough to size one by.
|
|
43
|
+
*/
|
|
44
|
+
function payloadBytes(task) {
|
|
45
|
+
try {
|
|
46
|
+
return Buffer.byteLength(JSON.stringify(task.payload) ?? "");
|
|
47
|
+
}
|
|
48
|
+
catch {
|
|
49
|
+
// Unmeasurable, and it came out of the store, so it is already resident:
|
|
50
|
+
// charging nothing under-counts, but failing the claim over an accounting
|
|
51
|
+
// detail would drop a task the worker can otherwise run.
|
|
52
|
+
return 0;
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* Give back one call's unit of a counted budget. Deleting at zero is what keeps
|
|
57
|
+
* the map to the keys actually in flight, so an idle worker holds no entries at
|
|
58
|
+
* all — and both budgets (a name's own concurrency, a resource's capacity)
|
|
59
|
+
* settle the same way, from one place.
|
|
60
|
+
*/
|
|
61
|
+
function release(counts, key) {
|
|
62
|
+
if (key == null)
|
|
63
|
+
return;
|
|
64
|
+
const rest = (counts.get(key) ?? 1) - 1;
|
|
65
|
+
if (rest > 0)
|
|
66
|
+
counts.set(key, rest);
|
|
67
|
+
else
|
|
68
|
+
counts.delete(key);
|
|
69
|
+
}
|
|
44
70
|
export class Worker {
|
|
45
71
|
store;
|
|
46
72
|
queues;
|
|
47
73
|
opts;
|
|
48
74
|
handlers = new Map();
|
|
49
75
|
workerId = newId("worker");
|
|
76
|
+
/** Payload bytes charged to running handlers — see maxInFlightBytes. */
|
|
77
|
+
inFlightBytes = 0;
|
|
78
|
+
/** Calls in flight, for the names that cap their own concurrency. */
|
|
79
|
+
callsInFlight = new Map();
|
|
80
|
+
/**
|
|
81
|
+
* Calls holding units of each declared resource. A resource is the same shape
|
|
82
|
+
* of budget as a name's own `concurrency` — a ceiling on calls — differing
|
|
83
|
+
* only in who draws from it: several names rather than one. That is what
|
|
84
|
+
* expresses "these handlers share one GPU" without inventing a queue per
|
|
85
|
+
* resource.
|
|
86
|
+
*/
|
|
87
|
+
resourceCalls = new Map();
|
|
88
|
+
/** Rotates which source is offered the free budget first — see loop(). */
|
|
89
|
+
claimCursor = 0;
|
|
90
|
+
/** Invalidated by task(); see schedule(). */
|
|
91
|
+
scheduleCache = null;
|
|
92
|
+
/**
|
|
93
|
+
* Retry backoff, resolved once. Both settlement paths read these — the
|
|
94
|
+
* worker's own `safeFail` and the TaskContext it hands a handler — so
|
|
95
|
+
* resolving the defaults per call site is how the two drift apart.
|
|
96
|
+
*/
|
|
97
|
+
backoffMs;
|
|
98
|
+
backoffMaxMs;
|
|
50
99
|
stopped = false;
|
|
51
100
|
stopWake;
|
|
52
101
|
// Resolved once by stop(); every sleep races against it. A stopped worker
|
|
@@ -62,6 +111,21 @@ export class Worker {
|
|
|
62
111
|
if (opts.maxRunMs != null && opts.maxRunMs <= 0) {
|
|
63
112
|
throw new Error(`maxRunMs must be > 0, got ${opts.maxRunMs}`);
|
|
64
113
|
}
|
|
114
|
+
// 0 would make the budget permanently spent, so the worker would claim
|
|
115
|
+
// nothing and look hung. Rejected here, as the Python SDK does.
|
|
116
|
+
if (opts.maxInFlightBytes != null && opts.maxInFlightBytes <= 0) {
|
|
117
|
+
throw new Error(`maxInFlightBytes must be > 0, got ${opts.maxInFlightBytes}`);
|
|
118
|
+
}
|
|
119
|
+
for (const [resource, capacity] of Object.entries(opts.resources ?? {})) {
|
|
120
|
+
if (!Number.isInteger(capacity) || capacity < 1) {
|
|
121
|
+
throw new Error(`resources[${JSON.stringify(resource)}] must be an integer >= 1, got ${capacity}`);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
if (opts.maxQueueDepth != null) {
|
|
125
|
+
store.useBackpressure(opts);
|
|
126
|
+
}
|
|
127
|
+
this.backoffMs = opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
|
|
128
|
+
this.backoffMaxMs = opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS;
|
|
65
129
|
}
|
|
66
130
|
static sqlite(path, opts = {}) {
|
|
67
131
|
const { queues = ["default"], busyTimeoutMs, ...rest } = opts;
|
|
@@ -80,7 +144,30 @@ export class Worker {
|
|
|
80
144
|
get id() {
|
|
81
145
|
return this.workerId;
|
|
82
146
|
}
|
|
83
|
-
task(arg,
|
|
147
|
+
task(arg, second, third) {
|
|
148
|
+
// Option form: (name | def, { batch?, concurrency?, resource? }, handler).
|
|
149
|
+
// Peel the options off and fall through, so name resolution and registration
|
|
150
|
+
// stay single-sited.
|
|
151
|
+
let batch;
|
|
152
|
+
let concurrency;
|
|
153
|
+
let resource;
|
|
154
|
+
let handler = second;
|
|
155
|
+
if (second != null && typeof second === "object") {
|
|
156
|
+
if (second.batch != null) {
|
|
157
|
+
if (!Number.isInteger(second.batch) || second.batch < 1) {
|
|
158
|
+
throw new Error(`batch must be an integer >= 1, got ${second.batch}`);
|
|
159
|
+
}
|
|
160
|
+
batch = second.batch;
|
|
161
|
+
}
|
|
162
|
+
if (second.concurrency != null) {
|
|
163
|
+
if (!Number.isInteger(second.concurrency) || second.concurrency < 1) {
|
|
164
|
+
throw new Error(`concurrency must be an integer >= 1, got ${second.concurrency}`);
|
|
165
|
+
}
|
|
166
|
+
concurrency = second.concurrency;
|
|
167
|
+
}
|
|
168
|
+
resource = second.resource;
|
|
169
|
+
handler = third;
|
|
170
|
+
}
|
|
84
171
|
let name;
|
|
85
172
|
let fn;
|
|
86
173
|
if (typeof arg === "function") {
|
|
@@ -99,7 +186,16 @@ export class Worker {
|
|
|
99
186
|
name = taskName(arg);
|
|
100
187
|
fn = handler;
|
|
101
188
|
}
|
|
102
|
-
|
|
189
|
+
// Loudly, at registration: an undeclared resource would otherwise read as
|
|
190
|
+
// an unbounded one, so a typo would silently remove the ceiling the caller
|
|
191
|
+
// asked for — the failure this option exists to prevent.
|
|
192
|
+
if (resource != null && this.opts.resources?.[resource] == null) {
|
|
193
|
+
const known = Object.keys(this.opts.resources ?? {}).sort().join(", ") || "none";
|
|
194
|
+
throw new Error(`task ${JSON.stringify(name)} declares resource ${JSON.stringify(resource)}, ` +
|
|
195
|
+
`which is not in WorkerOptions.resources; declared: ${known}`);
|
|
196
|
+
}
|
|
197
|
+
this.handlers.set(name, { fn, batch, concurrency, resource });
|
|
198
|
+
this.scheduleCache = null;
|
|
103
199
|
return this;
|
|
104
200
|
}
|
|
105
201
|
stop() {
|
|
@@ -130,11 +226,12 @@ export class Worker {
|
|
|
130
226
|
// beyond even stop()'s reach.
|
|
131
227
|
const concurrency = Math.max(1, opts.concurrency ?? this.opts.concurrency ?? 1);
|
|
132
228
|
const leaseMs = this.opts.leaseMs ?? 30_000;
|
|
133
|
-
const batch = this.opts.claimBatch ?? concurrency;
|
|
134
229
|
await this.store.connect();
|
|
230
|
+
// The calls in flight — `concurrency` counts these, so the set's size is the
|
|
231
|
+
// budget. How many tasks they carry is `maxInFlightBytes`'s business.
|
|
135
232
|
const running = new Set();
|
|
136
233
|
try {
|
|
137
|
-
await this.loop(concurrency,
|
|
234
|
+
await this.loop(concurrency, leaseMs, running);
|
|
138
235
|
}
|
|
139
236
|
finally {
|
|
140
237
|
// Whatever ends the loop — stop(), or something unexpected out of the body
|
|
@@ -144,28 +241,175 @@ export class Worker {
|
|
|
144
241
|
await Promise.all([...running]);
|
|
145
242
|
}
|
|
146
243
|
}
|
|
147
|
-
|
|
244
|
+
/**
|
|
245
|
+
* Split one claim into handler calls, each with the registration to run it.
|
|
246
|
+
*
|
|
247
|
+
* A claim is filtered by queue and by the names this worker handles, so it
|
|
248
|
+
* comes back mixed; batch size is per name (one embedding call wants 256
|
|
249
|
+
* texts, one Docling parse wants exactly 1). So group by name, then chunk each
|
|
250
|
+
* group by that name's size. Names registered without `batch` come back as
|
|
251
|
+
* one-task calls, as do names not registered at all — reachable only if a
|
|
252
|
+
* handler is unregistered mid-run, and dispatched to failNoHandler().
|
|
253
|
+
*
|
|
254
|
+
* The registration rides along because this is where it was resolved; looking
|
|
255
|
+
* it up again at the call site would put "is this name batched" in two places.
|
|
256
|
+
*/
|
|
257
|
+
deliveries(claimed) {
|
|
258
|
+
const byName = new Map();
|
|
259
|
+
for (const task of claimed) {
|
|
260
|
+
const group = byName.get(task.name);
|
|
261
|
+
if (group)
|
|
262
|
+
group.push(task);
|
|
263
|
+
else
|
|
264
|
+
byName.set(task.name, [task]);
|
|
265
|
+
}
|
|
266
|
+
const out = [];
|
|
267
|
+
for (const [name, group] of byName) {
|
|
268
|
+
const reg = this.handlers.get(name);
|
|
269
|
+
const size = reg?.batch ?? 1;
|
|
270
|
+
for (let i = 0; i < group.length; i += size)
|
|
271
|
+
out.push([reg, group.slice(i, i + size)]);
|
|
272
|
+
}
|
|
273
|
+
return out;
|
|
274
|
+
}
|
|
275
|
+
context(task, leaseMs) {
|
|
276
|
+
return new TaskContext(this.store, task, this.workerId, leaseMs, {
|
|
277
|
+
retryBackoffMs: this.backoffMs,
|
|
278
|
+
retryBackoffMaxMs: this.backoffMaxMs,
|
|
279
|
+
});
|
|
280
|
+
}
|
|
281
|
+
/**
|
|
282
|
+
* How this poll's claim is split into per-name quotas, plus the union of names
|
|
283
|
+
* the probe spans.
|
|
284
|
+
*
|
|
285
|
+
* Cached and invalidated by `task()`, rather than rebuilt each poll: handlers
|
|
286
|
+
* may be registered after run() started, but only there, and this otherwise
|
|
287
|
+
* allocates a source per name on every tick for a worker's whole lifetime.
|
|
288
|
+
*
|
|
289
|
+
* A name that limits itself — by `batch`, by its own `concurrency`, or by a
|
|
290
|
+
* `resource` — needs a quota the shared draw cannot express, so it gets a
|
|
291
|
+
* source of its own; every other name shares one, where a task is a call.
|
|
292
|
+
*
|
|
293
|
+
* A resource is deliberately *not* one source spanning its names: `batch` is
|
|
294
|
+
* per name, and a single source carries one batch size, so two members that
|
|
295
|
+
* batch differently could not share a draw. Keeping a source per name and
|
|
296
|
+
* letting several of them draw down one shared ceiling composes with batching
|
|
297
|
+
* instead of excluding it.
|
|
298
|
+
*/
|
|
299
|
+
schedule() {
|
|
300
|
+
if (this.scheduleCache)
|
|
301
|
+
return this.scheduleCache;
|
|
302
|
+
const sources = [];
|
|
303
|
+
const shared = [];
|
|
304
|
+
for (const [name, reg] of this.handlers) {
|
|
305
|
+
if (reg.batch != null || reg.concurrency != null || reg.resource != null) {
|
|
306
|
+
sources.push({
|
|
307
|
+
key: reg.concurrency == null ? undefined : name,
|
|
308
|
+
names: [name],
|
|
309
|
+
batch: reg.batch ?? 1,
|
|
310
|
+
concurrency: reg.concurrency,
|
|
311
|
+
resource: reg.resource,
|
|
312
|
+
});
|
|
313
|
+
}
|
|
314
|
+
else
|
|
315
|
+
shared.push(name);
|
|
316
|
+
}
|
|
317
|
+
if (shared.length)
|
|
318
|
+
sources.push({ names: shared, batch: 1 });
|
|
319
|
+
return (this.scheduleCache = { sources, names: [...this.handlers.keys()] });
|
|
320
|
+
}
|
|
321
|
+
/**
|
|
322
|
+
* A source's own call ceiling for one poll, or undefined when only the
|
|
323
|
+
* worker-wide budget applies.
|
|
324
|
+
*
|
|
325
|
+
* Three independent ceilings, whichever binds first: the name's own concurrency
|
|
326
|
+
* less what it is already running; its resource's capacity less what is running
|
|
327
|
+
* *and* what earlier draws in this same poll already took (`taken`); and
|
|
328
|
+
* `claimBatch`. The last is a ceiling on *rows* per poll, so it converts at this
|
|
329
|
+
* source's batch size — and never below one call, or a `claimBatch` under some
|
|
330
|
+
* name's batch would stall that name outright.
|
|
331
|
+
*/
|
|
332
|
+
sourceCalls(src, taken) {
|
|
333
|
+
const rows = this.opts.claimBatch;
|
|
334
|
+
const byRows = rows == null ? undefined : Math.max(1, Math.floor(rows / src.batch));
|
|
335
|
+
const byName = src.concurrency == null
|
|
336
|
+
? undefined
|
|
337
|
+
: Math.max(0, src.concurrency - (this.callsInFlight.get(src.key) ?? 0));
|
|
338
|
+
const byResource = src.resource == null
|
|
339
|
+
? undefined
|
|
340
|
+
: Math.max(0, this.opts.resources[src.resource] -
|
|
341
|
+
(this.resourceCalls.get(src.resource) ?? 0) -
|
|
342
|
+
(taken.get(src.resource) ?? 0));
|
|
343
|
+
const limits = [byRows, byName, byResource].filter((n) => n != null);
|
|
344
|
+
return limits.length ? Math.min(...limits) : undefined;
|
|
345
|
+
}
|
|
346
|
+
async loop(concurrency, leaseMs, running) {
|
|
148
347
|
const pollMs = this.opts.pollIntervalMs ?? 500;
|
|
348
|
+
const byteBudget = this.opts.maxInFlightBytes;
|
|
149
349
|
while (!this.stopped) {
|
|
350
|
+
// `concurrency` counts calls, so the calls in flight *are* the running
|
|
351
|
+
// promises — a batch holding 256 tasks is one of them.
|
|
150
352
|
const free = concurrency - running.size;
|
|
151
|
-
|
|
152
|
-
|
|
353
|
+
// Two ceilings, either of which stops the claim: calls in flight and
|
|
354
|
+
// resident payload bytes. The byte arm is guarded on running.size because
|
|
355
|
+
// it must never be the reason we race an empty set — Promise.race([]) is
|
|
356
|
+
// pending forever, past even stop(). With nothing running, nothing is
|
|
357
|
+
// resident, so the budget cannot be the thing holding us back anyway.
|
|
358
|
+
const overBudget = byteBudget != null && this.inFlightBytes >= byteBudget;
|
|
359
|
+
if (running.size > 0 && (free <= 0 || overBudget)) {
|
|
360
|
+
// Wait for a slot rather than spinning. runCall() never rejects, so
|
|
153
361
|
// racing these is safe.
|
|
154
362
|
await Promise.race([...running]);
|
|
155
363
|
continue;
|
|
156
364
|
}
|
|
365
|
+
const { sources, names } = this.schedule();
|
|
366
|
+
if (!sources.length) {
|
|
367
|
+
await this.idle(pollMs);
|
|
368
|
+
continue;
|
|
369
|
+
}
|
|
370
|
+
// Round-robin the starting point. The draws are served in order, so without
|
|
371
|
+
// rotating it the first source would take every free slot and the rest
|
|
372
|
+
// would starve behind its backlog.
|
|
373
|
+
const cursor = this.claimCursor % sources.length;
|
|
374
|
+
const order = [...sources.slice(cursor), ...sources.slice(0, cursor)];
|
|
375
|
+
this.claimCursor = (cursor + 1) % sources.length;
|
|
157
376
|
let claimed;
|
|
158
377
|
try {
|
|
159
|
-
claimed = await this.store.
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
378
|
+
claimed = await this.store.claimSession(
|
|
379
|
+
// Only what this worker can run. Queues do not partition work by task
|
|
380
|
+
// name, so another worker's tasks would otherwise be claimed here and
|
|
381
|
+
// failed for want of a handler.
|
|
382
|
+
{ queues: this.queues, workerId: this.workerId, leaseMs, names }, async (claim) => {
|
|
383
|
+
const drawn = [];
|
|
384
|
+
let left = free;
|
|
385
|
+
// Resource units this poll has already drawn. resourceCalls only
|
|
386
|
+
// moves when a call is dispatched, which happens after this whole
|
|
387
|
+
// plan returns — so without a local tally two sources sharing a
|
|
388
|
+
// resource would each see its full ceiling and together overshoot
|
|
389
|
+
// it. Same shape as `left`, one budget down.
|
|
390
|
+
const taken = new Map();
|
|
391
|
+
for (const src of order) {
|
|
392
|
+
if (left <= 0)
|
|
393
|
+
break;
|
|
394
|
+
const quota = Math.min(this.sourceCalls(src, taken) ?? left, left);
|
|
395
|
+
if (quota <= 0)
|
|
396
|
+
continue;
|
|
397
|
+
const rows = await claim(src.names, src.batch * quota);
|
|
398
|
+
if (!rows.length)
|
|
399
|
+
continue;
|
|
400
|
+
// deliveries() is what actually turns rows into handler calls, so
|
|
401
|
+
// spending the budget against its result is the only way the two
|
|
402
|
+
// cannot disagree. A source with nothing queued costs nothing,
|
|
403
|
+
// which is why the budget is spent here, draw by draw, rather than
|
|
404
|
+
// divided up before the claim.
|
|
405
|
+
const calls = this.deliveries(rows);
|
|
406
|
+
drawn.push({ src, calls });
|
|
407
|
+
left -= calls.length;
|
|
408
|
+
if (src.resource != null) {
|
|
409
|
+
taken.set(src.resource, (taken.get(src.resource) ?? 0) + calls.length);
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
return drawn;
|
|
169
413
|
});
|
|
170
414
|
}
|
|
171
415
|
catch (err) {
|
|
@@ -175,13 +419,35 @@ export class Worker {
|
|
|
175
419
|
await this.sleepOrStop(CLAIM_ERROR_BACKOFF_MS);
|
|
176
420
|
continue;
|
|
177
421
|
}
|
|
178
|
-
if (claimed
|
|
422
|
+
if (!claimed?.length) {
|
|
179
423
|
await this.idle(pollMs);
|
|
180
424
|
continue;
|
|
181
425
|
}
|
|
182
|
-
for (const
|
|
183
|
-
const
|
|
184
|
-
|
|
426
|
+
for (const { src, calls } of claimed) {
|
|
427
|
+
for (const [reg, group] of calls) {
|
|
428
|
+
// Charged before the handler starts and refunded when it settles, so
|
|
429
|
+
// the budgets cover exactly the span the call holds its slot and its
|
|
430
|
+
// payloads stay pinned in memory.
|
|
431
|
+
const bytes = byteBudget == null ? 0 : group.reduce((sum, t) => sum + payloadBytes(t), 0);
|
|
432
|
+
this.inFlightBytes += bytes;
|
|
433
|
+
const key = src.key;
|
|
434
|
+
const resource = src.resource;
|
|
435
|
+
if (key != null)
|
|
436
|
+
this.callsInFlight.set(key, (this.callsInFlight.get(key) ?? 0) + 1);
|
|
437
|
+
if (resource != null) {
|
|
438
|
+
this.resourceCalls.set(resource, (this.resourceCalls.get(resource) ?? 0) + 1);
|
|
439
|
+
}
|
|
440
|
+
// An unregistered name has nothing to run, so it never starts a handler
|
|
441
|
+
// or a heartbeat; everything else is one call, batched or not.
|
|
442
|
+
const call = reg == null ? this.failNoHandler(group[0], leaseMs) : this.runCall(reg, group, leaseMs);
|
|
443
|
+
const p = call.finally(() => {
|
|
444
|
+
this.inFlightBytes -= bytes;
|
|
445
|
+
release(this.callsInFlight, key);
|
|
446
|
+
release(this.resourceCalls, resource);
|
|
447
|
+
running.delete(p);
|
|
448
|
+
});
|
|
449
|
+
running.add(p);
|
|
450
|
+
}
|
|
185
451
|
}
|
|
186
452
|
}
|
|
187
453
|
}
|
|
@@ -215,74 +481,78 @@ export class Worker {
|
|
|
215
481
|
}
|
|
216
482
|
}
|
|
217
483
|
/**
|
|
218
|
-
*
|
|
219
|
-
*
|
|
220
|
-
*
|
|
484
|
+
* Record a claimed task this worker cannot run. Reachable only if a name is
|
|
485
|
+
* unregistered mid-run — the claim filters on the registered names — so it does
|
|
486
|
+
* not start a handler or a heartbeat for a task it will not run.
|
|
221
487
|
*/
|
|
222
|
-
async
|
|
223
|
-
|
|
224
|
-
|
|
488
|
+
async failNoHandler(task, leaseMs) {
|
|
489
|
+
try {
|
|
490
|
+
await this.safeFail(this.context(task, leaseMs), errorEnvelope({
|
|
491
|
+
type: "NoHandler",
|
|
492
|
+
code: "no_handler",
|
|
493
|
+
message: `no handler registered for ${task.name}`,
|
|
494
|
+
retryable: false,
|
|
495
|
+
}), false);
|
|
496
|
+
}
|
|
497
|
+
catch (err) {
|
|
498
|
+
this.report(err, { phase: "execute", taskId: task.id });
|
|
499
|
+
}
|
|
500
|
+
}
|
|
501
|
+
/**
|
|
502
|
+
* Run one handler call to completion — one task, or a whole batch. Never
|
|
503
|
+
* rejects: a task-level failure is reported through onError and the loop moves
|
|
504
|
+
* on. (It used to reject into a promise nobody awaited — an unhandled rejection
|
|
505
|
+
* that took the process down.)
|
|
506
|
+
*
|
|
507
|
+
* One lifecycle for both delivery modes, because a single-task handler *is* the
|
|
508
|
+
* one-element case: the same heartbeat covers the call, the same classifier
|
|
509
|
+
* reads its error, and the same rule settles what is left. Only two things vary
|
|
510
|
+
* — how the handler is invoked, and where a leftover task's result comes from —
|
|
511
|
+
* so those are the only two branches below.
|
|
512
|
+
*
|
|
513
|
+
* The contract is single: **when the handler returns, every task it did not
|
|
514
|
+
* settle itself is settled by how the call ended.** Returning succeeds them,
|
|
515
|
+
* throwing fails them (retryably, or as the thrown TaskError says). That is
|
|
516
|
+
* what keeps the ordinary cases free of bookkeeping — a handler that just
|
|
517
|
+
* returns has finished 256 tasks — while still letting it pick individual
|
|
518
|
+
* tasks off with `item.succeed()` / `item.fail()` as it goes.
|
|
519
|
+
*
|
|
520
|
+
* For a batch, a returned map of task id -> result fills in results for the
|
|
521
|
+
* tasks left over; anything else returned is ignored, and unmentioned tasks
|
|
522
|
+
* succeed with no result (the common shape, where the handler's output went to
|
|
523
|
+
* a database rather than into the task row). For a single task the return value
|
|
524
|
+
* simply is the result.
|
|
525
|
+
*/
|
|
526
|
+
async runCall(reg, tasks, leaseMs) {
|
|
527
|
+
const ctxs = tasks.map((t) => this.context(t, leaseMs));
|
|
528
|
+
const batched = reg.batch != null;
|
|
529
|
+
const hb = this.startHeartbeat(ctxs, leaseMs);
|
|
225
530
|
try {
|
|
226
|
-
const handler = this.handlers.get(task.name);
|
|
227
|
-
if (!handler) {
|
|
228
|
-
await this.safeFail(task, errorEnvelope({
|
|
229
|
-
type: "NoHandler",
|
|
230
|
-
code: "no_handler",
|
|
231
|
-
message: `no handler registered for ${task.name}`,
|
|
232
|
-
retryable: false,
|
|
233
|
-
}), false);
|
|
234
|
-
return;
|
|
235
|
-
}
|
|
236
531
|
let result;
|
|
237
532
|
try {
|
|
238
|
-
result = await this.attempt(
|
|
533
|
+
result = await this.attempt(() => batched
|
|
534
|
+
? reg.fn(ctxs)
|
|
535
|
+
: reg.fn(ctxs[0], tasks[0].payload), ctxs);
|
|
239
536
|
}
|
|
240
537
|
catch (err) {
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
// Recorded as a retryable failure, so backoff / maxAttempts /
|
|
245
|
-
// cancel-wins all apply exactly as for a thrown error.
|
|
246
|
-
await this.safeFail(task, timeoutEnvelope(task.name, err.maxRunMs), true);
|
|
247
|
-
return;
|
|
248
|
-
}
|
|
249
|
-
if (err instanceof TaskError) {
|
|
250
|
-
await this.safeFail(task, err.envelope(), err.retryable);
|
|
251
|
-
}
|
|
252
|
-
else {
|
|
253
|
-
await this.safeFail(task, exceptionEnvelope(err), true);
|
|
254
|
-
}
|
|
538
|
+
const outcome = this.outcomeOf(err, tasks[0].name);
|
|
539
|
+
if (outcome)
|
|
540
|
+
await this.settleEach(ctxs, (c) => this.safeFail(c, outcome[0], outcome[1]));
|
|
255
541
|
return;
|
|
256
542
|
}
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
if (err instanceof SerializationError) {
|
|
268
|
-
// The handler succeeded but its return value can't cross the JSON
|
|
269
|
-
// protocol (BigInt, non-finite number, circular). Deterministic, so
|
|
270
|
-
// fail fast and permanently — the alternative is sitting `running`
|
|
271
|
-
// until lease expiry redelivers a task that fails the same way every
|
|
272
|
-
// attempt.
|
|
273
|
-
await this.safeFail(task, errorEnvelope({
|
|
274
|
-
type: "SerializationError",
|
|
275
|
-
code: "unserializable_result",
|
|
276
|
-
message: `handler result is not JSON-serializable: ${err.message}`,
|
|
277
|
-
retryable: false,
|
|
278
|
-
}), false);
|
|
279
|
-
return;
|
|
280
|
-
}
|
|
281
|
-
throw err;
|
|
282
|
-
}
|
|
543
|
+
// A batch handler's return maps task id -> result; a single-task handler's
|
|
544
|
+
// return *is* the result. Anything else a batch returns is ignored, so its
|
|
545
|
+
// leftovers succeed with no result — hence the empty map rather than
|
|
546
|
+
// falling through to `result`.
|
|
547
|
+
const results = batched
|
|
548
|
+
? result != null && typeof result === "object"
|
|
549
|
+
? result
|
|
550
|
+
: {}
|
|
551
|
+
: null;
|
|
552
|
+
await this.settleEach(ctxs, (c) => this.succeedOne(c, results ? (results[c.taskId] ?? null) : result));
|
|
283
553
|
}
|
|
284
554
|
catch (err) {
|
|
285
|
-
this.report(err, { phase: "execute", taskId:
|
|
555
|
+
this.report(err, { phase: "execute", taskId: tasks[0].id });
|
|
286
556
|
}
|
|
287
557
|
finally {
|
|
288
558
|
hb.cancel();
|
|
@@ -290,20 +560,117 @@ export class Worker {
|
|
|
290
560
|
}
|
|
291
561
|
}
|
|
292
562
|
/**
|
|
293
|
-
*
|
|
294
|
-
*
|
|
295
|
-
*
|
|
296
|
-
*
|
|
297
|
-
*
|
|
298
|
-
*
|
|
563
|
+
* How an attempt that ended badly is recorded: [envelope, retryable], or null
|
|
564
|
+
* when there is nothing to record.
|
|
565
|
+
*
|
|
566
|
+
* One classifier for both delivery modes, so a handler error cannot mean
|
|
567
|
+
* different things depending on how its task happened to be delivered. That
|
|
568
|
+
* includes LostLease, which is not an outcome at all: it means a write through
|
|
569
|
+
* this context was already rejected, so recording anything more would be
|
|
570
|
+
* rejected too. Both modes then leave the task alone — the single-task one has
|
|
571
|
+
* nothing else to do, and a batch lets its remaining tasks fall to lease expiry
|
|
572
|
+
* and redelivery rather than stamping them with a failure the handler never
|
|
573
|
+
* reported.
|
|
299
574
|
*/
|
|
300
|
-
|
|
575
|
+
outcomeOf(err, name) {
|
|
576
|
+
if (err instanceof LostLease)
|
|
577
|
+
return null;
|
|
578
|
+
// Retryable, so backoff / maxAttempts / cancel-wins all apply exactly as for
|
|
579
|
+
// a thrown error.
|
|
580
|
+
if (err instanceof AttemptTimeout)
|
|
581
|
+
return [timeoutEnvelope(name, err.maxRunMs), true];
|
|
582
|
+
// Everything else is what a handler could equally have passed to ctx.fail(),
|
|
583
|
+
// so it goes through the same normalizer — a TaskError keeps its own
|
|
584
|
+
// retryability, anything else is retryable. Non-Error throws reach
|
|
585
|
+
// exceptionEnvelope the same way, since only ctx.fail can supply a ready
|
|
586
|
+
// envelope and a thrown object is not one.
|
|
587
|
+
if (err !== null && typeof err === "object" && !(err instanceof Error)) {
|
|
588
|
+
return [exceptionEnvelope(err), true];
|
|
589
|
+
}
|
|
590
|
+
return asEnvelope(err, true);
|
|
591
|
+
}
|
|
592
|
+
// Both leftover paths write through the store rather than through the context.
|
|
593
|
+
// `TaskContext.owned` short-circuits once a lease is known lost, which is there
|
|
594
|
+
// to stop a *zombie handler* writing after its attempt was abandoned — but
|
|
595
|
+
// these run after the handler is done, on the worker's own authority, exactly
|
|
596
|
+
// as execute()'s completion and failure arms do. Ownership is still enforced by
|
|
597
|
+
// each statement, so a task whose lease really was lost writes nothing either
|
|
598
|
+
// way.
|
|
599
|
+
/**
|
|
600
|
+
* Finalize one task the handler left for the worker to decide — the tail of
|
|
601
|
+
* both delivery modes.
|
|
602
|
+
*
|
|
603
|
+
* Includes the unserializable-result rule: the handler succeeded but its value
|
|
604
|
+
* cannot cross the JSON protocol (BigInt, non-finite number, circular), which
|
|
605
|
+
* is deterministic, so it fails permanently rather than being redelivered to
|
|
606
|
+
* fail the same way every attempt.
|
|
607
|
+
*/
|
|
608
|
+
async succeedOne(ctx, result) {
|
|
609
|
+
if (ctx.settled)
|
|
610
|
+
return;
|
|
611
|
+
try {
|
|
612
|
+
// complete (not succeed): finalizes as canceled if a cancel was requested
|
|
613
|
+
// while the handler ran, else succeeded.
|
|
614
|
+
await this.store.complete({ taskId: ctx.taskId, workerId: this.workerId, result });
|
|
615
|
+
ctx.markSettled();
|
|
616
|
+
}
|
|
617
|
+
catch (err) {
|
|
618
|
+
if (err instanceof LostLease) {
|
|
619
|
+
ctx.markLeaseLost();
|
|
620
|
+
return;
|
|
621
|
+
}
|
|
622
|
+
if (err instanceof SerializationError) {
|
|
623
|
+
await this.safeFail(ctx, errorEnvelope({
|
|
624
|
+
type: "SerializationError",
|
|
625
|
+
code: "unserializable_result",
|
|
626
|
+
message: `handler result is not JSON-serializable: ${err.message}`,
|
|
627
|
+
retryable: false,
|
|
628
|
+
}), false);
|
|
629
|
+
return;
|
|
630
|
+
}
|
|
631
|
+
throw err;
|
|
632
|
+
}
|
|
633
|
+
}
|
|
634
|
+
/**
|
|
635
|
+
* Settle every task the handler did not settle itself, concurrently, reporting
|
|
636
|
+
* rather than throwing. Each task keeps its own attempt count and backoff —
|
|
637
|
+
* they are separate tasks that happened to be delivered together.
|
|
638
|
+
*
|
|
639
|
+
* allSettled, because one task's write failing must not abandon the rest of the
|
|
640
|
+
* batch mid-settlement — the others still hold leases and would sit `running`
|
|
641
|
+
* until expiry. Each outcome is reported against the task it belongs to;
|
|
642
|
+
* without that, an operator learns a settlement failed somewhere in a batch of
|
|
643
|
+
* 256.
|
|
644
|
+
*/
|
|
645
|
+
async settleEach(ctxs, settle) {
|
|
646
|
+
const left = ctxs.filter((c) => !c.settled);
|
|
647
|
+
if (!left.length)
|
|
648
|
+
return;
|
|
649
|
+
const outcomes = await Promise.allSettled(left.map(settle));
|
|
650
|
+
outcomes.forEach((outcome, i) => {
|
|
651
|
+
if (outcome.status === "rejected") {
|
|
652
|
+
this.report(outcome.reason, { phase: "execute", taskId: left[i].taskId });
|
|
653
|
+
}
|
|
654
|
+
});
|
|
655
|
+
}
|
|
656
|
+
/**
|
|
657
|
+
* Run one attempt — of one task or of a whole batch — bounded by maxRunMs when
|
|
658
|
+
* set.
|
|
659
|
+
*
|
|
660
|
+
* On timeout the attempt is abandoned: every context it covers is flagged
|
|
661
|
+
* lease-lost first (ctx.signal aborts, and a handler that keeps running can
|
|
662
|
+
* never write again, nor settle anything behind the worker's back — see
|
|
663
|
+
* TaskContext.owned), then the still-pending promise is left to settle on its
|
|
664
|
+
* own, its outcome discarded. The caller records the handler_timeout failure;
|
|
665
|
+
* lease recovery is NOT involved, so redelivery is immediate.
|
|
666
|
+
*/
|
|
667
|
+
async attempt(invoke, ctxs) {
|
|
301
668
|
const maxRunMs = this.opts.maxRunMs;
|
|
302
669
|
if (maxRunMs == null)
|
|
303
|
-
return
|
|
670
|
+
return invoke();
|
|
304
671
|
// As a real promise: the race needs one (a handler may return a plain
|
|
305
672
|
// value), and the timeout path .catch()es it.
|
|
306
|
-
const run = (async () =>
|
|
673
|
+
const run = (async () => invoke())();
|
|
307
674
|
let timer;
|
|
308
675
|
const winner = await Promise.race([
|
|
309
676
|
run,
|
|
@@ -311,13 +678,27 @@ export class Worker {
|
|
|
311
678
|
]).finally(() => clearTimeout(timer));
|
|
312
679
|
if (winner !== TIMED_OUT)
|
|
313
680
|
return winner;
|
|
314
|
-
ctx
|
|
681
|
+
for (const ctx of ctxs)
|
|
682
|
+
ctx.markLeaseLost();
|
|
315
683
|
// The zombie may still reject later; that must not become an unhandled
|
|
316
684
|
// rejection — its outcome was already decided to be handler_timeout.
|
|
317
685
|
run.catch(() => { });
|
|
318
686
|
throw new AttemptTimeout(maxRunMs);
|
|
319
687
|
}
|
|
320
|
-
|
|
688
|
+
/**
|
|
689
|
+
* One statement per beat, however many tasks the call covers — a single-task
|
|
690
|
+
* handler is just the one-element case.
|
|
691
|
+
*
|
|
692
|
+
* Only tasks still in play are renewed. A task the handler already settled is
|
|
693
|
+
* terminal, and re-leasing it would be a write against a row nobody owns; that
|
|
694
|
+
* is also why an absence is only read as lease loss after re-checking
|
|
695
|
+
* `settled`, since the handler may have finalized the task while this beat was
|
|
696
|
+
* in flight, which takes the row out of `running` and out of the reply. A task
|
|
697
|
+
* genuinely missing lost its lease (another worker recovered it), so its
|
|
698
|
+
* context is flagged and the handler stops being able to write through it —
|
|
699
|
+
* only that one, never its neighbours.
|
|
700
|
+
*/
|
|
701
|
+
startHeartbeat(ctxs, leaseMs) {
|
|
321
702
|
let active = true;
|
|
322
703
|
let wake = null;
|
|
323
704
|
// lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
|
|
@@ -336,14 +717,27 @@ export class Worker {
|
|
|
336
717
|
wake = null;
|
|
337
718
|
if (!active)
|
|
338
719
|
break;
|
|
720
|
+
const live = ctxs.filter((c) => !c.settled && !c.lostLease);
|
|
721
|
+
if (!live.length)
|
|
722
|
+
break;
|
|
339
723
|
try {
|
|
340
|
-
await
|
|
724
|
+
const renewed = await this.store.heartbeatBatch({
|
|
725
|
+
taskIds: live.map((c) => c.taskId),
|
|
726
|
+
workerId: this.workerId,
|
|
727
|
+
leaseMs,
|
|
728
|
+
});
|
|
729
|
+
for (const ctx of live) {
|
|
730
|
+
const cancelRequested = renewed.get(ctx.taskId);
|
|
731
|
+
// Cancellation rides along on the write we were making anyway, so
|
|
732
|
+
// ctx.canceled() stays free here too.
|
|
733
|
+
if (cancelRequested !== undefined)
|
|
734
|
+
ctx.observeCancel(cancelRequested);
|
|
735
|
+
else if (!ctx.settled)
|
|
736
|
+
ctx.markLeaseLost();
|
|
737
|
+
}
|
|
341
738
|
}
|
|
342
739
|
catch (err) {
|
|
343
|
-
|
|
344
|
-
if (err instanceof LostLease)
|
|
345
|
-
break;
|
|
346
|
-
this.report(err, { phase: "execute", taskId: ctx.taskId });
|
|
740
|
+
this.report(err, { phase: "execute", taskId: live[0].taskId });
|
|
347
741
|
}
|
|
348
742
|
}
|
|
349
743
|
})();
|
|
@@ -356,18 +750,16 @@ export class Worker {
|
|
|
356
750
|
done,
|
|
357
751
|
};
|
|
358
752
|
}
|
|
359
|
-
async safeFail(
|
|
360
|
-
const delayMs = retryable
|
|
361
|
-
? retryDelayMs(task.attempt, this.opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS, this.opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS)
|
|
362
|
-
: 0;
|
|
753
|
+
async safeFail(ctx, envelope, retryable) {
|
|
363
754
|
try {
|
|
364
755
|
await this.store.fail({
|
|
365
|
-
taskId:
|
|
756
|
+
taskId: ctx.taskId,
|
|
366
757
|
workerId: this.workerId,
|
|
367
758
|
error: envelope,
|
|
368
759
|
retryable,
|
|
369
|
-
delayMs,
|
|
760
|
+
delayMs: failDelayMs(ctx.attempt, retryable, this.backoffMs, this.backoffMaxMs),
|
|
370
761
|
});
|
|
762
|
+
ctx.markSettled();
|
|
371
763
|
}
|
|
372
764
|
catch (err) {
|
|
373
765
|
if (!(err instanceof LostLease))
|