cairnq 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +11 -5
  2. package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
  3. package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
  4. package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
  5. package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
  6. package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
  7. package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
  8. package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
  9. package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
  10. package/dist/backoff.d.ts +31 -0
  11. package/dist/backoff.js +40 -0
  12. package/dist/client.d.ts +28 -2
  13. package/dist/client.js +30 -3
  14. package/dist/context.d.ts +48 -2
  15. package/dist/context.js +101 -10
  16. package/dist/errors.d.ts +60 -4
  17. package/dist/errors.js +94 -9
  18. package/dist/index.d.ts +6 -2
  19. package/dist/index.js +2 -1
  20. package/dist/retention.d.ts +60 -0
  21. package/dist/retention.js +115 -0
  22. package/dist/store/base.d.ts +50 -1
  23. package/dist/store/base.js +114 -19
  24. package/dist/store/sqlite.js +4 -1
  25. package/dist/wait.d.ts +20 -5
  26. package/dist/wait.js +34 -9
  27. package/dist/worker.d.ts +214 -16
  28. package/dist/worker.js +500 -131
  29. package/package.json +1 -1
  30. package/src/backoff.ts +53 -0
  31. package/src/client.ts +43 -5
  32. package/src/context.ts +116 -9
  33. package/src/errors.ts +101 -9
  34. package/src/index.ts +6 -1
  35. package/src/retention.ts +136 -0
  36. package/src/store/base.ts +121 -17
  37. package/src/store/sqlite.ts +4 -1
  38. package/src/wait.ts +60 -16
  39. package/src/worker.ts +640 -146
package/dist/worker.js CHANGED
@@ -1,29 +1,16 @@
1
+ import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
1
2
  import { TaskContext } from "./context.js";
2
- import { errorEnvelope, LostLease, SerializationError, TaskError } from "./errors.js";
3
+ import { asEnvelope, errorEnvelope, EventLoopBlocked, exceptionEnvelope, LostLease, SerializationError, } from "./errors.js";
3
4
  import { newId } from "./ids.js";
4
5
  import { SQLiteStore } from "./store/sqlite.js";
5
6
  import { PostgresStore } from "./store/postgres.js";
6
7
  import { taskName } from "./task.js";
7
- const DEFAULT_RETRY_BACKOFF_MS = 1_000;
8
- const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
8
+ // Re-exported: they moved to backoff.ts so TaskContext.fail could share them
9
+ // without context.ts importing the module that imports it. They stay importable
10
+ // from this module, which is where they used to live.
11
+ export { retryDelayMs, DEFAULT_RETRY_BACKOFF_MS, DEFAULT_RETRY_BACKOFF_MAX_MS } from "./backoff.js";
9
12
  /** Wait after a failed claim, so a broken database is not polled in a tight loop. */
10
13
  const CLAIM_ERROR_BACKOFF_MS = 250;
11
- /** Exponential backoff for the next attempt of a task that just failed. */
12
- export function retryDelayMs(attempt, baseMs, maxMs) {
13
- if (baseMs <= 0)
14
- return 0;
15
- const exponent = Math.max(0, attempt - 1);
16
- return Math.min(maxMs, baseMs * 2 ** exponent);
17
- }
18
- function exceptionEnvelope(err) {
19
- const e = err;
20
- return errorEnvelope({
21
- type: e?.name ?? "Error",
22
- code: "handler_error",
23
- message: String(e?.message ?? err),
24
- retryable: true,
25
- });
26
- }
27
14
  /** Internal: an attempt outran maxRunMs and was abandoned. */
28
15
  class AttemptTimeout extends Error {
29
16
  maxRunMs;
@@ -65,6 +52,21 @@ function payloadBytes(task) {
65
52
  return 0;
66
53
  }
67
54
  }
55
+ /**
56
+ * Give back one call's unit of a counted budget. Deleting at zero is what keeps
57
+ * the map to the keys actually in flight, so an idle worker holds no entries at
58
+ * all — and both budgets (a name's own concurrency, a resource's capacity)
59
+ * settle the same way, from one place.
60
+ */
61
+ function release(counts, key) {
62
+ if (key == null)
63
+ return;
64
+ const rest = (counts.get(key) ?? 1) - 1;
65
+ if (rest > 0)
66
+ counts.set(key, rest);
67
+ else
68
+ counts.delete(key);
69
+ }
68
70
  export class Worker {
69
71
  store;
70
72
  queues;
@@ -73,6 +75,27 @@ export class Worker {
73
75
  workerId = newId("worker");
74
76
  /** Payload bytes charged to running handlers — see maxInFlightBytes. */
75
77
  inFlightBytes = 0;
78
+ /** Calls in flight, for the names that cap their own concurrency. */
79
+ callsInFlight = new Map();
80
+ /**
81
+ * Calls holding units of each declared resource. A resource is the same shape
82
+ * of budget as a name's own `concurrency` — a ceiling on calls — differing
83
+ * only in who draws from it: several names rather than one. That is what
84
+ * expresses "these handlers share one GPU" without inventing a queue per
85
+ * resource.
86
+ */
87
+ resourceCalls = new Map();
88
+ /** Rotates which source is offered the free budget first — see loop(). */
89
+ claimCursor = 0;
90
+ /** Invalidated by task(); see schedule(). */
91
+ scheduleCache = null;
92
+ /**
93
+ * Retry backoff, resolved once. Both settlement paths read these — the
94
+ * worker's own `safeFail` and the TaskContext it hands a handler — so
95
+ * resolving the defaults per call site is how the two drift apart.
96
+ */
97
+ backoffMs;
98
+ backoffMaxMs;
76
99
  stopped = false;
77
100
  stopWake;
78
101
  // Resolved once by stop(); every sleep races against it. A stopped worker
@@ -93,9 +116,16 @@ export class Worker {
93
116
  if (opts.maxInFlightBytes != null && opts.maxInFlightBytes <= 0) {
94
117
  throw new Error(`maxInFlightBytes must be > 0, got ${opts.maxInFlightBytes}`);
95
118
  }
119
+ for (const [resource, capacity] of Object.entries(opts.resources ?? {})) {
120
+ if (!Number.isInteger(capacity) || capacity < 1) {
121
+ throw new Error(`resources[${JSON.stringify(resource)}] must be an integer >= 1, got ${capacity}`);
122
+ }
123
+ }
96
124
  if (opts.maxQueueDepth != null) {
97
125
  store.useBackpressure(opts);
98
126
  }
127
+ this.backoffMs = opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
128
+ this.backoffMaxMs = opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS;
99
129
  }
100
130
  static sqlite(path, opts = {}) {
101
131
  const { queues = ["default"], busyTimeoutMs, ...rest } = opts;
@@ -114,7 +144,30 @@ export class Worker {
114
144
  get id() {
115
145
  return this.workerId;
116
146
  }
117
- task(arg, handler) {
147
+ task(arg, second, third) {
148
+ // Option form: (name | def, { batch?, concurrency?, resource? }, handler).
149
+ // Peel the options off and fall through, so name resolution and registration
150
+ // stay single-sited.
151
+ let batch;
152
+ let concurrency;
153
+ let resource;
154
+ let handler = second;
155
+ if (second != null && typeof second === "object") {
156
+ if (second.batch != null) {
157
+ if (!Number.isInteger(second.batch) || second.batch < 1) {
158
+ throw new Error(`batch must be an integer >= 1, got ${second.batch}`);
159
+ }
160
+ batch = second.batch;
161
+ }
162
+ if (second.concurrency != null) {
163
+ if (!Number.isInteger(second.concurrency) || second.concurrency < 1) {
164
+ throw new Error(`concurrency must be an integer >= 1, got ${second.concurrency}`);
165
+ }
166
+ concurrency = second.concurrency;
167
+ }
168
+ resource = second.resource;
169
+ handler = third;
170
+ }
118
171
  let name;
119
172
  let fn;
120
173
  if (typeof arg === "function") {
@@ -133,7 +186,16 @@ export class Worker {
133
186
  name = taskName(arg);
134
187
  fn = handler;
135
188
  }
136
- this.handlers.set(name, fn);
189
+ // Loudly, at registration: an undeclared resource would otherwise read as
190
+ // an unbounded one, so a typo would silently remove the ceiling the caller
191
+ // asked for — the failure this option exists to prevent.
192
+ if (resource != null && this.opts.resources?.[resource] == null) {
193
+ const known = Object.keys(this.opts.resources ?? {}).sort().join(", ") || "none";
194
+ throw new Error(`task ${JSON.stringify(name)} declares resource ${JSON.stringify(resource)}, ` +
195
+ `which is not in WorkerOptions.resources; declared: ${known}`);
196
+ }
197
+ this.handlers.set(name, { fn, batch, concurrency, resource });
198
+ this.scheduleCache = null;
137
199
  return this;
138
200
  }
139
201
  stop() {
@@ -164,11 +226,12 @@ export class Worker {
164
226
  // beyond even stop()'s reach.
165
227
  const concurrency = Math.max(1, opts.concurrency ?? this.opts.concurrency ?? 1);
166
228
  const leaseMs = this.opts.leaseMs ?? 30_000;
167
- const batch = this.opts.claimBatch ?? concurrency;
168
229
  await this.store.connect();
230
+ // The calls in flight — `concurrency` counts these, so the set's size is the
231
+ // budget. How many tasks they carry is `maxInFlightBytes`'s business.
169
232
  const running = new Set();
170
233
  try {
171
- await this.loop(concurrency, batch, leaseMs, running);
234
+ await this.loop(concurrency, leaseMs, running);
172
235
  }
173
236
  finally {
174
237
  // Whatever ends the loop — stop(), or something unexpected out of the body
@@ -178,35 +241,175 @@ export class Worker {
178
241
  await Promise.all([...running]);
179
242
  }
180
243
  }
181
- async loop(concurrency, batch, leaseMs, running) {
244
+ /**
245
+ * Split one claim into handler calls, each with the registration to run it.
246
+ *
247
+ * A claim is filtered by queue and by the names this worker handles, so it
248
+ * comes back mixed; batch size is per name (one embedding call wants 256
249
+ * texts, one Docling parse wants exactly 1). So group by name, then chunk each
250
+ * group by that name's size. Names registered without `batch` come back as
251
+ * one-task calls, as do names not registered at all — reachable only if a
252
+ * handler is unregistered mid-run, and dispatched to failNoHandler().
253
+ *
254
+ * The registration rides along because this is where it was resolved; looking
255
+ * it up again at the call site would put "is this name batched" in two places.
256
+ */
257
+ deliveries(claimed) {
258
+ const byName = new Map();
259
+ for (const task of claimed) {
260
+ const group = byName.get(task.name);
261
+ if (group)
262
+ group.push(task);
263
+ else
264
+ byName.set(task.name, [task]);
265
+ }
266
+ const out = [];
267
+ for (const [name, group] of byName) {
268
+ const reg = this.handlers.get(name);
269
+ const size = reg?.batch ?? 1;
270
+ for (let i = 0; i < group.length; i += size)
271
+ out.push([reg, group.slice(i, i + size)]);
272
+ }
273
+ return out;
274
+ }
275
+ context(task, leaseMs) {
276
+ return new TaskContext(this.store, task, this.workerId, leaseMs, {
277
+ retryBackoffMs: this.backoffMs,
278
+ retryBackoffMaxMs: this.backoffMaxMs,
279
+ });
280
+ }
281
+ /**
282
+ * How this poll's claim is split into per-name quotas, plus the union of names
283
+ * the probe spans.
284
+ *
285
+ * Cached and invalidated by `task()`, rather than rebuilt each poll: handlers
286
+ * may be registered after run() started, but only there, and this otherwise
287
+ * allocates a source per name on every tick for a worker's whole lifetime.
288
+ *
289
+ * A name that limits itself — by `batch`, by its own `concurrency`, or by a
290
+ * `resource` — needs a quota the shared draw cannot express, so it gets a
291
+ * source of its own; every other name shares one, where a task is a call.
292
+ *
293
+ * A resource is deliberately *not* one source spanning its names: `batch` is
294
+ * per name, and a single source carries one batch size, so two members that
295
+ * batch differently could not share a draw. Keeping a source per name and
296
+ * letting several of them draw down one shared ceiling composes with batching
297
+ * instead of excluding it.
298
+ */
299
+ schedule() {
300
+ if (this.scheduleCache)
301
+ return this.scheduleCache;
302
+ const sources = [];
303
+ const shared = [];
304
+ for (const [name, reg] of this.handlers) {
305
+ if (reg.batch != null || reg.concurrency != null || reg.resource != null) {
306
+ sources.push({
307
+ key: reg.concurrency == null ? undefined : name,
308
+ names: [name],
309
+ batch: reg.batch ?? 1,
310
+ concurrency: reg.concurrency,
311
+ resource: reg.resource,
312
+ });
313
+ }
314
+ else
315
+ shared.push(name);
316
+ }
317
+ if (shared.length)
318
+ sources.push({ names: shared, batch: 1 });
319
+ return (this.scheduleCache = { sources, names: [...this.handlers.keys()] });
320
+ }
321
+ /**
322
+ * A source's own call ceiling for one poll, or undefined when only the
323
+ * worker-wide budget applies.
324
+ *
325
+ * Three independent ceilings, whichever binds first: the name's own concurrency
326
+ * less what it is already running; its resource's capacity less what is running
327
+ * *and* what earlier draws in this same poll already took (`taken`); and
328
+ * `claimBatch`. The last is a ceiling on *rows* per poll, so it converts at this
329
+ * source's batch size — and never below one call, or a `claimBatch` under some
330
+ * name's batch would stall that name outright.
331
+ */
332
+ sourceCalls(src, taken) {
333
+ const rows = this.opts.claimBatch;
334
+ const byRows = rows == null ? undefined : Math.max(1, Math.floor(rows / src.batch));
335
+ const byName = src.concurrency == null
336
+ ? undefined
337
+ : Math.max(0, src.concurrency - (this.callsInFlight.get(src.key) ?? 0));
338
+ const byResource = src.resource == null
339
+ ? undefined
340
+ : Math.max(0, this.opts.resources[src.resource] -
341
+ (this.resourceCalls.get(src.resource) ?? 0) -
342
+ (taken.get(src.resource) ?? 0));
343
+ const limits = [byRows, byName, byResource].filter((n) => n != null);
344
+ return limits.length ? Math.min(...limits) : undefined;
345
+ }
346
+ async loop(concurrency, leaseMs, running) {
182
347
  const pollMs = this.opts.pollIntervalMs ?? 500;
183
348
  const byteBudget = this.opts.maxInFlightBytes;
184
349
  while (!this.stopped) {
350
+ // `concurrency` counts calls, so the calls in flight *are* the running
351
+ // promises — a batch holding 256 tasks is one of them.
185
352
  const free = concurrency - running.size;
186
- // Two ceilings, either of which stops the claim: task count and resident
187
- // payload bytes. The byte arm is guarded on running.size because it must
188
- // never be the reason we race an empty set — Promise.race([]) is pending
189
- // forever, past even stop(). With nothing running, nothing is resident, so
190
- // the budget cannot be the thing holding us back anyway.
353
+ // Two ceilings, either of which stops the claim: calls in flight and
354
+ // resident payload bytes. The byte arm is guarded on running.size because
355
+ // it must never be the reason we race an empty set — Promise.race([]) is
356
+ // pending forever, past even stop(). With nothing running, nothing is
357
+ // resident, so the budget cannot be the thing holding us back anyway.
191
358
  const overBudget = byteBudget != null && this.inFlightBytes >= byteBudget;
192
359
  if (running.size > 0 && (free <= 0 || overBudget)) {
193
- // Wait for a slot rather than spinning. execute() never rejects, so
360
+ // Wait for a slot rather than spinning. runCall() never rejects, so
194
361
  // racing these is safe.
195
362
  await Promise.race([...running]);
196
363
  continue;
197
364
  }
365
+ const { sources, names } = this.schedule();
366
+ if (!sources.length) {
367
+ await this.idle(pollMs);
368
+ continue;
369
+ }
370
+ // Round-robin the starting point. The draws are served in order, so without
371
+ // rotating it the first source would take every free slot and the rest
372
+ // would starve behind its backlog.
373
+ const cursor = this.claimCursor % sources.length;
374
+ const order = [...sources.slice(cursor), ...sources.slice(0, cursor)];
375
+ this.claimCursor = (cursor + 1) % sources.length;
198
376
  let claimed;
199
377
  try {
200
- claimed = await this.store.claim({
201
- queues: this.queues,
202
- // Only what this worker can run. Queues do not partition work by task
203
- // name, so another worker's tasks would otherwise be claimed here and
204
- // failed for want of a handler. Read each poll: handlers may be
205
- // registered after run() started.
206
- names: [...this.handlers.keys()],
207
- workerId: this.workerId,
208
- leaseMs,
209
- limit: Math.min(batch, free),
378
+ claimed = await this.store.claimSession(
379
+ // Only what this worker can run. Queues do not partition work by task
380
+ // name, so another worker's tasks would otherwise be claimed here and
381
+ // failed for want of a handler.
382
+ { queues: this.queues, workerId: this.workerId, leaseMs, names }, async (claim) => {
383
+ const drawn = [];
384
+ let left = free;
385
+ // Resource units this poll has already drawn. resourceCalls only
386
+ // moves when a call is dispatched, which happens after this whole
387
+ // plan returns — so without a local tally two sources sharing a
388
+ // resource would each see its full ceiling and together overshoot
389
+ // it. Same shape as `left`, one budget down.
390
+ const taken = new Map();
391
+ for (const src of order) {
392
+ if (left <= 0)
393
+ break;
394
+ const quota = Math.min(this.sourceCalls(src, taken) ?? left, left);
395
+ if (quota <= 0)
396
+ continue;
397
+ const rows = await claim(src.names, src.batch * quota);
398
+ if (!rows.length)
399
+ continue;
400
+ // deliveries() is what actually turns rows into handler calls, so
401
+ // spending the budget against its result is the only way the two
402
+ // cannot disagree. A source with nothing queued costs nothing,
403
+ // which is why the budget is spent here, draw by draw, rather than
404
+ // divided up before the claim.
405
+ const calls = this.deliveries(rows);
406
+ drawn.push({ src, calls });
407
+ left -= calls.length;
408
+ if (src.resource != null) {
409
+ taken.set(src.resource, (taken.get(src.resource) ?? 0) + calls.length);
410
+ }
411
+ }
412
+ return drawn;
210
413
  });
211
414
  }
212
415
  catch (err) {
@@ -216,20 +419,35 @@ export class Worker {
216
419
  await this.sleepOrStop(CLAIM_ERROR_BACKOFF_MS);
217
420
  continue;
218
421
  }
219
- if (claimed.length === 0) {
422
+ if (!claimed?.length) {
220
423
  await this.idle(pollMs);
221
424
  continue;
222
425
  }
223
- for (const task of claimed) {
224
- // Charged before the handler starts and refunded when it settles, so the
225
- // budget covers exactly the span the payload is pinned in memory.
226
- const bytes = byteBudget == null ? 0 : payloadBytes(task);
227
- this.inFlightBytes += bytes;
228
- const p = this.execute(task, leaseMs).finally(() => {
229
- this.inFlightBytes -= bytes;
230
- running.delete(p);
231
- });
232
- running.add(p);
426
+ for (const { src, calls } of claimed) {
427
+ for (const [reg, group] of calls) {
428
+ // Charged before the handler starts and refunded when it settles, so
429
+ // the budgets cover exactly the span the call holds its slot and its
430
+ // payloads stay pinned in memory.
431
+ const bytes = byteBudget == null ? 0 : group.reduce((sum, t) => sum + payloadBytes(t), 0);
432
+ this.inFlightBytes += bytes;
433
+ const key = src.key;
434
+ const resource = src.resource;
435
+ if (key != null)
436
+ this.callsInFlight.set(key, (this.callsInFlight.get(key) ?? 0) + 1);
437
+ if (resource != null) {
438
+ this.resourceCalls.set(resource, (this.resourceCalls.get(resource) ?? 0) + 1);
439
+ }
440
+ // An unregistered name has nothing to run, so it never starts a handler
441
+ // or a heartbeat; everything else is one call, batched or not.
442
+ const call = reg == null ? this.failNoHandler(group[0], leaseMs) : this.runCall(reg, group, leaseMs);
443
+ const p = call.finally(() => {
444
+ this.inFlightBytes -= bytes;
445
+ release(this.callsInFlight, key);
446
+ release(this.resourceCalls, resource);
447
+ running.delete(p);
448
+ });
449
+ running.add(p);
450
+ }
233
451
  }
234
452
  }
235
453
  }
@@ -263,74 +481,78 @@ export class Worker {
263
481
  }
264
482
  }
265
483
  /**
266
- * Run one task to completion. Never rejects: a task-level failure is reported
267
- * through onError and the loop moves on. (It used to reject into a promise
268
- * nobody awaited — an unhandled rejection that took the process down.)
484
+ * Record a claimed task this worker cannot run. Reachable only if a name is
485
+ * unregistered mid-run — the claim filters on the registered names — so it does
486
+ * not start a handler or a heartbeat for a task it will not run.
269
487
  */
270
- async execute(task, leaseMs) {
271
- const ctx = new TaskContext(this.store, task, this.workerId, leaseMs);
272
- const hb = this.startHeartbeat(ctx, leaseMs);
488
+ async failNoHandler(task, leaseMs) {
489
+ try {
490
+ await this.safeFail(this.context(task, leaseMs), errorEnvelope({
491
+ type: "NoHandler",
492
+ code: "no_handler",
493
+ message: `no handler registered for ${task.name}`,
494
+ retryable: false,
495
+ }), false);
496
+ }
497
+ catch (err) {
498
+ this.report(err, { phase: "execute", taskId: task.id });
499
+ }
500
+ }
501
+ /**
502
+ * Run one handler call to completion — one task, or a whole batch. Never
503
+ * rejects: a task-level failure is reported through onError and the loop moves
504
+ * on. (It used to reject into a promise nobody awaited — an unhandled rejection
505
+ * that took the process down.)
506
+ *
507
+ * One lifecycle for both delivery modes, because a single-task handler *is* the
508
+ * one-element case: the same heartbeat covers the call, the same classifier
509
+ * reads its error, and the same rule settles what is left. Only two things vary
510
+ * — how the handler is invoked, and where a leftover task's result comes from —
511
+ * so those are the only two branches below.
512
+ *
513
+ * The contract is single: **when the handler returns, every task it did not
514
+ * settle itself is settled by how the call ended.** Returning succeeds them,
515
+ * throwing fails them (retryably, or as the thrown TaskError says). That is
516
+ * what keeps the ordinary cases free of bookkeeping — a handler that just
517
+ * returns has finished 256 tasks — while still letting it pick individual
518
+ * tasks off with `item.succeed()` / `item.fail()` as it goes.
519
+ *
520
+ * For a batch, a returned map of task id -> result fills in results for the
521
+ * tasks left over; anything else returned is ignored, and unmentioned tasks
522
+ * succeed with no result (the common shape, where the handler's output went to
523
+ * a database rather than into the task row). For a single task the return value
524
+ * simply is the result.
525
+ */
526
+ async runCall(reg, tasks, leaseMs) {
527
+ const ctxs = tasks.map((t) => this.context(t, leaseMs));
528
+ const batched = reg.batch != null;
529
+ const hb = this.startHeartbeat(ctxs, leaseMs);
273
530
  try {
274
- const handler = this.handlers.get(task.name);
275
- if (!handler) {
276
- await this.safeFail(task, errorEnvelope({
277
- type: "NoHandler",
278
- code: "no_handler",
279
- message: `no handler registered for ${task.name}`,
280
- retryable: false,
281
- }), false);
282
- return;
283
- }
284
531
  let result;
285
532
  try {
286
- result = await this.attempt(handler, ctx, task);
533
+ result = await this.attempt(() => batched
534
+ ? reg.fn(ctxs)
535
+ : reg.fn(ctxs[0], tasks[0].payload), ctxs);
287
536
  }
288
537
  catch (err) {
289
- if (err instanceof LostLease)
290
- return;
291
- if (err instanceof AttemptTimeout) {
292
- // Recorded as a retryable failure, so backoff / maxAttempts /
293
- // cancel-wins all apply exactly as for a thrown error.
294
- await this.safeFail(task, timeoutEnvelope(task.name, err.maxRunMs), true);
295
- return;
296
- }
297
- if (err instanceof TaskError) {
298
- await this.safeFail(task, err.envelope(), err.retryable);
299
- }
300
- else {
301
- await this.safeFail(task, exceptionEnvelope(err), true);
302
- }
538
+ const outcome = this.outcomeOf(err, tasks[0].name);
539
+ if (outcome)
540
+ await this.settleEach(ctxs, (c) => this.safeFail(c, outcome[0], outcome[1]));
303
541
  return;
304
542
  }
305
- try {
306
- // complete (not succeed): finalizes as canceled if a cancel was
307
- // requested while the handler ran, else succeeded.
308
- await this.store.complete({ taskId: task.id, workerId: this.workerId, result });
309
- }
310
- catch (err) {
311
- if (err instanceof LostLease) {
312
- ctx.markLeaseLost();
313
- return;
314
- }
315
- if (err instanceof SerializationError) {
316
- // The handler succeeded but its return value can't cross the JSON
317
- // protocol (BigInt, non-finite number, circular). Deterministic, so
318
- // fail fast and permanently — the alternative is sitting `running`
319
- // until lease expiry redelivers a task that fails the same way every
320
- // attempt.
321
- await this.safeFail(task, errorEnvelope({
322
- type: "SerializationError",
323
- code: "unserializable_result",
324
- message: `handler result is not JSON-serializable: ${err.message}`,
325
- retryable: false,
326
- }), false);
327
- return;
328
- }
329
- throw err;
330
- }
543
+ // A batch handler's return maps task id -> result; a single-task handler's
544
+ // return *is* the result. Anything else a batch returns is ignored, so its
545
+ // leftovers succeed with no result — hence the empty map rather than
546
+ // falling through to `result`.
547
+ const results = batched
548
+ ? result != null && typeof result === "object"
549
+ ? result
550
+ : {}
551
+ : null;
552
+ await this.settleEach(ctxs, (c) => this.succeedOne(c, results ? (results[c.taskId] ?? null) : result));
331
553
  }
332
554
  catch (err) {
333
- this.report(err, { phase: "execute", taskId: task.id });
555
+ this.report(err, { phase: "execute", taskId: tasks[0].id });
334
556
  }
335
557
  finally {
336
558
  hb.cancel();
@@ -338,20 +560,117 @@ export class Worker {
338
560
  }
339
561
  }
340
562
  /**
341
- * Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
342
- * abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
343
- * a handler that keeps running can never write again — see
344
- * TaskContext.owned), then the still-pending promise is left to settle on
345
- * its own, its outcome discarded. The caller records the handler_timeout
346
- * failure; lease recovery is NOT involved, so redelivery is immediate.
563
+ * How an attempt that ended badly is recorded: [envelope, retryable], or null
564
+ * when there is nothing to record.
565
+ *
566
+ * One classifier for both delivery modes, so a handler error cannot mean
567
+ * different things depending on how its task happened to be delivered. That
568
+ * includes LostLease, which is not an outcome at all: it means a write through
569
+ * this context was already rejected, so recording anything more would be
570
+ * rejected too. Both modes then leave the task alone — the single-task one has
571
+ * nothing else to do, and a batch lets its remaining tasks fall to lease expiry
572
+ * and redelivery rather than stamping them with a failure the handler never
573
+ * reported.
347
574
  */
348
- async attempt(handler, ctx, task) {
575
+ outcomeOf(err, name) {
576
+ if (err instanceof LostLease)
577
+ return null;
578
+ // Retryable, so backoff / maxAttempts / cancel-wins all apply exactly as for
579
+ // a thrown error.
580
+ if (err instanceof AttemptTimeout)
581
+ return [timeoutEnvelope(name, err.maxRunMs), true];
582
+ // Everything else is what a handler could equally have passed to ctx.fail(),
583
+ // so it goes through the same normalizer — a TaskError keeps its own
584
+ // retryability, anything else is retryable. Non-Error throws reach
585
+ // exceptionEnvelope the same way, since only ctx.fail can supply a ready
586
+ // envelope and a thrown object is not one.
587
+ if (err !== null && typeof err === "object" && !(err instanceof Error)) {
588
+ return [exceptionEnvelope(err), true];
589
+ }
590
+ return asEnvelope(err, true);
591
+ }
592
+ // Both leftover paths write through the store rather than through the context.
593
+ // `TaskContext.owned` short-circuits once a lease is known lost, which is there
594
+ // to stop a *zombie handler* writing after its attempt was abandoned — but
595
+ // these run after the handler is done, on the worker's own authority, exactly
596
+ // as execute()'s completion and failure arms do. Ownership is still enforced by
597
+ // each statement, so a task whose lease really was lost writes nothing either
598
+ // way.
599
+ /**
600
+ * Finalize one task the handler left for the worker to decide — the tail of
601
+ * both delivery modes.
602
+ *
603
+ * Includes the unserializable-result rule: the handler succeeded but its value
604
+ * cannot cross the JSON protocol (BigInt, non-finite number, circular), which
605
+ * is deterministic, so it fails permanently rather than being redelivered to
606
+ * fail the same way every attempt.
607
+ */
608
+ async succeedOne(ctx, result) {
609
+ if (ctx.settled)
610
+ return;
611
+ try {
612
+ // complete (not succeed): finalizes as canceled if a cancel was requested
613
+ // while the handler ran, else succeeded.
614
+ await this.store.complete({ taskId: ctx.taskId, workerId: this.workerId, result });
615
+ ctx.markSettled();
616
+ }
617
+ catch (err) {
618
+ if (err instanceof LostLease) {
619
+ ctx.markLeaseLost();
620
+ return;
621
+ }
622
+ if (err instanceof SerializationError) {
623
+ await this.safeFail(ctx, errorEnvelope({
624
+ type: "SerializationError",
625
+ code: "unserializable_result",
626
+ message: `handler result is not JSON-serializable: ${err.message}`,
627
+ retryable: false,
628
+ }), false);
629
+ return;
630
+ }
631
+ throw err;
632
+ }
633
+ }
634
+ /**
635
+ * Settle every task the handler did not settle itself, concurrently, reporting
636
+ * rather than throwing. Each task keeps its own attempt count and backoff —
637
+ * they are separate tasks that happened to be delivered together.
638
+ *
639
+ * allSettled, because one task's write failing must not abandon the rest of the
640
+ * batch mid-settlement — the others still hold leases and would sit `running`
641
+ * until expiry. Each outcome is reported against the task it belongs to;
642
+ * without that, an operator learns a settlement failed somewhere in a batch of
643
+ * 256.
644
+ */
645
+ async settleEach(ctxs, settle) {
646
+ const left = ctxs.filter((c) => !c.settled);
647
+ if (!left.length)
648
+ return;
649
+ const outcomes = await Promise.allSettled(left.map(settle));
650
+ outcomes.forEach((outcome, i) => {
651
+ if (outcome.status === "rejected") {
652
+ this.report(outcome.reason, { phase: "execute", taskId: left[i].taskId });
653
+ }
654
+ });
655
+ }
656
+ /**
657
+ * Run one attempt — of one task or of a whole batch — bounded by maxRunMs when
658
+ * set.
659
+ *
660
+ * On timeout the attempt is abandoned: every context it covers is flagged
661
+ * lease-lost first (ctx.signal aborts, and a handler that keeps running can
662
+ * never write again, nor settle anything behind the worker's back — see
663
+ * TaskContext.owned), then the still-pending promise is left to settle on its
664
+ * own, its outcome discarded. The caller records the handler_timeout failure;
665
+ * lease recovery is NOT involved, so redelivery is immediate.
666
+ */
667
+ async attempt(invoke, ctxs) {
349
668
  const maxRunMs = this.opts.maxRunMs;
350
669
  if (maxRunMs == null)
351
- return handler(ctx, task.payload);
670
+ return invoke();
352
671
  // As a real promise: the race needs one (a handler may return a plain
353
672
  // value), and the timeout path .catch()es it.
354
- const run = (async () => handler(ctx, task.payload))();
673
+ const run = (async () => invoke())();
355
674
  let timer;
356
675
  const winner = await Promise.race([
357
676
  run,
@@ -359,17 +678,51 @@ export class Worker {
359
678
  ]).finally(() => clearTimeout(timer));
360
679
  if (winner !== TIMED_OUT)
361
680
  return winner;
362
- ctx.markLeaseLost();
681
+ for (const ctx of ctxs)
682
+ ctx.markLeaseLost();
363
683
  // The zombie may still reject later; that must not become an unhandled
364
684
  // rejection — its outcome was already decided to be handler_timeout.
365
685
  run.catch(() => { });
366
686
  throw new AttemptTimeout(maxRunMs);
367
687
  }
368
- startHeartbeat(ctx, leaseMs) {
688
+ /**
689
+ * One statement per beat, however many tasks the call covers — a single-task
690
+ * handler is just the one-element case.
691
+ *
692
+ * Only tasks still in play are renewed. A task the handler already settled is
693
+ * terminal, and re-leasing it would be a write against a row nobody owns; that
694
+ * is also why an absence is only read as lease loss after re-checking
695
+ * `settled`, since the handler may have finalized the task while this beat was
696
+ * in flight, which takes the row out of `running` and out of the reply. A task
697
+ * genuinely missing lost its lease (another worker recovered it), so its
698
+ * context is flagged and the handler stops being able to write through it —
699
+ * only that one, never its neighbours.
700
+ */
701
+ startHeartbeat(ctxs, leaseMs) {
369
702
  let active = true;
370
703
  let wake = null;
371
704
  // lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
372
705
  const interval = this.opts.heartbeatIntervalMs ?? Math.max(50, Math.floor(leaseMs / 3));
706
+ // When this heartbeat last ran. A loop cannot report its own absence — a
707
+ // handler that blocks for its whole attempt never lets the timer fire at all
708
+ // — so the check lives outside the loop and cancel() runs it too.
709
+ let lastBeatAt = Date.now();
710
+ /** Report a heartbeat that has not run for more than two intervals: a whole
711
+ * beat missed, which at lease/3 means the next such block loses the lease
712
+ * outright. Fires while the lease still holds — after it expires the only
713
+ * evidence is a task that ran twice, in two workers' logs, with no error in
714
+ * either. */
715
+ const checkBeat = () => {
716
+ const now = Date.now();
717
+ const lateMs = now - lastBeatAt - interval;
718
+ if (lateMs > interval) {
719
+ this.report(new EventLoopBlocked(lateMs, interval, leaseMs), {
720
+ phase: "execute",
721
+ taskId: ctxs[0].taskId,
722
+ });
723
+ }
724
+ lastBeatAt = now;
725
+ };
373
726
  const done = (async () => {
374
727
  while (active) {
375
728
  // Cancellable sleep: cancel() resolves this immediately and clears the
@@ -384,14 +737,28 @@ export class Worker {
384
737
  wake = null;
385
738
  if (!active)
386
739
  break;
740
+ checkBeat();
741
+ const live = ctxs.filter((c) => !c.settled && !c.lostLease);
742
+ if (!live.length)
743
+ break;
387
744
  try {
388
- await ctx.heartbeat();
745
+ const renewed = await this.store.heartbeatBatch({
746
+ taskIds: live.map((c) => c.taskId),
747
+ workerId: this.workerId,
748
+ leaseMs,
749
+ });
750
+ for (const ctx of live) {
751
+ const cancelRequested = renewed.get(ctx.taskId);
752
+ // Cancellation rides along on the write we were making anyway, so
753
+ // ctx.canceled() stays free here too.
754
+ if (cancelRequested !== undefined)
755
+ ctx.observeCancel(cancelRequested);
756
+ else if (!ctx.settled)
757
+ ctx.markLeaseLost();
758
+ }
389
759
  }
390
760
  catch (err) {
391
- // ctx.heartbeat() already flagged the lease as lost for the handler.
392
- if (err instanceof LostLease)
393
- break;
394
- this.report(err, { phase: "execute", taskId: ctx.taskId });
761
+ this.report(err, { phase: "execute", taskId: live[0].taskId });
395
762
  }
396
763
  }
397
764
  })();
@@ -400,22 +767,24 @@ export class Worker {
400
767
  active = false;
401
768
  if (wake)
402
769
  wake();
770
+ // The attempt is over, so this is the last chance to notice that the
771
+ // heartbeat never got one — the whole-attempt block, which is also the
772
+ // one where the lease is already gone.
773
+ checkBeat();
403
774
  },
404
775
  done,
405
776
  };
406
777
  }
407
- async safeFail(task, envelope, retryable) {
408
- const delayMs = retryable
409
- ? retryDelayMs(task.attempt, this.opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS, this.opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS)
410
- : 0;
778
+ async safeFail(ctx, envelope, retryable) {
411
779
  try {
412
780
  await this.store.fail({
413
- taskId: task.id,
781
+ taskId: ctx.taskId,
414
782
  workerId: this.workerId,
415
783
  error: envelope,
416
784
  retryable,
417
- delayMs,
785
+ delayMs: failDelayMs(ctx.attempt, retryable, this.backoffMs, this.backoffMaxMs),
418
786
  });
787
+ ctx.markSettled();
419
788
  }
420
789
  catch (err) {
421
790
  if (!(err instanceof LostLease))