cairnq 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +11 -5
  2. package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
  3. package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
  4. package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
  5. package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
  6. package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
  7. package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
  8. package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
  9. package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
  10. package/dist/backoff.d.ts +31 -0
  11. package/dist/backoff.js +40 -0
  12. package/dist/client.d.ts +28 -2
  13. package/dist/client.js +30 -3
  14. package/dist/context.d.ts +48 -2
  15. package/dist/context.js +101 -10
  16. package/dist/errors.d.ts +60 -4
  17. package/dist/errors.js +94 -9
  18. package/dist/index.d.ts +6 -2
  19. package/dist/index.js +2 -1
  20. package/dist/retention.d.ts +60 -0
  21. package/dist/retention.js +115 -0
  22. package/dist/store/base.d.ts +50 -1
  23. package/dist/store/base.js +114 -19
  24. package/dist/store/sqlite.js +4 -1
  25. package/dist/wait.d.ts +20 -5
  26. package/dist/wait.js +34 -9
  27. package/dist/worker.d.ts +214 -16
  28. package/dist/worker.js +500 -131
  29. package/package.json +1 -1
  30. package/src/backoff.ts +53 -0
  31. package/src/client.ts +43 -5
  32. package/src/context.ts +116 -9
  33. package/src/errors.ts +101 -9
  34. package/src/index.ts +6 -1
  35. package/src/retention.ts +136 -0
  36. package/src/store/base.ts +121 -17
  37. package/src/store/sqlite.ts +4 -1
  38. package/src/wait.ts +60 -16
  39. package/src/worker.ts +640 -146
package/src/worker.ts CHANGED
@@ -1,6 +1,15 @@
1
+ import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
1
2
  import type { BackpressureOptions } from "./backpressure.js";
2
3
  import { TaskContext } from "./context.js";
3
- import { errorEnvelope, LostLease, SerializationError, TaskError } from "./errors.js";
4
+ import {
5
+ asEnvelope,
6
+ errorEnvelope,
7
+ EventLoopBlocked,
8
+ exceptionEnvelope,
9
+ type FailReason,
10
+ LostLease,
11
+ SerializationError,
12
+ } from "./errors.js";
4
13
  import { newId } from "./ids.js";
5
14
  import { type Task } from "./models.js";
6
15
  import { SQLiteStore } from "./store/sqlite.js";
@@ -8,10 +17,75 @@ import { PostgresStore } from "./store/postgres.js";
8
17
  import type { TaskStore } from "./store/base.js";
9
18
  import { type TaskDef, taskName } from "./task.js";
10
19
 
20
+ // Re-exported: they moved to backoff.ts so TaskContext.fail could share them
21
+ // without context.ts importing the module that imports it. They stay importable
22
+ // from this module, which is where they used to live.
23
+ export { retryDelayMs, DEFAULT_RETRY_BACKOFF_MS, DEFAULT_RETRY_BACKOFF_MAX_MS } from "./backoff.js";
24
+
11
25
  export type Handler = (ctx: TaskContext, payload: any) => unknown | Promise<unknown>;
12
26
  /** Handler typed against a TaskDef<P, R>: payload is P, the return is R. */
13
27
  export type TypedHandler<P, R> = (ctx: TaskContext, payload: P) => R | Promise<R>;
14
28
 
29
+ /**
30
+ * A batch handler takes one argument: the list of contexts. There is no payload
31
+ * shortcut to pair with it — payloads are per task, so they are read off the
32
+ * items (`item.payload`), which is also what a handler must hold to settle one
33
+ * of them.
34
+ *
35
+ * Returning a map of task id -> result fills in results for the tasks the
36
+ * handler did not settle itself; anything else returned is ignored.
37
+ */
38
+ export type BatchHandler = (
39
+ items: TaskContext[],
40
+ ) => void | Record<string, unknown> | Promise<void | Record<string, unknown>>;
41
+
42
+ /** What `worker.task` recorded for one task name. */
43
+ interface Registration {
44
+ fn: Handler | BatchHandler;
45
+ /** Tasks per handler call, or undefined for one-at-a-time delivery. */
46
+ batch?: number;
47
+ /** Concurrent handler calls allowed for this name, or undefined for no limit
48
+ * beyond the worker's own. */
49
+ concurrency?: number;
50
+ /** Resource this name's calls draw from, or undefined to draw from nothing
51
+ * but the worker budget. Declared in `WorkerOptions.resources`. */
52
+ resource?: string;
53
+ }
54
+
55
+ /**
56
+ * One draw's worth of quota: a set of names and how many handler calls they may
57
+ * start. A name that limits itself — by `batch`, by its own `concurrency`, or by
58
+ * a `resource` it shares with other names — gets a source to itself, because its
59
+ * quota cannot be expressed in a draw shared with names that count differently.
60
+ * Everything else shares one, where a task is a call and the worker's own budget
61
+ * is the only ceiling.
62
+ */
63
+ interface ClaimSource {
64
+ /**
65
+ * Counts calls in flight, and set only when this source caps its own
66
+ * concurrency — nothing else reads the count, so nothing else pays for it.
67
+ * Such a source always holds exactly one name, so this is that name.
68
+ */
69
+ key?: string;
70
+ names: string[];
71
+ /** Tasks per call — 1 for the shared source. */
72
+ batch: number;
73
+ /** Calls allowed for this source, or undefined for the worker budget alone. */
74
+ concurrency?: number;
75
+ /**
76
+ * Resource this source draws from, or undefined. Unlike `concurrency`, the
77
+ * ceiling it names is shared with the other sources that declare it, which is
78
+ * what keeps two names off one scarce thing at the same time.
79
+ */
80
+ resource?: string;
81
+ }
82
+
83
+ /** What one poll's claim draws from, and the names the probe spans. */
84
+ interface Schedule {
85
+ sources: ClaimSource[];
86
+ names: string[];
87
+ }
88
+
15
89
  /** Where an error the worker recovered from came from. */
16
90
  export type ErrorPhase = "claim" | "execute";
17
91
 
@@ -21,6 +95,12 @@ export type ErrorPhase = "claim" | "execute";
21
95
  * there is usually no CairnQ handle to have configured the store.
22
96
  */
23
97
  export interface WorkerOptions extends Partial<BackpressureOptions> {
98
+ /**
99
+ * Handler calls allowed to run at once. A batch call counts as one, however
100
+ * many tasks it carries — size it for how much work you want in flight, not
101
+ * for how many tasks that comes to. Per-name limits refine it; `maxInFlightBytes`
102
+ * bounds memory, which task counts never did.
103
+ */
24
104
  concurrency?: number;
25
105
  leaseMs?: number;
26
106
  heartbeatIntervalMs?: number;
@@ -49,15 +129,29 @@ export interface WorkerOptions extends Partial<BackpressureOptions> {
49
129
  * between megabytes and gigabytes resident. Once the budget is spent the
50
130
  * worker stops claiming until running handlers give it back.
51
131
  *
52
- * The bound is on tasks already executing. A claim commits to a whole batch
53
- * before any size is known, so one batch can overshoot by up to `claimBatch`
54
- * payloads; lower `claimBatch` to tighten that. A single payload larger than
55
- * the entire budget still runs — alone, rather than deadlocking the worker.
132
+ * The bound is on tasks already executing: it is read between claims, never
133
+ * during one, and a claim commits to its rows before any size is known. One
134
+ * poll can therefore overshoot by up to `claimBatch` rows per registered name
135
+ * (or one whole `batch`, whichever is larger). Lower `claimBatch`, or the batch
136
+ * sizes, to tighten that. A single payload larger than the entire budget still
137
+ * runs — alone, rather than deadlocking the worker.
56
138
  *
57
139
  * Costs one JSON serialization per task to measure, so it is only computed
58
140
  * when set. Unset disables the budget.
59
141
  */
60
142
  maxInFlightBytes?: number;
143
+ /**
144
+ * Call ceilings that several names can draw from, by name — `{ gpu: 1 }`.
145
+ * A handler joins one with `task(name, { resource: "gpu" }, fn)`.
146
+ *
147
+ * `concurrency` caps a name against itself, which cannot say what usually
148
+ * binds a worker doing heavy local work: several *different* handlers
149
+ * contending for one scarce thing — a GPU, an index that tolerates a single
150
+ * writer. The limit belongs to that thing rather than to any one name, so it
151
+ * is declared here, once, and at capacity 1 it is mutual exclusion across the
152
+ * names that join it.
153
+ */
154
+ resources?: Record<string, number>;
61
155
  /**
62
156
  * Called for errors the worker survived — a claim that threw, a store write
63
157
  * that failed while finalizing a task. Without it these are silent: the run
@@ -67,28 +161,9 @@ export interface WorkerOptions extends Partial<BackpressureOptions> {
67
161
  onError?: (err: unknown, info: { phase: ErrorPhase; taskId?: string }) => void;
68
162
  }
69
163
 
70
- const DEFAULT_RETRY_BACKOFF_MS = 1_000;
71
- const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
72
164
  /** Wait after a failed claim, so a broken database is not polled in a tight loop. */
73
165
  const CLAIM_ERROR_BACKOFF_MS = 250;
74
166
 
75
- /** Exponential backoff for the next attempt of a task that just failed. */
76
- export function retryDelayMs(attempt: number, baseMs: number, maxMs: number): number {
77
- if (baseMs <= 0) return 0;
78
- const exponent = Math.max(0, attempt - 1);
79
- return Math.min(maxMs, baseMs * 2 ** exponent);
80
- }
81
-
82
- function exceptionEnvelope(err: unknown): Record<string, unknown> {
83
- const e = err as { name?: string; message?: string };
84
- return errorEnvelope({
85
- type: e?.name ?? "Error",
86
- code: "handler_error",
87
- message: String(e?.message ?? err),
88
- retryable: true,
89
- });
90
- }
91
-
92
167
  /** Internal: an attempt outran maxRunMs and was abandoned. */
93
168
  class AttemptTimeout extends Error {
94
169
  constructor(readonly maxRunMs: number) {
@@ -131,11 +206,45 @@ function payloadBytes(task: Task): number {
131
206
  }
132
207
  }
133
208
 
209
+ /**
210
+ * Give back one call's unit of a counted budget. Deleting at zero is what keeps
211
+ * the map to the keys actually in flight, so an idle worker holds no entries at
212
+ * all — and both budgets (a name's own concurrency, a resource's capacity)
213
+ * settle the same way, from one place.
214
+ */
215
+ function release(counts: Map<string, number>, key: string | undefined): void {
216
+ if (key == null) return;
217
+ const rest = (counts.get(key) ?? 1) - 1;
218
+ if (rest > 0) counts.set(key, rest);
219
+ else counts.delete(key);
220
+ }
221
+
134
222
  export class Worker {
135
- private readonly handlers = new Map<string, Handler>();
223
+ private readonly handlers = new Map<string, Registration>();
136
224
  private readonly workerId = newId("worker");
137
225
  /** Payload bytes charged to running handlers — see maxInFlightBytes. */
138
226
  private inFlightBytes = 0;
227
+ /** Calls in flight, for the names that cap their own concurrency. */
228
+ private readonly callsInFlight = new Map<string, number>();
229
+ /**
230
+ * Calls holding units of each declared resource. A resource is the same shape
231
+ * of budget as a name's own `concurrency` — a ceiling on calls — differing
232
+ * only in who draws from it: several names rather than one. That is what
233
+ * expresses "these handlers share one GPU" without inventing a queue per
234
+ * resource.
235
+ */
236
+ private readonly resourceCalls = new Map<string, number>();
237
+ /** Rotates which source is offered the free budget first — see loop(). */
238
+ private claimCursor = 0;
239
+ /** Invalidated by task(); see schedule(). */
240
+ private scheduleCache: Schedule | null = null;
241
+ /**
242
+ * Retry backoff, resolved once. Both settlement paths read these — the
243
+ * worker's own `safeFail` and the TaskContext it hands a handler — so
244
+ * resolving the defaults per call site is how the two drift apart.
245
+ */
246
+ private readonly backoffMs: number;
247
+ private readonly backoffMaxMs: number;
139
248
  private stopped = false;
140
249
  private stopWake!: () => void;
141
250
  // Resolved once by stop(); every sleep races against it. A stopped worker
@@ -158,9 +267,18 @@ export class Worker {
158
267
  if (opts.maxInFlightBytes != null && opts.maxInFlightBytes <= 0) {
159
268
  throw new Error(`maxInFlightBytes must be > 0, got ${opts.maxInFlightBytes}`);
160
269
  }
270
+ for (const [resource, capacity] of Object.entries(opts.resources ?? {})) {
271
+ if (!Number.isInteger(capacity) || capacity < 1) {
272
+ throw new Error(
273
+ `resources[${JSON.stringify(resource)}] must be an integer >= 1, got ${capacity}`,
274
+ );
275
+ }
276
+ }
161
277
  if (opts.maxQueueDepth != null) {
162
278
  store.useBackpressure(opts as BackpressureOptions);
163
279
  }
280
+ this.backoffMs = opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
281
+ this.backoffMaxMs = opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS;
164
282
  }
165
283
 
166
284
  static sqlite(
@@ -192,9 +310,61 @@ export class Worker {
192
310
  task(handler: Handler): this;
193
311
  task(name: string, handler: Handler): this;
194
312
  task<P, R>(def: TaskDef<P, R>, handler: TypedHandler<P, R>): this;
195
- task(arg: string | Handler | TaskDef, handler?: Handler): this {
313
+ /**
314
+ * Batch delivery: the handler takes one argument, a `TaskContext[]` of up to
315
+ * `batch` tasks, instead of `(ctx, payload)`. Use it when the work itself is
316
+ * batched — one embedding call over 256 texts rather than 256 calls — and size
317
+ * it by what the downstream API wants, not by the queue.
318
+ *
319
+ * `concurrency` caps the calls this name may run at once, under the worker's
320
+ * own. Use it to keep one expensive name from taking the whole worker.
321
+ *
322
+ * `resource` draws each call from a ceiling declared in
323
+ * `WorkerOptions.resources` and shared with every other name that names it —
324
+ * at capacity 1, mutual exclusion across those names.
325
+ */
326
+ task(
327
+ name: string | TaskDef,
328
+ opts: { batch: number; concurrency?: number; resource?: string },
329
+ handler: BatchHandler,
330
+ ): this;
331
+ /** Per-name concurrency or a shared resource, without batching: the handler
332
+ * still takes `(ctx, payload)`. */
333
+ task(
334
+ name: string | TaskDef,
335
+ opts: { concurrency?: number; resource?: string },
336
+ handler: Handler,
337
+ ): this;
338
+ task(
339
+ arg: string | Handler | TaskDef,
340
+ second?: Handler | { batch?: number; concurrency?: number; resource?: string },
341
+ third?: Handler | BatchHandler,
342
+ ): this {
343
+ // Option form: (name | def, { batch?, concurrency?, resource? }, handler).
344
+ // Peel the options off and fall through, so name resolution and registration
345
+ // stay single-sited.
346
+ let batch: number | undefined;
347
+ let concurrency: number | undefined;
348
+ let resource: string | undefined;
349
+ let handler = second as Handler | BatchHandler | undefined;
350
+ if (second != null && typeof second === "object") {
351
+ if (second.batch != null) {
352
+ if (!Number.isInteger(second.batch) || second.batch < 1) {
353
+ throw new Error(`batch must be an integer >= 1, got ${second.batch}`);
354
+ }
355
+ batch = second.batch;
356
+ }
357
+ if (second.concurrency != null) {
358
+ if (!Number.isInteger(second.concurrency) || second.concurrency < 1) {
359
+ throw new Error(`concurrency must be an integer >= 1, got ${second.concurrency}`);
360
+ }
361
+ concurrency = second.concurrency;
362
+ }
363
+ resource = second.resource;
364
+ handler = third;
365
+ }
196
366
  let name: string;
197
- let fn: Handler;
367
+ let fn: Handler | BatchHandler;
198
368
  if (typeof arg === "function") {
199
369
  // Bare form: worker.task(fn) — registered under the function's name.
200
370
  // Strip the "bound " prefix .bind() stamps on it: otherwise a bound
@@ -212,7 +382,18 @@ export class Worker {
212
382
  name = taskName(arg);
213
383
  fn = handler!;
214
384
  }
215
- this.handlers.set(name, fn);
385
+ // Loudly, at registration: an undeclared resource would otherwise read as
386
+ // an unbounded one, so a typo would silently remove the ceiling the caller
387
+ // asked for — the failure this option exists to prevent.
388
+ if (resource != null && this.opts.resources?.[resource] == null) {
389
+ const known = Object.keys(this.opts.resources ?? {}).sort().join(", ") || "none";
390
+ throw new Error(
391
+ `task ${JSON.stringify(name)} declares resource ${JSON.stringify(resource)}, ` +
392
+ `which is not in WorkerOptions.resources; declared: ${known}`,
393
+ );
394
+ }
395
+ this.handlers.set(name, { fn, batch, concurrency, resource });
396
+ this.scheduleCache = null;
216
397
  return this;
217
398
  }
218
399
 
@@ -246,11 +427,12 @@ export class Worker {
246
427
  // beyond even stop()'s reach.
247
428
  const concurrency = Math.max(1, opts.concurrency ?? this.opts.concurrency ?? 1);
248
429
  const leaseMs = this.opts.leaseMs ?? 30_000;
249
- const batch = this.opts.claimBatch ?? concurrency;
250
430
  await this.store.connect();
431
+ // The calls in flight — `concurrency` counts these, so the set's size is the
432
+ // budget. How many tasks they carry is `maxInFlightBytes`'s business.
251
433
  const running = new Set<Promise<void>>();
252
434
  try {
253
- await this.loop(concurrency, batch, leaseMs, running);
435
+ await this.loop(concurrency, leaseMs, running);
254
436
  } finally {
255
437
  // Whatever ends the loop — stop(), or something unexpected out of the body
256
438
  // — nothing this worker started may outlive run(). serve() closes the store
@@ -260,41 +442,181 @@ export class Worker {
260
442
  }
261
443
  }
262
444
 
445
+ /**
446
+ * Split one claim into handler calls, each with the registration to run it.
447
+ *
448
+ * A claim is filtered by queue and by the names this worker handles, so it
449
+ * comes back mixed; batch size is per name (one embedding call wants 256
450
+ * texts, one Docling parse wants exactly 1). So group by name, then chunk each
451
+ * group by that name's size. Names registered without `batch` come back as
452
+ * one-task calls, as do names not registered at all — reachable only if a
453
+ * handler is unregistered mid-run, and dispatched to failNoHandler().
454
+ *
455
+ * The registration rides along because this is where it was resolved; looking
456
+ * it up again at the call site would put "is this name batched" in two places.
457
+ */
458
+ private deliveries(claimed: Task[]): [Registration | undefined, Task[]][] {
459
+ const byName = new Map<string, Task[]>();
460
+ for (const task of claimed) {
461
+ const group = byName.get(task.name);
462
+ if (group) group.push(task);
463
+ else byName.set(task.name, [task]);
464
+ }
465
+ const out: [Registration | undefined, Task[]][] = [];
466
+ for (const [name, group] of byName) {
467
+ const reg = this.handlers.get(name);
468
+ const size = reg?.batch ?? 1;
469
+ for (let i = 0; i < group.length; i += size) out.push([reg, group.slice(i, i + size)]);
470
+ }
471
+ return out;
472
+ }
473
+
474
+ private context(task: Task, leaseMs: number): TaskContext {
475
+ return new TaskContext(this.store, task, this.workerId, leaseMs, {
476
+ retryBackoffMs: this.backoffMs,
477
+ retryBackoffMaxMs: this.backoffMaxMs,
478
+ });
479
+ }
480
+
481
+ /**
482
+ * How this poll's claim is split into per-name quotas, plus the union of names
483
+ * the probe spans.
484
+ *
485
+ * Cached and invalidated by `task()`, rather than rebuilt each poll: handlers
486
+ * may be registered after run() started, but only there, and this otherwise
487
+ * allocates a source per name on every tick for a worker's whole lifetime.
488
+ *
489
+ * A name that limits itself — by `batch`, by its own `concurrency`, or by a
490
+ * `resource` — needs a quota the shared draw cannot express, so it gets a
491
+ * source of its own; every other name shares one, where a task is a call.
492
+ *
493
+ * A resource is deliberately *not* one source spanning its names: `batch` is
494
+ * per name, and a single source carries one batch size, so two members that
495
+ * batch differently could not share a draw. Keeping a source per name and
496
+ * letting several of them draw down one shared ceiling composes with batching
497
+ * instead of excluding it.
498
+ */
499
+ private schedule(): Schedule {
500
+ if (this.scheduleCache) return this.scheduleCache;
501
+ const sources: ClaimSource[] = [];
502
+ const shared: string[] = [];
503
+ for (const [name, reg] of this.handlers) {
504
+ if (reg.batch != null || reg.concurrency != null || reg.resource != null) {
505
+ sources.push({
506
+ key: reg.concurrency == null ? undefined : name,
507
+ names: [name],
508
+ batch: reg.batch ?? 1,
509
+ concurrency: reg.concurrency,
510
+ resource: reg.resource,
511
+ });
512
+ } else shared.push(name);
513
+ }
514
+ if (shared.length) sources.push({ names: shared, batch: 1 });
515
+ return (this.scheduleCache = { sources, names: [...this.handlers.keys()] });
516
+ }
517
+
518
+ /**
519
+ * A source's own call ceiling for one poll, or undefined when only the
520
+ * worker-wide budget applies.
521
+ *
522
+ * Three independent ceilings, whichever binds first: the name's own concurrency
523
+ * less what it is already running; its resource's capacity less what is running
524
+ * *and* what earlier draws in this same poll already took (`taken`); and
525
+ * `claimBatch`. The last is a ceiling on *rows* per poll, so it converts at this
526
+ * source's batch size — and never below one call, or a `claimBatch` under some
527
+ * name's batch would stall that name outright.
528
+ */
529
+ private sourceCalls(src: ClaimSource, taken: Map<string, number>): number | undefined {
530
+ const rows = this.opts.claimBatch;
531
+ const byRows = rows == null ? undefined : Math.max(1, Math.floor(rows / src.batch));
532
+ const byName =
533
+ src.concurrency == null
534
+ ? undefined
535
+ : Math.max(0, src.concurrency - (this.callsInFlight.get(src.key!) ?? 0));
536
+ const byResource =
537
+ src.resource == null
538
+ ? undefined
539
+ : Math.max(
540
+ 0,
541
+ this.opts.resources![src.resource] -
542
+ (this.resourceCalls.get(src.resource) ?? 0) -
543
+ (taken.get(src.resource) ?? 0),
544
+ );
545
+ const limits = [byRows, byName, byResource].filter((n): n is number => n != null);
546
+ return limits.length ? Math.min(...limits) : undefined;
547
+ }
548
+
263
549
  private async loop(
264
550
  concurrency: number,
265
- batch: number,
266
551
  leaseMs: number,
267
552
  running: Set<Promise<void>>,
268
553
  ): Promise<void> {
269
554
  const pollMs = this.opts.pollIntervalMs ?? 500;
270
555
  const byteBudget = this.opts.maxInFlightBytes;
271
556
  while (!this.stopped) {
557
+ // `concurrency` counts calls, so the calls in flight *are* the running
558
+ // promises — a batch holding 256 tasks is one of them.
272
559
  const free = concurrency - running.size;
273
- // Two ceilings, either of which stops the claim: task count and resident
274
- // payload bytes. The byte arm is guarded on running.size because it must
275
- // never be the reason we race an empty set — Promise.race([]) is pending
276
- // forever, past even stop(). With nothing running, nothing is resident, so
277
- // the budget cannot be the thing holding us back anyway.
560
+ // Two ceilings, either of which stops the claim: calls in flight and
561
+ // resident payload bytes. The byte arm is guarded on running.size because
562
+ // it must never be the reason we race an empty set — Promise.race([]) is
563
+ // pending forever, past even stop(). With nothing running, nothing is
564
+ // resident, so the budget cannot be the thing holding us back anyway.
278
565
  const overBudget = byteBudget != null && this.inFlightBytes >= byteBudget;
279
566
  if (running.size > 0 && (free <= 0 || overBudget)) {
280
- // Wait for a slot rather than spinning. execute() never rejects, so
567
+ // Wait for a slot rather than spinning. runCall() never rejects, so
281
568
  // racing these is safe.
282
569
  await Promise.race([...running]);
283
570
  continue;
284
571
  }
285
- let claimed: Task[];
572
+ const { sources, names } = this.schedule();
573
+ if (!sources.length) {
574
+ await this.idle(pollMs);
575
+ continue;
576
+ }
577
+ // Round-robin the starting point. The draws are served in order, so without
578
+ // rotating it the first source would take every free slot and the rest
579
+ // would starve behind its backlog.
580
+ const cursor = this.claimCursor % sources.length;
581
+ const order = [...sources.slice(cursor), ...sources.slice(0, cursor)];
582
+ this.claimCursor = (cursor + 1) % sources.length;
583
+ let claimed: { src: ClaimSource; calls: [Registration | undefined, Task[]][] }[] | undefined;
286
584
  try {
287
- claimed = await this.store.claim({
288
- queues: this.queues,
585
+ claimed = await this.store.claimSession(
289
586
  // Only what this worker can run. Queues do not partition work by task
290
587
  // name, so another worker's tasks would otherwise be claimed here and
291
- // failed for want of a handler. Read each poll: handlers may be
292
- // registered after run() started.
293
- names: [...this.handlers.keys()],
294
- workerId: this.workerId,
295
- leaseMs,
296
- limit: Math.min(batch, free),
297
- });
588
+ // failed for want of a handler.
589
+ { queues: this.queues, workerId: this.workerId, leaseMs, names },
590
+ async (claim) => {
591
+ const drawn: { src: ClaimSource; calls: [Registration | undefined, Task[]][] }[] = [];
592
+ let left = free;
593
+ // Resource units this poll has already drawn. resourceCalls only
594
+ // moves when a call is dispatched, which happens after this whole
595
+ // plan returns — so without a local tally two sources sharing a
596
+ // resource would each see its full ceiling and together overshoot
597
+ // it. Same shape as `left`, one budget down.
598
+ const taken = new Map<string, number>();
599
+ for (const src of order) {
600
+ if (left <= 0) break;
601
+ const quota = Math.min(this.sourceCalls(src, taken) ?? left, left);
602
+ if (quota <= 0) continue;
603
+ const rows = await claim(src.names, src.batch * quota);
604
+ if (!rows.length) continue;
605
+ // deliveries() is what actually turns rows into handler calls, so
606
+ // spending the budget against its result is the only way the two
607
+ // cannot disagree. A source with nothing queued costs nothing,
608
+ // which is why the budget is spent here, draw by draw, rather than
609
+ // divided up before the claim.
610
+ const calls = this.deliveries(rows);
611
+ drawn.push({ src, calls });
612
+ left -= calls.length;
613
+ if (src.resource != null) {
614
+ taken.set(src.resource, (taken.get(src.resource) ?? 0) + calls.length);
615
+ }
616
+ }
617
+ return drawn;
618
+ },
619
+ );
298
620
  } catch (err) {
299
621
  // A claim can fail transiently (lock contention, a dropped connection).
300
622
  // Report it and keep polling — one bad poll must not end the worker.
@@ -302,20 +624,36 @@ export class Worker {
302
624
  await this.sleepOrStop(CLAIM_ERROR_BACKOFF_MS);
303
625
  continue;
304
626
  }
305
- if (claimed.length === 0) {
627
+ if (!claimed?.length) {
306
628
  await this.idle(pollMs);
307
629
  continue;
308
630
  }
309
- for (const task of claimed) {
310
- // Charged before the handler starts and refunded when it settles, so the
311
- // budget covers exactly the span the payload is pinned in memory.
312
- const bytes = byteBudget == null ? 0 : payloadBytes(task);
313
- this.inFlightBytes += bytes;
314
- const p = this.execute(task, leaseMs).finally(() => {
315
- this.inFlightBytes -= bytes;
316
- running.delete(p);
317
- });
318
- running.add(p);
631
+ for (const { src, calls } of claimed) {
632
+ for (const [reg, group] of calls) {
633
+ // Charged before the handler starts and refunded when it settles, so
634
+ // the budgets cover exactly the span the call holds its slot and its
635
+ // payloads stay pinned in memory.
636
+ const bytes =
637
+ byteBudget == null ? 0 : group.reduce((sum, t) => sum + payloadBytes(t), 0);
638
+ this.inFlightBytes += bytes;
639
+ const key = src.key;
640
+ const resource = src.resource;
641
+ if (key != null) this.callsInFlight.set(key, (this.callsInFlight.get(key) ?? 0) + 1);
642
+ if (resource != null) {
643
+ this.resourceCalls.set(resource, (this.resourceCalls.get(resource) ?? 0) + 1);
644
+ }
645
+ // An unregistered name has nothing to run, so it never starts a handler
646
+ // or a heartbeat; everything else is one call, batched or not.
647
+ const call =
648
+ reg == null ? this.failNoHandler(group[0], leaseMs) : this.runCall(reg, group, leaseMs);
649
+ const p = call.finally(() => {
650
+ this.inFlightBytes -= bytes;
651
+ release(this.callsInFlight, key);
652
+ release(this.resourceCalls, resource);
653
+ running.delete(p);
654
+ });
655
+ running.add(p);
656
+ }
319
657
  }
320
658
  }
321
659
  }
@@ -350,77 +688,85 @@ export class Worker {
350
688
  }
351
689
 
352
690
  /**
353
- * Run one task to completion. Never rejects: a task-level failure is reported
354
- * through onError and the loop moves on. (It used to reject into a promise
355
- * nobody awaited — an unhandled rejection that took the process down.)
691
+ * Record a claimed task this worker cannot run. Reachable only if a name is
692
+ * unregistered mid-run — the claim filters on the registered names — so it does
693
+ * not start a handler or a heartbeat for a task it will not run.
356
694
  */
357
- private async execute(task: Task, leaseMs: number): Promise<void> {
358
- const ctx = new TaskContext(this.store, task, this.workerId, leaseMs);
359
- const hb = this.startHeartbeat(ctx, leaseMs);
695
+ private async failNoHandler(task: Task, leaseMs: number): Promise<void> {
696
+ try {
697
+ await this.safeFail(
698
+ this.context(task, leaseMs),
699
+ errorEnvelope({
700
+ type: "NoHandler",
701
+ code: "no_handler",
702
+ message: `no handler registered for ${task.name}`,
703
+ retryable: false,
704
+ }),
705
+ false,
706
+ );
707
+ } catch (err) {
708
+ this.report(err, { phase: "execute", taskId: task.id });
709
+ }
710
+ }
711
+
712
+ /**
713
+ * Run one handler call to completion — one task, or a whole batch. Never
714
+ * rejects: a task-level failure is reported through onError and the loop moves
715
+ * on. (It used to reject into a promise nobody awaited — an unhandled rejection
716
+ * that took the process down.)
717
+ *
718
+ * One lifecycle for both delivery modes, because a single-task handler *is* the
719
+ * one-element case: the same heartbeat covers the call, the same classifier
720
+ * reads its error, and the same rule settles what is left. Only two things vary
721
+ * — how the handler is invoked, and where a leftover task's result comes from —
722
+ * so those are the only two branches below.
723
+ *
724
+ * The contract is single: **when the handler returns, every task it did not
725
+ * settle itself is settled by how the call ended.** Returning succeeds them,
726
+ * throwing fails them (retryably, or as the thrown TaskError says). That is
727
+ * what keeps the ordinary cases free of bookkeeping — a handler that just
728
+ * returns has finished 256 tasks — while still letting it pick individual
729
+ * tasks off with `item.succeed()` / `item.fail()` as it goes.
730
+ *
731
+ * For a batch, a returned map of task id -> result fills in results for the
732
+ * tasks left over; anything else returned is ignored, and unmentioned tasks
733
+ * succeed with no result (the common shape, where the handler's output went to
734
+ * a database rather than into the task row). For a single task the return value
735
+ * simply is the result.
736
+ */
737
+ private async runCall(reg: Registration, tasks: Task[], leaseMs: number): Promise<void> {
738
+ const ctxs = tasks.map((t) => this.context(t, leaseMs));
739
+ const batched = reg.batch != null;
740
+ const hb = this.startHeartbeat(ctxs, leaseMs);
360
741
  try {
361
- const handler = this.handlers.get(task.name);
362
- if (!handler) {
363
- await this.safeFail(
364
- task,
365
- errorEnvelope({
366
- type: "NoHandler",
367
- code: "no_handler",
368
- message: `no handler registered for ${task.name}`,
369
- retryable: false,
370
- }),
371
- false,
372
- );
373
- return;
374
- }
375
742
  let result: unknown;
376
743
  try {
377
- result = await this.attempt(handler, ctx, task);
744
+ result = await this.attempt(
745
+ () =>
746
+ batched
747
+ ? (reg.fn as BatchHandler)(ctxs)
748
+ : (reg.fn as Handler)(ctxs[0], tasks[0].payload),
749
+ ctxs,
750
+ );
378
751
  } catch (err) {
379
- if (err instanceof LostLease) return;
380
- if (err instanceof AttemptTimeout) {
381
- // Recorded as a retryable failure, so backoff / maxAttempts /
382
- // cancel-wins all apply exactly as for a thrown error.
383
- await this.safeFail(task, timeoutEnvelope(task.name, err.maxRunMs), true);
384
- return;
385
- }
386
- if (err instanceof TaskError) {
387
- await this.safeFail(task, err.envelope(), err.retryable);
388
- } else {
389
- await this.safeFail(task, exceptionEnvelope(err), true);
390
- }
752
+ const outcome = this.outcomeOf(err, tasks[0].name);
753
+ if (outcome) await this.settleEach(ctxs, (c) => this.safeFail(c, outcome[0], outcome[1]));
391
754
  return;
392
755
  }
393
- try {
394
- // complete (not succeed): finalizes as canceled if a cancel was
395
- // requested while the handler ran, else succeeded.
396
- await this.store.complete({ taskId: task.id, workerId: this.workerId, result });
397
- } catch (err) {
398
- if (err instanceof LostLease) {
399
- ctx.markLeaseLost();
400
- return;
401
- }
402
- if (err instanceof SerializationError) {
403
- // The handler succeeded but its return value can't cross the JSON
404
- // protocol (BigInt, non-finite number, circular). Deterministic, so
405
- // fail fast and permanently — the alternative is sitting `running`
406
- // until lease expiry redelivers a task that fails the same way every
407
- // attempt.
408
- await this.safeFail(
409
- task,
410
- errorEnvelope({
411
- type: "SerializationError",
412
- code: "unserializable_result",
413
- message: `handler result is not JSON-serializable: ${err.message}`,
414
- retryable: false,
415
- }),
416
- false,
417
- );
418
- return;
419
- }
420
- throw err;
421
- }
756
+ // A batch handler's return maps task id -> result; a single-task handler's
757
+ // return *is* the result. Anything else a batch returns is ignored, so its
758
+ // leftovers succeed with no result — hence the empty map rather than
759
+ // falling through to `result`.
760
+ const results = batched
761
+ ? result != null && typeof result === "object"
762
+ ? (result as Record<string, unknown>)
763
+ : {}
764
+ : null;
765
+ await this.settleEach(ctxs, (c) =>
766
+ this.succeedOne(c, results ? (results[c.taskId] ?? null) : result),
767
+ );
422
768
  } catch (err) {
423
- this.report(err, { phase: "execute", taskId: task.id });
769
+ this.report(err, { phase: "execute", taskId: tasks[0].id });
424
770
  } finally {
425
771
  hb.cancel();
426
772
  await hb.done;
@@ -428,40 +774,178 @@ export class Worker {
428
774
  }
429
775
 
430
776
  /**
431
- * Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
432
- * abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
433
- * a handler that keeps running can never write again — see
434
- * TaskContext.owned), then the still-pending promise is left to settle on
435
- * its own, its outcome discarded. The caller records the handler_timeout
436
- * failure; lease recovery is NOT involved, so redelivery is immediate.
777
+ * How an attempt that ended badly is recorded: [envelope, retryable], or null
778
+ * when there is nothing to record.
779
+ *
780
+ * One classifier for both delivery modes, so a handler error cannot mean
781
+ * different things depending on how its task happened to be delivered. That
782
+ * includes LostLease, which is not an outcome at all: it means a write through
783
+ * this context was already rejected, so recording anything more would be
784
+ * rejected too. Both modes then leave the task alone — the single-task one has
785
+ * nothing else to do, and a batch lets its remaining tasks fall to lease expiry
786
+ * and redelivery rather than stamping them with a failure the handler never
787
+ * reported.
788
+ */
789
+ private outcomeOf(err: unknown, name: string): [Record<string, unknown>, boolean] | null {
790
+ if (err instanceof LostLease) return null;
791
+ // Retryable, so backoff / maxAttempts / cancel-wins all apply exactly as for
792
+ // a thrown error.
793
+ if (err instanceof AttemptTimeout) return [timeoutEnvelope(name, err.maxRunMs), true];
794
+ // Everything else is what a handler could equally have passed to ctx.fail(),
795
+ // so it goes through the same normalizer — a TaskError keeps its own
796
+ // retryability, anything else is retryable. Non-Error throws reach
797
+ // exceptionEnvelope the same way, since only ctx.fail can supply a ready
798
+ // envelope and a thrown object is not one.
799
+ if (err !== null && typeof err === "object" && !(err instanceof Error)) {
800
+ return [exceptionEnvelope(err), true];
801
+ }
802
+ return asEnvelope(err as FailReason, true);
803
+ }
804
+
805
+ // Both leftover paths write through the store rather than through the context.
806
+ // `TaskContext.owned` short-circuits once a lease is known lost, which is there
807
+ // to stop a *zombie handler* writing after its attempt was abandoned — but
808
+ // these run after the handler is done, on the worker's own authority, exactly
809
+ // as execute()'s completion and failure arms do. Ownership is still enforced by
810
+ // each statement, so a task whose lease really was lost writes nothing either
811
+ // way.
812
+
813
+ /**
814
+ * Finalize one task the handler left for the worker to decide — the tail of
815
+ * both delivery modes.
816
+ *
817
+ * Includes the unserializable-result rule: the handler succeeded but its value
818
+ * cannot cross the JSON protocol (BigInt, non-finite number, circular), which
819
+ * is deterministic, so it fails permanently rather than being redelivered to
820
+ * fail the same way every attempt.
437
821
  */
438
- private async attempt(handler: Handler, ctx: TaskContext, task: Task): Promise<unknown> {
822
+ private async succeedOne(ctx: TaskContext, result: unknown): Promise<void> {
823
+ if (ctx.settled) return;
824
+ try {
825
+ // complete (not succeed): finalizes as canceled if a cancel was requested
826
+ // while the handler ran, else succeeded.
827
+ await this.store.complete({ taskId: ctx.taskId, workerId: this.workerId, result });
828
+ ctx.markSettled();
829
+ } catch (err) {
830
+ if (err instanceof LostLease) {
831
+ ctx.markLeaseLost();
832
+ return;
833
+ }
834
+ if (err instanceof SerializationError) {
835
+ await this.safeFail(
836
+ ctx,
837
+ errorEnvelope({
838
+ type: "SerializationError",
839
+ code: "unserializable_result",
840
+ message: `handler result is not JSON-serializable: ${err.message}`,
841
+ retryable: false,
842
+ }),
843
+ false,
844
+ );
845
+ return;
846
+ }
847
+ throw err;
848
+ }
849
+ }
850
+
851
+ /**
852
+ * Settle every task the handler did not settle itself, concurrently, reporting
853
+ * rather than throwing. Each task keeps its own attempt count and backoff —
854
+ * they are separate tasks that happened to be delivered together.
855
+ *
856
+ * allSettled, because one task's write failing must not abandon the rest of the
857
+ * batch mid-settlement — the others still hold leases and would sit `running`
858
+ * until expiry. Each outcome is reported against the task it belongs to;
859
+ * without that, an operator learns a settlement failed somewhere in a batch of
860
+ * 256.
861
+ */
862
+ private async settleEach(
863
+ ctxs: TaskContext[],
864
+ settle: (ctx: TaskContext) => Promise<void>,
865
+ ): Promise<void> {
866
+ const left = ctxs.filter((c) => !c.settled);
867
+ if (!left.length) return;
868
+ const outcomes = await Promise.allSettled(left.map(settle));
869
+ outcomes.forEach((outcome, i) => {
870
+ if (outcome.status === "rejected") {
871
+ this.report(outcome.reason, { phase: "execute", taskId: left[i].taskId });
872
+ }
873
+ });
874
+ }
875
+
876
+ /**
877
+ * Run one attempt — of one task or of a whole batch — bounded by maxRunMs when
878
+ * set.
879
+ *
880
+ * On timeout the attempt is abandoned: every context it covers is flagged
881
+ * lease-lost first (ctx.signal aborts, and a handler that keeps running can
882
+ * never write again, nor settle anything behind the worker's back — see
883
+ * TaskContext.owned), then the still-pending promise is left to settle on its
884
+ * own, its outcome discarded. The caller records the handler_timeout failure;
885
+ * lease recovery is NOT involved, so redelivery is immediate.
886
+ */
887
+ private async attempt(invoke: () => unknown, ctxs: TaskContext[]): Promise<unknown> {
439
888
  const maxRunMs = this.opts.maxRunMs;
440
- if (maxRunMs == null) return handler(ctx, task.payload);
889
+ if (maxRunMs == null) return invoke();
441
890
  // As a real promise: the race needs one (a handler may return a plain
442
891
  // value), and the timeout path .catch()es it.
443
- const run = (async () => handler(ctx, task.payload))();
892
+ const run = (async () => invoke())();
444
893
  let timer!: NodeJS.Timeout;
445
894
  const winner = await Promise.race([
446
895
  run,
447
896
  new Promise<typeof TIMED_OUT>((r) => (timer = setTimeout(() => r(TIMED_OUT), maxRunMs))),
448
897
  ]).finally(() => clearTimeout(timer));
449
898
  if (winner !== TIMED_OUT) return winner;
450
- ctx.markLeaseLost();
899
+ for (const ctx of ctxs) ctx.markLeaseLost();
451
900
  // The zombie may still reject later; that must not become an unhandled
452
901
  // rejection — its outcome was already decided to be handler_timeout.
453
902
  run.catch(() => {});
454
903
  throw new AttemptTimeout(maxRunMs);
455
904
  }
456
905
 
906
+ /**
907
+ * One statement per beat, however many tasks the call covers — a single-task
908
+ * handler is just the one-element case.
909
+ *
910
+ * Only tasks still in play are renewed. A task the handler already settled is
911
+ * terminal, and re-leasing it would be a write against a row nobody owns; that
912
+ * is also why an absence is only read as lease loss after re-checking
913
+ * `settled`, since the handler may have finalized the task while this beat was
914
+ * in flight, which takes the row out of `running` and out of the reply. A task
915
+ * genuinely missing lost its lease (another worker recovered it), so its
916
+ * context is flagged and the handler stops being able to write through it —
917
+ * only that one, never its neighbours.
918
+ */
457
919
  private startHeartbeat(
458
- ctx: TaskContext,
920
+ ctxs: TaskContext[],
459
921
  leaseMs: number,
460
922
  ): { cancel: () => void; done: Promise<void> } {
461
923
  let active = true;
462
924
  let wake: (() => void) | null = null;
463
925
  // lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
464
926
  const interval = this.opts.heartbeatIntervalMs ?? Math.max(50, Math.floor(leaseMs / 3));
927
+
928
+ // When this heartbeat last ran. A loop cannot report its own absence — a
929
+ // handler that blocks for its whole attempt never lets the timer fire at all
930
+ // — so the check lives outside the loop and cancel() runs it too.
931
+ let lastBeatAt = Date.now();
932
+ /** Report a heartbeat that has not run for more than two intervals: a whole
933
+ * beat missed, which at lease/3 means the next such block loses the lease
934
+ * outright. Fires while the lease still holds — after it expires the only
935
+ * evidence is a task that ran twice, in two workers' logs, with no error in
936
+ * either. */
937
+ const checkBeat = (): void => {
938
+ const now = Date.now();
939
+ const lateMs = now - lastBeatAt - interval;
940
+ if (lateMs > interval) {
941
+ this.report(new EventLoopBlocked(lateMs, interval, leaseMs), {
942
+ phase: "execute",
943
+ taskId: ctxs[0].taskId,
944
+ });
945
+ }
946
+ lastBeatAt = now;
947
+ };
948
+
465
949
  const done = (async () => {
466
950
  while (active) {
467
951
  // Cancellable sleep: cancel() resolves this immediately and clears the
@@ -475,12 +959,24 @@ export class Worker {
475
959
  });
476
960
  wake = null;
477
961
  if (!active) break;
962
+ checkBeat();
963
+ const live = ctxs.filter((c) => !c.settled && !c.lostLease);
964
+ if (!live.length) break;
478
965
  try {
479
- await ctx.heartbeat();
966
+ const renewed = await this.store.heartbeatBatch({
967
+ taskIds: live.map((c) => c.taskId),
968
+ workerId: this.workerId,
969
+ leaseMs,
970
+ });
971
+ for (const ctx of live) {
972
+ const cancelRequested = renewed.get(ctx.taskId);
973
+ // Cancellation rides along on the write we were making anyway, so
974
+ // ctx.canceled() stays free here too.
975
+ if (cancelRequested !== undefined) ctx.observeCancel(cancelRequested);
976
+ else if (!ctx.settled) ctx.markLeaseLost();
977
+ }
480
978
  } catch (err) {
481
- // ctx.heartbeat() already flagged the lease as lost for the handler.
482
- if (err instanceof LostLease) break;
483
- this.report(err, { phase: "execute", taskId: ctx.taskId });
979
+ this.report(err, { phase: "execute", taskId: live[0].taskId });
484
980
  }
485
981
  }
486
982
  })();
@@ -488,31 +984,29 @@ export class Worker {
488
984
  cancel: () => {
489
985
  active = false;
490
986
  if (wake) wake();
987
+ // The attempt is over, so this is the last chance to notice that the
988
+ // heartbeat never got one — the whole-attempt block, which is also the
989
+ // one where the lease is already gone.
990
+ checkBeat();
491
991
  },
492
992
  done,
493
993
  };
494
994
  }
495
995
 
496
996
  private async safeFail(
497
- task: Task,
997
+ ctx: TaskContext,
498
998
  envelope: Record<string, unknown>,
499
999
  retryable: boolean,
500
1000
  ): Promise<void> {
501
- const delayMs = retryable
502
- ? retryDelayMs(
503
- task.attempt,
504
- this.opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS,
505
- this.opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS,
506
- )
507
- : 0;
508
1001
  try {
509
1002
  await this.store.fail({
510
- taskId: task.id,
1003
+ taskId: ctx.taskId,
511
1004
  workerId: this.workerId,
512
1005
  error: envelope,
513
1006
  retryable,
514
- delayMs,
1007
+ delayMs: failDelayMs(ctx.attempt, retryable, this.backoffMs, this.backoffMaxMs),
515
1008
  });
1009
+ ctx.markSettled();
516
1010
  } catch (err) {
517
1011
  if (!(err instanceof LostLease)) throw err;
518
1012
  }