cairnq 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +3 -2
  2. package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
  3. package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
  4. package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
  5. package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
  6. package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
  7. package/dist/_protocol/sql/postgres/queue_depth.sql +22 -0
  8. package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
  9. package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
  10. package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
  11. package/dist/_protocol/sql/sqlite/queue_depth.sql +26 -0
  12. package/dist/backoff.d.ts +31 -0
  13. package/dist/backoff.js +40 -0
  14. package/dist/backpressure.d.ts +59 -0
  15. package/dist/backpressure.js +122 -0
  16. package/dist/client.d.ts +16 -3
  17. package/dist/client.js +19 -5
  18. package/dist/context.d.ts +48 -2
  19. package/dist/context.js +101 -10
  20. package/dist/errors.d.ts +37 -0
  21. package/dist/errors.js +60 -0
  22. package/dist/index.d.ts +7 -3
  23. package/dist/index.js +2 -1
  24. package/dist/store/base.d.ts +73 -0
  25. package/dist/store/base.js +124 -14
  26. package/dist/store/sqlite.d.ts +35 -0
  27. package/dist/store/sqlite.js +163 -3
  28. package/dist/worker.d.ts +238 -13
  29. package/dist/worker.js +512 -120
  30. package/package.json +2 -1
  31. package/src/backoff.ts +53 -0
  32. package/src/backpressure.ts +140 -0
  33. package/src/client.ts +33 -5
  34. package/src/context.ts +116 -9
  35. package/src/errors.ts +66 -0
  36. package/src/index.ts +7 -2
  37. package/src/store/base.ts +136 -13
  38. package/src/store/sqlite.ts +168 -2
  39. package/src/worker.ts +671 -132
package/src/worker.ts CHANGED
@@ -1,5 +1,14 @@
1
+ import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
2
+ import type { BackpressureOptions } from "./backpressure.js";
1
3
  import { TaskContext } from "./context.js";
2
- import { errorEnvelope, LostLease, SerializationError, TaskError } from "./errors.js";
4
+ import {
5
+ asEnvelope,
6
+ errorEnvelope,
7
+ exceptionEnvelope,
8
+ type FailReason,
9
+ LostLease,
10
+ SerializationError,
11
+ } from "./errors.js";
3
12
  import { newId } from "./ids.js";
4
13
  import { type Task } from "./models.js";
5
14
  import { SQLiteStore } from "./store/sqlite.js";
@@ -7,14 +16,90 @@ import { PostgresStore } from "./store/postgres.js";
7
16
  import type { TaskStore } from "./store/base.js";
8
17
  import { type TaskDef, taskName } from "./task.js";
9
18
 
19
+ // Re-exported: they moved to backoff.ts so TaskContext.fail could share them
20
+ // without context.ts importing the module that imports it. They stay importable
21
+ // from this module, which is where they used to live.
22
+ export { retryDelayMs, DEFAULT_RETRY_BACKOFF_MS, DEFAULT_RETRY_BACKOFF_MAX_MS } from "./backoff.js";
23
+
10
24
  export type Handler = (ctx: TaskContext, payload: any) => unknown | Promise<unknown>;
11
25
  /** Handler typed against a TaskDef<P, R>: payload is P, the return is R. */
12
26
  export type TypedHandler<P, R> = (ctx: TaskContext, payload: P) => R | Promise<R>;
13
27
 
28
+ /**
29
+ * A batch handler takes one argument: the list of contexts. There is no payload
30
+ * shortcut to pair with it — payloads are per task, so they are read off the
31
+ * items (`item.payload`), which is also what a handler must hold to settle one
32
+ * of them.
33
+ *
34
+ * Returning a map of task id -> result fills in results for the tasks the
35
+ * handler did not settle itself; anything else returned is ignored.
36
+ */
37
+ export type BatchHandler = (
38
+ items: TaskContext[],
39
+ ) => void | Record<string, unknown> | Promise<void | Record<string, unknown>>;
40
+
41
+ /** What `worker.task` recorded for one task name. */
42
+ interface Registration {
43
+ fn: Handler | BatchHandler;
44
+ /** Tasks per handler call, or undefined for one-at-a-time delivery. */
45
+ batch?: number;
46
+ /** Concurrent handler calls allowed for this name, or undefined for no limit
47
+ * beyond the worker's own. */
48
+ concurrency?: number;
49
+ /** Resource this name's calls draw from, or undefined to draw from nothing
50
+ * but the worker budget. Declared in `WorkerOptions.resources`. */
51
+ resource?: string;
52
+ }
53
+
54
+ /**
55
+ * One draw's worth of quota: a set of names and how many handler calls they may
56
+ * start. A name that limits itself — by `batch`, by its own `concurrency`, or by
57
+ * a `resource` it shares with other names — gets a source to itself, because its
58
+ * quota cannot be expressed in a draw shared with names that count differently.
59
+ * Everything else shares one, where a task is a call and the worker's own budget
60
+ * is the only ceiling.
61
+ */
62
+ interface ClaimSource {
63
+ /**
64
+ * Counts calls in flight, and set only when this source caps its own
65
+ * concurrency — nothing else reads the count, so nothing else pays for it.
66
+ * Such a source always holds exactly one name, so this is that name.
67
+ */
68
+ key?: string;
69
+ names: string[];
70
+ /** Tasks per call — 1 for the shared source. */
71
+ batch: number;
72
+ /** Calls allowed for this source, or undefined for the worker budget alone. */
73
+ concurrency?: number;
74
+ /**
75
+ * Resource this source draws from, or undefined. Unlike `concurrency`, the
76
+ * ceiling it names is shared with the other sources that declare it, which is
77
+ * what keeps two names off one scarce thing at the same time.
78
+ */
79
+ resource?: string;
80
+ }
81
+
82
+ /** What one poll's claim draws from, and the names the probe spans. */
83
+ interface Schedule {
84
+ sources: ClaimSource[];
85
+ names: string[];
86
+ }
87
+
14
88
  /** Where an error the worker recovered from came from. */
15
89
  export type ErrorPhase = "claim" | "execute";
16
90
 
17
- export interface WorkerOptions {
91
+ /**
92
+ * Backpressure is accepted here too, not only on CairnQ: a handler spawning
93
+ * children through TaskContext.submit is a producer, and in a worker process
94
+ * there is usually no CairnQ handle to have configured the store.
95
+ */
96
+ export interface WorkerOptions extends Partial<BackpressureOptions> {
97
+ /**
98
+ * Handler calls allowed to run at once. A batch call counts as one, however
99
+ * many tasks it carries — size it for how much work you want in flight, not
100
+ * for how many tasks that comes to. Per-name limits refine it; `maxInFlightBytes`
101
+ * bounds memory, which task counts never did.
102
+ */
18
103
  concurrency?: number;
19
104
  leaseMs?: number;
20
105
  heartbeatIntervalMs?: number;
@@ -33,6 +118,39 @@ export interface WorkerOptions {
33
118
  * a retryable `handler_timeout` failure. Unset disables the ceiling.
34
119
  */
35
120
  maxRunMs?: number;
121
+ /**
122
+ * Resident payload bytes allowed across running handlers, independent of
123
+ * their count.
124
+ *
125
+ * `concurrency` bounds tasks, not memory, so a worker sized for small payloads
126
+ * holds concurrency * largest-payload bytes the moment a batch of big ones
127
+ * arrives — for payloads that carry media inline, that is the difference
128
+ * between megabytes and gigabytes resident. Once the budget is spent the
129
+ * worker stops claiming until running handlers give it back.
130
+ *
131
+ * The bound is on tasks already executing: it is read between claims, never
132
+ * during one, and a claim commits to its rows before any size is known. One
133
+ * poll can therefore overshoot by up to `claimBatch` rows per registered name
134
+ * (or one whole `batch`, whichever is larger). Lower `claimBatch`, or the batch
135
+ * sizes, to tighten that. A single payload larger than the entire budget still
136
+ * runs — alone, rather than deadlocking the worker.
137
+ *
138
+ * Costs one JSON serialization per task to measure, so it is only computed
139
+ * when set. Unset disables the budget.
140
+ */
141
+ maxInFlightBytes?: number;
142
+ /**
143
+ * Call ceilings that several names can draw from, by name — `{ gpu: 1 }`.
144
+ * A handler joins one with `task(name, { resource: "gpu" }, fn)`.
145
+ *
146
+ * `concurrency` caps a name against itself, which cannot say what usually
147
+ * binds a worker doing heavy local work: several *different* handlers
148
+ * contending for one scarce thing — a GPU, an index that tolerates a single
149
+ * writer. The limit belongs to that thing rather than to any one name, so it
150
+ * is declared here, once, and at capacity 1 it is mutual exclusion across the
151
+ * names that join it.
152
+ */
153
+ resources?: Record<string, number>;
36
154
  /**
37
155
  * Called for errors the worker survived — a claim that threw, a store write
38
156
  * that failed while finalizing a task. Without it these are silent: the run
@@ -42,28 +160,9 @@ export interface WorkerOptions {
42
160
  onError?: (err: unknown, info: { phase: ErrorPhase; taskId?: string }) => void;
43
161
  }
44
162
 
45
- const DEFAULT_RETRY_BACKOFF_MS = 1_000;
46
- const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
47
163
  /** Wait after a failed claim, so a broken database is not polled in a tight loop. */
48
164
  const CLAIM_ERROR_BACKOFF_MS = 250;
49
165
 
50
- /** Exponential backoff for the next attempt of a task that just failed. */
51
- export function retryDelayMs(attempt: number, baseMs: number, maxMs: number): number {
52
- if (baseMs <= 0) return 0;
53
- const exponent = Math.max(0, attempt - 1);
54
- return Math.min(maxMs, baseMs * 2 ** exponent);
55
- }
56
-
57
- function exceptionEnvelope(err: unknown): Record<string, unknown> {
58
- const e = err as { name?: string; message?: string };
59
- return errorEnvelope({
60
- type: e?.name ?? "Error",
61
- code: "handler_error",
62
- message: String(e?.message ?? err),
63
- retryable: true,
64
- });
65
- }
66
-
67
166
  /** Internal: an attempt outran maxRunMs and was abandoned. */
68
167
  class AttemptTimeout extends Error {
69
168
  constructor(readonly maxRunMs: number) {
@@ -82,9 +181,69 @@ function timeoutEnvelope(name: string, maxRunMs: number): Record<string, unknown
82
181
 
83
182
  const TIMED_OUT = Symbol("cairnq.timedOut");
84
183
 
184
+ /**
185
+ * Resident size of a task's payload, for the maxInFlightBytes budget.
186
+ *
187
+ * Re-serializes because by this point the wire form is gone: `pg` parses a jsonb
188
+ * column with JSON.parse and discards the text, so on Postgres there is nothing
189
+ * cheaper to read. On SQLite the column does arrive as a string that rowToTask
190
+ * sees before parsing — capturing its length there would make this free, at the
191
+ * cost of carrying a non-protocol field on Task in both SDKs. Left for when the
192
+ * measurement shows up in a profile.
193
+ *
194
+ * What the budget is really after is the memory a payload pins while its handler
195
+ * runs, and its JSON length tracks that closely enough to size one by.
196
+ */
197
+ function payloadBytes(task: Task): number {
198
+ try {
199
+ return Buffer.byteLength(JSON.stringify(task.payload) ?? "");
200
+ } catch {
201
+ // Unmeasurable, and it came out of the store, so it is already resident:
202
+ // charging nothing under-counts, but failing the claim over an accounting
203
+ // detail would drop a task the worker can otherwise run.
204
+ return 0;
205
+ }
206
+ }
207
+
208
+ /**
209
+ * Give back one call's unit of a counted budget. Deleting at zero is what keeps
210
+ * the map to the keys actually in flight, so an idle worker holds no entries at
211
+ * all — and both budgets (a name's own concurrency, a resource's capacity)
212
+ * settle the same way, from one place.
213
+ */
214
+ function release(counts: Map<string, number>, key: string | undefined): void {
215
+ if (key == null) return;
216
+ const rest = (counts.get(key) ?? 1) - 1;
217
+ if (rest > 0) counts.set(key, rest);
218
+ else counts.delete(key);
219
+ }
220
+
85
221
  export class Worker {
86
- private readonly handlers = new Map<string, Handler>();
222
+ private readonly handlers = new Map<string, Registration>();
87
223
  private readonly workerId = newId("worker");
224
+ /** Payload bytes charged to running handlers — see maxInFlightBytes. */
225
+ private inFlightBytes = 0;
226
+ /** Calls in flight, for the names that cap their own concurrency. */
227
+ private readonly callsInFlight = new Map<string, number>();
228
+ /**
229
+ * Calls holding units of each declared resource. A resource is the same shape
230
+ * of budget as a name's own `concurrency` — a ceiling on calls — differing
231
+ * only in who draws from it: several names rather than one. That is what
232
+ * expresses "these handlers share one GPU" without inventing a queue per
233
+ * resource.
234
+ */
235
+ private readonly resourceCalls = new Map<string, number>();
236
+ /** Rotates which source is offered the free budget first — see loop(). */
237
+ private claimCursor = 0;
238
+ /** Invalidated by task(); see schedule(). */
239
+ private scheduleCache: Schedule | null = null;
240
+ /**
241
+ * Retry backoff, resolved once. Both settlement paths read these — the
242
+ * worker's own `safeFail` and the TaskContext it hands a handler — so
243
+ * resolving the defaults per call site is how the two drift apart.
244
+ */
245
+ private readonly backoffMs: number;
246
+ private readonly backoffMaxMs: number;
88
247
  private stopped = false;
89
248
  private stopWake!: () => void;
90
249
  // Resolved once by stop(); every sleep races against it. A stopped worker
@@ -102,6 +261,23 @@ export class Worker {
102
261
  if (opts.maxRunMs != null && opts.maxRunMs <= 0) {
103
262
  throw new Error(`maxRunMs must be > 0, got ${opts.maxRunMs}`);
104
263
  }
264
+ // 0 would make the budget permanently spent, so the worker would claim
265
+ // nothing and look hung. Rejected here, as the Python SDK does.
266
+ if (opts.maxInFlightBytes != null && opts.maxInFlightBytes <= 0) {
267
+ throw new Error(`maxInFlightBytes must be > 0, got ${opts.maxInFlightBytes}`);
268
+ }
269
+ for (const [resource, capacity] of Object.entries(opts.resources ?? {})) {
270
+ if (!Number.isInteger(capacity) || capacity < 1) {
271
+ throw new Error(
272
+ `resources[${JSON.stringify(resource)}] must be an integer >= 1, got ${capacity}`,
273
+ );
274
+ }
275
+ }
276
+ if (opts.maxQueueDepth != null) {
277
+ store.useBackpressure(opts as BackpressureOptions);
278
+ }
279
+ this.backoffMs = opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
280
+ this.backoffMaxMs = opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS;
105
281
  }
106
282
 
107
283
  static sqlite(
@@ -133,9 +309,61 @@ export class Worker {
133
309
  task(handler: Handler): this;
134
310
  task(name: string, handler: Handler): this;
135
311
  task<P, R>(def: TaskDef<P, R>, handler: TypedHandler<P, R>): this;
136
- task(arg: string | Handler | TaskDef, handler?: Handler): this {
312
+ /**
313
+ * Batch delivery: the handler takes one argument, a `TaskContext[]` of up to
314
+ * `batch` tasks, instead of `(ctx, payload)`. Use it when the work itself is
315
+ * batched — one embedding call over 256 texts rather than 256 calls — and size
316
+ * it by what the downstream API wants, not by the queue.
317
+ *
318
+ * `concurrency` caps the calls this name may run at once, under the worker's
319
+ * own. Use it to keep one expensive name from taking the whole worker.
320
+ *
321
+ * `resource` draws each call from a ceiling declared in
322
+ * `WorkerOptions.resources` and shared with every other name that names it —
323
+ * at capacity 1, mutual exclusion across those names.
324
+ */
325
+ task(
326
+ name: string | TaskDef,
327
+ opts: { batch: number; concurrency?: number; resource?: string },
328
+ handler: BatchHandler,
329
+ ): this;
330
+ /** Per-name concurrency or a shared resource, without batching: the handler
331
+ * still takes `(ctx, payload)`. */
332
+ task(
333
+ name: string | TaskDef,
334
+ opts: { concurrency?: number; resource?: string },
335
+ handler: Handler,
336
+ ): this;
337
+ task(
338
+ arg: string | Handler | TaskDef,
339
+ second?: Handler | { batch?: number; concurrency?: number; resource?: string },
340
+ third?: Handler | BatchHandler,
341
+ ): this {
342
+ // Option form: (name | def, { batch?, concurrency?, resource? }, handler).
343
+ // Peel the options off and fall through, so name resolution and registration
344
+ // stay single-sited.
345
+ let batch: number | undefined;
346
+ let concurrency: number | undefined;
347
+ let resource: string | undefined;
348
+ let handler = second as Handler | BatchHandler | undefined;
349
+ if (second != null && typeof second === "object") {
350
+ if (second.batch != null) {
351
+ if (!Number.isInteger(second.batch) || second.batch < 1) {
352
+ throw new Error(`batch must be an integer >= 1, got ${second.batch}`);
353
+ }
354
+ batch = second.batch;
355
+ }
356
+ if (second.concurrency != null) {
357
+ if (!Number.isInteger(second.concurrency) || second.concurrency < 1) {
358
+ throw new Error(`concurrency must be an integer >= 1, got ${second.concurrency}`);
359
+ }
360
+ concurrency = second.concurrency;
361
+ }
362
+ resource = second.resource;
363
+ handler = third;
364
+ }
137
365
  let name: string;
138
- let fn: Handler;
366
+ let fn: Handler | BatchHandler;
139
367
  if (typeof arg === "function") {
140
368
  // Bare form: worker.task(fn) — registered under the function's name.
141
369
  // Strip the "bound " prefix .bind() stamps on it: otherwise a bound
@@ -153,7 +381,18 @@ export class Worker {
153
381
  name = taskName(arg);
154
382
  fn = handler!;
155
383
  }
156
- this.handlers.set(name, fn);
384
+ // Loudly, at registration: an undeclared resource would otherwise read as
385
+ // an unbounded one, so a typo would silently remove the ceiling the caller
386
+ // asked for — the failure this option exists to prevent.
387
+ if (resource != null && this.opts.resources?.[resource] == null) {
388
+ const known = Object.keys(this.opts.resources ?? {}).sort().join(", ") || "none";
389
+ throw new Error(
390
+ `task ${JSON.stringify(name)} declares resource ${JSON.stringify(resource)}, ` +
391
+ `which is not in WorkerOptions.resources; declared: ${known}`,
392
+ );
393
+ }
394
+ this.handlers.set(name, { fn, batch, concurrency, resource });
395
+ this.scheduleCache = null;
157
396
  return this;
158
397
  }
159
398
 
@@ -187,11 +426,12 @@ export class Worker {
187
426
  // beyond even stop()'s reach.
188
427
  const concurrency = Math.max(1, opts.concurrency ?? this.opts.concurrency ?? 1);
189
428
  const leaseMs = this.opts.leaseMs ?? 30_000;
190
- const batch = this.opts.claimBatch ?? concurrency;
191
429
  await this.store.connect();
430
+ // The calls in flight — `concurrency` counts these, so the set's size is the
431
+ // budget. How many tasks they carry is `maxInFlightBytes`'s business.
192
432
  const running = new Set<Promise<void>>();
193
433
  try {
194
- await this.loop(concurrency, batch, leaseMs, running);
434
+ await this.loop(concurrency, leaseMs, running);
195
435
  } finally {
196
436
  // Whatever ends the loop — stop(), or something unexpected out of the body
197
437
  // — nothing this worker started may outlive run(). serve() closes the store
@@ -201,34 +441,181 @@ export class Worker {
201
441
  }
202
442
  }
203
443
 
444
+ /**
445
+ * Split one claim into handler calls, each with the registration to run it.
446
+ *
447
+ * A claim is filtered by queue and by the names this worker handles, so it
448
+ * comes back mixed; batch size is per name (one embedding call wants 256
449
+ * texts, one Docling parse wants exactly 1). So group by name, then chunk each
450
+ * group by that name's size. Names registered without `batch` come back as
451
+ * one-task calls, as do names not registered at all — reachable only if a
452
+ * handler is unregistered mid-run, and dispatched to failNoHandler().
453
+ *
454
+ * The registration rides along because this is where it was resolved; looking
455
+ * it up again at the call site would put "is this name batched" in two places.
456
+ */
457
+ private deliveries(claimed: Task[]): [Registration | undefined, Task[]][] {
458
+ const byName = new Map<string, Task[]>();
459
+ for (const task of claimed) {
460
+ const group = byName.get(task.name);
461
+ if (group) group.push(task);
462
+ else byName.set(task.name, [task]);
463
+ }
464
+ const out: [Registration | undefined, Task[]][] = [];
465
+ for (const [name, group] of byName) {
466
+ const reg = this.handlers.get(name);
467
+ const size = reg?.batch ?? 1;
468
+ for (let i = 0; i < group.length; i += size) out.push([reg, group.slice(i, i + size)]);
469
+ }
470
+ return out;
471
+ }
472
+
473
+ private context(task: Task, leaseMs: number): TaskContext {
474
+ return new TaskContext(this.store, task, this.workerId, leaseMs, {
475
+ retryBackoffMs: this.backoffMs,
476
+ retryBackoffMaxMs: this.backoffMaxMs,
477
+ });
478
+ }
479
+
480
+ /**
481
+ * How this poll's claim is split into per-name quotas, plus the union of names
482
+ * the probe spans.
483
+ *
484
+ * Cached and invalidated by `task()`, rather than rebuilt each poll: handlers
485
+ * may be registered after run() started, but only there, and this otherwise
486
+ * allocates a source per name on every tick for a worker's whole lifetime.
487
+ *
488
+ * A name that limits itself — by `batch`, by its own `concurrency`, or by a
489
+ * `resource` — needs a quota the shared draw cannot express, so it gets a
490
+ * source of its own; every other name shares one, where a task is a call.
491
+ *
492
+ * A resource is deliberately *not* one source spanning its names: `batch` is
493
+ * per name, and a single source carries one batch size, so two members that
494
+ * batch differently could not share a draw. Keeping a source per name and
495
+ * letting several of them draw down one shared ceiling composes with batching
496
+ * instead of excluding it.
497
+ */
498
+ private schedule(): Schedule {
499
+ if (this.scheduleCache) return this.scheduleCache;
500
+ const sources: ClaimSource[] = [];
501
+ const shared: string[] = [];
502
+ for (const [name, reg] of this.handlers) {
503
+ if (reg.batch != null || reg.concurrency != null || reg.resource != null) {
504
+ sources.push({
505
+ key: reg.concurrency == null ? undefined : name,
506
+ names: [name],
507
+ batch: reg.batch ?? 1,
508
+ concurrency: reg.concurrency,
509
+ resource: reg.resource,
510
+ });
511
+ } else shared.push(name);
512
+ }
513
+ if (shared.length) sources.push({ names: shared, batch: 1 });
514
+ return (this.scheduleCache = { sources, names: [...this.handlers.keys()] });
515
+ }
516
+
517
+ /**
518
+ * A source's own call ceiling for one poll, or undefined when only the
519
+ * worker-wide budget applies.
520
+ *
521
+ * Three independent ceilings, whichever binds first: the name's own concurrency
522
+ * less what it is already running; its resource's capacity less what is running
523
+ * *and* what earlier draws in this same poll already took (`taken`); and
524
+ * `claimBatch`. The last is a ceiling on *rows* per poll, so it converts at this
525
+ * source's batch size — and never below one call, or a `claimBatch` under some
526
+ * name's batch would stall that name outright.
527
+ */
528
+ private sourceCalls(src: ClaimSource, taken: Map<string, number>): number | undefined {
529
+ const rows = this.opts.claimBatch;
530
+ const byRows = rows == null ? undefined : Math.max(1, Math.floor(rows / src.batch));
531
+ const byName =
532
+ src.concurrency == null
533
+ ? undefined
534
+ : Math.max(0, src.concurrency - (this.callsInFlight.get(src.key!) ?? 0));
535
+ const byResource =
536
+ src.resource == null
537
+ ? undefined
538
+ : Math.max(
539
+ 0,
540
+ this.opts.resources![src.resource] -
541
+ (this.resourceCalls.get(src.resource) ?? 0) -
542
+ (taken.get(src.resource) ?? 0),
543
+ );
544
+ const limits = [byRows, byName, byResource].filter((n): n is number => n != null);
545
+ return limits.length ? Math.min(...limits) : undefined;
546
+ }
547
+
204
548
  private async loop(
205
549
  concurrency: number,
206
- batch: number,
207
550
  leaseMs: number,
208
551
  running: Set<Promise<void>>,
209
552
  ): Promise<void> {
210
553
  const pollMs = this.opts.pollIntervalMs ?? 500;
554
+ const byteBudget = this.opts.maxInFlightBytes;
211
555
  while (!this.stopped) {
556
+ // `concurrency` counts calls, so the calls in flight *are* the running
557
+ // promises — a batch holding 256 tasks is one of them.
212
558
  const free = concurrency - running.size;
213
- if (free <= 0) {
214
- // Wait for a slot rather than spinning. execute() never rejects, so
559
+ // Two ceilings, either of which stops the claim: calls in flight and
560
+ // resident payload bytes. The byte arm is guarded on running.size because
561
+ // it must never be the reason we race an empty set — Promise.race([]) is
562
+ // pending forever, past even stop(). With nothing running, nothing is
563
+ // resident, so the budget cannot be the thing holding us back anyway.
564
+ const overBudget = byteBudget != null && this.inFlightBytes >= byteBudget;
565
+ if (running.size > 0 && (free <= 0 || overBudget)) {
566
+ // Wait for a slot rather than spinning. runCall() never rejects, so
215
567
  // racing these is safe.
216
568
  await Promise.race([...running]);
217
569
  continue;
218
570
  }
219
- let claimed: Task[];
571
+ const { sources, names } = this.schedule();
572
+ if (!sources.length) {
573
+ await this.idle(pollMs);
574
+ continue;
575
+ }
576
+ // Round-robin the starting point. The draws are served in order, so without
577
+ // rotating it the first source would take every free slot and the rest
578
+ // would starve behind its backlog.
579
+ const cursor = this.claimCursor % sources.length;
580
+ const order = [...sources.slice(cursor), ...sources.slice(0, cursor)];
581
+ this.claimCursor = (cursor + 1) % sources.length;
582
+ let claimed: { src: ClaimSource; calls: [Registration | undefined, Task[]][] }[] | undefined;
220
583
  try {
221
- claimed = await this.store.claim({
222
- queues: this.queues,
584
+ claimed = await this.store.claimSession(
223
585
  // Only what this worker can run. Queues do not partition work by task
224
586
  // name, so another worker's tasks would otherwise be claimed here and
225
- // failed for want of a handler. Read each poll: handlers may be
226
- // registered after run() started.
227
- names: [...this.handlers.keys()],
228
- workerId: this.workerId,
229
- leaseMs,
230
- limit: Math.min(batch, free),
231
- });
587
+ // failed for want of a handler.
588
+ { queues: this.queues, workerId: this.workerId, leaseMs, names },
589
+ async (claim) => {
590
+ const drawn: { src: ClaimSource; calls: [Registration | undefined, Task[]][] }[] = [];
591
+ let left = free;
592
+ // Resource units this poll has already drawn. resourceCalls only
593
+ // moves when a call is dispatched, which happens after this whole
594
+ // plan returns — so without a local tally two sources sharing a
595
+ // resource would each see its full ceiling and together overshoot
596
+ // it. Same shape as `left`, one budget down.
597
+ const taken = new Map<string, number>();
598
+ for (const src of order) {
599
+ if (left <= 0) break;
600
+ const quota = Math.min(this.sourceCalls(src, taken) ?? left, left);
601
+ if (quota <= 0) continue;
602
+ const rows = await claim(src.names, src.batch * quota);
603
+ if (!rows.length) continue;
604
+ // deliveries() is what actually turns rows into handler calls, so
605
+ // spending the budget against its result is the only way the two
606
+ // cannot disagree. A source with nothing queued costs nothing,
607
+ // which is why the budget is spent here, draw by draw, rather than
608
+ // divided up before the claim.
609
+ const calls = this.deliveries(rows);
610
+ drawn.push({ src, calls });
611
+ left -= calls.length;
612
+ if (src.resource != null) {
613
+ taken.set(src.resource, (taken.get(src.resource) ?? 0) + calls.length);
614
+ }
615
+ }
616
+ return drawn;
617
+ },
618
+ );
232
619
  } catch (err) {
233
620
  // A claim can fail transiently (lock contention, a dropped connection).
234
621
  // Report it and keep polling — one bad poll must not end the worker.
@@ -236,13 +623,36 @@ export class Worker {
236
623
  await this.sleepOrStop(CLAIM_ERROR_BACKOFF_MS);
237
624
  continue;
238
625
  }
239
- if (claimed.length === 0) {
626
+ if (!claimed?.length) {
240
627
  await this.idle(pollMs);
241
628
  continue;
242
629
  }
243
- for (const task of claimed) {
244
- const p = this.execute(task, leaseMs).finally(() => running.delete(p));
245
- running.add(p);
630
+ for (const { src, calls } of claimed) {
631
+ for (const [reg, group] of calls) {
632
+ // Charged before the handler starts and refunded when it settles, so
633
+ // the budgets cover exactly the span the call holds its slot and its
634
+ // payloads stay pinned in memory.
635
+ const bytes =
636
+ byteBudget == null ? 0 : group.reduce((sum, t) => sum + payloadBytes(t), 0);
637
+ this.inFlightBytes += bytes;
638
+ const key = src.key;
639
+ const resource = src.resource;
640
+ if (key != null) this.callsInFlight.set(key, (this.callsInFlight.get(key) ?? 0) + 1);
641
+ if (resource != null) {
642
+ this.resourceCalls.set(resource, (this.resourceCalls.get(resource) ?? 0) + 1);
643
+ }
644
+ // An unregistered name has nothing to run, so it never starts a handler
645
+ // or a heartbeat; everything else is one call, batched or not.
646
+ const call =
647
+ reg == null ? this.failNoHandler(group[0], leaseMs) : this.runCall(reg, group, leaseMs);
648
+ const p = call.finally(() => {
649
+ this.inFlightBytes -= bytes;
650
+ release(this.callsInFlight, key);
651
+ release(this.resourceCalls, resource);
652
+ running.delete(p);
653
+ });
654
+ running.add(p);
655
+ }
246
656
  }
247
657
  }
248
658
  }
@@ -277,77 +687,85 @@ export class Worker {
277
687
  }
278
688
 
279
689
  /**
280
- * Run one task to completion. Never rejects: a task-level failure is reported
281
- * through onError and the loop moves on. (It used to reject into a promise
282
- * nobody awaited — an unhandled rejection that took the process down.)
690
+ * Record a claimed task this worker cannot run. Reachable only if a name is
691
+ * unregistered mid-run — the claim filters on the registered names — so it does
692
+ * not start a handler or a heartbeat for a task it will not run.
283
693
  */
284
- private async execute(task: Task, leaseMs: number): Promise<void> {
285
- const ctx = new TaskContext(this.store, task, this.workerId, leaseMs);
286
- const hb = this.startHeartbeat(ctx, leaseMs);
694
+ private async failNoHandler(task: Task, leaseMs: number): Promise<void> {
695
+ try {
696
+ await this.safeFail(
697
+ this.context(task, leaseMs),
698
+ errorEnvelope({
699
+ type: "NoHandler",
700
+ code: "no_handler",
701
+ message: `no handler registered for ${task.name}`,
702
+ retryable: false,
703
+ }),
704
+ false,
705
+ );
706
+ } catch (err) {
707
+ this.report(err, { phase: "execute", taskId: task.id });
708
+ }
709
+ }
710
+
711
+ /**
712
+ * Run one handler call to completion — one task, or a whole batch. Never
713
+ * rejects: a task-level failure is reported through onError and the loop moves
714
+ * on. (It used to reject into a promise nobody awaited — an unhandled rejection
715
+ * that took the process down.)
716
+ *
717
+ * One lifecycle for both delivery modes, because a single-task handler *is* the
718
+ * one-element case: the same heartbeat covers the call, the same classifier
719
+ * reads its error, and the same rule settles what is left. Only two things vary
720
+ * — how the handler is invoked, and where a leftover task's result comes from —
721
+ * so those are the only two branches below.
722
+ *
723
+ * The contract is single: **when the handler returns, every task it did not
724
+ * settle itself is settled by how the call ended.** Returning succeeds them,
725
+ * throwing fails them (retryably, or as the thrown TaskError says). That is
726
+ * what keeps the ordinary cases free of bookkeeping — a handler that just
727
+ * returns has finished 256 tasks — while still letting it pick individual
728
+ * tasks off with `item.succeed()` / `item.fail()` as it goes.
729
+ *
730
+ * For a batch, a returned map of task id -> result fills in results for the
731
+ * tasks left over; anything else returned is ignored, and unmentioned tasks
732
+ * succeed with no result (the common shape, where the handler's output went to
733
+ * a database rather than into the task row). For a single task the return value
734
+ * simply is the result.
735
+ */
736
+ private async runCall(reg: Registration, tasks: Task[], leaseMs: number): Promise<void> {
737
+ const ctxs = tasks.map((t) => this.context(t, leaseMs));
738
+ const batched = reg.batch != null;
739
+ const hb = this.startHeartbeat(ctxs, leaseMs);
287
740
  try {
288
- const handler = this.handlers.get(task.name);
289
- if (!handler) {
290
- await this.safeFail(
291
- task,
292
- errorEnvelope({
293
- type: "NoHandler",
294
- code: "no_handler",
295
- message: `no handler registered for ${task.name}`,
296
- retryable: false,
297
- }),
298
- false,
299
- );
300
- return;
301
- }
302
741
  let result: unknown;
303
742
  try {
304
- result = await this.attempt(handler, ctx, task);
743
+ result = await this.attempt(
744
+ () =>
745
+ batched
746
+ ? (reg.fn as BatchHandler)(ctxs)
747
+ : (reg.fn as Handler)(ctxs[0], tasks[0].payload),
748
+ ctxs,
749
+ );
305
750
  } catch (err) {
306
- if (err instanceof LostLease) return;
307
- if (err instanceof AttemptTimeout) {
308
- // Recorded as a retryable failure, so backoff / maxAttempts /
309
- // cancel-wins all apply exactly as for a thrown error.
310
- await this.safeFail(task, timeoutEnvelope(task.name, err.maxRunMs), true);
311
- return;
312
- }
313
- if (err instanceof TaskError) {
314
- await this.safeFail(task, err.envelope(), err.retryable);
315
- } else {
316
- await this.safeFail(task, exceptionEnvelope(err), true);
317
- }
751
+ const outcome = this.outcomeOf(err, tasks[0].name);
752
+ if (outcome) await this.settleEach(ctxs, (c) => this.safeFail(c, outcome[0], outcome[1]));
318
753
  return;
319
754
  }
320
- try {
321
- // complete (not succeed): finalizes as canceled if a cancel was
322
- // requested while the handler ran, else succeeded.
323
- await this.store.complete({ taskId: task.id, workerId: this.workerId, result });
324
- } catch (err) {
325
- if (err instanceof LostLease) {
326
- ctx.markLeaseLost();
327
- return;
328
- }
329
- if (err instanceof SerializationError) {
330
- // The handler succeeded but its return value can't cross the JSON
331
- // protocol (BigInt, non-finite number, circular). Deterministic, so
332
- // fail fast and permanently — the alternative is sitting `running`
333
- // until lease expiry redelivers a task that fails the same way every
334
- // attempt.
335
- await this.safeFail(
336
- task,
337
- errorEnvelope({
338
- type: "SerializationError",
339
- code: "unserializable_result",
340
- message: `handler result is not JSON-serializable: ${err.message}`,
341
- retryable: false,
342
- }),
343
- false,
344
- );
345
- return;
346
- }
347
- throw err;
348
- }
755
+ // A batch handler's return maps task id -> result; a single-task handler's
756
+ // return *is* the result. Anything else a batch returns is ignored, so its
757
+ // leftovers succeed with no result — hence the empty map rather than
758
+ // falling through to `result`.
759
+ const results = batched
760
+ ? result != null && typeof result === "object"
761
+ ? (result as Record<string, unknown>)
762
+ : {}
763
+ : null;
764
+ await this.settleEach(ctxs, (c) =>
765
+ this.succeedOne(c, results ? (results[c.taskId] ?? null) : result),
766
+ );
349
767
  } catch (err) {
350
- this.report(err, { phase: "execute", taskId: task.id });
768
+ this.report(err, { phase: "execute", taskId: tasks[0].id });
351
769
  } finally {
352
770
  hb.cancel();
353
771
  await hb.done;
@@ -355,34 +773,150 @@ export class Worker {
355
773
  }
356
774
 
357
775
  /**
358
- * Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
359
- * abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
360
- * a handler that keeps running can never write again — see
361
- * TaskContext.owned), then the still-pending promise is left to settle on
362
- * its own, its outcome discarded. The caller records the handler_timeout
363
- * failure; lease recovery is NOT involved, so redelivery is immediate.
776
+ * How an attempt that ended badly is recorded: [envelope, retryable], or null
777
+ * when there is nothing to record.
778
+ *
779
+ * One classifier for both delivery modes, so a handler error cannot mean
780
+ * different things depending on how its task happened to be delivered. That
781
+ * includes LostLease, which is not an outcome at all: it means a write through
782
+ * this context was already rejected, so recording anything more would be
783
+ * rejected too. Both modes then leave the task alone — the single-task one has
784
+ * nothing else to do, and a batch lets its remaining tasks fall to lease expiry
785
+ * and redelivery rather than stamping them with a failure the handler never
786
+ * reported.
364
787
  */
365
- private async attempt(handler: Handler, ctx: TaskContext, task: Task): Promise<unknown> {
788
+ private outcomeOf(err: unknown, name: string): [Record<string, unknown>, boolean] | null {
789
+ if (err instanceof LostLease) return null;
790
+ // Retryable, so backoff / maxAttempts / cancel-wins all apply exactly as for
791
+ // a thrown error.
792
+ if (err instanceof AttemptTimeout) return [timeoutEnvelope(name, err.maxRunMs), true];
793
+ // Everything else is what a handler could equally have passed to ctx.fail(),
794
+ // so it goes through the same normalizer — a TaskError keeps its own
795
+ // retryability, anything else is retryable. Non-Error throws reach
796
+ // exceptionEnvelope the same way, since only ctx.fail can supply a ready
797
+ // envelope and a thrown object is not one.
798
+ if (err !== null && typeof err === "object" && !(err instanceof Error)) {
799
+ return [exceptionEnvelope(err), true];
800
+ }
801
+ return asEnvelope(err as FailReason, true);
802
+ }
803
+
804
+ // Both leftover paths write through the store rather than through the context.
805
+ // `TaskContext.owned` short-circuits once a lease is known lost, which is there
806
+ // to stop a *zombie handler* writing after its attempt was abandoned — but
807
+ // these run after the handler is done, on the worker's own authority, exactly
808
+ // as execute()'s completion and failure arms do. Ownership is still enforced by
809
+ // each statement, so a task whose lease really was lost writes nothing either
810
+ // way.
811
+
812
+ /**
813
+ * Finalize one task the handler left for the worker to decide — the tail of
814
+ * both delivery modes.
815
+ *
816
+ * Includes the unserializable-result rule: the handler succeeded but its value
817
+ * cannot cross the JSON protocol (BigInt, non-finite number, circular), which
818
+ * is deterministic, so it fails permanently rather than being redelivered to
819
+ * fail the same way every attempt.
820
+ */
821
+ private async succeedOne(ctx: TaskContext, result: unknown): Promise<void> {
822
+ if (ctx.settled) return;
823
+ try {
824
+ // complete (not succeed): finalizes as canceled if a cancel was requested
825
+ // while the handler ran, else succeeded.
826
+ await this.store.complete({ taskId: ctx.taskId, workerId: this.workerId, result });
827
+ ctx.markSettled();
828
+ } catch (err) {
829
+ if (err instanceof LostLease) {
830
+ ctx.markLeaseLost();
831
+ return;
832
+ }
833
+ if (err instanceof SerializationError) {
834
+ await this.safeFail(
835
+ ctx,
836
+ errorEnvelope({
837
+ type: "SerializationError",
838
+ code: "unserializable_result",
839
+ message: `handler result is not JSON-serializable: ${err.message}`,
840
+ retryable: false,
841
+ }),
842
+ false,
843
+ );
844
+ return;
845
+ }
846
+ throw err;
847
+ }
848
+ }
849
+
850
+ /**
851
+ * Settle every task the handler did not settle itself, concurrently, reporting
852
+ * rather than throwing. Each task keeps its own attempt count and backoff —
853
+ * they are separate tasks that happened to be delivered together.
854
+ *
855
+ * allSettled, because one task's write failing must not abandon the rest of the
856
+ * batch mid-settlement — the others still hold leases and would sit `running`
857
+ * until expiry. Each outcome is reported against the task it belongs to;
858
+ * without that, an operator learns a settlement failed somewhere in a batch of
859
+ * 256.
860
+ */
861
+ private async settleEach(
862
+ ctxs: TaskContext[],
863
+ settle: (ctx: TaskContext) => Promise<void>,
864
+ ): Promise<void> {
865
+ const left = ctxs.filter((c) => !c.settled);
866
+ if (!left.length) return;
867
+ const outcomes = await Promise.allSettled(left.map(settle));
868
+ outcomes.forEach((outcome, i) => {
869
+ if (outcome.status === "rejected") {
870
+ this.report(outcome.reason, { phase: "execute", taskId: left[i].taskId });
871
+ }
872
+ });
873
+ }
874
+
875
+ /**
876
+ * Run one attempt — of one task or of a whole batch — bounded by maxRunMs when
877
+ * set.
878
+ *
879
+ * On timeout the attempt is abandoned: every context it covers is flagged
880
+ * lease-lost first (ctx.signal aborts, and a handler that keeps running can
881
+ * never write again, nor settle anything behind the worker's back — see
882
+ * TaskContext.owned), then the still-pending promise is left to settle on its
883
+ * own, its outcome discarded. The caller records the handler_timeout failure;
884
+ * lease recovery is NOT involved, so redelivery is immediate.
885
+ */
886
+ private async attempt(invoke: () => unknown, ctxs: TaskContext[]): Promise<unknown> {
366
887
  const maxRunMs = this.opts.maxRunMs;
367
- if (maxRunMs == null) return handler(ctx, task.payload);
888
+ if (maxRunMs == null) return invoke();
368
889
  // As a real promise: the race needs one (a handler may return a plain
369
890
  // value), and the timeout path .catch()es it.
370
- const run = (async () => handler(ctx, task.payload))();
891
+ const run = (async () => invoke())();
371
892
  let timer!: NodeJS.Timeout;
372
893
  const winner = await Promise.race([
373
894
  run,
374
895
  new Promise<typeof TIMED_OUT>((r) => (timer = setTimeout(() => r(TIMED_OUT), maxRunMs))),
375
896
  ]).finally(() => clearTimeout(timer));
376
897
  if (winner !== TIMED_OUT) return winner;
377
- ctx.markLeaseLost();
898
+ for (const ctx of ctxs) ctx.markLeaseLost();
378
899
  // The zombie may still reject later; that must not become an unhandled
379
900
  // rejection — its outcome was already decided to be handler_timeout.
380
901
  run.catch(() => {});
381
902
  throw new AttemptTimeout(maxRunMs);
382
903
  }
383
904
 
905
+ /**
906
+ * One statement per beat, however many tasks the call covers — a single-task
907
+ * handler is just the one-element case.
908
+ *
909
+ * Only tasks still in play are renewed. A task the handler already settled is
910
+ * terminal, and re-leasing it would be a write against a row nobody owns; that
911
+ * is also why an absence is only read as lease loss after re-checking
912
+ * `settled`, since the handler may have finalized the task while this beat was
913
+ * in flight, which takes the row out of `running` and out of the reply. A task
914
+ * genuinely missing lost its lease (another worker recovered it), so its
915
+ * context is flagged and the handler stops being able to write through it —
916
+ * only that one, never its neighbours.
917
+ */
384
918
  private startHeartbeat(
385
- ctx: TaskContext,
919
+ ctxs: TaskContext[],
386
920
  leaseMs: number,
387
921
  ): { cancel: () => void; done: Promise<void> } {
388
922
  let active = true;
@@ -402,12 +936,23 @@ export class Worker {
402
936
  });
403
937
  wake = null;
404
938
  if (!active) break;
939
+ const live = ctxs.filter((c) => !c.settled && !c.lostLease);
940
+ if (!live.length) break;
405
941
  try {
406
- await ctx.heartbeat();
942
+ const renewed = await this.store.heartbeatBatch({
943
+ taskIds: live.map((c) => c.taskId),
944
+ workerId: this.workerId,
945
+ leaseMs,
946
+ });
947
+ for (const ctx of live) {
948
+ const cancelRequested = renewed.get(ctx.taskId);
949
+ // Cancellation rides along on the write we were making anyway, so
950
+ // ctx.canceled() stays free here too.
951
+ if (cancelRequested !== undefined) ctx.observeCancel(cancelRequested);
952
+ else if (!ctx.settled) ctx.markLeaseLost();
953
+ }
407
954
  } catch (err) {
408
- // ctx.heartbeat() already flagged the lease as lost for the handler.
409
- if (err instanceof LostLease) break;
410
- this.report(err, { phase: "execute", taskId: ctx.taskId });
955
+ this.report(err, { phase: "execute", taskId: live[0].taskId });
411
956
  }
412
957
  }
413
958
  })();
@@ -421,25 +966,19 @@ export class Worker {
421
966
  }
422
967
 
423
968
  private async safeFail(
424
- task: Task,
969
+ ctx: TaskContext,
425
970
  envelope: Record<string, unknown>,
426
971
  retryable: boolean,
427
972
  ): Promise<void> {
428
- const delayMs = retryable
429
- ? retryDelayMs(
430
- task.attempt,
431
- this.opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS,
432
- this.opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS,
433
- )
434
- : 0;
435
973
  try {
436
974
  await this.store.fail({
437
- taskId: task.id,
975
+ taskId: ctx.taskId,
438
976
  workerId: this.workerId,
439
977
  error: envelope,
440
978
  retryable,
441
- delayMs,
979
+ delayMs: failDelayMs(ctx.attempt, retryable, this.backoffMs, this.backoffMaxMs),
442
980
  });
981
+ ctx.markSettled();
443
982
  } catch (err) {
444
983
  if (!(err instanceof LostLease)) throw err;
445
984
  }