cairnq 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +11 -5
  2. package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
  3. package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
  4. package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
  5. package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
  6. package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
  7. package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
  8. package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
  9. package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
  10. package/dist/backoff.d.ts +31 -0
  11. package/dist/backoff.js +40 -0
  12. package/dist/client.d.ts +28 -2
  13. package/dist/client.js +30 -3
  14. package/dist/context.d.ts +48 -2
  15. package/dist/context.js +101 -10
  16. package/dist/errors.d.ts +60 -4
  17. package/dist/errors.js +94 -9
  18. package/dist/index.d.ts +6 -2
  19. package/dist/index.js +2 -1
  20. package/dist/retention.d.ts +60 -0
  21. package/dist/retention.js +115 -0
  22. package/dist/store/base.d.ts +50 -1
  23. package/dist/store/base.js +114 -19
  24. package/dist/store/sqlite.js +4 -1
  25. package/dist/wait.d.ts +20 -5
  26. package/dist/wait.js +34 -9
  27. package/dist/worker.d.ts +214 -16
  28. package/dist/worker.js +500 -131
  29. package/package.json +1 -1
  30. package/src/backoff.ts +53 -0
  31. package/src/client.ts +43 -5
  32. package/src/context.ts +116 -9
  33. package/src/errors.ts +101 -9
  34. package/src/index.ts +6 -1
  35. package/src/retention.ts +136 -0
  36. package/src/store/base.ts +121 -17
  37. package/src/store/sqlite.ts +4 -1
  38. package/src/wait.ts +60 -16
  39. package/src/worker.ts +640 -146
@@ -11,7 +11,7 @@ export declare function dumpJson(value: unknown): string;
11
11
  * The supported major is a protocol fact, not a dialect one — every backend
12
12
  * checks it here so the constant can't fork per store. */
13
13
  export declare function checkProtocolVersion(version: number): void;
14
- declare const CONFLICTS: readonly ["reuse", "reject", "replace"];
14
+ declare const CONFLICTS: readonly ["reuse", "reuse-succeeded", "reject", "replace"];
15
15
  export type Conflict = (typeof CONFLICTS)[number];
16
16
  /** The queue a submit lands on when it names none. Owned here, where the
17
17
  * default is applied, so nothing above has to re-derive it. */
@@ -174,11 +174,60 @@ export declare abstract class TaskStore {
174
174
  limit?: number;
175
175
  names?: string[];
176
176
  }): Promise<Task[]>;
177
+ /**
178
+ * Open one claim transaction and let the caller draw from it repeatedly.
179
+ *
180
+ * The transaction is what has to live here: the read-only probe that keeps an
181
+ * idle worker off SQLite's single write lock, the `recover_leases` whose
182
+ * reclaimed leases must be visible to the claims that follow and to nobody in
183
+ * between, and the write lock itself. *What* gets claimed under it is the
184
+ * caller's business — a worker drawing a separate quota per task name is
185
+ * scheduling policy, and this layer has no vocabulary for the "handler call"
186
+ * that policy is denominated in. It knows queues, names, limits and rows.
187
+ *
188
+ * `plan` is handed a `claim(names, limit)` it may call any number of times,
189
+ * each a separate statement under the same lock and the same recovery, and
190
+ * each free to size itself from what the previous one returned. That feedback
191
+ * is the reason this is a callback rather than a list of quotas: a caller
192
+ * dividing a budget up front has to guess, and every share handed to a name
193
+ * with nothing queued is a slot left idle until the next poll.
194
+ *
195
+ * `plan` runs with the write lock held, so it must await nothing but that
196
+ * callback.
197
+ *
198
+ * `names` is the union `plan` might ask for — the probe and the recovery are
199
+ * filtered by it. Returns undefined when the probe finds nothing claimable, in
200
+ * which case `plan` never runs and no transaction is opened.
201
+ */
202
+ claimSession<T>(input: {
203
+ queues: string[];
204
+ workerId: string;
205
+ leaseMs?: number;
206
+ names: string[] | null;
207
+ }, plan: (claim: (names: string[] | null, limit: number) => Promise<Task[]>) => Promise<T>): Promise<T | undefined>;
177
208
  heartbeat(input: {
178
209
  taskId: string;
179
210
  workerId: string;
180
211
  leaseMs?: number;
181
212
  }): Promise<Task>;
213
+ /**
214
+ * Renew several leases in one statement. Returns `taskId -> cancel requested`
215
+ * for the tasks this worker still holds.
216
+ *
217
+ * Deliberately not an ownedWrite: ownership is per task here, so there is no
218
+ * single answer to "did it work". A task **absent** from the result lost its
219
+ * lease, and the caller decides what that means for that one task rather than
220
+ * failing the whole beat.
221
+ *
222
+ * It returns flags rather than Tasks because nothing downstream needs a task:
223
+ * the caller renews leases and observes cancellation, and whole rows would drag
224
+ * every payload back on every beat for the life of the call.
225
+ */
226
+ heartbeatBatch(input: {
227
+ taskIds: string[];
228
+ workerId: string;
229
+ leaseMs?: number;
230
+ }): Promise<Map<string, boolean>>;
182
231
  progress(input: {
183
232
  taskId: string;
184
233
  workerId: string;
@@ -1,6 +1,6 @@
1
1
  import { newId } from "../ids.js";
2
2
  import { AlreadyExists, errorEnvelope, LostLease, ProtocolVersionMismatch, SerializationError, } from "../errors.js";
3
- import { rowToTask, STATUSES } from "../models.js";
3
+ import { rowToTask, STATUSES, TERMINAL } from "../models.js";
4
4
  import { QueueDepthGate } from "../backpressure.js";
5
5
  const rejectMangled = function (_key, v) {
6
6
  if (typeof v === "number" && !Number.isFinite(v)) {
@@ -51,7 +51,24 @@ export function checkProtocolVersion(version) {
51
51
  // CONFLICTS is the canonical declaration; the type derives from it so the
52
52
  // runtime guard in submit() and the type can't drift apart (same pattern as
53
53
  // STATUSES/TaskStatus in models.ts).
54
- const CONFLICTS = ["reuse", "reject", "replace"];
54
+ const CONFLICTS = ["reuse", "reuse-succeeded", "reject", "replace"];
55
+ /**
56
+ * Whether a keyed submit's strategy accepts the task the key already points at.
57
+ *
58
+ * Both reuse strategies deduplicate work that is still in play — that is what a
59
+ * key is for, and the answer cannot depend on the outcome of a task that has no
60
+ * outcome yet. They differ only on what a *finished* task means: `reuse` treats
61
+ * the key as free again, while `reuse-succeeded` reads a succeeded task as a
62
+ * cached result. Neither ever hands back a failed or canceled one, which would
63
+ * poison the key for every later submit (see PROTOCOL.md "Key conflict").
64
+ */
65
+ function reusable(conflict, status) {
66
+ if (conflict === "replace")
67
+ return false;
68
+ if (!TERMINAL.includes(status))
69
+ return true;
70
+ return conflict === "reuse-succeeded" && status === "succeeded";
71
+ }
55
72
  /** The queue a submit lands on when it names none. Owned here, where the
56
73
  * default is applied, so nothing above has to re-derive it. */
57
74
  export const DEFAULT_QUEUE = "default";
@@ -207,12 +224,17 @@ export class TaskStore {
207
224
  // free after all, whatever the strategy.
208
225
  const current = (await fetch("get", { id: existing[0].task_id }))[0];
209
226
  if (current) {
210
- if (conflict === "reuse")
211
- return rowToTask(current);
212
227
  if (conflict === "reject")
213
228
  throw new AlreadyExists(key);
214
- // "replace": cancel the recorded task, then repoint the key below.
215
- await fetch("cancel", { id: existing[0].task_id });
229
+ if (reusable(conflict, current.status))
230
+ return rowToTask(current);
231
+ // The strategy declined the recorded task, so the key repoints to the
232
+ // fresh one inserted below. Cancel only what is still live: a terminal
233
+ // task has nothing to stop, and cancelling it would rewrite a settled
234
+ // row (and hand a `canceled` back to whoever is waiting on it).
235
+ if (!TERMINAL.includes(current.status)) {
236
+ await fetch("cancel", { id: existing[0].task_id });
237
+ }
216
238
  }
217
239
  }
218
240
  const row = (await fetch("insert_task", ins))[0];
@@ -330,27 +352,77 @@ export class TaskStore {
330
352
  * array claims nothing.
331
353
  */
332
354
  async claim(input) {
333
- // One queue is the common case and gets its own statement: a list-valued queue
334
- // filter cannot be read in claim order, so the planner sorts every claimable
335
- // row to take LIMIT of them, and claim's cost grows with the queued backlog
336
- // while it holds the claim transaction. See claim_one_queue.sql.
355
+ const names = input.names ?? null;
356
+ const claimed = await this.claimSession({ queues: input.queues, workerId: input.workerId, leaseMs: input.leaseMs, names }, (claim) => claim(names, input.limit ?? 1));
357
+ return claimed ?? [];
358
+ }
359
+ /**
360
+ * Open one claim transaction and let the caller draw from it repeatedly.
361
+ *
362
+ * The transaction is what has to live here: the read-only probe that keeps an
363
+ * idle worker off SQLite's single write lock, the `recover_leases` whose
364
+ * reclaimed leases must be visible to the claims that follow and to nobody in
365
+ * between, and the write lock itself. *What* gets claimed under it is the
366
+ * caller's business — a worker drawing a separate quota per task name is
367
+ * scheduling policy, and this layer has no vocabulary for the "handler call"
368
+ * that policy is denominated in. It knows queues, names, limits and rows.
369
+ *
370
+ * `plan` is handed a `claim(names, limit)` it may call any number of times,
371
+ * each a separate statement under the same lock and the same recovery, and
372
+ * each free to size itself from what the previous one returned. That feedback
373
+ * is the reason this is a callback rather than a list of quotas: a caller
374
+ * dividing a budget up front has to guess, and every share handed to a name
375
+ * with nothing queued is a slot left idle until the next poll.
376
+ *
377
+ * `plan` runs with the write lock held, so it must await nothing but that
378
+ * callback.
379
+ *
380
+ * `names` is the union `plan` might ask for — the probe and the recovery are
381
+ * filtered by it. Returns undefined when the probe finds nothing claimable, in
382
+ * which case `plan` never runs and no transaction is opened.
383
+ */
384
+ async claimSession(input, plan) {
385
+ // A list-valued filter cannot be read in claim order, so the planner sorts
386
+ // every claimable row to take LIMIT of them and the claim's cost grows with
387
+ // the backlog while it holds the transaction. Both filters therefore have an
388
+ // equality form, picked per draw: one queue is the common deployment, and one
389
+ // name is every per-name quota. See claim_one_queue.sql and claim_one_name.sql.
337
390
  const oneQueue = input.queues.length === 1;
338
- const params = {
391
+ const base = {
339
392
  queues: input.queues,
340
393
  queue: oneQueue ? input.queues[0] : null,
341
- names: input.names ?? null,
394
+ names: input.names,
395
+ name: null,
342
396
  worker_id: input.workerId,
343
397
  lease_ms: input.leaseMs ?? 30_000,
344
- limit: input.limit ?? 1,
398
+ limit: 1,
345
399
  lease_expired_error: LEASE_EXPIRED_ERROR_JSON,
346
400
  };
347
- if (!(await this.hasClaimableWork(params)))
348
- return [];
349
- // Recovery must share the claim's transaction: a lease reclaimed here has to
350
- // be visible to the claim that follows, and to nobody in between.
401
+ if (!(await this.hasClaimableWork(base)))
402
+ return undefined;
351
403
  return this.tx(async (fetch) => {
352
- await fetch("recover_leases", params);
353
- return (await fetch(oneQueue ? "claim_one_queue" : "claim", params)).map(rowToTask);
404
+ await fetch("recover_leases", base);
405
+ return plan(async (names, limit) => {
406
+ // A draw asking for nothing, or filtered to no names, claims nothing —
407
+ // answer it here rather than spending a statement to learn that.
408
+ if (limit <= 0 || names?.length === 0)
409
+ return [];
410
+ const oneName = names?.length === 1;
411
+ const statement = oneName
412
+ ? oneQueue
413
+ ? "claim_one_queue_one_name"
414
+ : "claim_one_name"
415
+ : oneQueue
416
+ ? "claim_one_queue"
417
+ : "claim";
418
+ const rows = await fetch(statement, {
419
+ ...base,
420
+ names,
421
+ name: oneName ? names[0] : null,
422
+ limit,
423
+ });
424
+ return rows.map(rowToTask);
425
+ });
354
426
  });
355
427
  }
356
428
  async heartbeat(input) {
@@ -360,6 +432,29 @@ export class TaskStore {
360
432
  lease_ms: input.leaseMs ?? 30_000,
361
433
  });
362
434
  }
435
+ /**
436
+ * Renew several leases in one statement. Returns `taskId -> cancel requested`
437
+ * for the tasks this worker still holds.
438
+ *
439
+ * Deliberately not an ownedWrite: ownership is per task here, so there is no
440
+ * single answer to "did it work". A task **absent** from the result lost its
441
+ * lease, and the caller decides what that means for that one task rather than
442
+ * failing the whole beat.
443
+ *
444
+ * It returns flags rather than Tasks because nothing downstream needs a task:
445
+ * the caller renews leases and observes cancellation, and whole rows would drag
446
+ * every payload back on every beat for the life of the call.
447
+ */
448
+ async heartbeatBatch(input) {
449
+ if (!input.taskIds.length)
450
+ return new Map();
451
+ const rows = await this.fetch("heartbeat_batch", {
452
+ ids: input.taskIds,
453
+ worker_id: input.workerId,
454
+ lease_ms: input.leaseMs ?? 30_000,
455
+ });
456
+ return new Map(rows.map((r) => [r.id, r.cancel_requested_at_ms != null]));
457
+ }
363
458
  async progress(input) {
364
459
  return this.ownedWrite("progress", input.taskId, {
365
460
  id: input.taskId,
@@ -313,7 +313,10 @@ export class SQLiteStore extends TaskStore {
313
313
  bound[name] = now - params.older_than_ms;
314
314
  break;
315
315
  case "queues":
316
- bound[name] = JSON.stringify(params.queues);
316
+ case "ids":
317
+ // json_each needs a JSON array. Postgres binds the array itself as
318
+ // text[], so only this dialect encodes.
319
+ bound[name] = JSON.stringify(params[name]);
317
320
  break;
318
321
  case "names":
319
322
  // json_each needs a JSON array; null stays null so the SQL's
package/dist/wait.d.ts CHANGED
@@ -2,6 +2,11 @@ import { type Task } from "./models.js";
2
2
  import type { TaskStore } from "./store/base.js";
3
3
  export declare const DEFAULT_POLL_MS = 100;
4
4
  export declare const MAX_POLL_MS = 500;
5
+ export interface PollOptions {
6
+ timeoutMs: number;
7
+ pollMs?: number;
8
+ maxPollMs?: number;
9
+ }
5
10
  /**
6
11
  * Grow the polling interval towards the ceiling.
7
12
  *
@@ -14,8 +19,18 @@ export declare function nextPollMs(current: number, maxMs: number): number;
14
19
  /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
15
20
  * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
16
21
  * it backs off towards `maxPollMs`. */
17
- export declare function pollWait(store: TaskStore, taskId: string, { timeoutMs, pollMs, maxPollMs, }: {
18
- timeoutMs: number;
19
- pollMs?: number;
20
- maxPollMs?: number;
21
- }): Promise<Task>;
22
+ export declare function pollWait(store: TaskStore, taskId: string, opts: PollOptions): Promise<Task>;
23
+ /**
24
+ * The same wait, following a key instead of an id.
25
+ *
26
+ * The key is re-resolved on every read, because that is what a key means: a
27
+ * pointer to the task that is *current* under it. A `replace` landing mid-wait
28
+ * moves the wait onto the new task rather than reporting the cancellation of the
29
+ * old one, and a key that points at nothing yet is simply not finished — it
30
+ * polls until something appears, the same way waiting on an id that does not
31
+ * exist yet does.
32
+ *
33
+ * There is nothing to subscribe to before the key resolves, so those naps are
34
+ * plain sleeps; once it resolves, the store's push channel applies as usual.
35
+ */
36
+ export declare function pollWaitByKey(store: TaskStore, key: string, opts: PollOptions): Promise<Task>;
package/dist/wait.js CHANGED
@@ -4,6 +4,7 @@ import { isTerminal } from "./models.js";
4
4
  export const DEFAULT_POLL_MS = 100;
5
5
  export const MAX_POLL_MS = 500;
6
6
  const GROWTH = 1.5;
7
+ const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
7
8
  /**
8
9
  * Grow the polling interval towards the ceiling.
9
10
  *
@@ -15,22 +16,46 @@ const GROWTH = 1.5;
15
16
  export function nextPollMs(current, maxMs) {
16
17
  return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
17
18
  }
18
- /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
19
- * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
20
- * it backs off towards `maxPollMs`. */
21
- export async function pollWait(store, taskId, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS, }) {
19
+ /**
20
+ * Poll `read` until it yields a terminal task, or the timeout elapses.
21
+ *
22
+ * `wake` is what the loop sleeps on between reads: a store with a push channel
23
+ * (Postgres) cuts it short when the task goes terminal, but the re-read is the
24
+ * source of truth either way, so a plain sleep is always a correct answer.
25
+ */
26
+ async function poll(read, wake, subject, key, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS }) {
22
27
  const deadline = nowMs() + timeoutMs;
23
28
  let interval = pollMs;
24
29
  for (;;) {
25
- const task = await store.get(taskId);
30
+ const task = await read();
26
31
  if (task && isTerminal(task))
27
32
  return task;
28
33
  const remaining = deadline - nowMs();
29
34
  if (remaining <= 0)
30
- throw new TaskTimeout(taskId, { timeoutMs, task });
31
- // A store with a push channel (Postgres) cuts the sleep short when the task
32
- // goes terminal; the re-get above stays the source of truth either way.
33
- await store.taskDoneWake(taskId, Math.min(interval, remaining));
35
+ throw new TaskTimeout(task?.id ?? subject, { timeoutMs, task, key });
36
+ await wake(task, Math.min(interval, remaining));
34
37
  interval = nextPollMs(interval, maxPollMs);
35
38
  }
36
39
  }
40
+ /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
41
+ * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
42
+ * it backs off towards `maxPollMs`. */
43
+ export function pollWait(store, taskId, opts) {
44
+ return poll(() => store.get(taskId), (_task, ms) => store.taskDoneWake(taskId, ms), taskId, null, opts);
45
+ }
46
+ /**
47
+ * The same wait, following a key instead of an id.
48
+ *
49
+ * The key is re-resolved on every read, because that is what a key means: a
50
+ * pointer to the task that is *current* under it. A `replace` landing mid-wait
51
+ * moves the wait onto the new task rather than reporting the cancellation of the
52
+ * old one, and a key that points at nothing yet is simply not finished — it
53
+ * polls until something appears, the same way waiting on an id that does not
54
+ * exist yet does.
55
+ *
56
+ * There is nothing to subscribe to before the key resolves, so those naps are
57
+ * plain sleeps; once it resolves, the store's push channel applies as usual.
58
+ */
59
+ export function pollWaitByKey(store, key, opts) {
60
+ return poll(() => store.getByKey(key), (task, ms) => (task ? store.taskDoneWake(task.id, ms) : sleep(ms)), key, key, opts);
61
+ }
package/dist/worker.d.ts CHANGED
@@ -2,9 +2,20 @@ import type { BackpressureOptions } from "./backpressure.js";
2
2
  import { TaskContext } from "./context.js";
3
3
  import type { TaskStore } from "./store/base.js";
4
4
  import { type TaskDef } from "./task.js";
5
+ export { retryDelayMs, DEFAULT_RETRY_BACKOFF_MS, DEFAULT_RETRY_BACKOFF_MAX_MS } from "./backoff.js";
5
6
  export type Handler = (ctx: TaskContext, payload: any) => unknown | Promise<unknown>;
6
7
  /** Handler typed against a TaskDef<P, R>: payload is P, the return is R. */
7
8
  export type TypedHandler<P, R> = (ctx: TaskContext, payload: P) => R | Promise<R>;
9
+ /**
10
+ * A batch handler takes one argument: the list of contexts. There is no payload
11
+ * shortcut to pair with it — payloads are per task, so they are read off the
12
+ * items (`item.payload`), which is also what a handler must hold to settle one
13
+ * of them.
14
+ *
15
+ * Returning a map of task id -> result fills in results for the tasks the
16
+ * handler did not settle itself; anything else returned is ignored.
17
+ */
18
+ export type BatchHandler = (items: TaskContext[]) => void | Record<string, unknown> | Promise<void | Record<string, unknown>>;
8
19
  /** Where an error the worker recovered from came from. */
9
20
  export type ErrorPhase = "claim" | "execute";
10
21
  /**
@@ -13,6 +24,12 @@ export type ErrorPhase = "claim" | "execute";
13
24
  * there is usually no CairnQ handle to have configured the store.
14
25
  */
15
26
  export interface WorkerOptions extends Partial<BackpressureOptions> {
27
+ /**
28
+ * Handler calls allowed to run at once. A batch call counts as one, however
29
+ * many tasks it carries — size it for how much work you want in flight, not
30
+ * for how many tasks that comes to. Per-name limits refine it; `maxInFlightBytes`
31
+ * bounds memory, which task counts never did.
32
+ */
16
33
  concurrency?: number;
17
34
  leaseMs?: number;
18
35
  heartbeatIntervalMs?: number;
@@ -41,15 +58,29 @@ export interface WorkerOptions extends Partial<BackpressureOptions> {
41
58
  * between megabytes and gigabytes resident. Once the budget is spent the
42
59
  * worker stops claiming until running handlers give it back.
43
60
  *
44
- * The bound is on tasks already executing. A claim commits to a whole batch
45
- * before any size is known, so one batch can overshoot by up to `claimBatch`
46
- * payloads; lower `claimBatch` to tighten that. A single payload larger than
47
- * the entire budget still runs — alone, rather than deadlocking the worker.
61
+ * The bound is on tasks already executing: it is read between claims, never
62
+ * during one, and a claim commits to its rows before any size is known. One
63
+ * poll can therefore overshoot by up to `claimBatch` rows per registered name
64
+ * (or one whole `batch`, whichever is larger). Lower `claimBatch`, or the batch
65
+ * sizes, to tighten that. A single payload larger than the entire budget still
66
+ * runs — alone, rather than deadlocking the worker.
48
67
  *
49
68
  * Costs one JSON serialization per task to measure, so it is only computed
50
69
  * when set. Unset disables the budget.
51
70
  */
52
71
  maxInFlightBytes?: number;
72
+ /**
73
+ * Call ceilings that several names can draw from, by name — `{ gpu: 1 }`.
74
+ * A handler joins one with `task(name, { resource: "gpu" }, fn)`.
75
+ *
76
+ * `concurrency` caps a name against itself, which cannot say what usually
77
+ * binds a worker doing heavy local work: several *different* handlers
78
+ * contending for one scarce thing — a GPU, an index that tolerates a single
79
+ * writer. The limit belongs to that thing rather than to any one name, so it
80
+ * is declared here, once, and at capacity 1 it is mutual exclusion across the
81
+ * names that join it.
82
+ */
83
+ resources?: Record<string, number>;
53
84
  /**
54
85
  * Called for errors the worker survived — a claim that threw, a store write
55
86
  * that failed while finalizing a task. Without it these are silent: the run
@@ -61,8 +92,6 @@ export interface WorkerOptions extends Partial<BackpressureOptions> {
61
92
  taskId?: string;
62
93
  }) => void;
63
94
  }
64
- /** Exponential backoff for the next attempt of a task that just failed. */
65
- export declare function retryDelayMs(attempt: number, baseMs: number, maxMs: number): number;
66
95
  export declare class Worker {
67
96
  private readonly store;
68
97
  private readonly queues;
@@ -71,6 +100,27 @@ export declare class Worker {
71
100
  private readonly workerId;
72
101
  /** Payload bytes charged to running handlers — see maxInFlightBytes. */
73
102
  private inFlightBytes;
103
+ /** Calls in flight, for the names that cap their own concurrency. */
104
+ private readonly callsInFlight;
105
+ /**
106
+ * Calls holding units of each declared resource. A resource is the same shape
107
+ * of budget as a name's own `concurrency` — a ceiling on calls — differing
108
+ * only in who draws from it: several names rather than one. That is what
109
+ * expresses "these handlers share one GPU" without inventing a queue per
110
+ * resource.
111
+ */
112
+ private readonly resourceCalls;
113
+ /** Rotates which source is offered the free budget first — see loop(). */
114
+ private claimCursor;
115
+ /** Invalidated by task(); see schedule(). */
116
+ private scheduleCache;
117
+ /**
118
+ * Retry backoff, resolved once. Both settlement paths read these — the
119
+ * worker's own `safeFail` and the TaskContext it hands a handler — so
120
+ * resolving the defaults per call site is how the two drift apart.
121
+ */
122
+ private readonly backoffMs;
123
+ private readonly backoffMaxMs;
74
124
  private stopped;
75
125
  private stopWake;
76
126
  private readonly stopped$;
@@ -90,6 +140,30 @@ export declare class Worker {
90
140
  task(handler: Handler): this;
91
141
  task(name: string, handler: Handler): this;
92
142
  task<P, R>(def: TaskDef<P, R>, handler: TypedHandler<P, R>): this;
143
+ /**
144
+ * Batch delivery: the handler takes one argument, a `TaskContext[]` of up to
145
+ * `batch` tasks, instead of `(ctx, payload)`. Use it when the work itself is
146
+ * batched — one embedding call over 256 texts rather than 256 calls — and size
147
+ * it by what the downstream API wants, not by the queue.
148
+ *
149
+ * `concurrency` caps the calls this name may run at once, under the worker's
150
+ * own. Use it to keep one expensive name from taking the whole worker.
151
+ *
152
+ * `resource` draws each call from a ceiling declared in
153
+ * `WorkerOptions.resources` and shared with every other name that names it —
154
+ * at capacity 1, mutual exclusion across those names.
155
+ */
156
+ task(name: string | TaskDef, opts: {
157
+ batch: number;
158
+ concurrency?: number;
159
+ resource?: string;
160
+ }, handler: BatchHandler): this;
161
+ /** Per-name concurrency or a shared resource, without batching: the handler
162
+ * still takes `(ctx, payload)`. */
163
+ task(name: string | TaskDef, opts: {
164
+ concurrency?: number;
165
+ resource?: string;
166
+ }, handler: Handler): this;
93
167
  stop(): void;
94
168
  /** Close the underlying store connection. Call after run() returns. */
95
169
  close(): Promise<void>;
@@ -98,6 +172,52 @@ export declare class Worker {
98
172
  run(opts?: {
99
173
  concurrency?: number;
100
174
  }): Promise<void>;
175
+ /**
176
+ * Split one claim into handler calls, each with the registration to run it.
177
+ *
178
+ * A claim is filtered by queue and by the names this worker handles, so it
179
+ * comes back mixed; batch size is per name (one embedding call wants 256
180
+ * texts, one Docling parse wants exactly 1). So group by name, then chunk each
181
+ * group by that name's size. Names registered without `batch` come back as
182
+ * one-task calls, as do names not registered at all — reachable only if a
183
+ * handler is unregistered mid-run, and dispatched to failNoHandler().
184
+ *
185
+ * The registration rides along because this is where it was resolved; looking
186
+ * it up again at the call site would put "is this name batched" in two places.
187
+ */
188
+ private deliveries;
189
+ private context;
190
+ /**
191
+ * How this poll's claim is split into per-name quotas, plus the union of names
192
+ * the probe spans.
193
+ *
194
+ * Cached and invalidated by `task()`, rather than rebuilt each poll: handlers
195
+ * may be registered after run() started, but only there, and this otherwise
196
+ * allocates a source per name on every tick for a worker's whole lifetime.
197
+ *
198
+ * A name that limits itself — by `batch`, by its own `concurrency`, or by a
199
+ * `resource` — needs a quota the shared draw cannot express, so it gets a
200
+ * source of its own; every other name shares one, where a task is a call.
201
+ *
202
+ * A resource is deliberately *not* one source spanning its names: `batch` is
203
+ * per name, and a single source carries one batch size, so two members that
204
+ * batch differently could not share a draw. Keeping a source per name and
205
+ * letting several of them draw down one shared ceiling composes with batching
206
+ * instead of excluding it.
207
+ */
208
+ private schedule;
209
+ /**
210
+ * A source's own call ceiling for one poll, or undefined when only the
211
+ * worker-wide budget applies.
212
+ *
213
+ * Three independent ceilings, whichever binds first: the name's own concurrency
214
+ * less what it is already running; its resource's capacity less what is running
215
+ * *and* what earlier draws in this same poll already took (`taken`); and
216
+ * `claimBatch`. The last is a ceiling on *rows* per poll, so it converts at this
217
+ * source's batch size — and never below one call, or a `claimBatch` under some
218
+ * name's batch would stall that name outright.
219
+ */
220
+ private sourceCalls;
101
221
  private loop;
102
222
  /** Blocking-style entry point for a standalone worker process: run until
103
223
  * SIGINT/SIGTERM, then close the store. Use this at a script's top level;
@@ -110,20 +230,98 @@ export declare class Worker {
110
230
  concurrency?: number;
111
231
  }): Promise<T>;
112
232
  /**
113
- * Run one task to completion. Never rejects: a task-level failure is reported
114
- * through onError and the loop moves on. (It used to reject into a promise
115
- * nobody awaited — an unhandled rejection that took the process down.)
233
+ * Record a claimed task this worker cannot run. Reachable only if a name is
234
+ * unregistered mid-run — the claim filters on the registered names — so it does
235
+ * not start a handler or a heartbeat for a task it will not run.
236
+ */
237
+ private failNoHandler;
238
+ /**
239
+ * Run one handler call to completion — one task, or a whole batch. Never
240
+ * rejects: a task-level failure is reported through onError and the loop moves
241
+ * on. (It used to reject into a promise nobody awaited — an unhandled rejection
242
+ * that took the process down.)
243
+ *
244
+ * One lifecycle for both delivery modes, because a single-task handler *is* the
245
+ * one-element case: the same heartbeat covers the call, the same classifier
246
+ * reads its error, and the same rule settles what is left. Only two things vary
247
+ * — how the handler is invoked, and where a leftover task's result comes from —
248
+ * so those are the only two branches below.
249
+ *
250
+ * The contract is single: **when the handler returns, every task it did not
251
+ * settle itself is settled by how the call ended.** Returning succeeds them,
252
+ * throwing fails them (retryably, or as the thrown TaskError says). That is
253
+ * what keeps the ordinary cases free of bookkeeping — a handler that just
254
+ * returns has finished 256 tasks — while still letting it pick individual
255
+ * tasks off with `item.succeed()` / `item.fail()` as it goes.
256
+ *
257
+ * For a batch, a returned map of task id -> result fills in results for the
258
+ * tasks left over; anything else returned is ignored, and unmentioned tasks
259
+ * succeed with no result (the common shape, where the handler's output went to
260
+ * a database rather than into the task row). For a single task the return value
261
+ * simply is the result.
116
262
  */
117
- private execute;
263
+ private runCall;
118
264
  /**
119
- * Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
120
- * abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
121
- * a handler that keeps running can never write again — see
122
- * TaskContext.owned), then the still-pending promise is left to settle on
123
- * its own, its outcome discarded. The caller records the handler_timeout
124
- * failure; lease recovery is NOT involved, so redelivery is immediate.
265
+ * How an attempt that ended badly is recorded: [envelope, retryable], or null
266
+ * when there is nothing to record.
267
+ *
268
+ * One classifier for both delivery modes, so a handler error cannot mean
269
+ * different things depending on how its task happened to be delivered. That
270
+ * includes LostLease, which is not an outcome at all: it means a write through
271
+ * this context was already rejected, so recording anything more would be
272
+ * rejected too. Both modes then leave the task alone — the single-task one has
273
+ * nothing else to do, and a batch lets its remaining tasks fall to lease expiry
274
+ * and redelivery rather than stamping them with a failure the handler never
275
+ * reported.
276
+ */
277
+ private outcomeOf;
278
+ /**
279
+ * Finalize one task the handler left for the worker to decide — the tail of
280
+ * both delivery modes.
281
+ *
282
+ * Includes the unserializable-result rule: the handler succeeded but its value
283
+ * cannot cross the JSON protocol (BigInt, non-finite number, circular), which
284
+ * is deterministic, so it fails permanently rather than being redelivered to
285
+ * fail the same way every attempt.
286
+ */
287
+ private succeedOne;
288
+ /**
289
+ * Settle every task the handler did not settle itself, concurrently, reporting
290
+ * rather than throwing. Each task keeps its own attempt count and backoff —
291
+ * they are separate tasks that happened to be delivered together.
292
+ *
293
+ * allSettled, because one task's write failing must not abandon the rest of the
294
+ * batch mid-settlement — the others still hold leases and would sit `running`
295
+ * until expiry. Each outcome is reported against the task it belongs to;
296
+ * without that, an operator learns a settlement failed somewhere in a batch of
297
+ * 256.
298
+ */
299
+ private settleEach;
300
+ /**
301
+ * Run one attempt — of one task or of a whole batch — bounded by maxRunMs when
302
+ * set.
303
+ *
304
+ * On timeout the attempt is abandoned: every context it covers is flagged
305
+ * lease-lost first (ctx.signal aborts, and a handler that keeps running can
306
+ * never write again, nor settle anything behind the worker's back — see
307
+ * TaskContext.owned), then the still-pending promise is left to settle on its
308
+ * own, its outcome discarded. The caller records the handler_timeout failure;
309
+ * lease recovery is NOT involved, so redelivery is immediate.
125
310
  */
126
311
  private attempt;
312
+ /**
313
+ * One statement per beat, however many tasks the call covers — a single-task
314
+ * handler is just the one-element case.
315
+ *
316
+ * Only tasks still in play are renewed. A task the handler already settled is
317
+ * terminal, and re-leasing it would be a write against a row nobody owns; that
318
+ * is also why an absence is only read as lease loss after re-checking
319
+ * `settled`, since the handler may have finalized the task while this beat was
320
+ * in flight, which takes the row out of `running` and out of the reply. A task
321
+ * genuinely missing lost its lease (another worker recovered it), so its
322
+ * context is flagged and the handler stops being able to write through it —
323
+ * only that one, never its neighbours.
324
+ */
127
325
  private startHeartbeat;
128
326
  private safeFail;
129
327
  /**