cairnq 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +3 -2
  2. package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
  3. package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
  4. package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
  5. package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
  6. package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
  7. package/dist/_protocol/sql/postgres/queue_depth.sql +22 -0
  8. package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
  9. package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
  10. package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
  11. package/dist/_protocol/sql/sqlite/queue_depth.sql +26 -0
  12. package/dist/backoff.d.ts +31 -0
  13. package/dist/backoff.js +40 -0
  14. package/dist/backpressure.d.ts +59 -0
  15. package/dist/backpressure.js +122 -0
  16. package/dist/client.d.ts +16 -3
  17. package/dist/client.js +19 -5
  18. package/dist/context.d.ts +48 -2
  19. package/dist/context.js +101 -10
  20. package/dist/errors.d.ts +37 -0
  21. package/dist/errors.js +60 -0
  22. package/dist/index.d.ts +7 -3
  23. package/dist/index.js +2 -1
  24. package/dist/store/base.d.ts +73 -0
  25. package/dist/store/base.js +124 -14
  26. package/dist/store/sqlite.d.ts +35 -0
  27. package/dist/store/sqlite.js +163 -3
  28. package/dist/worker.d.ts +238 -13
  29. package/dist/worker.js +512 -120
  30. package/package.json +2 -1
  31. package/src/backoff.ts +53 -0
  32. package/src/backpressure.ts +140 -0
  33. package/src/client.ts +33 -5
  34. package/src/context.ts +116 -9
  35. package/src/errors.ts +66 -0
  36. package/src/index.ts +7 -2
  37. package/src/store/base.ts +136 -13
  38. package/src/store/sqlite.ts +168 -2
  39. package/src/worker.ts +671 -132
package/dist/worker.d.ts CHANGED
@@ -1,12 +1,35 @@
1
+ import type { BackpressureOptions } from "./backpressure.js";
1
2
  import { TaskContext } from "./context.js";
2
3
  import type { TaskStore } from "./store/base.js";
3
4
  import { type TaskDef } from "./task.js";
5
+ export { retryDelayMs, DEFAULT_RETRY_BACKOFF_MS, DEFAULT_RETRY_BACKOFF_MAX_MS } from "./backoff.js";
4
6
  export type Handler = (ctx: TaskContext, payload: any) => unknown | Promise<unknown>;
5
7
  /** Handler typed against a TaskDef<P, R>: payload is P, the return is R. */
6
8
  export type TypedHandler<P, R> = (ctx: TaskContext, payload: P) => R | Promise<R>;
9
+ /**
10
+ * A batch handler takes one argument: the list of contexts. There is no payload
11
+ * shortcut to pair with it — payloads are per task, so they are read off the
12
+ * items (`item.payload`), which is also what a handler must hold to settle one
13
+ * of them.
14
+ *
15
+ * Returning a map of task id -> result fills in results for the tasks the
16
+ * handler did not settle itself; anything else returned is ignored.
17
+ */
18
+ export type BatchHandler = (items: TaskContext[]) => void | Record<string, unknown> | Promise<void | Record<string, unknown>>;
7
19
  /** Where an error the worker recovered from came from. */
8
20
  export type ErrorPhase = "claim" | "execute";
9
- export interface WorkerOptions {
21
+ /**
22
+ * Backpressure is accepted here too, not only on CairnQ: a handler spawning
23
+ * children through TaskContext.submit is a producer, and in a worker process
24
+ * there is usually no CairnQ handle to have configured the store.
25
+ */
26
+ export interface WorkerOptions extends Partial<BackpressureOptions> {
27
+ /**
28
+ * Handler calls allowed to run at once. A batch call counts as one, however
29
+ * many tasks it carries — size it for how much work you want in flight, not
30
+ * for how many tasks that comes to. Per-name limits refine it; `maxInFlightBytes`
31
+ * bounds memory, which task counts never did.
32
+ */
10
33
  concurrency?: number;
11
34
  leaseMs?: number;
12
35
  heartbeatIntervalMs?: number;
@@ -25,6 +48,39 @@ export interface WorkerOptions {
25
48
  * a retryable `handler_timeout` failure. Unset disables the ceiling.
26
49
  */
27
50
  maxRunMs?: number;
51
+ /**
52
+ * Resident payload bytes allowed across running handlers, independent of
53
+ * their count.
54
+ *
55
+ * `concurrency` bounds tasks, not memory, so a worker sized for small payloads
56
+ * holds concurrency * largest-payload bytes the moment a batch of big ones
57
+ * arrives — for payloads that carry media inline, that is the difference
58
+ * between megabytes and gigabytes resident. Once the budget is spent the
59
+ * worker stops claiming until running handlers give it back.
60
+ *
61
+ * The bound is on tasks already executing: it is read between claims, never
62
+ * during one, and a claim commits to its rows before any size is known. One
63
+ * poll can therefore overshoot by up to `claimBatch` rows per registered name
64
+ * (or one whole `batch`, whichever is larger). Lower `claimBatch`, or the batch
65
+ * sizes, to tighten that. A single payload larger than the entire budget still
66
+ * runs — alone, rather than deadlocking the worker.
67
+ *
68
+ * Costs one JSON serialization per task to measure, so it is only computed
69
+ * when set. Unset disables the budget.
70
+ */
71
+ maxInFlightBytes?: number;
72
+ /**
73
+ * Call ceilings that several names can draw from, by name — `{ gpu: 1 }`.
74
+ * A handler joins one with `task(name, { resource: "gpu" }, fn)`.
75
+ *
76
+ * `concurrency` caps a name against itself, which cannot say what usually
77
+ * binds a worker doing heavy local work: several *different* handlers
78
+ * contending for one scarce thing — a GPU, an index that tolerates a single
79
+ * writer. The limit belongs to that thing rather than to any one name, so it
80
+ * is declared here, once, and at capacity 1 it is mutual exclusion across the
81
+ * names that join it.
82
+ */
83
+ resources?: Record<string, number>;
28
84
  /**
29
85
  * Called for errors the worker survived — a claim that threw, a store write
30
86
  * that failed while finalizing a task. Without it these are silent: the run
@@ -36,14 +92,35 @@ export interface WorkerOptions {
36
92
  taskId?: string;
37
93
  }) => void;
38
94
  }
39
- /** Exponential backoff for the next attempt of a task that just failed. */
40
- export declare function retryDelayMs(attempt: number, baseMs: number, maxMs: number): number;
41
95
  export declare class Worker {
42
96
  private readonly store;
43
97
  private readonly queues;
44
98
  private readonly opts;
45
99
  private readonly handlers;
46
100
  private readonly workerId;
101
+ /** Payload bytes charged to running handlers — see maxInFlightBytes. */
102
+ private inFlightBytes;
103
+ /** Calls in flight, for the names that cap their own concurrency. */
104
+ private readonly callsInFlight;
105
+ /**
106
+ * Calls holding units of each declared resource. A resource is the same shape
107
+ * of budget as a name's own `concurrency` — a ceiling on calls — differing
108
+ * only in who draws from it: several names rather than one. That is what
109
+ * expresses "these handlers share one GPU" without inventing a queue per
110
+ * resource.
111
+ */
112
+ private readonly resourceCalls;
113
+ /** Rotates which source is offered the free budget first — see loop(). */
114
+ private claimCursor;
115
+ /** Invalidated by task(); see schedule(). */
116
+ private scheduleCache;
117
+ /**
118
+ * Retry backoff, resolved once. Both settlement paths read these — the
119
+ * worker's own `safeFail` and the TaskContext it hands a handler — so
120
+ * resolving the defaults per call site is how the two drift apart.
121
+ */
122
+ private readonly backoffMs;
123
+ private readonly backoffMaxMs;
47
124
  private stopped;
48
125
  private stopWake;
49
126
  private readonly stopped$;
@@ -63,6 +140,30 @@ export declare class Worker {
63
140
  task(handler: Handler): this;
64
141
  task(name: string, handler: Handler): this;
65
142
  task<P, R>(def: TaskDef<P, R>, handler: TypedHandler<P, R>): this;
143
+ /**
144
+ * Batch delivery: the handler takes one argument, a `TaskContext[]` of up to
145
+ * `batch` tasks, instead of `(ctx, payload)`. Use it when the work itself is
146
+ * batched — one embedding call over 256 texts rather than 256 calls — and size
147
+ * it by what the downstream API wants, not by the queue.
148
+ *
149
+ * `concurrency` caps the calls this name may run at once, under the worker's
150
+ * own. Use it to keep one expensive name from taking the whole worker.
151
+ *
152
+ * `resource` draws each call from a ceiling declared in
153
+ * `WorkerOptions.resources` and shared with every other name that names it —
154
+ * at capacity 1, mutual exclusion across those names.
155
+ */
156
+ task(name: string | TaskDef, opts: {
157
+ batch: number;
158
+ concurrency?: number;
159
+ resource?: string;
160
+ }, handler: BatchHandler): this;
161
+ /** Per-name concurrency or a shared resource, without batching: the handler
162
+ * still takes `(ctx, payload)`. */
163
+ task(name: string | TaskDef, opts: {
164
+ concurrency?: number;
165
+ resource?: string;
166
+ }, handler: Handler): this;
66
167
  stop(): void;
67
168
  /** Close the underlying store connection. Call after run() returns. */
68
169
  close(): Promise<void>;
@@ -71,6 +172,52 @@ export declare class Worker {
71
172
  run(opts?: {
72
173
  concurrency?: number;
73
174
  }): Promise<void>;
175
+ /**
176
+ * Split one claim into handler calls, each with the registration to run it.
177
+ *
178
+ * A claim is filtered by queue and by the names this worker handles, so it
179
+ * comes back mixed; batch size is per name (one embedding call wants 256
180
+ * texts, one Docling parse wants exactly 1). So group by name, then chunk each
181
+ * group by that name's size. Names registered without `batch` come back as
182
+ * one-task calls, as do names not registered at all — reachable only if a
183
+ * handler is unregistered mid-run, and dispatched to failNoHandler().
184
+ *
185
+ * The registration rides along because this is where it was resolved; looking
186
+ * it up again at the call site would put "is this name batched" in two places.
187
+ */
188
+ private deliveries;
189
+ private context;
190
+ /**
191
+ * How this poll's claim is split into per-name quotas, plus the union of names
192
+ * the probe spans.
193
+ *
194
+ * Cached and invalidated by `task()`, rather than rebuilt each poll: handlers
195
+ * may be registered after run() started, but only there, and this otherwise
196
+ * allocates a source per name on every tick for a worker's whole lifetime.
197
+ *
198
+ * A name that limits itself — by `batch`, by its own `concurrency`, or by a
199
+ * `resource` — needs a quota the shared draw cannot express, so it gets a
200
+ * source of its own; every other name shares one, where a task is a call.
201
+ *
202
+ * A resource is deliberately *not* one source spanning its names: `batch` is
203
+ * per name, and a single source carries one batch size, so two members that
204
+ * batch differently could not share a draw. Keeping a source per name and
205
+ * letting several of them draw down one shared ceiling composes with batching
206
+ * instead of excluding it.
207
+ */
208
+ private schedule;
209
+ /**
210
+ * A source's own call ceiling for one poll, or undefined when only the
211
+ * worker-wide budget applies.
212
+ *
213
+ * Three independent ceilings, whichever binds first: the name's own concurrency
214
+ * less what it is already running; its resource's capacity less what is running
215
+ * *and* what earlier draws in this same poll already took (`taken`); and
216
+ * `claimBatch`. The last is a ceiling on *rows* per poll, so it converts at this
217
+ * source's batch size — and never below one call, or a `claimBatch` under some
218
+ * name's batch would stall that name outright.
219
+ */
220
+ private sourceCalls;
74
221
  private loop;
75
222
  /** Blocking-style entry point for a standalone worker process: run until
76
223
  * SIGINT/SIGTERM, then close the store. Use this at a script's top level;
@@ -83,20 +230,98 @@ export declare class Worker {
83
230
  concurrency?: number;
84
231
  }): Promise<T>;
85
232
  /**
86
- * Run one task to completion. Never rejects: a task-level failure is reported
87
- * through onError and the loop moves on. (It used to reject into a promise
88
- * nobody awaited — an unhandled rejection that took the process down.)
233
+ * Record a claimed task this worker cannot run. Reachable only if a name is
234
+ * unregistered mid-run — the claim filters on the registered names — so it does
235
+ * not start a handler or a heartbeat for a task it will not run.
236
+ */
237
+ private failNoHandler;
238
+ /**
239
+ * Run one handler call to completion — one task, or a whole batch. Never
240
+ * rejects: a task-level failure is reported through onError and the loop moves
241
+ * on. (It used to reject into a promise nobody awaited — an unhandled rejection
242
+ * that took the process down.)
243
+ *
244
+ * One lifecycle for both delivery modes, because a single-task handler *is* the
245
+ * one-element case: the same heartbeat covers the call, the same classifier
246
+ * reads its error, and the same rule settles what is left. Only two things vary
247
+ * — how the handler is invoked, and where a leftover task's result comes from —
248
+ * so those are the only two branches below.
249
+ *
250
+ * The contract is single: **when the handler returns, every task it did not
251
+ * settle itself is settled by how the call ended.** Returning succeeds them,
252
+ * throwing fails them (retryably, or as the thrown TaskError says). That is
253
+ * what keeps the ordinary cases free of bookkeeping — a handler that just
254
+ * returns has finished 256 tasks — while still letting it pick individual
255
+ * tasks off with `item.succeed()` / `item.fail()` as it goes.
256
+ *
257
+ * For a batch, a returned map of task id -> result fills in results for the
258
+ * tasks left over; anything else returned is ignored, and unmentioned tasks
259
+ * succeed with no result (the common shape, where the handler's output went to
260
+ * a database rather than into the task row). For a single task the return value
261
+ * simply is the result.
89
262
  */
90
- private execute;
263
+ private runCall;
91
264
  /**
92
- * Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
93
- * abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
94
- * a handler that keeps running can never write again — see
95
- * TaskContext.owned), then the still-pending promise is left to settle on
96
- * its own, its outcome discarded. The caller records the handler_timeout
97
- * failure; lease recovery is NOT involved, so redelivery is immediate.
265
+ * How an attempt that ended badly is recorded: [envelope, retryable], or null
266
+ * when there is nothing to record.
267
+ *
268
+ * One classifier for both delivery modes, so a handler error cannot mean
269
+ * different things depending on how its task happened to be delivered. That
270
+ * includes LostLease, which is not an outcome at all: it means a write through
271
+ * this context was already rejected, so recording anything more would be
272
+ * rejected too. Both modes then leave the task alone — the single-task one has
273
+ * nothing else to do, and a batch lets its remaining tasks fall to lease expiry
274
+ * and redelivery rather than stamping them with a failure the handler never
275
+ * reported.
276
+ */
277
+ private outcomeOf;
278
+ /**
279
+ * Finalize one task the handler left for the worker to decide — the tail of
280
+ * both delivery modes.
281
+ *
282
+ * Includes the unserializable-result rule: the handler succeeded but its value
283
+ * cannot cross the JSON protocol (BigInt, non-finite number, circular), which
284
+ * is deterministic, so it fails permanently rather than being redelivered to
285
+ * fail the same way every attempt.
286
+ */
287
+ private succeedOne;
288
+ /**
289
+ * Settle every task the handler did not settle itself, concurrently, reporting
290
+ * rather than throwing. Each task keeps its own attempt count and backoff —
291
+ * they are separate tasks that happened to be delivered together.
292
+ *
293
+ * allSettled, because one task's write failing must not abandon the rest of the
294
+ * batch mid-settlement — the others still hold leases and would sit `running`
295
+ * until expiry. Each outcome is reported against the task it belongs to;
296
+ * without that, an operator learns a settlement failed somewhere in a batch of
297
+ * 256.
298
+ */
299
+ private settleEach;
300
+ /**
301
+ * Run one attempt — of one task or of a whole batch — bounded by maxRunMs when
302
+ * set.
303
+ *
304
+ * On timeout the attempt is abandoned: every context it covers is flagged
305
+ * lease-lost first (ctx.signal aborts, and a handler that keeps running can
306
+ * never write again, nor settle anything behind the worker's back — see
307
+ * TaskContext.owned), then the still-pending promise is left to settle on its
308
+ * own, its outcome discarded. The caller records the handler_timeout failure;
309
+ * lease recovery is NOT involved, so redelivery is immediate.
98
310
  */
99
311
  private attempt;
312
+ /**
313
+ * One statement per beat, however many tasks the call covers — a single-task
314
+ * handler is just the one-element case.
315
+ *
316
+ * Only tasks still in play are renewed. A task the handler already settled is
317
+ * terminal, and re-leasing it would be a write against a row nobody owns; that
318
+ * is also why an absence is only read as lease loss after re-checking
319
+ * `settled`, since the handler may have finalized the task while this beat was
320
+ * in flight, which takes the row out of `running` and out of the reply. A task
321
+ * genuinely missing lost its lease (another worker recovered it), so its
322
+ * context is flagged and the handler stops being able to write through it —
323
+ * only that one, never its neighbours.
324
+ */
100
325
  private startHeartbeat;
101
326
  private safeFail;
102
327
  /**