cairnq 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -2
- package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
- package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
- package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
- package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
- package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
- package/dist/_protocol/sql/postgres/queue_depth.sql +22 -0
- package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
- package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
- package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
- package/dist/_protocol/sql/sqlite/queue_depth.sql +26 -0
- package/dist/backoff.d.ts +31 -0
- package/dist/backoff.js +40 -0
- package/dist/backpressure.d.ts +59 -0
- package/dist/backpressure.js +122 -0
- package/dist/client.d.ts +16 -3
- package/dist/client.js +19 -5
- package/dist/context.d.ts +48 -2
- package/dist/context.js +101 -10
- package/dist/errors.d.ts +37 -0
- package/dist/errors.js +60 -0
- package/dist/index.d.ts +7 -3
- package/dist/index.js +2 -1
- package/dist/store/base.d.ts +73 -0
- package/dist/store/base.js +124 -14
- package/dist/store/sqlite.d.ts +35 -0
- package/dist/store/sqlite.js +163 -3
- package/dist/worker.d.ts +238 -13
- package/dist/worker.js +512 -120
- package/package.json +2 -1
- package/src/backoff.ts +53 -0
- package/src/backpressure.ts +140 -0
- package/src/client.ts +33 -5
- package/src/context.ts +116 -9
- package/src/errors.ts +66 -0
- package/src/index.ts +7 -2
- package/src/store/base.ts +136 -13
- package/src/store/sqlite.ts +168 -2
- package/src/worker.ts +671 -132
package/dist/worker.d.ts
CHANGED
|
@@ -1,12 +1,35 @@
|
|
|
1
|
+
import type { BackpressureOptions } from "./backpressure.js";
|
|
1
2
|
import { TaskContext } from "./context.js";
|
|
2
3
|
import type { TaskStore } from "./store/base.js";
|
|
3
4
|
import { type TaskDef } from "./task.js";
|
|
5
|
+
export { retryDelayMs, DEFAULT_RETRY_BACKOFF_MS, DEFAULT_RETRY_BACKOFF_MAX_MS } from "./backoff.js";
|
|
4
6
|
export type Handler = (ctx: TaskContext, payload: any) => unknown | Promise<unknown>;
|
|
5
7
|
/** Handler typed against a TaskDef<P, R>: payload is P, the return is R. */
|
|
6
8
|
export type TypedHandler<P, R> = (ctx: TaskContext, payload: P) => R | Promise<R>;
|
|
9
|
+
/**
|
|
10
|
+
* A batch handler takes one argument: the list of contexts. There is no payload
|
|
11
|
+
* shortcut to pair with it — payloads are per task, so they are read off the
|
|
12
|
+
* items (`item.payload`), which is also what a handler must hold to settle one
|
|
13
|
+
* of them.
|
|
14
|
+
*
|
|
15
|
+
* Returning a map of task id -> result fills in results for the tasks the
|
|
16
|
+
* handler did not settle itself; anything else returned is ignored.
|
|
17
|
+
*/
|
|
18
|
+
export type BatchHandler = (items: TaskContext[]) => void | Record<string, unknown> | Promise<void | Record<string, unknown>>;
|
|
7
19
|
/** Where an error the worker recovered from came from. */
|
|
8
20
|
export type ErrorPhase = "claim" | "execute";
|
|
9
|
-
|
|
21
|
+
/**
|
|
22
|
+
* Backpressure is accepted here too, not only on CairnQ: a handler spawning
|
|
23
|
+
* children through TaskContext.submit is a producer, and in a worker process
|
|
24
|
+
* there is usually no CairnQ handle to have configured the store.
|
|
25
|
+
*/
|
|
26
|
+
export interface WorkerOptions extends Partial<BackpressureOptions> {
|
|
27
|
+
/**
|
|
28
|
+
* Handler calls allowed to run at once. A batch call counts as one, however
|
|
29
|
+
* many tasks it carries — size it for how much work you want in flight, not
|
|
30
|
+
* for how many tasks that comes to. Per-name limits refine it; `maxInFlightBytes`
|
|
31
|
+
* bounds memory, which task counts never did.
|
|
32
|
+
*/
|
|
10
33
|
concurrency?: number;
|
|
11
34
|
leaseMs?: number;
|
|
12
35
|
heartbeatIntervalMs?: number;
|
|
@@ -25,6 +48,39 @@ export interface WorkerOptions {
|
|
|
25
48
|
* a retryable `handler_timeout` failure. Unset disables the ceiling.
|
|
26
49
|
*/
|
|
27
50
|
maxRunMs?: number;
|
|
51
|
+
/**
|
|
52
|
+
* Resident payload bytes allowed across running handlers, independent of
|
|
53
|
+
* their count.
|
|
54
|
+
*
|
|
55
|
+
* `concurrency` bounds tasks, not memory, so a worker sized for small payloads
|
|
56
|
+
* holds concurrency * largest-payload bytes the moment a batch of big ones
|
|
57
|
+
* arrives — for payloads that carry media inline, that is the difference
|
|
58
|
+
* between megabytes and gigabytes resident. Once the budget is spent the
|
|
59
|
+
* worker stops claiming until running handlers give it back.
|
|
60
|
+
*
|
|
61
|
+
* The bound is on tasks already executing: it is read between claims, never
|
|
62
|
+
* during one, and a claim commits to its rows before any size is known. One
|
|
63
|
+
* poll can therefore overshoot by up to `claimBatch` rows per registered name
|
|
64
|
+
* (or one whole `batch`, whichever is larger). Lower `claimBatch`, or the batch
|
|
65
|
+
* sizes, to tighten that. A single payload larger than the entire budget still
|
|
66
|
+
* runs — alone, rather than deadlocking the worker.
|
|
67
|
+
*
|
|
68
|
+
* Costs one JSON serialization per task to measure, so it is only computed
|
|
69
|
+
* when set. Unset disables the budget.
|
|
70
|
+
*/
|
|
71
|
+
maxInFlightBytes?: number;
|
|
72
|
+
/**
|
|
73
|
+
* Call ceilings that several names can draw from, by name — `{ gpu: 1 }`.
|
|
74
|
+
* A handler joins one with `task(name, { resource: "gpu" }, fn)`.
|
|
75
|
+
*
|
|
76
|
+
* `concurrency` caps a name against itself, which cannot say what usually
|
|
77
|
+
* binds a worker doing heavy local work: several *different* handlers
|
|
78
|
+
* contending for one scarce thing — a GPU, an index that tolerates a single
|
|
79
|
+
* writer. The limit belongs to that thing rather than to any one name, so it
|
|
80
|
+
* is declared here, once, and at capacity 1 it is mutual exclusion across the
|
|
81
|
+
* names that join it.
|
|
82
|
+
*/
|
|
83
|
+
resources?: Record<string, number>;
|
|
28
84
|
/**
|
|
29
85
|
* Called for errors the worker survived — a claim that threw, a store write
|
|
30
86
|
* that failed while finalizing a task. Without it these are silent: the run
|
|
@@ -36,14 +92,35 @@ export interface WorkerOptions {
|
|
|
36
92
|
taskId?: string;
|
|
37
93
|
}) => void;
|
|
38
94
|
}
|
|
39
|
-
/** Exponential backoff for the next attempt of a task that just failed. */
|
|
40
|
-
export declare function retryDelayMs(attempt: number, baseMs: number, maxMs: number): number;
|
|
41
95
|
export declare class Worker {
|
|
42
96
|
private readonly store;
|
|
43
97
|
private readonly queues;
|
|
44
98
|
private readonly opts;
|
|
45
99
|
private readonly handlers;
|
|
46
100
|
private readonly workerId;
|
|
101
|
+
/** Payload bytes charged to running handlers — see maxInFlightBytes. */
|
|
102
|
+
private inFlightBytes;
|
|
103
|
+
/** Calls in flight, for the names that cap their own concurrency. */
|
|
104
|
+
private readonly callsInFlight;
|
|
105
|
+
/**
|
|
106
|
+
* Calls holding units of each declared resource. A resource is the same shape
|
|
107
|
+
* of budget as a name's own `concurrency` — a ceiling on calls — differing
|
|
108
|
+
* only in who draws from it: several names rather than one. That is what
|
|
109
|
+
* expresses "these handlers share one GPU" without inventing a queue per
|
|
110
|
+
* resource.
|
|
111
|
+
*/
|
|
112
|
+
private readonly resourceCalls;
|
|
113
|
+
/** Rotates which source is offered the free budget first — see loop(). */
|
|
114
|
+
private claimCursor;
|
|
115
|
+
/** Invalidated by task(); see schedule(). */
|
|
116
|
+
private scheduleCache;
|
|
117
|
+
/**
|
|
118
|
+
* Retry backoff, resolved once. Both settlement paths read these — the
|
|
119
|
+
* worker's own `safeFail` and the TaskContext it hands a handler — so
|
|
120
|
+
* resolving the defaults per call site is how the two drift apart.
|
|
121
|
+
*/
|
|
122
|
+
private readonly backoffMs;
|
|
123
|
+
private readonly backoffMaxMs;
|
|
47
124
|
private stopped;
|
|
48
125
|
private stopWake;
|
|
49
126
|
private readonly stopped$;
|
|
@@ -63,6 +140,30 @@ export declare class Worker {
|
|
|
63
140
|
task(handler: Handler): this;
|
|
64
141
|
task(name: string, handler: Handler): this;
|
|
65
142
|
task<P, R>(def: TaskDef<P, R>, handler: TypedHandler<P, R>): this;
|
|
143
|
+
/**
|
|
144
|
+
* Batch delivery: the handler takes one argument, a `TaskContext[]` of up to
|
|
145
|
+
* `batch` tasks, instead of `(ctx, payload)`. Use it when the work itself is
|
|
146
|
+
* batched — one embedding call over 256 texts rather than 256 calls — and size
|
|
147
|
+
* it by what the downstream API wants, not by the queue.
|
|
148
|
+
*
|
|
149
|
+
* `concurrency` caps the calls this name may run at once, under the worker's
|
|
150
|
+
* own. Use it to keep one expensive name from taking the whole worker.
|
|
151
|
+
*
|
|
152
|
+
* `resource` draws each call from a ceiling declared in
|
|
153
|
+
* `WorkerOptions.resources` and shared with every other name that names it —
|
|
154
|
+
* at capacity 1, mutual exclusion across those names.
|
|
155
|
+
*/
|
|
156
|
+
task(name: string | TaskDef, opts: {
|
|
157
|
+
batch: number;
|
|
158
|
+
concurrency?: number;
|
|
159
|
+
resource?: string;
|
|
160
|
+
}, handler: BatchHandler): this;
|
|
161
|
+
/** Per-name concurrency or a shared resource, without batching: the handler
|
|
162
|
+
* still takes `(ctx, payload)`. */
|
|
163
|
+
task(name: string | TaskDef, opts: {
|
|
164
|
+
concurrency?: number;
|
|
165
|
+
resource?: string;
|
|
166
|
+
}, handler: Handler): this;
|
|
66
167
|
stop(): void;
|
|
67
168
|
/** Close the underlying store connection. Call after run() returns. */
|
|
68
169
|
close(): Promise<void>;
|
|
@@ -71,6 +172,52 @@ export declare class Worker {
|
|
|
71
172
|
run(opts?: {
|
|
72
173
|
concurrency?: number;
|
|
73
174
|
}): Promise<void>;
|
|
175
|
+
/**
|
|
176
|
+
* Split one claim into handler calls, each with the registration to run it.
|
|
177
|
+
*
|
|
178
|
+
* A claim is filtered by queue and by the names this worker handles, so it
|
|
179
|
+
* comes back mixed; batch size is per name (one embedding call wants 256
|
|
180
|
+
* texts, one Docling parse wants exactly 1). So group by name, then chunk each
|
|
181
|
+
* group by that name's size. Names registered without `batch` come back as
|
|
182
|
+
* one-task calls, as do names not registered at all — reachable only if a
|
|
183
|
+
* handler is unregistered mid-run, and dispatched to failNoHandler().
|
|
184
|
+
*
|
|
185
|
+
* The registration rides along because this is where it was resolved; looking
|
|
186
|
+
* it up again at the call site would put "is this name batched" in two places.
|
|
187
|
+
*/
|
|
188
|
+
private deliveries;
|
|
189
|
+
private context;
|
|
190
|
+
/**
|
|
191
|
+
* How this poll's claim is split into per-name quotas, plus the union of names
|
|
192
|
+
* the probe spans.
|
|
193
|
+
*
|
|
194
|
+
* Cached and invalidated by `task()`, rather than rebuilt each poll: handlers
|
|
195
|
+
* may be registered after run() started, but only there, and this otherwise
|
|
196
|
+
* allocates a source per name on every tick for a worker's whole lifetime.
|
|
197
|
+
*
|
|
198
|
+
* A name that limits itself — by `batch`, by its own `concurrency`, or by a
|
|
199
|
+
* `resource` — needs a quota the shared draw cannot express, so it gets a
|
|
200
|
+
* source of its own; every other name shares one, where a task is a call.
|
|
201
|
+
*
|
|
202
|
+
* A resource is deliberately *not* one source spanning its names: `batch` is
|
|
203
|
+
* per name, and a single source carries one batch size, so two members that
|
|
204
|
+
* batch differently could not share a draw. Keeping a source per name and
|
|
205
|
+
* letting several of them draw down one shared ceiling composes with batching
|
|
206
|
+
* instead of excluding it.
|
|
207
|
+
*/
|
|
208
|
+
private schedule;
|
|
209
|
+
/**
|
|
210
|
+
* A source's own call ceiling for one poll, or undefined when only the
|
|
211
|
+
* worker-wide budget applies.
|
|
212
|
+
*
|
|
213
|
+
* Three independent ceilings, whichever binds first: the name's own concurrency
|
|
214
|
+
* less what it is already running; its resource's capacity less what is running
|
|
215
|
+
* *and* what earlier draws in this same poll already took (`taken`); and
|
|
216
|
+
* `claimBatch`. The last is a ceiling on *rows* per poll, so it converts at this
|
|
217
|
+
* source's batch size — and never below one call, or a `claimBatch` under some
|
|
218
|
+
* name's batch would stall that name outright.
|
|
219
|
+
*/
|
|
220
|
+
private sourceCalls;
|
|
74
221
|
private loop;
|
|
75
222
|
/** Blocking-style entry point for a standalone worker process: run until
|
|
76
223
|
* SIGINT/SIGTERM, then close the store. Use this at a script's top level;
|
|
@@ -83,20 +230,98 @@ export declare class Worker {
|
|
|
83
230
|
concurrency?: number;
|
|
84
231
|
}): Promise<T>;
|
|
85
232
|
/**
|
|
86
|
-
*
|
|
87
|
-
*
|
|
88
|
-
*
|
|
233
|
+
* Record a claimed task this worker cannot run. Reachable only if a name is
|
|
234
|
+
* unregistered mid-run — the claim filters on the registered names — so it does
|
|
235
|
+
* not start a handler or a heartbeat for a task it will not run.
|
|
236
|
+
*/
|
|
237
|
+
private failNoHandler;
|
|
238
|
+
/**
|
|
239
|
+
* Run one handler call to completion — one task, or a whole batch. Never
|
|
240
|
+
* rejects: a task-level failure is reported through onError and the loop moves
|
|
241
|
+
* on. (It used to reject into a promise nobody awaited — an unhandled rejection
|
|
242
|
+
* that took the process down.)
|
|
243
|
+
*
|
|
244
|
+
* One lifecycle for both delivery modes, because a single-task handler *is* the
|
|
245
|
+
* one-element case: the same heartbeat covers the call, the same classifier
|
|
246
|
+
* reads its error, and the same rule settles what is left. Only two things vary
|
|
247
|
+
* — how the handler is invoked, and where a leftover task's result comes from —
|
|
248
|
+
* so those are the only two branches below.
|
|
249
|
+
*
|
|
250
|
+
* The contract is single: **when the handler returns, every task it did not
|
|
251
|
+
* settle itself is settled by how the call ended.** Returning succeeds them,
|
|
252
|
+
* throwing fails them (retryably, or as the thrown TaskError says). That is
|
|
253
|
+
* what keeps the ordinary cases free of bookkeeping — a handler that just
|
|
254
|
+
* returns has finished 256 tasks — while still letting it pick individual
|
|
255
|
+
* tasks off with `item.succeed()` / `item.fail()` as it goes.
|
|
256
|
+
*
|
|
257
|
+
* For a batch, a returned map of task id -> result fills in results for the
|
|
258
|
+
* tasks left over; anything else returned is ignored, and unmentioned tasks
|
|
259
|
+
* succeed with no result (the common shape, where the handler's output went to
|
|
260
|
+
* a database rather than into the task row). For a single task the return value
|
|
261
|
+
* simply is the result.
|
|
89
262
|
*/
|
|
90
|
-
private
|
|
263
|
+
private runCall;
|
|
91
264
|
/**
|
|
92
|
-
*
|
|
93
|
-
*
|
|
94
|
-
*
|
|
95
|
-
*
|
|
96
|
-
*
|
|
97
|
-
*
|
|
265
|
+
* How an attempt that ended badly is recorded: [envelope, retryable], or null
|
|
266
|
+
* when there is nothing to record.
|
|
267
|
+
*
|
|
268
|
+
* One classifier for both delivery modes, so a handler error cannot mean
|
|
269
|
+
* different things depending on how its task happened to be delivered. That
|
|
270
|
+
* includes LostLease, which is not an outcome at all: it means a write through
|
|
271
|
+
* this context was already rejected, so recording anything more would be
|
|
272
|
+
* rejected too. Both modes then leave the task alone — the single-task one has
|
|
273
|
+
* nothing else to do, and a batch lets its remaining tasks fall to lease expiry
|
|
274
|
+
* and redelivery rather than stamping them with a failure the handler never
|
|
275
|
+
* reported.
|
|
276
|
+
*/
|
|
277
|
+
private outcomeOf;
|
|
278
|
+
/**
|
|
279
|
+
* Finalize one task the handler left for the worker to decide — the tail of
|
|
280
|
+
* both delivery modes.
|
|
281
|
+
*
|
|
282
|
+
* Includes the unserializable-result rule: the handler succeeded but its value
|
|
283
|
+
* cannot cross the JSON protocol (BigInt, non-finite number, circular), which
|
|
284
|
+
* is deterministic, so it fails permanently rather than being redelivered to
|
|
285
|
+
* fail the same way every attempt.
|
|
286
|
+
*/
|
|
287
|
+
private succeedOne;
|
|
288
|
+
/**
|
|
289
|
+
* Settle every task the handler did not settle itself, concurrently, reporting
|
|
290
|
+
* rather than throwing. Each task keeps its own attempt count and backoff —
|
|
291
|
+
* they are separate tasks that happened to be delivered together.
|
|
292
|
+
*
|
|
293
|
+
* allSettled, because one task's write failing must not abandon the rest of the
|
|
294
|
+
* batch mid-settlement — the others still hold leases and would sit `running`
|
|
295
|
+
* until expiry. Each outcome is reported against the task it belongs to;
|
|
296
|
+
* without that, an operator learns a settlement failed somewhere in a batch of
|
|
297
|
+
* 256.
|
|
298
|
+
*/
|
|
299
|
+
private settleEach;
|
|
300
|
+
/**
|
|
301
|
+
* Run one attempt — of one task or of a whole batch — bounded by maxRunMs when
|
|
302
|
+
* set.
|
|
303
|
+
*
|
|
304
|
+
* On timeout the attempt is abandoned: every context it covers is flagged
|
|
305
|
+
* lease-lost first (ctx.signal aborts, and a handler that keeps running can
|
|
306
|
+
* never write again, nor settle anything behind the worker's back — see
|
|
307
|
+
* TaskContext.owned), then the still-pending promise is left to settle on its
|
|
308
|
+
* own, its outcome discarded. The caller records the handler_timeout failure;
|
|
309
|
+
* lease recovery is NOT involved, so redelivery is immediate.
|
|
98
310
|
*/
|
|
99
311
|
private attempt;
|
|
312
|
+
/**
|
|
313
|
+
* One statement per beat, however many tasks the call covers — a single-task
|
|
314
|
+
* handler is just the one-element case.
|
|
315
|
+
*
|
|
316
|
+
* Only tasks still in play are renewed. A task the handler already settled is
|
|
317
|
+
* terminal, and re-leasing it would be a write against a row nobody owns; that
|
|
318
|
+
* is also why an absence is only read as lease loss after re-checking
|
|
319
|
+
* `settled`, since the handler may have finalized the task while this beat was
|
|
320
|
+
* in flight, which takes the row out of `running` and out of the reply. A task
|
|
321
|
+
* genuinely missing lost its lease (another worker recovered it), so its
|
|
322
|
+
* context is flagged and the handler stops being able to write through it —
|
|
323
|
+
* only that one, never its neighbours.
|
|
324
|
+
*/
|
|
100
325
|
private startHeartbeat;
|
|
101
326
|
private safeFail;
|
|
102
327
|
/**
|