cairnq 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -5
- package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
- package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
- package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
- package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
- package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
- package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
- package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
- package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
- package/dist/backoff.d.ts +31 -0
- package/dist/backoff.js +40 -0
- package/dist/client.d.ts +28 -2
- package/dist/client.js +30 -3
- package/dist/context.d.ts +48 -2
- package/dist/context.js +101 -10
- package/dist/errors.d.ts +60 -4
- package/dist/errors.js +94 -9
- package/dist/index.d.ts +6 -2
- package/dist/index.js +2 -1
- package/dist/retention.d.ts +60 -0
- package/dist/retention.js +115 -0
- package/dist/store/base.d.ts +50 -1
- package/dist/store/base.js +114 -19
- package/dist/store/sqlite.js +4 -1
- package/dist/wait.d.ts +20 -5
- package/dist/wait.js +34 -9
- package/dist/worker.d.ts +214 -16
- package/dist/worker.js +500 -131
- package/package.json +1 -1
- package/src/backoff.ts +53 -0
- package/src/client.ts +43 -5
- package/src/context.ts +116 -9
- package/src/errors.ts +101 -9
- package/src/index.ts +6 -1
- package/src/retention.ts +136 -0
- package/src/store/base.ts +121 -17
- package/src/store/sqlite.ts +4 -1
- package/src/wait.ts +60 -16
- package/src/worker.ts +640 -146
package/dist/store/base.d.ts
CHANGED
|
@@ -11,7 +11,7 @@ export declare function dumpJson(value: unknown): string;
|
|
|
11
11
|
* The supported major is a protocol fact, not a dialect one — every backend
|
|
12
12
|
* checks it here so the constant can't fork per store. */
|
|
13
13
|
export declare function checkProtocolVersion(version: number): void;
|
|
14
|
-
declare const CONFLICTS: readonly ["reuse", "reject", "replace"];
|
|
14
|
+
declare const CONFLICTS: readonly ["reuse", "reuse-succeeded", "reject", "replace"];
|
|
15
15
|
export type Conflict = (typeof CONFLICTS)[number];
|
|
16
16
|
/** The queue a submit lands on when it names none. Owned here, where the
|
|
17
17
|
* default is applied, so nothing above has to re-derive it. */
|
|
@@ -174,11 +174,60 @@ export declare abstract class TaskStore {
|
|
|
174
174
|
limit?: number;
|
|
175
175
|
names?: string[];
|
|
176
176
|
}): Promise<Task[]>;
|
|
177
|
+
/**
|
|
178
|
+
* Open one claim transaction and let the caller draw from it repeatedly.
|
|
179
|
+
*
|
|
180
|
+
* The transaction is what has to live here: the read-only probe that keeps an
|
|
181
|
+
* idle worker off SQLite's single write lock, the `recover_leases` whose
|
|
182
|
+
* reclaimed leases must be visible to the claims that follow and to nobody in
|
|
183
|
+
* between, and the write lock itself. *What* gets claimed under it is the
|
|
184
|
+
* caller's business — a worker drawing a separate quota per task name is
|
|
185
|
+
* scheduling policy, and this layer has no vocabulary for the "handler call"
|
|
186
|
+
* that policy is denominated in. It knows queues, names, limits and rows.
|
|
187
|
+
*
|
|
188
|
+
* `plan` is handed a `claim(names, limit)` it may call any number of times,
|
|
189
|
+
* each a separate statement under the same lock and the same recovery, and
|
|
190
|
+
* each free to size itself from what the previous one returned. That feedback
|
|
191
|
+
* is the reason this is a callback rather than a list of quotas: a caller
|
|
192
|
+
* dividing a budget up front has to guess, and every share handed to a name
|
|
193
|
+
* with nothing queued is a slot left idle until the next poll.
|
|
194
|
+
*
|
|
195
|
+
* `plan` runs with the write lock held, so it must await nothing but that
|
|
196
|
+
* callback.
|
|
197
|
+
*
|
|
198
|
+
* `names` is the union `plan` might ask for — the probe and the recovery are
|
|
199
|
+
* filtered by it. Returns undefined when the probe finds nothing claimable, in
|
|
200
|
+
* which case `plan` never runs and no transaction is opened.
|
|
201
|
+
*/
|
|
202
|
+
claimSession<T>(input: {
|
|
203
|
+
queues: string[];
|
|
204
|
+
workerId: string;
|
|
205
|
+
leaseMs?: number;
|
|
206
|
+
names: string[] | null;
|
|
207
|
+
}, plan: (claim: (names: string[] | null, limit: number) => Promise<Task[]>) => Promise<T>): Promise<T | undefined>;
|
|
177
208
|
heartbeat(input: {
|
|
178
209
|
taskId: string;
|
|
179
210
|
workerId: string;
|
|
180
211
|
leaseMs?: number;
|
|
181
212
|
}): Promise<Task>;
|
|
213
|
+
/**
|
|
214
|
+
* Renew several leases in one statement. Returns `taskId -> cancel requested`
|
|
215
|
+
* for the tasks this worker still holds.
|
|
216
|
+
*
|
|
217
|
+
* Deliberately not an ownedWrite: ownership is per task here, so there is no
|
|
218
|
+
* single answer to "did it work". A task **absent** from the result lost its
|
|
219
|
+
* lease, and the caller decides what that means for that one task rather than
|
|
220
|
+
* failing the whole beat.
|
|
221
|
+
*
|
|
222
|
+
* It returns flags rather than Tasks because nothing downstream needs a task:
|
|
223
|
+
* the caller renews leases and observes cancellation, and whole rows would drag
|
|
224
|
+
* every payload back on every beat for the life of the call.
|
|
225
|
+
*/
|
|
226
|
+
heartbeatBatch(input: {
|
|
227
|
+
taskIds: string[];
|
|
228
|
+
workerId: string;
|
|
229
|
+
leaseMs?: number;
|
|
230
|
+
}): Promise<Map<string, boolean>>;
|
|
182
231
|
progress(input: {
|
|
183
232
|
taskId: string;
|
|
184
233
|
workerId: string;
|
package/dist/store/base.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { newId } from "../ids.js";
|
|
2
2
|
import { AlreadyExists, errorEnvelope, LostLease, ProtocolVersionMismatch, SerializationError, } from "../errors.js";
|
|
3
|
-
import { rowToTask, STATUSES } from "../models.js";
|
|
3
|
+
import { rowToTask, STATUSES, TERMINAL } from "../models.js";
|
|
4
4
|
import { QueueDepthGate } from "../backpressure.js";
|
|
5
5
|
const rejectMangled = function (_key, v) {
|
|
6
6
|
if (typeof v === "number" && !Number.isFinite(v)) {
|
|
@@ -51,7 +51,24 @@ export function checkProtocolVersion(version) {
|
|
|
51
51
|
// CONFLICTS is the canonical declaration; the type derives from it so the
|
|
52
52
|
// runtime guard in submit() and the type can't drift apart (same pattern as
|
|
53
53
|
// STATUSES/TaskStatus in models.ts).
|
|
54
|
-
const CONFLICTS = ["reuse", "reject", "replace"];
|
|
54
|
+
const CONFLICTS = ["reuse", "reuse-succeeded", "reject", "replace"];
|
|
55
|
+
/**
|
|
56
|
+
* Whether a keyed submit's strategy accepts the task the key already points at.
|
|
57
|
+
*
|
|
58
|
+
* Both reuse strategies deduplicate work that is still in play — that is what a
|
|
59
|
+
* key is for, and the answer cannot depend on the outcome of a task that has no
|
|
60
|
+
* outcome yet. They differ only on what a *finished* task means: `reuse` treats
|
|
61
|
+
* the key as free again, while `reuse-succeeded` reads a succeeded task as a
|
|
62
|
+
* cached result. Neither ever hands back a failed or canceled one, which would
|
|
63
|
+
* poison the key for every later submit (see PROTOCOL.md "Key conflict").
|
|
64
|
+
*/
|
|
65
|
+
function reusable(conflict, status) {
|
|
66
|
+
if (conflict === "replace")
|
|
67
|
+
return false;
|
|
68
|
+
if (!TERMINAL.includes(status))
|
|
69
|
+
return true;
|
|
70
|
+
return conflict === "reuse-succeeded" && status === "succeeded";
|
|
71
|
+
}
|
|
55
72
|
/** The queue a submit lands on when it names none. Owned here, where the
|
|
56
73
|
* default is applied, so nothing above has to re-derive it. */
|
|
57
74
|
export const DEFAULT_QUEUE = "default";
|
|
@@ -207,12 +224,17 @@ export class TaskStore {
|
|
|
207
224
|
// free after all, whatever the strategy.
|
|
208
225
|
const current = (await fetch("get", { id: existing[0].task_id }))[0];
|
|
209
226
|
if (current) {
|
|
210
|
-
if (conflict === "reuse")
|
|
211
|
-
return rowToTask(current);
|
|
212
227
|
if (conflict === "reject")
|
|
213
228
|
throw new AlreadyExists(key);
|
|
214
|
-
|
|
215
|
-
|
|
229
|
+
if (reusable(conflict, current.status))
|
|
230
|
+
return rowToTask(current);
|
|
231
|
+
// The strategy declined the recorded task, so the key repoints to the
|
|
232
|
+
// fresh one inserted below. Cancel only what is still live: a terminal
|
|
233
|
+
// task has nothing to stop, and cancelling it would rewrite a settled
|
|
234
|
+
// row (and hand a `canceled` back to whoever is waiting on it).
|
|
235
|
+
if (!TERMINAL.includes(current.status)) {
|
|
236
|
+
await fetch("cancel", { id: existing[0].task_id });
|
|
237
|
+
}
|
|
216
238
|
}
|
|
217
239
|
}
|
|
218
240
|
const row = (await fetch("insert_task", ins))[0];
|
|
@@ -330,27 +352,77 @@ export class TaskStore {
|
|
|
330
352
|
* array claims nothing.
|
|
331
353
|
*/
|
|
332
354
|
async claim(input) {
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
355
|
+
const names = input.names ?? null;
|
|
356
|
+
const claimed = await this.claimSession({ queues: input.queues, workerId: input.workerId, leaseMs: input.leaseMs, names }, (claim) => claim(names, input.limit ?? 1));
|
|
357
|
+
return claimed ?? [];
|
|
358
|
+
}
|
|
359
|
+
/**
|
|
360
|
+
* Open one claim transaction and let the caller draw from it repeatedly.
|
|
361
|
+
*
|
|
362
|
+
* The transaction is what has to live here: the read-only probe that keeps an
|
|
363
|
+
* idle worker off SQLite's single write lock, the `recover_leases` whose
|
|
364
|
+
* reclaimed leases must be visible to the claims that follow and to nobody in
|
|
365
|
+
* between, and the write lock itself. *What* gets claimed under it is the
|
|
366
|
+
* caller's business — a worker drawing a separate quota per task name is
|
|
367
|
+
* scheduling policy, and this layer has no vocabulary for the "handler call"
|
|
368
|
+
* that policy is denominated in. It knows queues, names, limits and rows.
|
|
369
|
+
*
|
|
370
|
+
* `plan` is handed a `claim(names, limit)` it may call any number of times,
|
|
371
|
+
* each a separate statement under the same lock and the same recovery, and
|
|
372
|
+
* each free to size itself from what the previous one returned. That feedback
|
|
373
|
+
* is the reason this is a callback rather than a list of quotas: a caller
|
|
374
|
+
* dividing a budget up front has to guess, and every share handed to a name
|
|
375
|
+
* with nothing queued is a slot left idle until the next poll.
|
|
376
|
+
*
|
|
377
|
+
* `plan` runs with the write lock held, so it must await nothing but that
|
|
378
|
+
* callback.
|
|
379
|
+
*
|
|
380
|
+
* `names` is the union `plan` might ask for — the probe and the recovery are
|
|
381
|
+
* filtered by it. Returns undefined when the probe finds nothing claimable, in
|
|
382
|
+
* which case `plan` never runs and no transaction is opened.
|
|
383
|
+
*/
|
|
384
|
+
async claimSession(input, plan) {
|
|
385
|
+
// A list-valued filter cannot be read in claim order, so the planner sorts
|
|
386
|
+
// every claimable row to take LIMIT of them and the claim's cost grows with
|
|
387
|
+
// the backlog while it holds the transaction. Both filters therefore have an
|
|
388
|
+
// equality form, picked per draw: one queue is the common deployment, and one
|
|
389
|
+
// name is every per-name quota. See claim_one_queue.sql and claim_one_name.sql.
|
|
337
390
|
const oneQueue = input.queues.length === 1;
|
|
338
|
-
const
|
|
391
|
+
const base = {
|
|
339
392
|
queues: input.queues,
|
|
340
393
|
queue: oneQueue ? input.queues[0] : null,
|
|
341
|
-
names: input.names
|
|
394
|
+
names: input.names,
|
|
395
|
+
name: null,
|
|
342
396
|
worker_id: input.workerId,
|
|
343
397
|
lease_ms: input.leaseMs ?? 30_000,
|
|
344
|
-
limit:
|
|
398
|
+
limit: 1,
|
|
345
399
|
lease_expired_error: LEASE_EXPIRED_ERROR_JSON,
|
|
346
400
|
};
|
|
347
|
-
if (!(await this.hasClaimableWork(
|
|
348
|
-
return
|
|
349
|
-
// Recovery must share the claim's transaction: a lease reclaimed here has to
|
|
350
|
-
// be visible to the claim that follows, and to nobody in between.
|
|
401
|
+
if (!(await this.hasClaimableWork(base)))
|
|
402
|
+
return undefined;
|
|
351
403
|
return this.tx(async (fetch) => {
|
|
352
|
-
await fetch("recover_leases",
|
|
353
|
-
return (
|
|
404
|
+
await fetch("recover_leases", base);
|
|
405
|
+
return plan(async (names, limit) => {
|
|
406
|
+
// A draw asking for nothing, or filtered to no names, claims nothing —
|
|
407
|
+
// answer it here rather than spending a statement to learn that.
|
|
408
|
+
if (limit <= 0 || names?.length === 0)
|
|
409
|
+
return [];
|
|
410
|
+
const oneName = names?.length === 1;
|
|
411
|
+
const statement = oneName
|
|
412
|
+
? oneQueue
|
|
413
|
+
? "claim_one_queue_one_name"
|
|
414
|
+
: "claim_one_name"
|
|
415
|
+
: oneQueue
|
|
416
|
+
? "claim_one_queue"
|
|
417
|
+
: "claim";
|
|
418
|
+
const rows = await fetch(statement, {
|
|
419
|
+
...base,
|
|
420
|
+
names,
|
|
421
|
+
name: oneName ? names[0] : null,
|
|
422
|
+
limit,
|
|
423
|
+
});
|
|
424
|
+
return rows.map(rowToTask);
|
|
425
|
+
});
|
|
354
426
|
});
|
|
355
427
|
}
|
|
356
428
|
async heartbeat(input) {
|
|
@@ -360,6 +432,29 @@ export class TaskStore {
|
|
|
360
432
|
lease_ms: input.leaseMs ?? 30_000,
|
|
361
433
|
});
|
|
362
434
|
}
|
|
435
|
+
/**
|
|
436
|
+
* Renew several leases in one statement. Returns `taskId -> cancel requested`
|
|
437
|
+
* for the tasks this worker still holds.
|
|
438
|
+
*
|
|
439
|
+
* Deliberately not an ownedWrite: ownership is per task here, so there is no
|
|
440
|
+
* single answer to "did it work". A task **absent** from the result lost its
|
|
441
|
+
* lease, and the caller decides what that means for that one task rather than
|
|
442
|
+
* failing the whole beat.
|
|
443
|
+
*
|
|
444
|
+
* It returns flags rather than Tasks because nothing downstream needs a task:
|
|
445
|
+
* the caller renews leases and observes cancellation, and whole rows would drag
|
|
446
|
+
* every payload back on every beat for the life of the call.
|
|
447
|
+
*/
|
|
448
|
+
async heartbeatBatch(input) {
|
|
449
|
+
if (!input.taskIds.length)
|
|
450
|
+
return new Map();
|
|
451
|
+
const rows = await this.fetch("heartbeat_batch", {
|
|
452
|
+
ids: input.taskIds,
|
|
453
|
+
worker_id: input.workerId,
|
|
454
|
+
lease_ms: input.leaseMs ?? 30_000,
|
|
455
|
+
});
|
|
456
|
+
return new Map(rows.map((r) => [r.id, r.cancel_requested_at_ms != null]));
|
|
457
|
+
}
|
|
363
458
|
async progress(input) {
|
|
364
459
|
return this.ownedWrite("progress", input.taskId, {
|
|
365
460
|
id: input.taskId,
|
package/dist/store/sqlite.js
CHANGED
|
@@ -313,7 +313,10 @@ export class SQLiteStore extends TaskStore {
|
|
|
313
313
|
bound[name] = now - params.older_than_ms;
|
|
314
314
|
break;
|
|
315
315
|
case "queues":
|
|
316
|
-
|
|
316
|
+
case "ids":
|
|
317
|
+
// json_each needs a JSON array. Postgres binds the array itself as
|
|
318
|
+
// text[], so only this dialect encodes.
|
|
319
|
+
bound[name] = JSON.stringify(params[name]);
|
|
317
320
|
break;
|
|
318
321
|
case "names":
|
|
319
322
|
// json_each needs a JSON array; null stays null so the SQL's
|
package/dist/wait.d.ts
CHANGED
|
@@ -2,6 +2,11 @@ import { type Task } from "./models.js";
|
|
|
2
2
|
import type { TaskStore } from "./store/base.js";
|
|
3
3
|
export declare const DEFAULT_POLL_MS = 100;
|
|
4
4
|
export declare const MAX_POLL_MS = 500;
|
|
5
|
+
export interface PollOptions {
|
|
6
|
+
timeoutMs: number;
|
|
7
|
+
pollMs?: number;
|
|
8
|
+
maxPollMs?: number;
|
|
9
|
+
}
|
|
5
10
|
/**
|
|
6
11
|
* Grow the polling interval towards the ceiling.
|
|
7
12
|
*
|
|
@@ -14,8 +19,18 @@ export declare function nextPollMs(current: number, maxMs: number): number;
|
|
|
14
19
|
/** Poll get() until terminal or timeout. Returns the terminal Task (any status).
|
|
15
20
|
* Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
|
|
16
21
|
* it backs off towards `maxPollMs`. */
|
|
17
|
-
export declare function pollWait(store: TaskStore, taskId: string,
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
+
export declare function pollWait(store: TaskStore, taskId: string, opts: PollOptions): Promise<Task>;
|
|
23
|
+
/**
|
|
24
|
+
* The same wait, following a key instead of an id.
|
|
25
|
+
*
|
|
26
|
+
* The key is re-resolved on every read, because that is what a key means: a
|
|
27
|
+
* pointer to the task that is *current* under it. A `replace` landing mid-wait
|
|
28
|
+
* moves the wait onto the new task rather than reporting the cancellation of the
|
|
29
|
+
* old one, and a key that points at nothing yet is simply not finished — it
|
|
30
|
+
* polls until something appears, the same way waiting on an id that does not
|
|
31
|
+
* exist yet does.
|
|
32
|
+
*
|
|
33
|
+
* There is nothing to subscribe to before the key resolves, so those naps are
|
|
34
|
+
* plain sleeps; once it resolves, the store's push channel applies as usual.
|
|
35
|
+
*/
|
|
36
|
+
export declare function pollWaitByKey(store: TaskStore, key: string, opts: PollOptions): Promise<Task>;
|
package/dist/wait.js
CHANGED
|
@@ -4,6 +4,7 @@ import { isTerminal } from "./models.js";
|
|
|
4
4
|
export const DEFAULT_POLL_MS = 100;
|
|
5
5
|
export const MAX_POLL_MS = 500;
|
|
6
6
|
const GROWTH = 1.5;
|
|
7
|
+
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
|
|
7
8
|
/**
|
|
8
9
|
* Grow the polling interval towards the ceiling.
|
|
9
10
|
*
|
|
@@ -15,22 +16,46 @@ const GROWTH = 1.5;
|
|
|
15
16
|
export function nextPollMs(current, maxMs) {
|
|
16
17
|
return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
|
|
17
18
|
}
|
|
18
|
-
/**
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
|
|
19
|
+
/**
|
|
20
|
+
* Poll `read` until it yields a terminal task, or the timeout elapses.
|
|
21
|
+
*
|
|
22
|
+
* `wake` is what the loop sleeps on between reads: a store with a push channel
|
|
23
|
+
* (Postgres) cuts it short when the task goes terminal, but the re-read is the
|
|
24
|
+
* source of truth either way, so a plain sleep is always a correct answer.
|
|
25
|
+
*/
|
|
26
|
+
async function poll(read, wake, subject, key, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS }) {
|
|
22
27
|
const deadline = nowMs() + timeoutMs;
|
|
23
28
|
let interval = pollMs;
|
|
24
29
|
for (;;) {
|
|
25
|
-
const task = await
|
|
30
|
+
const task = await read();
|
|
26
31
|
if (task && isTerminal(task))
|
|
27
32
|
return task;
|
|
28
33
|
const remaining = deadline - nowMs();
|
|
29
34
|
if (remaining <= 0)
|
|
30
|
-
throw new TaskTimeout(
|
|
31
|
-
|
|
32
|
-
// goes terminal; the re-get above stays the source of truth either way.
|
|
33
|
-
await store.taskDoneWake(taskId, Math.min(interval, remaining));
|
|
35
|
+
throw new TaskTimeout(task?.id ?? subject, { timeoutMs, task, key });
|
|
36
|
+
await wake(task, Math.min(interval, remaining));
|
|
34
37
|
interval = nextPollMs(interval, maxPollMs);
|
|
35
38
|
}
|
|
36
39
|
}
|
|
40
|
+
/** Poll get() until terminal or timeout. Returns the terminal Task (any status).
|
|
41
|
+
* Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
|
|
42
|
+
* it backs off towards `maxPollMs`. */
|
|
43
|
+
export function pollWait(store, taskId, opts) {
|
|
44
|
+
return poll(() => store.get(taskId), (_task, ms) => store.taskDoneWake(taskId, ms), taskId, null, opts);
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* The same wait, following a key instead of an id.
|
|
48
|
+
*
|
|
49
|
+
* The key is re-resolved on every read, because that is what a key means: a
|
|
50
|
+
* pointer to the task that is *current* under it. A `replace` landing mid-wait
|
|
51
|
+
* moves the wait onto the new task rather than reporting the cancellation of the
|
|
52
|
+
* old one, and a key that points at nothing yet is simply not finished — it
|
|
53
|
+
* polls until something appears, the same way waiting on an id that does not
|
|
54
|
+
* exist yet does.
|
|
55
|
+
*
|
|
56
|
+
* There is nothing to subscribe to before the key resolves, so those naps are
|
|
57
|
+
* plain sleeps; once it resolves, the store's push channel applies as usual.
|
|
58
|
+
*/
|
|
59
|
+
export function pollWaitByKey(store, key, opts) {
|
|
60
|
+
return poll(() => store.getByKey(key), (task, ms) => (task ? store.taskDoneWake(task.id, ms) : sleep(ms)), key, key, opts);
|
|
61
|
+
}
|
package/dist/worker.d.ts
CHANGED
|
@@ -2,9 +2,20 @@ import type { BackpressureOptions } from "./backpressure.js";
|
|
|
2
2
|
import { TaskContext } from "./context.js";
|
|
3
3
|
import type { TaskStore } from "./store/base.js";
|
|
4
4
|
import { type TaskDef } from "./task.js";
|
|
5
|
+
export { retryDelayMs, DEFAULT_RETRY_BACKOFF_MS, DEFAULT_RETRY_BACKOFF_MAX_MS } from "./backoff.js";
|
|
5
6
|
export type Handler = (ctx: TaskContext, payload: any) => unknown | Promise<unknown>;
|
|
6
7
|
/** Handler typed against a TaskDef<P, R>: payload is P, the return is R. */
|
|
7
8
|
export type TypedHandler<P, R> = (ctx: TaskContext, payload: P) => R | Promise<R>;
|
|
9
|
+
/**
|
|
10
|
+
* A batch handler takes one argument: the list of contexts. There is no payload
|
|
11
|
+
* shortcut to pair with it — payloads are per task, so they are read off the
|
|
12
|
+
* items (`item.payload`), which is also what a handler must hold to settle one
|
|
13
|
+
* of them.
|
|
14
|
+
*
|
|
15
|
+
* Returning a map of task id -> result fills in results for the tasks the
|
|
16
|
+
* handler did not settle itself; anything else returned is ignored.
|
|
17
|
+
*/
|
|
18
|
+
export type BatchHandler = (items: TaskContext[]) => void | Record<string, unknown> | Promise<void | Record<string, unknown>>;
|
|
8
19
|
/** Where an error the worker recovered from came from. */
|
|
9
20
|
export type ErrorPhase = "claim" | "execute";
|
|
10
21
|
/**
|
|
@@ -13,6 +24,12 @@ export type ErrorPhase = "claim" | "execute";
|
|
|
13
24
|
* there is usually no CairnQ handle to have configured the store.
|
|
14
25
|
*/
|
|
15
26
|
export interface WorkerOptions extends Partial<BackpressureOptions> {
|
|
27
|
+
/**
|
|
28
|
+
* Handler calls allowed to run at once. A batch call counts as one, however
|
|
29
|
+
* many tasks it carries — size it for how much work you want in flight, not
|
|
30
|
+
* for how many tasks that comes to. Per-name limits refine it; `maxInFlightBytes`
|
|
31
|
+
* bounds memory, which task counts never did.
|
|
32
|
+
*/
|
|
16
33
|
concurrency?: number;
|
|
17
34
|
leaseMs?: number;
|
|
18
35
|
heartbeatIntervalMs?: number;
|
|
@@ -41,15 +58,29 @@ export interface WorkerOptions extends Partial<BackpressureOptions> {
|
|
|
41
58
|
* between megabytes and gigabytes resident. Once the budget is spent the
|
|
42
59
|
* worker stops claiming until running handlers give it back.
|
|
43
60
|
*
|
|
44
|
-
* The bound is on tasks already executing
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
61
|
+
* The bound is on tasks already executing: it is read between claims, never
|
|
62
|
+
* during one, and a claim commits to its rows before any size is known. One
|
|
63
|
+
* poll can therefore overshoot by up to `claimBatch` rows per registered name
|
|
64
|
+
* (or one whole `batch`, whichever is larger). Lower `claimBatch`, or the batch
|
|
65
|
+
* sizes, to tighten that. A single payload larger than the entire budget still
|
|
66
|
+
* runs — alone, rather than deadlocking the worker.
|
|
48
67
|
*
|
|
49
68
|
* Costs one JSON serialization per task to measure, so it is only computed
|
|
50
69
|
* when set. Unset disables the budget.
|
|
51
70
|
*/
|
|
52
71
|
maxInFlightBytes?: number;
|
|
72
|
+
/**
|
|
73
|
+
* Call ceilings that several names can draw from, by name — `{ gpu: 1 }`.
|
|
74
|
+
* A handler joins one with `task(name, { resource: "gpu" }, fn)`.
|
|
75
|
+
*
|
|
76
|
+
* `concurrency` caps a name against itself, which cannot say what usually
|
|
77
|
+
* binds a worker doing heavy local work: several *different* handlers
|
|
78
|
+
* contending for one scarce thing — a GPU, an index that tolerates a single
|
|
79
|
+
* writer. The limit belongs to that thing rather than to any one name, so it
|
|
80
|
+
* is declared here, once, and at capacity 1 it is mutual exclusion across the
|
|
81
|
+
* names that join it.
|
|
82
|
+
*/
|
|
83
|
+
resources?: Record<string, number>;
|
|
53
84
|
/**
|
|
54
85
|
* Called for errors the worker survived — a claim that threw, a store write
|
|
55
86
|
* that failed while finalizing a task. Without it these are silent: the run
|
|
@@ -61,8 +92,6 @@ export interface WorkerOptions extends Partial<BackpressureOptions> {
|
|
|
61
92
|
taskId?: string;
|
|
62
93
|
}) => void;
|
|
63
94
|
}
|
|
64
|
-
/** Exponential backoff for the next attempt of a task that just failed. */
|
|
65
|
-
export declare function retryDelayMs(attempt: number, baseMs: number, maxMs: number): number;
|
|
66
95
|
export declare class Worker {
|
|
67
96
|
private readonly store;
|
|
68
97
|
private readonly queues;
|
|
@@ -71,6 +100,27 @@ export declare class Worker {
|
|
|
71
100
|
private readonly workerId;
|
|
72
101
|
/** Payload bytes charged to running handlers — see maxInFlightBytes. */
|
|
73
102
|
private inFlightBytes;
|
|
103
|
+
/** Calls in flight, for the names that cap their own concurrency. */
|
|
104
|
+
private readonly callsInFlight;
|
|
105
|
+
/**
|
|
106
|
+
* Calls holding units of each declared resource. A resource is the same shape
|
|
107
|
+
* of budget as a name's own `concurrency` — a ceiling on calls — differing
|
|
108
|
+
* only in who draws from it: several names rather than one. That is what
|
|
109
|
+
* expresses "these handlers share one GPU" without inventing a queue per
|
|
110
|
+
* resource.
|
|
111
|
+
*/
|
|
112
|
+
private readonly resourceCalls;
|
|
113
|
+
/** Rotates which source is offered the free budget first — see loop(). */
|
|
114
|
+
private claimCursor;
|
|
115
|
+
/** Invalidated by task(); see schedule(). */
|
|
116
|
+
private scheduleCache;
|
|
117
|
+
/**
|
|
118
|
+
* Retry backoff, resolved once. Both settlement paths read these — the
|
|
119
|
+
* worker's own `safeFail` and the TaskContext it hands a handler — so
|
|
120
|
+
* resolving the defaults per call site is how the two drift apart.
|
|
121
|
+
*/
|
|
122
|
+
private readonly backoffMs;
|
|
123
|
+
private readonly backoffMaxMs;
|
|
74
124
|
private stopped;
|
|
75
125
|
private stopWake;
|
|
76
126
|
private readonly stopped$;
|
|
@@ -90,6 +140,30 @@ export declare class Worker {
|
|
|
90
140
|
task(handler: Handler): this;
|
|
91
141
|
task(name: string, handler: Handler): this;
|
|
92
142
|
task<P, R>(def: TaskDef<P, R>, handler: TypedHandler<P, R>): this;
|
|
143
|
+
/**
|
|
144
|
+
* Batch delivery: the handler takes one argument, a `TaskContext[]` of up to
|
|
145
|
+
* `batch` tasks, instead of `(ctx, payload)`. Use it when the work itself is
|
|
146
|
+
* batched — one embedding call over 256 texts rather than 256 calls — and size
|
|
147
|
+
* it by what the downstream API wants, not by the queue.
|
|
148
|
+
*
|
|
149
|
+
* `concurrency` caps the calls this name may run at once, under the worker's
|
|
150
|
+
* own. Use it to keep one expensive name from taking the whole worker.
|
|
151
|
+
*
|
|
152
|
+
* `resource` draws each call from a ceiling declared in
|
|
153
|
+
* `WorkerOptions.resources` and shared with every other name that names it —
|
|
154
|
+
* at capacity 1, mutual exclusion across those names.
|
|
155
|
+
*/
|
|
156
|
+
task(name: string | TaskDef, opts: {
|
|
157
|
+
batch: number;
|
|
158
|
+
concurrency?: number;
|
|
159
|
+
resource?: string;
|
|
160
|
+
}, handler: BatchHandler): this;
|
|
161
|
+
/** Per-name concurrency or a shared resource, without batching: the handler
|
|
162
|
+
* still takes `(ctx, payload)`. */
|
|
163
|
+
task(name: string | TaskDef, opts: {
|
|
164
|
+
concurrency?: number;
|
|
165
|
+
resource?: string;
|
|
166
|
+
}, handler: Handler): this;
|
|
93
167
|
stop(): void;
|
|
94
168
|
/** Close the underlying store connection. Call after run() returns. */
|
|
95
169
|
close(): Promise<void>;
|
|
@@ -98,6 +172,52 @@ export declare class Worker {
|
|
|
98
172
|
run(opts?: {
|
|
99
173
|
concurrency?: number;
|
|
100
174
|
}): Promise<void>;
|
|
175
|
+
/**
|
|
176
|
+
* Split one claim into handler calls, each with the registration to run it.
|
|
177
|
+
*
|
|
178
|
+
* A claim is filtered by queue and by the names this worker handles, so it
|
|
179
|
+
* comes back mixed; batch size is per name (one embedding call wants 256
|
|
180
|
+
* texts, one Docling parse wants exactly 1). So group by name, then chunk each
|
|
181
|
+
* group by that name's size. Names registered without `batch` come back as
|
|
182
|
+
* one-task calls, as do names not registered at all — reachable only if a
|
|
183
|
+
* handler is unregistered mid-run, and dispatched to failNoHandler().
|
|
184
|
+
*
|
|
185
|
+
* The registration rides along because this is where it was resolved; looking
|
|
186
|
+
* it up again at the call site would put "is this name batched" in two places.
|
|
187
|
+
*/
|
|
188
|
+
private deliveries;
|
|
189
|
+
private context;
|
|
190
|
+
/**
|
|
191
|
+
* How this poll's claim is split into per-name quotas, plus the union of names
|
|
192
|
+
* the probe spans.
|
|
193
|
+
*
|
|
194
|
+
* Cached and invalidated by `task()`, rather than rebuilt each poll: handlers
|
|
195
|
+
* may be registered after run() started, but only there, and this otherwise
|
|
196
|
+
* allocates a source per name on every tick for a worker's whole lifetime.
|
|
197
|
+
*
|
|
198
|
+
* A name that limits itself — by `batch`, by its own `concurrency`, or by a
|
|
199
|
+
* `resource` — needs a quota the shared draw cannot express, so it gets a
|
|
200
|
+
* source of its own; every other name shares one, where a task is a call.
|
|
201
|
+
*
|
|
202
|
+
* A resource is deliberately *not* one source spanning its names: `batch` is
|
|
203
|
+
* per name, and a single source carries one batch size, so two members that
|
|
204
|
+
* batch differently could not share a draw. Keeping a source per name and
|
|
205
|
+
* letting several of them draw down one shared ceiling composes with batching
|
|
206
|
+
* instead of excluding it.
|
|
207
|
+
*/
|
|
208
|
+
private schedule;
|
|
209
|
+
/**
|
|
210
|
+
* A source's own call ceiling for one poll, or undefined when only the
|
|
211
|
+
* worker-wide budget applies.
|
|
212
|
+
*
|
|
213
|
+
* Three independent ceilings, whichever binds first: the name's own concurrency
|
|
214
|
+
* less what it is already running; its resource's capacity less what is running
|
|
215
|
+
* *and* what earlier draws in this same poll already took (`taken`); and
|
|
216
|
+
* `claimBatch`. The last is a ceiling on *rows* per poll, so it converts at this
|
|
217
|
+
* source's batch size — and never below one call, or a `claimBatch` under some
|
|
218
|
+
* name's batch would stall that name outright.
|
|
219
|
+
*/
|
|
220
|
+
private sourceCalls;
|
|
101
221
|
private loop;
|
|
102
222
|
/** Blocking-style entry point for a standalone worker process: run until
|
|
103
223
|
* SIGINT/SIGTERM, then close the store. Use this at a script's top level;
|
|
@@ -110,20 +230,98 @@ export declare class Worker {
|
|
|
110
230
|
concurrency?: number;
|
|
111
231
|
}): Promise<T>;
|
|
112
232
|
/**
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
*
|
|
233
|
+
* Record a claimed task this worker cannot run. Reachable only if a name is
|
|
234
|
+
* unregistered mid-run — the claim filters on the registered names — so it does
|
|
235
|
+
* not start a handler or a heartbeat for a task it will not run.
|
|
236
|
+
*/
|
|
237
|
+
private failNoHandler;
|
|
238
|
+
/**
|
|
239
|
+
* Run one handler call to completion — one task, or a whole batch. Never
|
|
240
|
+
* rejects: a task-level failure is reported through onError and the loop moves
|
|
241
|
+
* on. (It used to reject into a promise nobody awaited — an unhandled rejection
|
|
242
|
+
* that took the process down.)
|
|
243
|
+
*
|
|
244
|
+
* One lifecycle for both delivery modes, because a single-task handler *is* the
|
|
245
|
+
* one-element case: the same heartbeat covers the call, the same classifier
|
|
246
|
+
* reads its error, and the same rule settles what is left. Only two things vary
|
|
247
|
+
* — how the handler is invoked, and where a leftover task's result comes from —
|
|
248
|
+
* so those are the only two branches below.
|
|
249
|
+
*
|
|
250
|
+
* The contract is single: **when the handler returns, every task it did not
|
|
251
|
+
* settle itself is settled by how the call ended.** Returning succeeds them,
|
|
252
|
+
* throwing fails them (retryably, or as the thrown TaskError says). That is
|
|
253
|
+
* what keeps the ordinary cases free of bookkeeping — a handler that just
|
|
254
|
+
* returns has finished 256 tasks — while still letting it pick individual
|
|
255
|
+
* tasks off with `item.succeed()` / `item.fail()` as it goes.
|
|
256
|
+
*
|
|
257
|
+
* For a batch, a returned map of task id -> result fills in results for the
|
|
258
|
+
* tasks left over; anything else returned is ignored, and unmentioned tasks
|
|
259
|
+
* succeed with no result (the common shape, where the handler's output went to
|
|
260
|
+
* a database rather than into the task row). For a single task the return value
|
|
261
|
+
* simply is the result.
|
|
116
262
|
*/
|
|
117
|
-
private
|
|
263
|
+
private runCall;
|
|
118
264
|
/**
|
|
119
|
-
*
|
|
120
|
-
*
|
|
121
|
-
*
|
|
122
|
-
*
|
|
123
|
-
*
|
|
124
|
-
*
|
|
265
|
+
* How an attempt that ended badly is recorded: [envelope, retryable], or null
|
|
266
|
+
* when there is nothing to record.
|
|
267
|
+
*
|
|
268
|
+
* One classifier for both delivery modes, so a handler error cannot mean
|
|
269
|
+
* different things depending on how its task happened to be delivered. That
|
|
270
|
+
* includes LostLease, which is not an outcome at all: it means a write through
|
|
271
|
+
* this context was already rejected, so recording anything more would be
|
|
272
|
+
* rejected too. Both modes then leave the task alone — the single-task one has
|
|
273
|
+
* nothing else to do, and a batch lets its remaining tasks fall to lease expiry
|
|
274
|
+
* and redelivery rather than stamping them with a failure the handler never
|
|
275
|
+
* reported.
|
|
276
|
+
*/
|
|
277
|
+
private outcomeOf;
|
|
278
|
+
/**
|
|
279
|
+
* Finalize one task the handler left for the worker to decide — the tail of
|
|
280
|
+
* both delivery modes.
|
|
281
|
+
*
|
|
282
|
+
* Includes the unserializable-result rule: the handler succeeded but its value
|
|
283
|
+
* cannot cross the JSON protocol (BigInt, non-finite number, circular), which
|
|
284
|
+
* is deterministic, so it fails permanently rather than being redelivered to
|
|
285
|
+
* fail the same way every attempt.
|
|
286
|
+
*/
|
|
287
|
+
private succeedOne;
|
|
288
|
+
/**
|
|
289
|
+
* Settle every task the handler did not settle itself, concurrently, reporting
|
|
290
|
+
* rather than throwing. Each task keeps its own attempt count and backoff —
|
|
291
|
+
* they are separate tasks that happened to be delivered together.
|
|
292
|
+
*
|
|
293
|
+
* allSettled, because one task's write failing must not abandon the rest of the
|
|
294
|
+
* batch mid-settlement — the others still hold leases and would sit `running`
|
|
295
|
+
* until expiry. Each outcome is reported against the task it belongs to;
|
|
296
|
+
* without that, an operator learns a settlement failed somewhere in a batch of
|
|
297
|
+
* 256.
|
|
298
|
+
*/
|
|
299
|
+
private settleEach;
|
|
300
|
+
/**
|
|
301
|
+
* Run one attempt — of one task or of a whole batch — bounded by maxRunMs when
|
|
302
|
+
* set.
|
|
303
|
+
*
|
|
304
|
+
* On timeout the attempt is abandoned: every context it covers is flagged
|
|
305
|
+
* lease-lost first (ctx.signal aborts, and a handler that keeps running can
|
|
306
|
+
* never write again, nor settle anything behind the worker's back — see
|
|
307
|
+
* TaskContext.owned), then the still-pending promise is left to settle on its
|
|
308
|
+
* own, its outcome discarded. The caller records the handler_timeout failure;
|
|
309
|
+
* lease recovery is NOT involved, so redelivery is immediate.
|
|
125
310
|
*/
|
|
126
311
|
private attempt;
|
|
312
|
+
/**
|
|
313
|
+
* One statement per beat, however many tasks the call covers — a single-task
|
|
314
|
+
* handler is just the one-element case.
|
|
315
|
+
*
|
|
316
|
+
* Only tasks still in play are renewed. A task the handler already settled is
|
|
317
|
+
* terminal, and re-leasing it would be a write against a row nobody owns; that
|
|
318
|
+
* is also why an absence is only read as lease loss after re-checking
|
|
319
|
+
* `settled`, since the handler may have finalized the task while this beat was
|
|
320
|
+
* in flight, which takes the row out of `running` and out of the reply. A task
|
|
321
|
+
* genuinely missing lost its lease (another worker recovered it), so its
|
|
322
|
+
* context is flagged and the handler stops being able to write through it —
|
|
323
|
+
* only that one, never its neighbours.
|
|
324
|
+
*/
|
|
127
325
|
private startHeartbeat;
|
|
128
326
|
private safeFail;
|
|
129
327
|
/**
|