cairnq 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -2
- package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
- package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
- package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
- package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
- package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
- package/dist/_protocol/sql/postgres/queue_depth.sql +22 -0
- package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
- package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
- package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
- package/dist/_protocol/sql/sqlite/queue_depth.sql +26 -0
- package/dist/backoff.d.ts +31 -0
- package/dist/backoff.js +40 -0
- package/dist/backpressure.d.ts +59 -0
- package/dist/backpressure.js +122 -0
- package/dist/client.d.ts +16 -3
- package/dist/client.js +19 -5
- package/dist/context.d.ts +48 -2
- package/dist/context.js +101 -10
- package/dist/errors.d.ts +37 -0
- package/dist/errors.js +60 -0
- package/dist/index.d.ts +7 -3
- package/dist/index.js +2 -1
- package/dist/store/base.d.ts +73 -0
- package/dist/store/base.js +124 -14
- package/dist/store/sqlite.d.ts +35 -0
- package/dist/store/sqlite.js +163 -3
- package/dist/worker.d.ts +238 -13
- package/dist/worker.js +512 -120
- package/package.json +2 -1
- package/src/backoff.ts +53 -0
- package/src/backpressure.ts +140 -0
- package/src/client.ts +33 -5
- package/src/context.ts +116 -9
- package/src/errors.ts +66 -0
- package/src/index.ts +7 -2
- package/src/store/base.ts +136 -13
- package/src/store/sqlite.ts +168 -2
- package/src/worker.ts +671 -132
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "cairnq",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.6.0",
|
|
4
4
|
"description": "SQLite-first, cross-language, storage-centered durable task runtime",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Jannchie <jannchie@gmail.com>",
|
|
@@ -34,6 +34,7 @@
|
|
|
34
34
|
"engines": { "node": ">=22" },
|
|
35
35
|
"scripts": {
|
|
36
36
|
"bench": "tsx bench/run.ts",
|
|
37
|
+
"bench:sweep": "tsx bench/sweep.ts",
|
|
37
38
|
"build": "tsc -p tsconfig.json",
|
|
38
39
|
"test": "vitest run",
|
|
39
40
|
"typecheck": "tsc -p tsconfig.json --noEmit"
|
package/src/backoff.ts
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Retry backoff, in its own module because two callers need it.
|
|
3
|
+
*
|
|
4
|
+
* The worker computes it when a handler's failure ends an attempt; TaskContext
|
|
5
|
+
* computes it when a handler fails one task of a batch itself. Keeping it in
|
|
6
|
+
* worker.ts would make context.ts import the module that imports it.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
export const DEFAULT_RETRY_BACKOFF_MS = 1_000;
|
|
10
|
+
export const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Exponential backoff with equal jitter: the window doubles per attempt up to
|
|
14
|
+
* `maxMs`, and the delay lands uniformly in its upper half, `[w/2, w)`.
|
|
15
|
+
*
|
|
16
|
+
* The jitter is what keeps a fleet from retrying in lockstep. Failures align
|
|
17
|
+
* when the downstream fails fast enough that a whole concurrency batch raises
|
|
18
|
+
* at once (connection refused, DNS gone), and capped exponential backoff then
|
|
19
|
+
* *preserves* that alignment — once every task sits at `maxMs`, they all retry
|
|
20
|
+
* on the same beat forever. Spreading over half the window breaks it; keeping
|
|
21
|
+
* the lower half as a floor means jitter never shortens the wait to less than
|
|
22
|
+
* half of what plain exponential backoff would have asked for.
|
|
23
|
+
*
|
|
24
|
+
* `rand` is injected so tests can pin an exact delay.
|
|
25
|
+
*/
|
|
26
|
+
export function retryDelayMs(
|
|
27
|
+
attempt: number,
|
|
28
|
+
baseMs: number,
|
|
29
|
+
maxMs: number,
|
|
30
|
+
rand: () => number = Math.random,
|
|
31
|
+
): number {
|
|
32
|
+
if (baseMs <= 0) return 0;
|
|
33
|
+
const exponent = Math.max(0, attempt - 1);
|
|
34
|
+
const window = Math.min(maxMs, baseMs * 2 ** exponent);
|
|
35
|
+
const floor = Math.floor(window / 2);
|
|
36
|
+
return floor + Math.floor(rand() * (window - floor));
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* The delay a `fail` write should carry. Not just the backoff: a permanent
|
|
41
|
+
* failure is never re-run, so it always delays 0. Both settlement paths — the
|
|
42
|
+
* worker's and a handler's `ctx.fail` — go through this, so they cannot end up
|
|
43
|
+
* backing off differently.
|
|
44
|
+
*/
|
|
45
|
+
export function failDelayMs(
|
|
46
|
+
attempt: number,
|
|
47
|
+
retryable: boolean,
|
|
48
|
+
baseMs: number,
|
|
49
|
+
maxMs: number,
|
|
50
|
+
rand: () => number = Math.random,
|
|
51
|
+
): number {
|
|
52
|
+
return retryable ? retryDelayMs(attempt, baseMs, maxMs, rand) : 0;
|
|
53
|
+
}
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
import { setTimeout as delay } from "node:timers/promises";
|
|
2
|
+
import { QueueFull } from "./errors.js";
|
|
3
|
+
import type { TaskStore } from "./store/base.js";
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Most tasks a producer may enqueue on one probe's word.
|
|
7
|
+
*
|
|
8
|
+
* The gate probes only when its headroom runs out, so this is what the check
|
|
9
|
+
* costs amortized: one bounded index read per MAX_GRANT submits. It also bounds
|
|
10
|
+
* how far the limit can be overshot — see the class docstring on why several
|
|
11
|
+
* producers make this a soft limit, and why that overshoot is (N-1) * MAX_GRANT
|
|
12
|
+
* rather than unbounded.
|
|
13
|
+
*/
|
|
14
|
+
const MAX_GRANT = 64;
|
|
15
|
+
|
|
16
|
+
// Named for probing, not polling: wait.ts exports DEFAULT_POLL_MS / MAX_POLL_MS
|
|
17
|
+
// for the get() loop behind wait(), an order of magnitude tighter and answering
|
|
18
|
+
// a different question. Two constants of the same name in one SDK would be read
|
|
19
|
+
// as one policy.
|
|
20
|
+
const INITIAL_PROBE_INTERVAL_MS = 250;
|
|
21
|
+
const MAX_PROBE_INTERVAL_MS = 5_000;
|
|
22
|
+
const DEFAULT_MAX_WAIT_MS = 600_000;
|
|
23
|
+
|
|
24
|
+
/** Per-queue depth limits. A number applies one limit to every queue; a record
|
|
25
|
+
* gates only the queues it names and leaves the rest unbounded. */
|
|
26
|
+
export type QueueDepthLimit = number | Record<string, number>;
|
|
27
|
+
|
|
28
|
+
export interface BackpressureOptions {
|
|
29
|
+
/** Queued tasks a queue may hold before `submit` blocks. */
|
|
30
|
+
maxQueueDepth: QueueDepthLimit;
|
|
31
|
+
/** How long a blocked submit waits before raising QueueFull. Default 600_000. */
|
|
32
|
+
maxQueueWaitMs?: number;
|
|
33
|
+
/** First backoff between depth probes; doubles to a 5s ceiling. Default 250. */
|
|
34
|
+
queuePollIntervalMs?: number;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Blocks `submit` while a queue is at its depth limit.
|
|
39
|
+
*
|
|
40
|
+
* Without one of these a producer that outruns its workers is only bounded by
|
|
41
|
+
* disk: the backlog grows, every task's queue wait grows with it, and the
|
|
42
|
+
* failure is a database that filled up rather than a producer that slowed down.
|
|
43
|
+
* A queue is the wrong place to buffer an overload — pushing back on the
|
|
44
|
+
* producer is the point.
|
|
45
|
+
*
|
|
46
|
+
* **A soft limit under several producers.** The check is a read followed by a
|
|
47
|
+
* write that other producers can interleave with, and each holds its own grant,
|
|
48
|
+
* so N producers can overshoot the limit by up to (N-1) * MAX_GRANT tasks. Made
|
|
49
|
+
* exact it would need the depth check inside insert_task's transaction, which
|
|
50
|
+
* puts an unbounded-scan predicate on the hot path of every submit and turns
|
|
51
|
+
* concurrent submits into lock contention — a steep price for a bound whose
|
|
52
|
+
* whole purpose is approximate. Size the limit for the pushback you want, not as
|
|
53
|
+
* a capacity assertion.
|
|
54
|
+
*/
|
|
55
|
+
export class QueueDepthGate {
|
|
56
|
+
/** Remaining grant per queue: submits allowed before the next probe. */
|
|
57
|
+
private readonly headroom = new Map<string, number>();
|
|
58
|
+
/** In-flight probe per queue, so concurrent submits share one read rather
|
|
59
|
+
* than each issuing their own against a queue that is already known full. */
|
|
60
|
+
private readonly probing = new Map<string, Promise<void>>();
|
|
61
|
+
private readonly limits: QueueDepthLimit;
|
|
62
|
+
private readonly maxWaitMs: number;
|
|
63
|
+
private readonly initialProbeMs: number;
|
|
64
|
+
|
|
65
|
+
constructor(
|
|
66
|
+
private readonly store: TaskStore,
|
|
67
|
+
opts: BackpressureOptions,
|
|
68
|
+
) {
|
|
69
|
+
this.limits = opts.maxQueueDepth;
|
|
70
|
+
this.maxWaitMs = opts.maxQueueWaitMs ?? DEFAULT_MAX_WAIT_MS;
|
|
71
|
+
this.initialProbeMs = opts.queuePollIntervalMs ?? INITIAL_PROBE_INTERVAL_MS;
|
|
72
|
+
if (typeof this.limits === "number") this.validate("*", this.limits);
|
|
73
|
+
else for (const [q, v] of Object.entries(this.limits)) this.validate(q, v);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
private validate(queue: string, limit: number): void {
|
|
77
|
+
// A limit of 0 would block every submit forever, which is never what a
|
|
78
|
+
// caller means; catching it here beats a first submit that hangs for
|
|
79
|
+
// maxQueueWaitMs and then raises.
|
|
80
|
+
if (!Number.isInteger(limit) || limit < 1) {
|
|
81
|
+
throw new Error(`maxQueueDepth for ${queue} must be an integer >= 1, got ${limit}`);
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** The limit for `queue`, or null when it is not gated. */
|
|
86
|
+
limitFor(queue: string): number | null {
|
|
87
|
+
if (typeof this.limits === "number") return this.limits;
|
|
88
|
+
return this.limits[queue] ?? null;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Consume one unit of headroom for `queue`, waiting for room if it is full.
|
|
93
|
+
* Returns immediately for an ungated queue. Raises QueueFull on timeout,
|
|
94
|
+
* having enqueued nothing.
|
|
95
|
+
*/
|
|
96
|
+
async acquire(queue: string): Promise<void> {
|
|
97
|
+
const limit = this.limitFor(queue);
|
|
98
|
+
if (limit == null) return;
|
|
99
|
+
|
|
100
|
+
const startedAt = Date.now();
|
|
101
|
+
let waitMs = this.initialProbeMs;
|
|
102
|
+
for (;;) {
|
|
103
|
+
const left = this.headroom.get(queue) ?? 0;
|
|
104
|
+
if (left > 0) {
|
|
105
|
+
this.headroom.set(queue, left - 1);
|
|
106
|
+
return;
|
|
107
|
+
}
|
|
108
|
+
await this.probe(queue, limit);
|
|
109
|
+
if ((this.headroom.get(queue) ?? 0) > 0) continue;
|
|
110
|
+
|
|
111
|
+
const waited = Date.now() - startedAt;
|
|
112
|
+
if (waited >= this.maxWaitMs) throw new QueueFull(queue, limit, waited);
|
|
113
|
+
// Back off: a queue at its limit will not drain within one poll interval,
|
|
114
|
+
// and re-probing tightly adds read load to a database already behind.
|
|
115
|
+
await delay(Math.min(waitMs, this.maxWaitMs - waited));
|
|
116
|
+
waitMs = Math.min(waitMs * 2, MAX_PROBE_INTERVAL_MS);
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Refresh `queue`'s grant from the store, at most one probe in flight.
|
|
122
|
+
*
|
|
123
|
+
* Callers re-read `headroom` afterwards rather than using a returned value:
|
|
124
|
+
* only the caller that started the probe writes the grant, so waiters that
|
|
125
|
+
* joined it cannot overwrite the units already handed out.
|
|
126
|
+
*/
|
|
127
|
+
private probe(queue: string, limit: number): Promise<void> {
|
|
128
|
+
let p = this.probing.get(queue);
|
|
129
|
+
if (!p) {
|
|
130
|
+
p = this.store
|
|
131
|
+
.queueDepth(queue, limit)
|
|
132
|
+
.then((headroom) => {
|
|
133
|
+
this.headroom.set(queue, Math.min(headroom, MAX_GRANT));
|
|
134
|
+
})
|
|
135
|
+
.finally(() => this.probing.delete(queue));
|
|
136
|
+
this.probing.set(queue, p);
|
|
137
|
+
}
|
|
138
|
+
return p;
|
|
139
|
+
}
|
|
140
|
+
}
|
package/src/client.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { BackpressureOptions } from "./backpressure.js";
|
|
1
2
|
import { TaskCanceled, TaskFailed } from "./errors.js";
|
|
2
3
|
import { isFailed, isSucceeded, type Task, type TaskStatus } from "./models.js";
|
|
3
4
|
import { SQLiteStore } from "./store/sqlite.js";
|
|
@@ -12,18 +13,33 @@ export interface CallOptions extends SubmitOptions {
|
|
|
12
13
|
pollMs?: number;
|
|
13
14
|
}
|
|
14
15
|
|
|
16
|
+
/** Options this handle configures on the store it wraps, rather than the
|
|
17
|
+
* store's own constructor arguments. */
|
|
18
|
+
export type ClientOptions = Partial<BackpressureOptions>;
|
|
19
|
+
|
|
15
20
|
/** API-side handle. Thin wrapper over a TaskStore + SDK-orchestrated wait/call. */
|
|
16
21
|
export class CairnQ {
|
|
17
|
-
constructor(
|
|
22
|
+
constructor(
|
|
23
|
+
private readonly _store: TaskStore,
|
|
24
|
+
opts: ClientOptions = {},
|
|
25
|
+
) {
|
|
26
|
+
// Installed on the store, not held here: every submit path goes through the
|
|
27
|
+
// store, including TaskContext.submit, which this handle never sees.
|
|
28
|
+
if (opts.maxQueueDepth != null) {
|
|
29
|
+
_store.useBackpressure(opts as BackpressureOptions);
|
|
30
|
+
}
|
|
31
|
+
}
|
|
18
32
|
|
|
19
|
-
static sqlite(path: string, opts
|
|
20
|
-
|
|
33
|
+
static sqlite(path: string, opts: { busyTimeoutMs?: number } & ClientOptions = {}): CairnQ {
|
|
34
|
+
const { busyTimeoutMs, ...client } = opts;
|
|
35
|
+
return new CairnQ(new SQLiteStore(path, { busyTimeoutMs }), client);
|
|
21
36
|
}
|
|
22
37
|
|
|
23
38
|
/** Multi-host backend. `dsn` is a libpq connection string; requires the
|
|
24
39
|
* optional `pg` package. */
|
|
25
|
-
static postgres(dsn: string, opts
|
|
26
|
-
|
|
40
|
+
static postgres(dsn: string, opts: { max?: number } & ClientOptions = {}): CairnQ {
|
|
41
|
+
const { max, ...client } = opts;
|
|
42
|
+
return new CairnQ(new PostgresStore(dsn, { max }), client);
|
|
27
43
|
}
|
|
28
44
|
|
|
29
45
|
get store(): TaskStore {
|
|
@@ -38,12 +54,24 @@ export class CairnQ {
|
|
|
38
54
|
return this._store.close();
|
|
39
55
|
}
|
|
40
56
|
|
|
57
|
+
/** Enqueue a task. With `maxQueueDepth` configured this blocks while the
|
|
58
|
+
* target queue is at its limit, and raises QueueFull if it stays there for
|
|
59
|
+
* `maxQueueWaitMs` — see QueueDepthGate for why that bound is approximate
|
|
60
|
+
* across several producers. */
|
|
41
61
|
submit(name: string, payload?: unknown, opts?: SubmitOptions): Promise<Task>;
|
|
42
62
|
submit<P, R>(task: TaskDef<P, R>, payload?: P, opts?: SubmitOptions): Promise<Task>;
|
|
43
63
|
submit(task: string | TaskDef, payload?: unknown, opts: SubmitOptions = {}): Promise<Task> {
|
|
44
64
|
return this._store.submit({ name: taskName(task), payload, ...opts });
|
|
45
65
|
}
|
|
46
66
|
|
|
67
|
+
/** How many more tasks fit on `queue` under `maxDepth` — 0 once it is full.
|
|
68
|
+
* The non-blocking read behind `maxQueueDepth`, for a producer that would
|
|
69
|
+
* rather shed load or pick another queue than wait. Cheaper than `stats()`:
|
|
70
|
+
* bounded at `maxDepth` index entries instead of aggregating the table. */
|
|
71
|
+
queueDepth(queue: string, maxDepth: number): Promise<number> {
|
|
72
|
+
return this._store.queueDepth(queue, maxDepth);
|
|
73
|
+
}
|
|
74
|
+
|
|
47
75
|
get(taskId: string): Promise<Task | null> {
|
|
48
76
|
return this._store.get(taskId);
|
|
49
77
|
}
|
package/src/context.ts
CHANGED
|
@@ -1,24 +1,48 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
|
|
2
|
+
import { asEnvelope, type FailReason, LostLease } from "./errors.js";
|
|
2
3
|
import { cancelRequested, type Task } from "./models.js";
|
|
3
4
|
import type { SubmitOptions } from "./client.js";
|
|
4
5
|
import type { TaskStore } from "./store/base.js";
|
|
5
6
|
import { type TaskDef, taskName } from "./task.js";
|
|
6
7
|
import { pollWait } from "./wait.js";
|
|
7
8
|
|
|
8
|
-
|
|
9
|
+
export interface TaskContextOptions {
|
|
10
|
+
retryBackoffMs?: number;
|
|
11
|
+
retryBackoffMaxMs?: number;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Handed to a task handler. Worker-side capabilities mirror the Python SDK.
|
|
16
|
+
*
|
|
17
|
+
* One of these per task, whether a handler is delivered one task or a batch: a
|
|
18
|
+
* batch handler receives a `TaskContext[]`, so a single-task handler's `ctx` is
|
|
19
|
+
* literally the batch-of-one element. Lease, cancellation and settlement are per
|
|
20
|
+
* task, which is why they live here rather than on anything batch-shaped.
|
|
21
|
+
*/
|
|
9
22
|
export class TaskContext {
|
|
10
23
|
private readonly abort = new AbortController();
|
|
11
24
|
private leaseLost = false;
|
|
12
25
|
// Cancellation is monotonic: once the DB has told us a cancel was requested it
|
|
13
26
|
// can't be taken back, so canceled() can answer from this without a re-read.
|
|
14
27
|
private cancelSeen = false;
|
|
28
|
+
// Set once this task reached a terminal state through succeed()/fail(). The
|
|
29
|
+
// worker reads it to know which tasks a batch handler already decided, so it
|
|
30
|
+
// neither settles them twice nor keeps renewing their leases — the bookkeeping
|
|
31
|
+
// every ack/nack-style handler otherwise has to carry itself.
|
|
32
|
+
private isSettled = false;
|
|
33
|
+
private readonly backoffMs: number;
|
|
34
|
+
private readonly backoffMaxMs: number;
|
|
15
35
|
|
|
16
36
|
constructor(
|
|
17
37
|
private readonly store: TaskStore,
|
|
18
38
|
private readonly task: Task,
|
|
19
39
|
public readonly workerId: string,
|
|
20
40
|
private readonly leaseMs: number,
|
|
21
|
-
|
|
41
|
+
opts: TaskContextOptions = {},
|
|
42
|
+
) {
|
|
43
|
+
this.backoffMs = opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
|
|
44
|
+
this.backoffMaxMs = opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS;
|
|
45
|
+
}
|
|
22
46
|
|
|
23
47
|
get taskId(): string {
|
|
24
48
|
return this.task.id;
|
|
@@ -44,6 +68,19 @@ export class TaskContext {
|
|
|
44
68
|
get payload(): any {
|
|
45
69
|
return this.task.payload;
|
|
46
70
|
}
|
|
71
|
+
/**
|
|
72
|
+
* True once this task reached a terminal state — whether the handler settled
|
|
73
|
+
* it with succeed()/fail() or the worker settled it on the handler's behalf.
|
|
74
|
+
* The heartbeat and the settlement paths both read it.
|
|
75
|
+
*/
|
|
76
|
+
get settled(): boolean {
|
|
77
|
+
return this.isSettled;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** @internal Called by the worker when it finalizes this task itself. */
|
|
81
|
+
markSettled(): void {
|
|
82
|
+
this.isSettled = true;
|
|
83
|
+
}
|
|
47
84
|
|
|
48
85
|
/**
|
|
49
86
|
* True once this worker has lost the task's lease — it expired and another
|
|
@@ -70,17 +107,36 @@ export class TaskContext {
|
|
|
70
107
|
// Every owned write returns the current row, so cancellation and lease loss
|
|
71
108
|
// ride along on writes the handler was making anyway.
|
|
72
109
|
private observe(task: Task): Task {
|
|
73
|
-
|
|
110
|
+
this.observeCancel(cancelRequested(task));
|
|
74
111
|
return task;
|
|
75
112
|
}
|
|
76
113
|
|
|
114
|
+
/**
|
|
115
|
+
* @internal The same observation from just the flag, for a caller that read it
|
|
116
|
+
* without materializing a Task — the shared heartbeat, whose statement returns
|
|
117
|
+
* only the id and the cancel column precisely so it does not have to drag
|
|
118
|
+
* every payload back on every beat.
|
|
119
|
+
*/
|
|
120
|
+
observeCancel(cancelRequested: boolean): void {
|
|
121
|
+
if (cancelRequested) this.cancelSeen = true;
|
|
122
|
+
}
|
|
123
|
+
|
|
77
124
|
private async owned(write: () => Promise<Task>): Promise<Task> {
|
|
78
|
-
//
|
|
79
|
-
//
|
|
80
|
-
//
|
|
81
|
-
//
|
|
82
|
-
//
|
|
125
|
+
// One gate for every write through this context, so "may I still write?" is
|
|
126
|
+
// answered in one place rather than at each call site.
|
|
127
|
+
//
|
|
128
|
+
// Lease lost: nothing this context writes may be recorded any more. Checked
|
|
129
|
+
// locally, not just via the store's ownership check — after an abandoned
|
|
130
|
+
// (timed-out) attempt the same worker may re-claim this task under the same
|
|
131
|
+
// workerId, and a zombie handler's write would then pass ownership against
|
|
132
|
+
// the NEW attempt.
|
|
83
133
|
if (this.leaseLost) throw new LostLease(this.task.id);
|
|
134
|
+
// Settled: the task is terminal, so the statement would match no row and come
|
|
135
|
+
// back as a lost lease — telling the handler "another worker took this" when
|
|
136
|
+
// the truth is "you already finished it", and flipping lostLease on the way.
|
|
137
|
+
// Refuse here instead, without the round trip and without corrupting the
|
|
138
|
+
// lease state.
|
|
139
|
+
if (this.isSettled) throw new LostLease(this.task.id);
|
|
84
140
|
try {
|
|
85
141
|
return this.observe(await write());
|
|
86
142
|
} catch (err) {
|
|
@@ -119,6 +175,57 @@ export class TaskContext {
|
|
|
119
175
|
return this.cancelSeen || t.status === "canceled";
|
|
120
176
|
}
|
|
121
177
|
|
|
178
|
+
// ------------------------------------------------------------- settlement
|
|
179
|
+
// Finalizing a task is normally the worker's job, decided by whether the
|
|
180
|
+
// handler returned or threw. These two let a handler decide one task itself,
|
|
181
|
+
// which is what a batch needs: four of 256 tasks failing for four different
|
|
182
|
+
// reasons is the ordinary case, not the edge one, and it cannot be expressed
|
|
183
|
+
// by a single return value or a single throw.
|
|
184
|
+
//
|
|
185
|
+
// Settling twice is a no-op rather than an error. Handlers built on ack/nack
|
|
186
|
+
// queues all end up carrying a `finalizedIds` set to guarantee exactly that;
|
|
187
|
+
// holding it here instead is the point.
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Finalize this task as succeeded, now, without waiting for the handler to
|
|
191
|
+
* return. `complete` semantics: a cancel requested while it ran wins and the
|
|
192
|
+
* task finalizes as canceled instead, its result discarded. Returns null if
|
|
193
|
+
* this task was already settled.
|
|
194
|
+
*/
|
|
195
|
+
async succeed(result: unknown = null): Promise<Task | null> {
|
|
196
|
+
if (this.isSettled) return null;
|
|
197
|
+
const task = await this.owned(() =>
|
|
198
|
+
this.store.complete({ taskId: this.task.id, workerId: this.workerId, result }),
|
|
199
|
+
);
|
|
200
|
+
this.markSettled();
|
|
201
|
+
return task;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Finalize this task as failed, now. `error` may be a string reason, an Error,
|
|
206
|
+
* a TaskError (which carries its own retryability), or a ready envelope.
|
|
207
|
+
* Retryable failures get the worker's backoff and are re-queued while attempts
|
|
208
|
+
* remain, exactly as a thrown error would be. Returns null if already settled.
|
|
209
|
+
*/
|
|
210
|
+
async fail(
|
|
211
|
+
error: FailReason = "task failed",
|
|
212
|
+
opts: { retryable?: boolean } = {},
|
|
213
|
+
): Promise<Task | null> {
|
|
214
|
+
if (this.isSettled) return null;
|
|
215
|
+
const [envelope, retryable] = asEnvelope(error, opts.retryable ?? true);
|
|
216
|
+
const task = await this.owned(() =>
|
|
217
|
+
this.store.fail({
|
|
218
|
+
taskId: this.task.id,
|
|
219
|
+
workerId: this.workerId,
|
|
220
|
+
error: envelope,
|
|
221
|
+
retryable,
|
|
222
|
+
delayMs: failDelayMs(this.task.attempt, retryable, this.backoffMs, this.backoffMaxMs),
|
|
223
|
+
}),
|
|
224
|
+
);
|
|
225
|
+
this.markSettled();
|
|
226
|
+
return task;
|
|
227
|
+
}
|
|
228
|
+
|
|
122
229
|
/** Submit a child task; parent/root/correlation are wired automatically. */
|
|
123
230
|
submit(name: string, payload?: unknown, opts?: SubmitOptions): Promise<Task>;
|
|
124
231
|
submit<P, R>(task: TaskDef<P, R>, payload?: P, opts?: SubmitOptions): Promise<Task>;
|
package/src/errors.ts
CHANGED
|
@@ -20,6 +20,24 @@ export function errorEnvelope(e: {
|
|
|
20
20
|
};
|
|
21
21
|
}
|
|
22
22
|
|
|
23
|
+
/**
|
|
24
|
+
* How an arbitrary thrown value becomes an envelope. Split out from `asEnvelope`
|
|
25
|
+
* below because the worker also reaches it directly, for a thrown plain object —
|
|
26
|
+
* which `asEnvelope` reads as a ready envelope, the right call for `ctx.fail` and
|
|
27
|
+
* the wrong one for something that was thrown. Both must agree on `code` and on
|
|
28
|
+
* deriving `type` from the error's name, or the same error reads differently
|
|
29
|
+
* depending on which way it was recorded.
|
|
30
|
+
*/
|
|
31
|
+
export function exceptionEnvelope(err: unknown, retryable = true): Record<string, unknown> {
|
|
32
|
+
const e = err as { name?: string; message?: string };
|
|
33
|
+
return errorEnvelope({
|
|
34
|
+
type: e?.name ?? "Error",
|
|
35
|
+
code: "handler_error",
|
|
36
|
+
message: String(e?.message ?? err),
|
|
37
|
+
retryable,
|
|
38
|
+
});
|
|
39
|
+
}
|
|
40
|
+
|
|
23
41
|
export class CairnQError extends Error {
|
|
24
42
|
constructor(message?: string) {
|
|
25
43
|
super(message);
|
|
@@ -37,6 +55,24 @@ export class AlreadyExists extends CairnQError {
|
|
|
37
55
|
}
|
|
38
56
|
}
|
|
39
57
|
|
|
58
|
+
/** A gated submit waited out `maxWaitMs` without the queue draining below its
|
|
59
|
+
* depth limit. Nothing was enqueued. Distinct from a slow submit on purpose: a
|
|
60
|
+
* queue this far behind is a capacity problem, and a caller that silently
|
|
61
|
+
* retries forever converts it into an invisible one. */
|
|
62
|
+
export class QueueFull extends CairnQError {
|
|
63
|
+
constructor(
|
|
64
|
+
public queue: string,
|
|
65
|
+
public maxDepth: number,
|
|
66
|
+
public waitedMs: number,
|
|
67
|
+
) {
|
|
68
|
+
super(
|
|
69
|
+
`queue ${queue} still holds ${maxDepth} or more queued tasks after ` +
|
|
70
|
+
`${waitedMs}ms; refusing to enqueue more`,
|
|
71
|
+
);
|
|
72
|
+
this.name = "QueueFull";
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
|
|
40
76
|
/** One line of "why hasn't this finished" from the last snapshot wait()
|
|
41
77
|
* observed. No worker running, no handler for the name, wrong queue, and two
|
|
42
78
|
* processes on different database files all look identical from the API side —
|
|
@@ -165,3 +201,33 @@ export class TaskError extends CairnQError {
|
|
|
165
201
|
});
|
|
166
202
|
}
|
|
167
203
|
}
|
|
204
|
+
|
|
205
|
+
/** What a handler may pass to `ctx.fail`. */
|
|
206
|
+
export type FailReason = string | Error | TaskError | Record<string, unknown>;
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* Normalize anything that can end a task into [envelope, retryable].
|
|
210
|
+
*
|
|
211
|
+
* Shared by both ways a failure is recorded — a handler passing a reason to
|
|
212
|
+
* `ctx.fail`, and the worker classifying an error that ended an attempt — so the
|
|
213
|
+
* two cannot disagree about what a given error means. It lives here, beside the
|
|
214
|
+
* envelope constructors it dispatches to, rather than in the module that happens
|
|
215
|
+
* to expose it to handlers.
|
|
216
|
+
*
|
|
217
|
+
* A handler failing one task of a batch has a reason, not an exception object:
|
|
218
|
+
* `item.fail("no source records", { retryable: false })` is the shape the real
|
|
219
|
+
* code wants. A TaskError carries its own retryability and wins over the option;
|
|
220
|
+
* everything else takes the caller's. A ready envelope passes through, which is
|
|
221
|
+
* how the worker hands in the ones it composes itself.
|
|
222
|
+
*/
|
|
223
|
+
export function asEnvelope(
|
|
224
|
+
error: FailReason,
|
|
225
|
+
retryable: boolean,
|
|
226
|
+
): [Record<string, unknown>, boolean] {
|
|
227
|
+
if (error instanceof TaskError) return [error.envelope(), error.retryable];
|
|
228
|
+
if (error instanceof Error) return [exceptionEnvelope(error, retryable), retryable];
|
|
229
|
+
if (typeof error === "object" && error !== null) return [error, retryable];
|
|
230
|
+
// A bare reason is a TaskError in everything but the throwing, so let
|
|
231
|
+
// TaskError own its own type/code defaults rather than restating them.
|
|
232
|
+
return [new TaskError(String(error), { retryable }).envelope(), retryable];
|
|
233
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -1,8 +1,11 @@
|
|
|
1
1
|
export { CairnQ } from "./client.js";
|
|
2
|
-
export type { CallOptions, SubmitOptions } from "./client.js";
|
|
2
|
+
export type { CallOptions, ClientOptions, SubmitOptions } from "./client.js";
|
|
3
|
+
export { QueueDepthGate } from "./backpressure.js";
|
|
4
|
+
export type { BackpressureOptions, QueueDepthLimit } from "./backpressure.js";
|
|
3
5
|
export { Worker } from "./worker.js";
|
|
4
|
-
export type { Handler, TypedHandler, WorkerOptions } from "./worker.js";
|
|
6
|
+
export type { BatchHandler, Handler, TypedHandler, WorkerOptions } from "./worker.js";
|
|
5
7
|
export { TaskContext } from "./context.js";
|
|
8
|
+
export type { TaskContextOptions } from "./context.js";
|
|
6
9
|
export { defineTask } from "./task.js";
|
|
7
10
|
export type { TaskDef } from "./task.js";
|
|
8
11
|
export { SQLiteStore } from "./store/sqlite.js";
|
|
@@ -23,6 +26,7 @@ export {
|
|
|
23
26
|
export {
|
|
24
27
|
CairnQError,
|
|
25
28
|
AlreadyExists,
|
|
29
|
+
QueueFull,
|
|
26
30
|
TaskTimeout,
|
|
27
31
|
TaskFailed,
|
|
28
32
|
TaskCanceled,
|
|
@@ -31,3 +35,4 @@ export {
|
|
|
31
35
|
ProtocolVersionMismatch,
|
|
32
36
|
SerializationError,
|
|
33
37
|
} from "./errors.js";
|
|
38
|
+
export type { FailReason } from "./errors.js";
|