cairnq 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -0
- package/dist/_protocol/migrations/postgres/0001_init.sql +3 -1
- package/dist/_protocol/migrations/postgres/0002_purge_index.sql +6 -0
- package/dist/_protocol/migrations/postgres/0003_notify.sql +38 -0
- package/dist/_protocol/migrations/postgres/0004_lease_index.sql +16 -0
- package/dist/_protocol/migrations/postgres/0005_clear_terminal_lease.sql +17 -0
- package/dist/_protocol/migrations/sqlite/0001_init.sql +3 -1
- package/dist/_protocol/migrations/sqlite/0002_purge_index.sql +6 -0
- package/dist/_protocol/migrations/sqlite/0004_lease_index.sql +22 -0
- package/dist/_protocol/migrations/sqlite/0005_clear_terminal_lease.sql +17 -0
- package/dist/_protocol/sql/postgres/claim.sql +18 -5
- package/dist/_protocol/sql/postgres/claim_one_queue.sql +35 -0
- package/dist/_protocol/sql/postgres/complete.sql +3 -0
- package/dist/_protocol/sql/postgres/fail.sql +30 -8
- package/dist/_protocol/sql/postgres/insert_task.sql +6 -3
- package/dist/_protocol/sql/postgres/list.sql +3 -1
- package/dist/_protocol/sql/postgres/lock_key.sql +9 -0
- package/dist/_protocol/sql/postgres/progress.sql +4 -3
- package/dist/_protocol/sql/postgres/protocol_version.sql +4 -0
- package/dist/_protocol/sql/postgres/purge.sql +25 -0
- package/dist/_protocol/sql/postgres/recover_leases.sql +49 -14
- package/dist/_protocol/sql/postgres/retry.sql +3 -0
- package/dist/_protocol/sql/postgres/stats.sql +8 -0
- package/dist/_protocol/sql/postgres/succeed.sql +4 -0
- package/dist/_protocol/sql/sqlite/claim.sql +13 -2
- package/dist/_protocol/sql/sqlite/claim_one_queue.sql +36 -0
- package/dist/_protocol/sql/sqlite/claimable_probe.sql +6 -2
- package/dist/_protocol/sql/sqlite/complete.sql +3 -0
- package/dist/_protocol/sql/sqlite/fail.sql +32 -8
- package/dist/_protocol/sql/sqlite/list.sql +3 -1
- package/dist/_protocol/sql/sqlite/lock_key.sql +5 -0
- package/dist/_protocol/sql/sqlite/progress.sql +6 -2
- package/dist/_protocol/sql/sqlite/protocol_version.sql +4 -0
- package/dist/_protocol/sql/sqlite/purge.sql +18 -0
- package/dist/_protocol/sql/sqlite/recover_leases.sql +25 -7
- package/dist/_protocol/sql/sqlite/retry.sql +3 -0
- package/dist/_protocol/sql/sqlite/stats.sql +8 -0
- package/dist/_protocol/sql/sqlite/succeed.sql +4 -0
- package/dist/client.d.ts +10 -2
- package/dist/client.js +12 -0
- package/dist/context.d.ts +17 -1
- package/dist/context.js +60 -6
- package/dist/errors.d.ts +18 -2
- package/dist/errors.js +49 -3
- package/dist/index.d.ts +3 -2
- package/dist/index.js +2 -1
- package/dist/sql.js +16 -9
- package/dist/store/base.d.ts +114 -9
- package/dist/store/base.js +376 -1
- package/dist/store/postgres.d.ts +62 -63
- package/dist/store/postgres.js +245 -222
- package/dist/store/sqlite.d.ts +83 -60
- package/dist/store/sqlite.js +370 -234
- package/dist/wait.d.ts +15 -2
- package/dist/wait.js +23 -5
- package/dist/worker.d.ts +53 -1
- package/dist/worker.js +202 -42
- package/package.json +9 -2
- package/src/client.ts +16 -2
- package/src/context.ts +70 -13
- package/src/errors.ts +59 -4
- package/src/index.ts +3 -1
- package/src/sql.ts +15 -8
- package/src/store/base.ts +443 -27
- package/src/store/postgres.ts +243 -267
- package/src/store/sqlite.ts +378 -263
- package/src/wait.ts +28 -5
- package/src/worker.ts +242 -42
package/dist/wait.d.ts
CHANGED
|
@@ -1,8 +1,21 @@
|
|
|
1
1
|
import { type Task } from "./models.js";
|
|
2
2
|
import type { TaskStore } from "./store/base.js";
|
|
3
|
+
export declare const DEFAULT_POLL_MS = 100;
|
|
4
|
+
export declare const MAX_POLL_MS = 500;
|
|
5
|
+
/**
|
|
6
|
+
* Grow the polling interval towards the ceiling.
|
|
7
|
+
*
|
|
8
|
+
* wait() has no idea whether the task takes 50ms or an hour. Starting tight keeps
|
|
9
|
+
* short tasks snappy; growing keeps a long wait from costing a read every 100ms
|
|
10
|
+
* for its whole duration. The +1 keeps truncation from pinning tiny intervals:
|
|
11
|
+
* Math.floor(1 * 1.5) === 1 would otherwise never grow past 1.
|
|
12
|
+
*/
|
|
13
|
+
export declare function nextPollMs(current: number, maxMs: number): number;
|
|
3
14
|
/** Poll get() until terminal or timeout. Returns the terminal Task (any status).
|
|
4
|
-
* Throws TaskTimeout, leaving the task running.
|
|
5
|
-
|
|
15
|
+
* Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
|
|
16
|
+
* it backs off towards `maxPollMs`. */
|
|
17
|
+
export declare function pollWait(store: TaskStore, taskId: string, { timeoutMs, pollMs, maxPollMs, }: {
|
|
6
18
|
timeoutMs: number;
|
|
7
19
|
pollMs?: number;
|
|
20
|
+
maxPollMs?: number;
|
|
8
21
|
}): Promise<Task>;
|
package/dist/wait.js
CHANGED
|
@@ -1,18 +1,36 @@
|
|
|
1
1
|
import { TaskTimeout } from "./errors.js";
|
|
2
2
|
import { nowMs } from "./ids.js";
|
|
3
3
|
import { isTerminal } from "./models.js";
|
|
4
|
-
const
|
|
4
|
+
export const DEFAULT_POLL_MS = 100;
|
|
5
|
+
export const MAX_POLL_MS = 500;
|
|
6
|
+
const GROWTH = 1.5;
|
|
7
|
+
/**
|
|
8
|
+
* Grow the polling interval towards the ceiling.
|
|
9
|
+
*
|
|
10
|
+
* wait() has no idea whether the task takes 50ms or an hour. Starting tight keeps
|
|
11
|
+
* short tasks snappy; growing keeps a long wait from costing a read every 100ms
|
|
12
|
+
* for its whole duration. The +1 keeps truncation from pinning tiny intervals:
|
|
13
|
+
* Math.floor(1 * 1.5) === 1 would otherwise never grow past 1.
|
|
14
|
+
*/
|
|
15
|
+
export function nextPollMs(current, maxMs) {
|
|
16
|
+
return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
|
|
17
|
+
}
|
|
5
18
|
/** Poll get() until terminal or timeout. Returns the terminal Task (any status).
|
|
6
|
-
* Throws TaskTimeout, leaving the task running.
|
|
7
|
-
|
|
19
|
+
* Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
|
|
20
|
+
* it backs off towards `maxPollMs`. */
|
|
21
|
+
export async function pollWait(store, taskId, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS, }) {
|
|
8
22
|
const deadline = nowMs() + timeoutMs;
|
|
23
|
+
let interval = pollMs;
|
|
9
24
|
for (;;) {
|
|
10
25
|
const task = await store.get(taskId);
|
|
11
26
|
if (task && isTerminal(task))
|
|
12
27
|
return task;
|
|
13
28
|
const remaining = deadline - nowMs();
|
|
14
29
|
if (remaining <= 0)
|
|
15
|
-
throw new TaskTimeout(taskId);
|
|
16
|
-
|
|
30
|
+
throw new TaskTimeout(taskId, { timeoutMs, task });
|
|
31
|
+
// A store with a push channel (Postgres) cuts the sleep short when the task
|
|
32
|
+
// goes terminal; the re-get above stays the source of truth either way.
|
|
33
|
+
await store.taskDoneWake(taskId, Math.min(interval, remaining));
|
|
34
|
+
interval = nextPollMs(interval, maxPollMs);
|
|
17
35
|
}
|
|
18
36
|
}
|
package/dist/worker.d.ts
CHANGED
|
@@ -4,13 +4,40 @@ import { type TaskDef } from "./task.js";
|
|
|
4
4
|
export type Handler = (ctx: TaskContext, payload: any) => unknown | Promise<unknown>;
|
|
5
5
|
/** Handler typed against a TaskDef<P, R>: payload is P, the return is R. */
|
|
6
6
|
export type TypedHandler<P, R> = (ctx: TaskContext, payload: P) => R | Promise<R>;
|
|
7
|
+
/** Where an error the worker recovered from came from. */
|
|
8
|
+
export type ErrorPhase = "claim" | "execute";
|
|
7
9
|
export interface WorkerOptions {
|
|
8
10
|
concurrency?: number;
|
|
9
11
|
leaseMs?: number;
|
|
10
12
|
heartbeatIntervalMs?: number;
|
|
11
13
|
pollIntervalMs?: number;
|
|
12
14
|
claimBatch?: number;
|
|
15
|
+
/** Base delay before re-running a failed attempt; doubles per attempt. 0 disables. */
|
|
16
|
+
retryBackoffMs?: number;
|
|
17
|
+
/** Ceiling for the doubling. */
|
|
18
|
+
retryBackoffMaxMs?: number;
|
|
19
|
+
/**
|
|
20
|
+
* Wall-clock ceiling for one attempt. The heartbeat renews the lease for as
|
|
21
|
+
* long as a handler runs, so a hung handler would otherwise hold its task
|
|
22
|
+
* `running` (and its concurrency slot) forever — cancel can't help,
|
|
23
|
+
* cooperative checks need a live handler. On expiry the worker abandons the
|
|
24
|
+
* attempt (ctx.signal aborts, further ctx writes throw LostLease) and records
|
|
25
|
+
* a retryable `handler_timeout` failure. Unset disables the ceiling.
|
|
26
|
+
*/
|
|
27
|
+
maxRunMs?: number;
|
|
28
|
+
/**
|
|
29
|
+
* Called for errors the worker survived — a claim that threw, a store write
|
|
30
|
+
* that failed while finalizing a task. Without it these are silent: the run
|
|
31
|
+
* loop carries on either way, so this is the only place an operator learns a
|
|
32
|
+
* worker is limping. Must not throw.
|
|
33
|
+
*/
|
|
34
|
+
onError?: (err: unknown, info: {
|
|
35
|
+
phase: ErrorPhase;
|
|
36
|
+
taskId?: string;
|
|
37
|
+
}) => void;
|
|
13
38
|
}
|
|
39
|
+
/** Exponential backoff for the next attempt of a task that just failed. */
|
|
40
|
+
export declare function retryDelayMs(attempt: number, baseMs: number, maxMs: number): number;
|
|
14
41
|
export declare class Worker {
|
|
15
42
|
private readonly store;
|
|
16
43
|
private readonly queues;
|
|
@@ -18,7 +45,8 @@ export declare class Worker {
|
|
|
18
45
|
private readonly handlers;
|
|
19
46
|
private readonly workerId;
|
|
20
47
|
private stopped;
|
|
21
|
-
private
|
|
48
|
+
private stopWake;
|
|
49
|
+
private readonly stopped$;
|
|
22
50
|
private ownsStore;
|
|
23
51
|
constructor(store: TaskStore, queues: string[], opts?: WorkerOptions);
|
|
24
52
|
static sqlite(path: string, opts?: WorkerOptions & {
|
|
@@ -39,9 +67,11 @@ export declare class Worker {
|
|
|
39
67
|
/** Close the underlying store connection. Call after run() returns. */
|
|
40
68
|
close(): Promise<void>;
|
|
41
69
|
private closeIfOwned;
|
|
70
|
+
private report;
|
|
42
71
|
run(opts?: {
|
|
43
72
|
concurrency?: number;
|
|
44
73
|
}): Promise<void>;
|
|
74
|
+
private loop;
|
|
45
75
|
/** Blocking-style entry point for a standalone worker process: run until
|
|
46
76
|
* SIGINT/SIGTERM, then close the store. Use this at a script's top level;
|
|
47
77
|
* use run() / background() when you manage the event loop yourself. */
|
|
@@ -52,9 +82,31 @@ export declare class Worker {
|
|
|
52
82
|
background<T>(fn: () => Promise<T>, opts?: {
|
|
53
83
|
concurrency?: number;
|
|
54
84
|
}): Promise<T>;
|
|
85
|
+
/**
|
|
86
|
+
* Run one task to completion. Never rejects: a task-level failure is reported
|
|
87
|
+
* through onError and the loop moves on. (It used to reject into a promise
|
|
88
|
+
* nobody awaited — an unhandled rejection that took the process down.)
|
|
89
|
+
*/
|
|
55
90
|
private execute;
|
|
91
|
+
/**
|
|
92
|
+
* Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
|
|
93
|
+
* abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
|
|
94
|
+
* a handler that keeps running can never write again — see
|
|
95
|
+
* TaskContext.owned), then the still-pending promise is left to settle on
|
|
96
|
+
* its own, its outcome discarded. The caller records the handler_timeout
|
|
97
|
+
* failure; lease recovery is NOT involved, so redelivery is immediate.
|
|
98
|
+
*/
|
|
99
|
+
private attempt;
|
|
56
100
|
private startHeartbeat;
|
|
57
101
|
private safeFail;
|
|
102
|
+
/**
|
|
103
|
+
* The empty-poll sleep. A store with a push channel (Postgres LISTEN/NOTIFY)
|
|
104
|
+
* cuts it short when a task on this worker's queues becomes claimable;
|
|
105
|
+
* stop() interrupts it either way, and sleepOrStop bounds it at `ms` so the
|
|
106
|
+
* poll fallback — which also drives lease recovery — never stretches.
|
|
107
|
+
*/
|
|
108
|
+
private idle;
|
|
58
109
|
private sleepOrStop;
|
|
110
|
+
/** Take SIGINT/SIGTERM for the duration of serve(). Returns the undo. */
|
|
59
111
|
private installSignals;
|
|
60
112
|
}
|
package/dist/worker.js
CHANGED
|
@@ -1,9 +1,20 @@
|
|
|
1
1
|
import { TaskContext } from "./context.js";
|
|
2
|
-
import { errorEnvelope, LostLease, TaskError } from "./errors.js";
|
|
2
|
+
import { errorEnvelope, LostLease, SerializationError, TaskError } from "./errors.js";
|
|
3
3
|
import { newId } from "./ids.js";
|
|
4
4
|
import { SQLiteStore } from "./store/sqlite.js";
|
|
5
5
|
import { PostgresStore } from "./store/postgres.js";
|
|
6
6
|
import { taskName } from "./task.js";
|
|
7
|
+
const DEFAULT_RETRY_BACKOFF_MS = 1_000;
|
|
8
|
+
const DEFAULT_RETRY_BACKOFF_MAX_MS = 30_000;
|
|
9
|
+
/** Wait after a failed claim, so a broken database is not polled in a tight loop. */
|
|
10
|
+
const CLAIM_ERROR_BACKOFF_MS = 250;
|
|
11
|
+
/** Exponential backoff for the next attempt of a task that just failed. */
|
|
12
|
+
export function retryDelayMs(attempt, baseMs, maxMs) {
|
|
13
|
+
if (baseMs <= 0)
|
|
14
|
+
return 0;
|
|
15
|
+
const exponent = Math.max(0, attempt - 1);
|
|
16
|
+
return Math.min(maxMs, baseMs * 2 ** exponent);
|
|
17
|
+
}
|
|
7
18
|
function exceptionEnvelope(err) {
|
|
8
19
|
const e = err;
|
|
9
20
|
return errorEnvelope({
|
|
@@ -13,6 +24,23 @@ function exceptionEnvelope(err) {
|
|
|
13
24
|
retryable: true,
|
|
14
25
|
});
|
|
15
26
|
}
|
|
27
|
+
/** Internal: an attempt outran maxRunMs and was abandoned. */
|
|
28
|
+
class AttemptTimeout extends Error {
|
|
29
|
+
maxRunMs;
|
|
30
|
+
constructor(maxRunMs) {
|
|
31
|
+
super(`attempt exceeded ${maxRunMs}ms`);
|
|
32
|
+
this.maxRunMs = maxRunMs;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
function timeoutEnvelope(name, maxRunMs) {
|
|
36
|
+
return errorEnvelope({
|
|
37
|
+
type: "HandlerTimeout",
|
|
38
|
+
code: "handler_timeout",
|
|
39
|
+
message: `handler for ${name} exceeded maxRunMs=${maxRunMs}ms; the attempt was abandoned`,
|
|
40
|
+
retryable: true,
|
|
41
|
+
});
|
|
42
|
+
}
|
|
43
|
+
const TIMED_OUT = Symbol("cairnq.timedOut");
|
|
16
44
|
export class Worker {
|
|
17
45
|
store;
|
|
18
46
|
queues;
|
|
@@ -20,7 +48,10 @@ export class Worker {
|
|
|
20
48
|
handlers = new Map();
|
|
21
49
|
workerId = newId("worker");
|
|
22
50
|
stopped = false;
|
|
23
|
-
|
|
51
|
+
stopWake;
|
|
52
|
+
// Resolved once by stop(); every sleep races against it. A stopped worker
|
|
53
|
+
// never restarts, so one promise serves the instance's lifetime.
|
|
54
|
+
stopped$ = new Promise((r) => (this.stopWake = r));
|
|
24
55
|
// True only when this worker created its own store (via Worker.sqlite); an
|
|
25
56
|
// injected store may be shared, so serve()/background() must not close it.
|
|
26
57
|
ownsStore = false;
|
|
@@ -28,6 +59,9 @@ export class Worker {
|
|
|
28
59
|
this.store = store;
|
|
29
60
|
this.queues = queues;
|
|
30
61
|
this.opts = opts;
|
|
62
|
+
if (opts.maxRunMs != null && opts.maxRunMs <= 0) {
|
|
63
|
+
throw new Error(`maxRunMs must be > 0, got ${opts.maxRunMs}`);
|
|
64
|
+
}
|
|
31
65
|
}
|
|
32
66
|
static sqlite(path, opts = {}) {
|
|
33
67
|
const { queues = ["default"], busyTimeoutMs, ...rest } = opts;
|
|
@@ -51,8 +85,10 @@ export class Worker {
|
|
|
51
85
|
let fn;
|
|
52
86
|
if (typeof arg === "function") {
|
|
53
87
|
// Bare form: worker.task(fn) — registered under the function's name.
|
|
88
|
+
// Strip the "bound " prefix .bind() stamps on it: otherwise a bound
|
|
89
|
+
// method registers under "bound process", a name no submit ever uses.
|
|
54
90
|
fn = arg;
|
|
55
|
-
name = fn.name;
|
|
91
|
+
name = fn.name.replace(/^(bound )+/, "");
|
|
56
92
|
if (!name) {
|
|
57
93
|
throw new Error("worker.task(fn): the handler is anonymous; pass a name explicitly, " +
|
|
58
94
|
"e.g. worker.task('summary.create', fn)");
|
|
@@ -68,10 +104,7 @@ export class Worker {
|
|
|
68
104
|
}
|
|
69
105
|
stop() {
|
|
70
106
|
this.stopped = true;
|
|
71
|
-
|
|
72
|
-
this.stopResolvers = [];
|
|
73
|
-
for (const r of resolvers)
|
|
74
|
-
r();
|
|
107
|
+
this.stopWake();
|
|
75
108
|
}
|
|
76
109
|
/** Close the underlying store connection. Call after run() returns. */
|
|
77
110
|
async close() {
|
|
@@ -84,28 +117,66 @@ export class Worker {
|
|
|
84
117
|
if (this.ownsStore)
|
|
85
118
|
await this.close();
|
|
86
119
|
}
|
|
120
|
+
report(err, info) {
|
|
121
|
+
try {
|
|
122
|
+
this.opts.onError?.(err, info);
|
|
123
|
+
}
|
|
124
|
+
catch {
|
|
125
|
+
// A reporting hook must never take the worker down with it.
|
|
126
|
+
}
|
|
127
|
+
}
|
|
87
128
|
async run(opts = {}) {
|
|
88
|
-
|
|
129
|
+
// Clamped: at 0 the loop would await Promise.race([]) — pending forever,
|
|
130
|
+
// beyond even stop()'s reach.
|
|
131
|
+
const concurrency = Math.max(1, opts.concurrency ?? this.opts.concurrency ?? 1);
|
|
89
132
|
const leaseMs = this.opts.leaseMs ?? 30_000;
|
|
90
|
-
const pollMs = this.opts.pollIntervalMs ?? 500;
|
|
91
133
|
const batch = this.opts.claimBatch ?? concurrency;
|
|
92
134
|
await this.store.connect();
|
|
93
|
-
this.installSignals();
|
|
94
135
|
const running = new Set();
|
|
136
|
+
try {
|
|
137
|
+
await this.loop(concurrency, batch, leaseMs, running);
|
|
138
|
+
}
|
|
139
|
+
finally {
|
|
140
|
+
// Whatever ends the loop — stop(), or something unexpected out of the body
|
|
141
|
+
// — nothing this worker started may outlive run(). serve() closes the store
|
|
142
|
+
// as soon as run() settles, and a handler still holding the connection
|
|
143
|
+
// would fault on it.
|
|
144
|
+
await Promise.all([...running]);
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
async loop(concurrency, batch, leaseMs, running) {
|
|
148
|
+
const pollMs = this.opts.pollIntervalMs ?? 500;
|
|
95
149
|
while (!this.stopped) {
|
|
96
150
|
const free = concurrency - running.size;
|
|
97
151
|
if (free <= 0) {
|
|
98
|
-
|
|
152
|
+
// Wait for a slot rather than spinning. execute() never rejects, so
|
|
153
|
+
// racing these is safe.
|
|
154
|
+
await Promise.race([...running]);
|
|
155
|
+
continue;
|
|
156
|
+
}
|
|
157
|
+
let claimed;
|
|
158
|
+
try {
|
|
159
|
+
claimed = await this.store.claim({
|
|
160
|
+
queues: this.queues,
|
|
161
|
+
// Only what this worker can run. Queues do not partition work by task
|
|
162
|
+
// name, so another worker's tasks would otherwise be claimed here and
|
|
163
|
+
// failed for want of a handler. Read each poll: handlers may be
|
|
164
|
+
// registered after run() started.
|
|
165
|
+
names: [...this.handlers.keys()],
|
|
166
|
+
workerId: this.workerId,
|
|
167
|
+
leaseMs,
|
|
168
|
+
limit: Math.min(batch, free),
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
catch (err) {
|
|
172
|
+
// A claim can fail transiently (lock contention, a dropped connection).
|
|
173
|
+
// Report it and keep polling — one bad poll must not end the worker.
|
|
174
|
+
this.report(err, { phase: "claim" });
|
|
175
|
+
await this.sleepOrStop(CLAIM_ERROR_BACKOFF_MS);
|
|
99
176
|
continue;
|
|
100
177
|
}
|
|
101
|
-
const claimed = await this.store.claim({
|
|
102
|
-
queues: this.queues,
|
|
103
|
-
workerId: this.workerId,
|
|
104
|
-
leaseMs,
|
|
105
|
-
limit: Math.min(batch, free),
|
|
106
|
-
});
|
|
107
178
|
if (claimed.length === 0) {
|
|
108
|
-
await this.
|
|
179
|
+
await this.idle(pollMs);
|
|
109
180
|
continue;
|
|
110
181
|
}
|
|
111
182
|
for (const task of claimed) {
|
|
@@ -113,16 +184,21 @@ export class Worker {
|
|
|
113
184
|
running.add(p);
|
|
114
185
|
}
|
|
115
186
|
}
|
|
116
|
-
await Promise.all([...running]);
|
|
117
187
|
}
|
|
118
188
|
/** Blocking-style entry point for a standalone worker process: run until
|
|
119
189
|
* SIGINT/SIGTERM, then close the store. Use this at a script's top level;
|
|
120
190
|
* use run() / background() when you manage the event loop yourself. */
|
|
121
191
|
async serve(opts = {}) {
|
|
192
|
+
// Signals are installed here rather than in run(): serve() is the entry point
|
|
193
|
+
// that owns the process. run()/background() embed the worker in someone
|
|
194
|
+
// else's process, where a leftover listener suppresses Node's default Ctrl-C
|
|
195
|
+
// handling for the host long after the worker is done.
|
|
196
|
+
const removeSignalHandlers = this.installSignals();
|
|
122
197
|
try {
|
|
123
198
|
await this.run(opts);
|
|
124
199
|
}
|
|
125
200
|
finally {
|
|
201
|
+
removeSignalHandlers();
|
|
126
202
|
await this.closeIfOwned();
|
|
127
203
|
}
|
|
128
204
|
}
|
|
@@ -138,13 +214,18 @@ export class Worker {
|
|
|
138
214
|
await this.closeIfOwned();
|
|
139
215
|
}
|
|
140
216
|
}
|
|
217
|
+
/**
|
|
218
|
+
* Run one task to completion. Never rejects: a task-level failure is reported
|
|
219
|
+
* through onError and the loop moves on. (It used to reject into a promise
|
|
220
|
+
* nobody awaited — an unhandled rejection that took the process down.)
|
|
221
|
+
*/
|
|
141
222
|
async execute(task, leaseMs) {
|
|
142
223
|
const ctx = new TaskContext(this.store, task, this.workerId, leaseMs);
|
|
143
224
|
const hb = this.startHeartbeat(ctx, leaseMs);
|
|
144
225
|
try {
|
|
145
226
|
const handler = this.handlers.get(task.name);
|
|
146
227
|
if (!handler) {
|
|
147
|
-
await this.safeFail(task
|
|
228
|
+
await this.safeFail(task, errorEnvelope({
|
|
148
229
|
type: "NoHandler",
|
|
149
230
|
code: "no_handler",
|
|
150
231
|
message: `no handler registered for ${task.name}`,
|
|
@@ -154,16 +235,22 @@ export class Worker {
|
|
|
154
235
|
}
|
|
155
236
|
let result;
|
|
156
237
|
try {
|
|
157
|
-
result = await handler
|
|
238
|
+
result = await this.attempt(handler, ctx, task);
|
|
158
239
|
}
|
|
159
240
|
catch (err) {
|
|
160
241
|
if (err instanceof LostLease)
|
|
161
242
|
return;
|
|
243
|
+
if (err instanceof AttemptTimeout) {
|
|
244
|
+
// Recorded as a retryable failure, so backoff / maxAttempts /
|
|
245
|
+
// cancel-wins all apply exactly as for a thrown error.
|
|
246
|
+
await this.safeFail(task, timeoutEnvelope(task.name, err.maxRunMs), true);
|
|
247
|
+
return;
|
|
248
|
+
}
|
|
162
249
|
if (err instanceof TaskError) {
|
|
163
|
-
await this.safeFail(task
|
|
250
|
+
await this.safeFail(task, err.envelope(), err.retryable);
|
|
164
251
|
}
|
|
165
252
|
else {
|
|
166
|
-
await this.safeFail(task
|
|
253
|
+
await this.safeFail(task, exceptionEnvelope(err), true);
|
|
167
254
|
}
|
|
168
255
|
return;
|
|
169
256
|
}
|
|
@@ -173,20 +260,68 @@ export class Worker {
|
|
|
173
260
|
await this.store.complete({ taskId: task.id, workerId: this.workerId, result });
|
|
174
261
|
}
|
|
175
262
|
catch (err) {
|
|
176
|
-
if (err instanceof LostLease)
|
|
263
|
+
if (err instanceof LostLease) {
|
|
264
|
+
ctx.markLeaseLost();
|
|
177
265
|
return;
|
|
266
|
+
}
|
|
267
|
+
if (err instanceof SerializationError) {
|
|
268
|
+
// The handler succeeded but its return value can't cross the JSON
|
|
269
|
+
// protocol (BigInt, non-finite number, circular). Deterministic, so
|
|
270
|
+
// fail fast and permanently — the alternative is sitting `running`
|
|
271
|
+
// until lease expiry redelivers a task that fails the same way every
|
|
272
|
+
// attempt.
|
|
273
|
+
await this.safeFail(task, errorEnvelope({
|
|
274
|
+
type: "SerializationError",
|
|
275
|
+
code: "unserializable_result",
|
|
276
|
+
message: `handler result is not JSON-serializable: ${err.message}`,
|
|
277
|
+
retryable: false,
|
|
278
|
+
}), false);
|
|
279
|
+
return;
|
|
280
|
+
}
|
|
178
281
|
throw err;
|
|
179
282
|
}
|
|
180
283
|
}
|
|
284
|
+
catch (err) {
|
|
285
|
+
this.report(err, { phase: "execute", taskId: task.id });
|
|
286
|
+
}
|
|
181
287
|
finally {
|
|
182
288
|
hb.cancel();
|
|
183
289
|
await hb.done;
|
|
184
290
|
}
|
|
185
291
|
}
|
|
292
|
+
/**
|
|
293
|
+
* Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
|
|
294
|
+
* abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
|
|
295
|
+
* a handler that keeps running can never write again — see
|
|
296
|
+
* TaskContext.owned), then the still-pending promise is left to settle on
|
|
297
|
+
* its own, its outcome discarded. The caller records the handler_timeout
|
|
298
|
+
* failure; lease recovery is NOT involved, so redelivery is immediate.
|
|
299
|
+
*/
|
|
300
|
+
async attempt(handler, ctx, task) {
|
|
301
|
+
const maxRunMs = this.opts.maxRunMs;
|
|
302
|
+
if (maxRunMs == null)
|
|
303
|
+
return handler(ctx, task.payload);
|
|
304
|
+
// As a real promise: the race needs one (a handler may return a plain
|
|
305
|
+
// value), and the timeout path .catch()es it.
|
|
306
|
+
const run = (async () => handler(ctx, task.payload))();
|
|
307
|
+
let timer;
|
|
308
|
+
const winner = await Promise.race([
|
|
309
|
+
run,
|
|
310
|
+
new Promise((r) => (timer = setTimeout(() => r(TIMED_OUT), maxRunMs))),
|
|
311
|
+
]).finally(() => clearTimeout(timer));
|
|
312
|
+
if (winner !== TIMED_OUT)
|
|
313
|
+
return winner;
|
|
314
|
+
ctx.markLeaseLost();
|
|
315
|
+
// The zombie may still reject later; that must not become an unhandled
|
|
316
|
+
// rejection — its outcome was already decided to be handler_timeout.
|
|
317
|
+
run.catch(() => { });
|
|
318
|
+
throw new AttemptTimeout(maxRunMs);
|
|
319
|
+
}
|
|
186
320
|
startHeartbeat(ctx, leaseMs) {
|
|
187
321
|
let active = true;
|
|
188
322
|
let wake = null;
|
|
189
|
-
|
|
323
|
+
// lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
|
|
324
|
+
const interval = this.opts.heartbeatIntervalMs ?? Math.max(50, Math.floor(leaseMs / 3));
|
|
190
325
|
const done = (async () => {
|
|
191
326
|
while (active) {
|
|
192
327
|
// Cancellable sleep: cancel() resolves this immediately and clears the
|
|
@@ -205,8 +340,10 @@ export class Worker {
|
|
|
205
340
|
await ctx.heartbeat();
|
|
206
341
|
}
|
|
207
342
|
catch (err) {
|
|
343
|
+
// ctx.heartbeat() already flagged the lease as lost for the handler.
|
|
208
344
|
if (err instanceof LostLease)
|
|
209
345
|
break;
|
|
346
|
+
this.report(err, { phase: "execute", taskId: ctx.taskId });
|
|
210
347
|
}
|
|
211
348
|
}
|
|
212
349
|
})();
|
|
@@ -219,34 +356,57 @@ export class Worker {
|
|
|
219
356
|
done,
|
|
220
357
|
};
|
|
221
358
|
}
|
|
222
|
-
async safeFail(
|
|
359
|
+
async safeFail(task, envelope, retryable) {
|
|
360
|
+
const delayMs = retryable
|
|
361
|
+
? retryDelayMs(task.attempt, this.opts.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS, this.opts.retryBackoffMaxMs ?? DEFAULT_RETRY_BACKOFF_MAX_MS)
|
|
362
|
+
: 0;
|
|
223
363
|
try {
|
|
224
|
-
await this.store.fail({
|
|
364
|
+
await this.store.fail({
|
|
365
|
+
taskId: task.id,
|
|
366
|
+
workerId: this.workerId,
|
|
367
|
+
error: envelope,
|
|
368
|
+
retryable,
|
|
369
|
+
delayMs,
|
|
370
|
+
});
|
|
225
371
|
}
|
|
226
372
|
catch (err) {
|
|
227
373
|
if (!(err instanceof LostLease))
|
|
228
374
|
throw err;
|
|
229
375
|
}
|
|
230
376
|
}
|
|
377
|
+
/**
|
|
378
|
+
* The empty-poll sleep. A store with a push channel (Postgres LISTEN/NOTIFY)
|
|
379
|
+
* cuts it short when a task on this worker's queues becomes claimable;
|
|
380
|
+
* stop() interrupts it either way, and sleepOrStop bounds it at `ms` so the
|
|
381
|
+
* poll fallback — which also drives lease recovery — never stretches.
|
|
382
|
+
*/
|
|
383
|
+
idle(ms) {
|
|
384
|
+
return Promise.race([this.sleepOrStop(ms), this.store.claimWake(this.queues, ms)]);
|
|
385
|
+
}
|
|
231
386
|
sleepOrStop(ms) {
|
|
232
387
|
if (this.stopped)
|
|
233
388
|
return Promise.resolve();
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
done = true;
|
|
240
|
-
clearTimeout(timer);
|
|
241
|
-
resolve();
|
|
242
|
-
};
|
|
243
|
-
const timer = setTimeout(finish, ms);
|
|
244
|
-
this.stopResolvers.push(finish);
|
|
245
|
-
});
|
|
389
|
+
let timer;
|
|
390
|
+
const nap = new Promise((r) => (timer = setTimeout(r, ms)));
|
|
391
|
+
// Clear the timer whichever side wins, so a stop is never followed by a
|
|
392
|
+
// leftover poll timer holding the process open.
|
|
393
|
+
return Promise.race([nap, this.stopped$]).finally(() => clearTimeout(timer));
|
|
246
394
|
}
|
|
395
|
+
/** Take SIGINT/SIGTERM for the duration of serve(). Returns the undo. */
|
|
247
396
|
installSignals() {
|
|
248
|
-
const
|
|
249
|
-
|
|
250
|
-
|
|
397
|
+
const remove = () => {
|
|
398
|
+
process.off("SIGINT", handler);
|
|
399
|
+
process.off("SIGTERM", handler);
|
|
400
|
+
};
|
|
401
|
+
const handler = () => {
|
|
402
|
+
// Stand down after the first signal, so a second Ctrl-C reaches Node's
|
|
403
|
+
// default and kills a worker that will not drain. `once` would only drop
|
|
404
|
+
// whichever signal fired and leave the other suppressing the default.
|
|
405
|
+
remove();
|
|
406
|
+
this.stop();
|
|
407
|
+
};
|
|
408
|
+
process.on("SIGINT", handler);
|
|
409
|
+
process.on("SIGTERM", handler);
|
|
410
|
+
return remove;
|
|
251
411
|
}
|
|
252
412
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "cairnq",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.0",
|
|
4
4
|
"description": "SQLite-first, cross-language, storage-centered durable task runtime",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Jannchie <jannchie@gmail.com>",
|
|
@@ -33,6 +33,7 @@
|
|
|
33
33
|
"files": ["dist", "src"],
|
|
34
34
|
"engines": { "node": ">=20" },
|
|
35
35
|
"scripts": {
|
|
36
|
+
"bench": "tsx bench/run.ts",
|
|
36
37
|
"build": "tsc -p tsconfig.json",
|
|
37
38
|
"test": "vitest run",
|
|
38
39
|
"typecheck": "tsc -p tsconfig.json --noEmit"
|
|
@@ -40,13 +41,19 @@
|
|
|
40
41
|
"dependencies": {
|
|
41
42
|
"better-sqlite3": "^11.3.0"
|
|
42
43
|
},
|
|
43
|
-
"
|
|
44
|
+
"peerDependencies": {
|
|
44
45
|
"pg": "^8.13.0"
|
|
45
46
|
},
|
|
47
|
+
"peerDependenciesMeta": {
|
|
48
|
+
"pg": {
|
|
49
|
+
"optional": true
|
|
50
|
+
}
|
|
51
|
+
},
|
|
46
52
|
"devDependencies": {
|
|
47
53
|
"@types/better-sqlite3": "^7.6.11",
|
|
48
54
|
"@types/node": "^22.7.0",
|
|
49
55
|
"@types/pg": "^8.11.10",
|
|
56
|
+
"pg": "^8.13.0",
|
|
50
57
|
"tsx": "^4.19.0",
|
|
51
58
|
"typescript": "^5.6.0",
|
|
52
59
|
"vitest": "^2.1.0"
|
package/src/client.ts
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { TaskCanceled, TaskFailed } from "./errors.js";
|
|
2
|
-
import { isFailed, isSucceeded, type Task } from "./models.js";
|
|
2
|
+
import { isFailed, isSucceeded, type Task, type TaskStatus } from "./models.js";
|
|
3
3
|
import { SQLiteStore } from "./store/sqlite.js";
|
|
4
4
|
import { PostgresStore } from "./store/postgres.js";
|
|
5
|
-
import type { ListInput, SubmitInput, TaskStore } from "./store/base.js";
|
|
5
|
+
import type { ListInput, PurgeInput, SubmitInput, TaskStore } from "./store/base.js";
|
|
6
6
|
import { type TaskDef, taskName } from "./task.js";
|
|
7
7
|
import { pollWait } from "./wait.js";
|
|
8
8
|
|
|
@@ -72,6 +72,20 @@ export class CairnQ {
|
|
|
72
72
|
return this._store.retryByKey(key, opts);
|
|
73
73
|
}
|
|
74
74
|
|
|
75
|
+
/** Delete terminal tasks that finished more than `olderThanMs` ago and return
|
|
76
|
+
* their ids. Nothing else in CairnQ removes rows, so a long-lived database
|
|
77
|
+
* needs this on a schedule. Each call is bounded by `limit` to keep the write
|
|
78
|
+
* short; loop until it returns fewer than `limit`. */
|
|
79
|
+
purge(input?: PurgeInput): Promise<string[]> {
|
|
80
|
+
return this._store.purge(input);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/** Task counts per queue, keyed by status and zero-filled across all statuses
|
|
84
|
+
* — `(await stats()).default.queued` is the backlog of a queue. */
|
|
85
|
+
stats(): Promise<Record<string, Record<TaskStatus, number>>> {
|
|
86
|
+
return this._store.stats();
|
|
87
|
+
}
|
|
88
|
+
|
|
75
89
|
wait(
|
|
76
90
|
taskId: string,
|
|
77
91
|
opts: { timeoutMs?: number; pollMs?: number } = {},
|