cairnq 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/errors.ts CHANGED
@@ -77,8 +77,12 @@ export class QueueFull extends CairnQError {
77
77
  * observed. No worker running, no handler for the name, wrong queue, and two
78
78
  * processes on different database files all look identical from the API side —
79
79
  * queued, never claimed — so that case names the likely causes. */
80
- function timeoutDetail(task: Task | null): string {
81
- if (!task) return "task not found — wrong database file, or already purged?";
80
+ function timeoutDetail(task: Task | null, key: string | null): string {
81
+ if (!task) {
82
+ return key === null
83
+ ? "task not found — wrong database file, or already purged?"
84
+ : "no task under this key — never submitted, or already purged?";
85
+ }
82
86
  if (isQueued(task)) {
83
87
  const delayMs = task.run_at_ms - nowMs();
84
88
  if (task.attempt === 0 && delayMs <= 0) {
@@ -94,23 +98,33 @@ function timeoutDetail(task: Task | null): string {
94
98
  return `still running (attempt ${task.attempt}/${task.max_attempts})`;
95
99
  }
96
100
 
97
- /** wait/call did not reach a terminal status in time. The task keeps running.
98
- * `task` is the last snapshot wait() observed (null if get() found nothing), and
99
- * the message says what state it was stuck in — a queued-never-claimed task is
100
- * the classic first-run failure (no worker, no handler, wrong queue or file). */
101
+ /** wait/call did not reach a terminal status in time. The task keeps running, so
102
+ * `taskId` is the handle for picking the wait back up — `wait(err.taskId)`
103
+ * re-attaches to the same task from anywhere that can reach the store. `task` is
104
+ * the last snapshot wait() observed (null if the lookup found nothing), and the
105
+ * message says what state it was stuck in — a queued-never-claimed task is the
106
+ * classic first-run failure (no worker, no handler, wrong queue or file).
107
+ *
108
+ * `key` is set when the wait watched a key rather than an id; `taskId` is then
109
+ * the task the key pointed at, or the key itself when it pointed at nothing —
110
+ * there was no id to report. */
101
111
  export class TaskTimeout extends CairnQError {
102
112
  readonly task: Task | null;
113
+ readonly key: string | null;
103
114
  constructor(
104
115
  public taskId: string,
105
- opts: { timeoutMs?: number; task?: Task | null } = {},
116
+ opts: { timeoutMs?: number; task?: Task | null; key?: string | null } = {},
106
117
  ) {
118
+ const key = opts.key ?? null;
119
+ const subject = key === null ? `task ${taskId}` : `key ${key}`;
107
120
  super(
108
121
  opts.timeoutMs == null
109
- ? `task ${taskId} did not finish in time`
110
- : `task ${taskId} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null)}`,
122
+ ? `${subject} did not finish in time`
123
+ : `${subject} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null, key)}`,
111
124
  );
112
125
  this.name = "TaskTimeout";
113
126
  this.task = opts.task ?? null;
127
+ this.key = key;
114
128
  }
115
129
  }
116
130
 
@@ -146,6 +160,39 @@ export class TaskCanceled extends CairnQError {
146
160
  }
147
161
  }
148
162
 
163
+ /**
164
+ * A heartbeat beat came back later than its own interval allowed.
165
+ *
166
+ * The heartbeat shares the event loop with the handlers whose leases it renews,
167
+ * so a handler that blocks the loop stops the renewal with it: the lease expires,
168
+ * the task is recovered and redelivered, and a second worker starts computing
169
+ * what the first is still computing — one task, billed twice, with no error
170
+ * anywhere. Nothing inside the blocked handler can observe that, which is why it
171
+ * is reported through `onError` alongside the other things the run loop survived.
172
+ *
173
+ * The usual cause is synchronous work in a handler: a tight loop, a large
174
+ * JSON.parse, a `*Sync` filesystem or crypto call. Node has one loop and no way
175
+ * to preempt it — move the work to a worker thread, a child process, or an async
176
+ * API that yields. The other cause is a worker simply oversubscribed for its
177
+ * `leaseMs` — nothing is blocking, there is just more work than turns — which the
178
+ * same report covers, because the lease is at equal risk either way.
179
+ */
180
+ export class EventLoopBlocked extends CairnQError {
181
+ constructor(
182
+ readonly lateMs: number,
183
+ readonly intervalMs: number,
184
+ readonly leaseMs: number,
185
+ ) {
186
+ super(
187
+ `heartbeat beat was ${lateMs}ms late (interval ${intervalMs}ms, lease ${leaseMs}ms): ` +
188
+ `the event loop was blocked long enough to miss a beat, so this worker's leases ` +
189
+ `are at risk. Usually synchronous work in a handler (move it off the loop); ` +
190
+ `otherwise the worker is oversubscribed for its leaseMs.`,
191
+ );
192
+ this.name = "EventLoopBlocked";
193
+ }
194
+ }
195
+
149
196
  /** A worker write affected 0 rows: the lease expired and was reclaimed. */
150
197
  export class LostLease extends CairnQError {
151
198
  constructor(public taskId: string) {
package/src/index.ts CHANGED
@@ -1,7 +1,9 @@
1
1
  export { CairnQ } from "./client.js";
2
- export type { CallOptions, ClientOptions, SubmitOptions } from "./client.js";
2
+ export type { CallOptions, ClientOptions, SubmitOptions, WaitOptions } from "./client.js";
3
3
  export { QueueDepthGate } from "./backpressure.js";
4
4
  export type { BackpressureOptions, QueueDepthLimit } from "./backpressure.js";
5
+ export { RetentionSweeper } from "./retention.js";
6
+ export type { RetentionCutoffs, RetentionOptions } from "./retention.js";
5
7
  export { Worker } from "./worker.js";
6
8
  export type { BatchHandler, Handler, TypedHandler, WorkerOptions } from "./worker.js";
7
9
  export { TaskContext } from "./context.js";
@@ -12,10 +14,11 @@ export { SQLiteStore } from "./store/sqlite.js";
12
14
  export { PostgresStore } from "./store/postgres.js";
13
15
  export { TaskStore } from "./store/base.js";
14
16
  export type { ListInput, PurgeInput, SubmitInput, Conflict } from "./store/base.js";
15
- export type { Task, TaskStatus } from "./models.js";
17
+ export type { Task, TaskRef, TaskStatus, TerminalStatus } from "./models.js";
16
18
  export {
17
19
  STATUSES,
18
20
  isTerminal,
21
+ isTerminalStatus,
19
22
  cancelRequested,
20
23
  isQueued,
21
24
  isRunning,
@@ -32,6 +35,7 @@ export {
32
35
  TaskCanceled,
33
36
  TaskError,
34
37
  LostLease,
38
+ EventLoopBlocked,
35
39
  ProtocolVersionMismatch,
36
40
  SerializationError,
37
41
  } from "./errors.js";
package/src/models.ts CHANGED
@@ -31,8 +31,28 @@ export interface Task {
31
31
  completed_at_ms: number | null;
32
32
  }
33
33
 
34
+ /** The id + status pair the wait loop polls on (see get_status.sql) — a probe,
35
+ * not a snapshot: everything else about the task is deliberately not read. */
36
+ export interface TaskRef {
37
+ id: string;
38
+ status: TaskStatus;
39
+ }
40
+
34
41
  const JSON_COLUMNS = ["payload", "result", "error", "metadata"] as const;
35
- export const TERMINAL: TaskStatus[] = ["succeeded", "failed", "canceled"];
42
+ // As a const tuple so TerminalStatus derives from it — the same declare-once
43
+ // pattern as STATUSES/TaskStatus above.
44
+ export const TERMINAL = ["succeeded", "failed", "canceled"] as const;
45
+ export type TerminalStatus = (typeof TERMINAL)[number];
46
+
47
+ export function isTerminalStatus(status: TaskStatus): status is TerminalStatus {
48
+ return (TERMINAL as readonly TaskStatus[]).includes(status);
49
+ }
50
+
51
+ /** Map a probe row (see get_status.sql) to a TaskRef — the ref twin of
52
+ * rowToTask, so the row shape stays models' knowledge alone. */
53
+ export function rowToRef(row: Record<string, unknown>): TaskRef {
54
+ return { id: row.id as string, status: row.status as TaskStatus };
55
+ }
36
56
 
37
57
  export function rowToTask(row: Record<string, unknown>): Task {
38
58
  const t: Record<string, unknown> = { ...row };
@@ -46,8 +66,9 @@ export function rowToTask(row: Record<string, unknown>): Task {
46
66
  return t as unknown as Task;
47
67
  }
48
68
 
49
- export function isTerminal(task: Task): boolean {
50
- return TERMINAL.includes(task.status);
69
+ /** Accepts anything carrying a status — a Task or a TaskRef probe. */
70
+ export function isTerminal(task: Pick<Task, "status">): boolean {
71
+ return isTerminalStatus(task.status);
51
72
  }
52
73
 
53
74
  export function cancelRequested(task: Task): boolean {
@@ -0,0 +1,166 @@
1
+ import type { TaskStatus, TerminalStatus } from "./models.js";
2
+ import { validatePurgeInput, type PurgeInput, type TaskStore } from "./store/base.js";
3
+
4
+ /** Sweep every hour unless asked otherwise — often enough that a queue with a
5
+ * day of retention never carries more than an hour of extra rows, rare enough
6
+ * that the sweep is invisible next to the task traffic. */
7
+ const DEFAULT_INTERVAL_MS = 3_600_000;
8
+ /** Rows per purge statement. The same bound `purge` defaults to: big enough that
9
+ * a backlog drains in few statements, small enough that each is a short write. */
10
+ const DEFAULT_LIMIT = 1_000;
11
+
12
+ /** Per-status cutoffs. A status left out is never swept — granular retention is
13
+ * an explicit statement of what may go, not a default for what wasn't named. */
14
+ export type RetentionCutoffs = Partial<Record<TerminalStatus, number>>;
15
+
16
+ export interface RetentionOptions {
17
+ /**
18
+ * How long a terminal task is kept after it finished. Required: there is no
19
+ * safe default for how long someone else's results stay readable.
20
+ *
21
+ * A number keeps every terminal status the same time. Retention needs are
22
+ * often tiered — a succeeded row is spent once its result is consumed, while
23
+ * a failed one is worth keeping for diagnosis — so a per-status map sets a
24
+ * cutoff per status instead: `{ succeeded: 300_000, failed: 86_400_000 }`.
25
+ */
26
+ olderThanMs: number | RetentionCutoffs;
27
+ /** Time between sweeps. Default 3_600_000 (one hour). */
28
+ intervalMs?: number;
29
+ /** Rows deleted per statement while draining. Default 1_000. */
30
+ limit?: number;
31
+ /**
32
+ * Called for a sweep that threw. The next sweep runs on schedule regardless —
33
+ * a purge that failed because the database was busy is not a reason to stop
34
+ * retaining — so without this a store quietly stops being swept. Must not throw.
35
+ */
36
+ onError?: (err: unknown) => void;
37
+ }
38
+
39
+ /**
40
+ * Deletes terminal tasks on a schedule, for as long as the handle is open.
41
+ *
42
+ * `purge` exists because nothing else in CairnQ removes rows, and a queue whose
43
+ * payloads carry real data — an image, a document, a batch of embeddings — turns
44
+ * that into a disk leak measured in gigabytes per backfill. Every deployment
45
+ * that runs longer than a demo needs the sweep; leaving it to an external
46
+ * scheduler means the leak is the default and remembering is the opt-in.
47
+ *
48
+ * It sweeps in bounded batches with a yield between them, so draining a backlog
49
+ * that accumulated while nothing was sweeping stays a sequence of short writes
50
+ * rather than one long one — on SQLite that matters, since a long write holds
51
+ * the single write lock against every producer and worker on the file.
52
+ */
53
+ export class RetentionSweeper {
54
+ /** Whether the scheduled loop is running. */
55
+ private active = false;
56
+ /** Set by stop(), so a drain in progress can cut itself short too. */
57
+ private stopping = false;
58
+ /** Resolves the current sleep early, so stop() need not wait out an interval. */
59
+ private wake: (() => void) | null = null;
60
+ /** The loop itself, awaited by stop() so no purge outlives the store. */
61
+ private loop: Promise<void> | null = null;
62
+ private readonly intervalMs: number;
63
+ /** Rows per purge statement while draining — see DEFAULT_LIMIT. */
64
+ private readonly limit: number;
65
+ /** One purge per cutoff: a lone entry for a number, one per status for a map. */
66
+ private readonly purgeInputs: PurgeInput[];
67
+
68
+ constructor(
69
+ private readonly store: TaskStore,
70
+ private readonly opts: RetentionOptions,
71
+ ) {
72
+ this.intervalMs = opts.intervalMs ?? DEFAULT_INTERVAL_MS;
73
+ if (!Number.isFinite(this.intervalMs) || this.intervalMs < 1) {
74
+ throw new Error(`retention.intervalMs must be >= 1, got ${this.intervalMs}`);
75
+ }
76
+ this.limit = opts.limit ?? DEFAULT_LIMIT;
77
+ const cutoffs: [TaskStatus | undefined, number][] =
78
+ typeof opts.olderThanMs === "number"
79
+ ? [[undefined, opts.olderThanMs]]
80
+ : (Object.entries(opts.olderThanMs) as [TaskStatus, number][]);
81
+ // An empty map retains nothing and sweeps nothing — almost certainly a bug
82
+ // upstream of this call, so refuse it rather than silently never purging.
83
+ if (!cutoffs.length) {
84
+ throw new Error("retention.olderThanMs must name at least one status");
85
+ }
86
+ this.purgeInputs = cutoffs.map(([status, ms]) => ({
87
+ olderThanMs: ms,
88
+ status,
89
+ limit: this.limit,
90
+ }));
91
+ // Fail fast on the store's own purge rules (terminal status, cutoff >= 0):
92
+ // the sweep runs an hour from now, and its errors only surface via onError.
93
+ for (const input of this.purgeInputs) validatePurgeInput(input);
94
+ }
95
+
96
+ start(): void {
97
+ if (this.active) return;
98
+ this.active = true;
99
+ this.stopping = false;
100
+ this.loop = this.run();
101
+ }
102
+
103
+ /** Stop sweeping and wait for the sweep in flight, if any. */
104
+ async stop(): Promise<void> {
105
+ this.stopping = true;
106
+ this.active = false;
107
+ this.wake?.();
108
+ await this.loop;
109
+ this.loop = null;
110
+ }
111
+
112
+ private async run(): Promise<void> {
113
+ // Sleep first: a process that restarts often would otherwise purge on every
114
+ // boot, which is a write burst exactly when the store is busiest.
115
+ while (!this.stopping) {
116
+ await this.sleep(this.intervalMs);
117
+ if (this.stopping) return;
118
+ try {
119
+ await this.sweep();
120
+ } catch (err) {
121
+ try {
122
+ this.opts.onError?.(err);
123
+ } catch {
124
+ // A reporting hook must never take the sweep down with it — the same
125
+ // rule the worker's onError follows.
126
+ }
127
+ }
128
+ }
129
+ }
130
+
131
+ /**
132
+ * Delete everything past the cutoff now, in bounded batches, and return how
133
+ * many rows went. The scheduled loop calls this; call it directly to drain on
134
+ * demand — after a backfill, or from a maintenance command.
135
+ */
136
+ async sweep(): Promise<number> {
137
+ let deleted = 0;
138
+ for (const input of this.purgeInputs) {
139
+ for (;;) {
140
+ const ids = await this.store.purge(input);
141
+ deleted += ids.length;
142
+ if (this.stopping) return deleted;
143
+ if (ids.length < this.limit) break;
144
+ // Hand the loop back between batches: a large drain must not starve the
145
+ // submits and claims sharing this process.
146
+ await this.sleep(0);
147
+ }
148
+ }
149
+ return deleted;
150
+ }
151
+
152
+ /** Sleep, interruptible by stop(). Unref'd: retention is housekeeping, and a
153
+ * pending sweep must never be the reason a process refuses to exit. */
154
+ private sleep(ms: number): Promise<void> {
155
+ return new Promise<void>((resolve) => {
156
+ const timer = setTimeout(resolve, ms);
157
+ timer.unref?.();
158
+ this.wake = () => {
159
+ clearTimeout(timer);
160
+ resolve();
161
+ };
162
+ }).finally(() => {
163
+ this.wake = null;
164
+ });
165
+ }
166
+ }
package/src/store/base.ts CHANGED
@@ -6,7 +6,15 @@ import {
6
6
  ProtocolVersionMismatch,
7
7
  SerializationError,
8
8
  } from "../errors.js";
9
- import { rowToTask, STATUSES, type Task, type TaskStatus } from "../models.js";
9
+ import {
10
+ isTerminalStatus,
11
+ rowToRef,
12
+ rowToTask,
13
+ STATUSES,
14
+ type Task,
15
+ type TaskRef,
16
+ type TaskStatus,
17
+ } from "../models.js";
10
18
  import { type BackpressureOptions, QueueDepthGate } from "../backpressure.js";
11
19
 
12
20
  const rejectMangled = function (this: unknown, _key: string, v: unknown): unknown {
@@ -62,9 +70,25 @@ export function checkProtocolVersion(version: number): void {
62
70
  // CONFLICTS is the canonical declaration; the type derives from it so the
63
71
  // runtime guard in submit() and the type can't drift apart (same pattern as
64
72
  // STATUSES/TaskStatus in models.ts).
65
- const CONFLICTS = ["reuse", "reject", "replace"] as const;
73
+ const CONFLICTS = ["reuse", "reuse-succeeded", "reject", "replace"] as const;
66
74
  export type Conflict = (typeof CONFLICTS)[number];
67
75
 
76
+ /**
77
+ * Whether a keyed submit's strategy accepts the task the key already points at.
78
+ *
79
+ * Both reuse strategies deduplicate work that is still in play — that is what a
80
+ * key is for, and the answer cannot depend on the outcome of a task that has no
81
+ * outcome yet. They differ only on what a *finished* task means: `reuse` treats
82
+ * the key as free again, while `reuse-succeeded` reads a succeeded task as a
83
+ * cached result. Neither ever hands back a failed or canceled one, which would
84
+ * poison the key for every later submit (see PROTOCOL.md "Key conflict").
85
+ */
86
+ function reusable(conflict: Conflict, status: TaskStatus): boolean {
87
+ if (conflict === "replace") return false;
88
+ if (!isTerminalStatus(status)) return true;
89
+ return conflict === "reuse-succeeded" && status === "succeeded";
90
+ }
91
+
68
92
  /** The queue a submit lands on when it names none. Owned here, where the
69
93
  * default is applied, so nothing above has to re-derive it. */
70
94
  export const DEFAULT_QUEUE = "default";
@@ -94,8 +118,32 @@ export interface ListInput {
94
118
  offset?: number;
95
119
  }
96
120
 
121
+ /** Validate a purge's inputs. Shared with RetentionSweeper, which fail-fasts at
122
+ * construction on the same rules an hourly sweep would otherwise only surface
123
+ * through its onError hook — one statement of the rules, two callers. */
124
+ export function validatePurgeInput(input: PurgeInput): void {
125
+ if (input.olderThanMs != null && (!Number.isFinite(input.olderThanMs) || input.olderThanMs < 0)) {
126
+ throw new Error(`olderThanMs must be >= 0, got ${input.olderThanMs}`);
127
+ }
128
+ if (input.limit != null && input.limit < 1) {
129
+ throw new Error(`limit must be >= 1, got ${input.limit}`);
130
+ }
131
+ // Terminal only: purge never deletes live work, so accepting `queued` here
132
+ // would be accepting a filter that silently matches nothing.
133
+ if (input.status != null && !isTerminalStatus(input.status)) {
134
+ throw new Error(`status must be terminal, got ${input.status}`);
135
+ }
136
+ }
137
+
97
138
  export interface PurgeInput {
98
139
  olderThanMs?: number;
140
+ /** Restrict the sweep to one terminal status. Retention needs are tiered —
141
+ * succeeded rows are spent once their result is consumed, failed ones are
142
+ * worth keeping for diagnosis — and without this the shortest-lived tier
143
+ * sets the retention for every row. Absent means all terminal statuses. */
144
+ status?: TaskStatus;
145
+ /** Restrict the sweep to one task name. Absent means all names. */
146
+ name?: string;
99
147
  limit?: number;
100
148
  }
101
149
 
@@ -218,6 +266,10 @@ export abstract class TaskStore {
218
266
  return rows.length ? rowToTask(rows[0]) : null;
219
267
  }
220
268
 
269
+ private static oneRef(rows: any[]): TaskRef | null {
270
+ return rows.length ? rowToRef(rows[0]) : null;
271
+ }
272
+
221
273
  // ------------------------------------------------------------- client side
222
274
  /**
223
275
  * Bound how deep a queue may get before `submit` blocks. Off unless set.
@@ -283,10 +335,15 @@ export abstract class TaskStore {
283
335
  // free after all, whatever the strategy.
284
336
  const current = (await fetch("get", { id: existing[0].task_id }))[0];
285
337
  if (current) {
286
- if (conflict === "reuse") return rowToTask(current);
287
338
  if (conflict === "reject") throw new AlreadyExists(key);
288
- // "replace": cancel the recorded task, then repoint the key below.
289
- await fetch("cancel", { id: existing[0].task_id });
339
+ if (reusable(conflict, current.status as TaskStatus)) return rowToTask(current);
340
+ // The strategy declined the recorded task, so the key repoints to the
341
+ // fresh one inserted below. Cancel only what is still live: a terminal
342
+ // task has nothing to stop, and cancelling it would rewrite a settled
343
+ // row (and hand a `canceled` back to whoever is waiting on it).
344
+ if (!isTerminalStatus(current.status as TaskStatus)) {
345
+ await fetch("cancel", { id: existing[0].task_id });
346
+ }
290
347
  }
291
348
  }
292
349
  const row = (await fetch("insert_task", ins))[0];
@@ -303,6 +360,18 @@ export abstract class TaskStore {
303
360
  return TaskStore.one(await this.fetch("get_by_key", { key }));
304
361
  }
305
362
 
363
+ /** The wait loop's probe: id + status alone, so polling a task with a large
364
+ * payload does not re-read and re-parse that payload on every beat. */
365
+ async getStatus(taskId: string): Promise<TaskRef | null> {
366
+ return TaskStore.oneRef(await this.fetch("get_status", { id: taskId }));
367
+ }
368
+
369
+ /** getStatus, following a key instead of an id — re-resolved per call, so a
370
+ * `replace` moves the probe onto the new task. */
371
+ async getStatusByKey(key: string): Promise<TaskRef | null> {
372
+ return TaskStore.oneRef(await this.fetch("get_status_by_key", { key }));
373
+ }
374
+
306
375
  async list(input: ListInput = {}): Promise<Task[]> {
307
376
  // Validate up front, like submit's conflict guard: a typo'd status otherwise
308
377
  // matches nothing and returns [] indistinguishably from "no such tasks".
@@ -364,14 +433,11 @@ export abstract class TaskStore {
364
433
  * call it in a loop until it returns fewer than `limit`.
365
434
  */
366
435
  async purge(input: PurgeInput = {}): Promise<string[]> {
367
- if (input.olderThanMs != null && input.olderThanMs < 0) {
368
- throw new Error(`olderThanMs must be >= 0, got ${input.olderThanMs}`);
369
- }
370
- if (input.limit != null && input.limit < 1) {
371
- throw new Error(`limit must be >= 1, got ${input.limit}`);
372
- }
436
+ validatePurgeInput(input);
373
437
  const rows = await this.fetch("purge", {
374
438
  older_than_ms: input.olderThanMs ?? 0,
439
+ status: input.status ?? null,
440
+ name: input.name ?? null,
375
441
  limit: input.limit ?? 1_000,
376
442
  });
377
443
  return rows.map((r) => r.id as string);
package/src/wait.ts CHANGED
@@ -1,12 +1,25 @@
1
1
  import { TaskTimeout } from "./errors.js";
2
2
  import { nowMs } from "./ids.js";
3
- import { isTerminal, type Task } from "./models.js";
3
+ import { isTerminal, type Task, type TaskRef } from "./models.js";
4
4
  import type { TaskStore } from "./store/base.js";
5
5
 
6
+ export const DEFAULT_WAIT_TIMEOUT_MS = 30_000;
6
7
  export const DEFAULT_POLL_MS = 100;
7
8
  export const MAX_POLL_MS = 500;
8
9
  const GROWTH = 1.5;
9
10
 
11
+ const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
12
+
13
+ export interface PollOptions {
14
+ timeoutMs: number;
15
+ /** The first poll interval (default 100). */
16
+ pollMs?: number;
17
+ /** Ceiling the poll interval backs off to (default 500). Worth raising for a
18
+ * task known to take minutes — fewer reads — or lowering when shaving the
19
+ * average half-interval of completion-detection latency matters. */
20
+ maxPollMs?: number;
21
+ }
22
+
10
23
  /**
11
24
  * Grow the polling interval towards the ceiling.
12
25
  *
@@ -19,28 +32,80 @@ export function nextPollMs(current: number, maxMs: number): number {
19
32
  return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
20
33
  }
21
34
 
22
- /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
23
- * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
24
- * it backs off towards `maxPollMs`. */
25
- export async function pollWait(
26
- store: TaskStore,
27
- taskId: string,
28
- {
29
- timeoutMs,
30
- pollMs = DEFAULT_POLL_MS,
31
- maxPollMs = MAX_POLL_MS,
32
- }: { timeoutMs: number; pollMs?: number; maxPollMs?: number },
35
+ /**
36
+ * Poll `probe` until it reports a terminal status, then return the full task
37
+ * via `read`; or throw once the timeout elapses.
38
+ *
39
+ * The loop's repeated read is the status-only `probe` (see get_status.sql): a
40
+ * waiting caller asks nothing but "is it finished yet", and re-reading the whole
41
+ * row would drag the payload back — and re-parse it — on every beat for the life
42
+ * of the wait. The full row is read once, when the probe turns terminal or, on
43
+ * the timeout beat, for the error's snapshot. Between the probe and that read
44
+ * the row can vanish (purge) or the key repoint (`replace`); a read that comes
45
+ * back empty or non-terminal is simply not finished, and the loop keeps polling.
46
+ *
47
+ * `wake` is what the loop sleeps on between reads: a store with a push channel
48
+ * (Postgres) cuts it short when the task goes terminal, but the re-probe is the
49
+ * source of truth either way, so a plain sleep is always a correct answer.
50
+ */
51
+ async function poll(
52
+ probe: () => Promise<TaskRef | null>,
53
+ read: () => Promise<Task | null>,
54
+ wake: (ref: TaskRef | null, ms: number) => Promise<void>,
55
+ subject: string,
56
+ key: string | null,
57
+ { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS }: PollOptions,
33
58
  ): Promise<Task> {
34
59
  const deadline = nowMs() + timeoutMs;
35
60
  let interval = pollMs;
36
61
  for (;;) {
37
- const task = await store.get(taskId);
38
- if (task && isTerminal(task)) return task;
62
+ const ref = await probe();
39
63
  const remaining = deadline - nowMs();
40
- if (remaining <= 0) throw new TaskTimeout(taskId, { timeoutMs, task });
41
- // A store with a push channel (Postgres) cuts the sleep short when the task
42
- // goes terminal; the re-get above stays the source of truth either way.
43
- await store.taskDoneWake(taskId, Math.min(interval, remaining));
64
+ // The one full-read site: when the probe says finished, or on the timeout
65
+ // beat for the error's stuck-in-what-state snapshot. No ref means no row,
66
+ // so there is nothing for a read to add to either case.
67
+ const task = ref && (isTerminal(ref) || remaining <= 0) ? await read() : null;
68
+ if (task && isTerminal(task)) return task;
69
+ if (remaining <= 0) throw new TaskTimeout(ref?.id ?? subject, { timeoutMs, task, key });
70
+ await wake(ref, Math.min(interval, remaining));
44
71
  interval = nextPollMs(interval, maxPollMs);
45
72
  }
46
73
  }
74
+
75
+ /** Poll the task's status until terminal or timeout. Returns the terminal Task
76
+ * (any status). Throws TaskTimeout, leaving the task running. `pollMs` is the
77
+ * *first* interval; it backs off towards `maxPollMs`. */
78
+ export function pollWait(store: TaskStore, taskId: string, opts: PollOptions): Promise<Task> {
79
+ return poll(
80
+ () => store.getStatus(taskId),
81
+ () => store.get(taskId),
82
+ (_ref, ms) => store.taskDoneWake(taskId, ms),
83
+ taskId,
84
+ null,
85
+ opts,
86
+ );
87
+ }
88
+
89
+ /**
90
+ * The same wait, following a key instead of an id.
91
+ *
92
+ * The key is re-resolved on every probe, because that is what a key means: a
93
+ * pointer to the task that is *current* under it. A `replace` landing mid-wait
94
+ * moves the wait onto the new task rather than reporting the cancellation of the
95
+ * old one, and a key that points at nothing yet is simply not finished — it
96
+ * polls until something appears, the same way waiting on an id that does not
97
+ * exist yet does.
98
+ *
99
+ * There is nothing to subscribe to before the key resolves, so those naps are
100
+ * plain sleeps; once it resolves, the store's push channel applies as usual.
101
+ */
102
+ export function pollWaitByKey(store: TaskStore, key: string, opts: PollOptions): Promise<Task> {
103
+ return poll(
104
+ () => store.getStatusByKey(key),
105
+ () => store.getByKey(key),
106
+ (ref, ms) => (ref ? store.taskDoneWake(ref.id, ms) : sleep(ms)),
107
+ key,
108
+ key,
109
+ opts,
110
+ );
111
+ }
package/src/worker.ts CHANGED
@@ -4,6 +4,7 @@ import { TaskContext } from "./context.js";
4
4
  import {
5
5
  asEnvelope,
6
6
  errorEnvelope,
7
+ EventLoopBlocked,
7
8
  exceptionEnvelope,
8
9
  type FailReason,
9
10
  LostLease,
@@ -923,6 +924,28 @@ export class Worker {
923
924
  let wake: (() => void) | null = null;
924
925
  // lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
925
926
  const interval = this.opts.heartbeatIntervalMs ?? Math.max(50, Math.floor(leaseMs / 3));
927
+
928
+ // When this heartbeat last ran. A loop cannot report its own absence — a
929
+ // handler that blocks for its whole attempt never lets the timer fire at all
930
+ // — so the check lives outside the loop and cancel() runs it too.
931
+ let lastBeatAt = Date.now();
932
+ /** Report a heartbeat that has not run for more than two intervals: a whole
933
+ * beat missed, which at lease/3 means the next such block loses the lease
934
+ * outright. Fires while the lease still holds — after it expires the only
935
+ * evidence is a task that ran twice, in two workers' logs, with no error in
936
+ * either. */
937
+ const checkBeat = (): void => {
938
+ const now = Date.now();
939
+ const lateMs = now - lastBeatAt - interval;
940
+ if (lateMs > interval) {
941
+ this.report(new EventLoopBlocked(lateMs, interval, leaseMs), {
942
+ phase: "execute",
943
+ taskId: ctxs[0].taskId,
944
+ });
945
+ }
946
+ lastBeatAt = now;
947
+ };
948
+
926
949
  const done = (async () => {
927
950
  while (active) {
928
951
  // Cancellable sleep: cancel() resolves this immediately and clears the
@@ -936,6 +959,7 @@ export class Worker {
936
959
  });
937
960
  wake = null;
938
961
  if (!active) break;
962
+ checkBeat();
939
963
  const live = ctxs.filter((c) => !c.settled && !c.lostLease);
940
964
  if (!live.length) break;
941
965
  try {
@@ -960,6 +984,10 @@ export class Worker {
960
984
  cancel: () => {
961
985
  active = false;
962
986
  if (wake) wake();
987
+ // The attempt is over, so this is the last chance to notice that the
988
+ // heartbeat never got one — the whole-attempt block, which is also the
989
+ // one where the lease is already gone.
990
+ checkBeat();
963
991
  },
964
992
  done,
965
993
  };