cairnq 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,137 @@
1
+ import { validatePurgeInput } from "./store/base.js";
2
+ /** Sweep every hour unless asked otherwise — often enough that a queue with a
3
+ * day of retention never carries more than an hour of extra rows, rare enough
4
+ * that the sweep is invisible next to the task traffic. */
5
+ const DEFAULT_INTERVAL_MS = 3_600_000;
6
+ /** Rows per purge statement. The same bound `purge` defaults to: big enough that
7
+ * a backlog drains in few statements, small enough that each is a short write. */
8
+ const DEFAULT_LIMIT = 1_000;
9
+ /**
10
+ * Deletes terminal tasks on a schedule, for as long as the handle is open.
11
+ *
12
+ * `purge` exists because nothing else in CairnQ removes rows, and a queue whose
13
+ * payloads carry real data — an image, a document, a batch of embeddings — turns
14
+ * that into a disk leak measured in gigabytes per backfill. Every deployment
15
+ * that runs longer than a demo needs the sweep; leaving it to an external
16
+ * scheduler means the leak is the default and remembering is the opt-in.
17
+ *
18
+ * It sweeps in bounded batches with a yield between them, so draining a backlog
19
+ * that accumulated while nothing was sweeping stays a sequence of short writes
20
+ * rather than one long one — on SQLite that matters, since a long write holds
21
+ * the single write lock against every producer and worker on the file.
22
+ */
23
+ export class RetentionSweeper {
24
+ store;
25
+ opts;
26
+ /** Whether the scheduled loop is running. */
27
+ active = false;
28
+ /** Set by stop(), so a drain in progress can cut itself short too. */
29
+ stopping = false;
30
+ /** Resolves the current sleep early, so stop() need not wait out an interval. */
31
+ wake = null;
32
+ /** The loop itself, awaited by stop() so no purge outlives the store. */
33
+ loop = null;
34
+ intervalMs;
35
+ /** Rows per purge statement while draining — see DEFAULT_LIMIT. */
36
+ limit;
37
+ /** One purge per cutoff: a lone entry for a number, one per status for a map. */
38
+ purgeInputs;
39
+ constructor(store, opts) {
40
+ this.store = store;
41
+ this.opts = opts;
42
+ this.intervalMs = opts.intervalMs ?? DEFAULT_INTERVAL_MS;
43
+ if (!Number.isFinite(this.intervalMs) || this.intervalMs < 1) {
44
+ throw new Error(`retention.intervalMs must be >= 1, got ${this.intervalMs}`);
45
+ }
46
+ this.limit = opts.limit ?? DEFAULT_LIMIT;
47
+ const cutoffs = typeof opts.olderThanMs === "number"
48
+ ? [[undefined, opts.olderThanMs]]
49
+ : Object.entries(opts.olderThanMs);
50
+ // An empty map retains nothing and sweeps nothing — almost certainly a bug
51
+ // upstream of this call, so refuse it rather than silently never purging.
52
+ if (!cutoffs.length) {
53
+ throw new Error("retention.olderThanMs must name at least one status");
54
+ }
55
+ this.purgeInputs = cutoffs.map(([status, ms]) => ({
56
+ olderThanMs: ms,
57
+ status,
58
+ limit: this.limit,
59
+ }));
60
+ // Fail fast on the store's own purge rules (terminal status, cutoff >= 0):
61
+ // the sweep runs an hour from now, and its errors only surface via onError.
62
+ for (const input of this.purgeInputs)
63
+ validatePurgeInput(input);
64
+ }
65
+ start() {
66
+ if (this.active)
67
+ return;
68
+ this.active = true;
69
+ this.stopping = false;
70
+ this.loop = this.run();
71
+ }
72
+ /** Stop sweeping and wait for the sweep in flight, if any. */
73
+ async stop() {
74
+ this.stopping = true;
75
+ this.active = false;
76
+ this.wake?.();
77
+ await this.loop;
78
+ this.loop = null;
79
+ }
80
+ async run() {
81
+ // Sleep first: a process that restarts often would otherwise purge on every
82
+ // boot, which is a write burst exactly when the store is busiest.
83
+ while (!this.stopping) {
84
+ await this.sleep(this.intervalMs);
85
+ if (this.stopping)
86
+ return;
87
+ try {
88
+ await this.sweep();
89
+ }
90
+ catch (err) {
91
+ try {
92
+ this.opts.onError?.(err);
93
+ }
94
+ catch {
95
+ // A reporting hook must never take the sweep down with it — the same
96
+ // rule the worker's onError follows.
97
+ }
98
+ }
99
+ }
100
+ }
101
+ /**
102
+ * Delete everything past the cutoff now, in bounded batches, and return how
103
+ * many rows went. The scheduled loop calls this; call it directly to drain on
104
+ * demand — after a backfill, or from a maintenance command.
105
+ */
106
+ async sweep() {
107
+ let deleted = 0;
108
+ for (const input of this.purgeInputs) {
109
+ for (;;) {
110
+ const ids = await this.store.purge(input);
111
+ deleted += ids.length;
112
+ if (this.stopping)
113
+ return deleted;
114
+ if (ids.length < this.limit)
115
+ break;
116
+ // Hand the loop back between batches: a large drain must not starve the
117
+ // submits and claims sharing this process.
118
+ await this.sleep(0);
119
+ }
120
+ }
121
+ return deleted;
122
+ }
123
+ /** Sleep, interruptible by stop(). Unref'd: retention is housekeeping, and a
124
+ * pending sweep must never be the reason a process refuses to exit. */
125
+ sleep(ms) {
126
+ return new Promise((resolve) => {
127
+ const timer = setTimeout(resolve, ms);
128
+ timer.unref?.();
129
+ this.wake = () => {
130
+ clearTimeout(timer);
131
+ resolve();
132
+ };
133
+ }).finally(() => {
134
+ this.wake = null;
135
+ });
136
+ }
137
+ }
@@ -1,4 +1,4 @@
1
- import { type Task, type TaskStatus } from "../models.js";
1
+ import { type Task, type TaskRef, type TaskStatus } from "../models.js";
2
2
  import { type BackpressureOptions } from "../backpressure.js";
3
3
  /** Encode a value for a protocol JSON column, raising SerializationError on
4
4
  * anything JSON cannot represent. Refuses what JSON.stringify would silently
@@ -11,7 +11,7 @@ export declare function dumpJson(value: unknown): string;
11
11
  * The supported major is a protocol fact, not a dialect one — every backend
12
12
  * checks it here so the constant can't fork per store. */
13
13
  export declare function checkProtocolVersion(version: number): void;
14
- declare const CONFLICTS: readonly ["reuse", "reject", "replace"];
14
+ declare const CONFLICTS: readonly ["reuse", "reuse-succeeded", "reject", "replace"];
15
15
  export type Conflict = (typeof CONFLICTS)[number];
16
16
  /** The queue a submit lands on when it names none. Owned here, where the
17
17
  * default is applied, so nothing above has to re-derive it. */
@@ -39,8 +39,19 @@ export interface ListInput {
39
39
  limit?: number;
40
40
  offset?: number;
41
41
  }
42
+ /** Validate a purge's inputs. Shared with RetentionSweeper, which fail-fasts at
43
+ * construction on the same rules an hourly sweep would otherwise only surface
44
+ * through its onError hook — one statement of the rules, two callers. */
45
+ export declare function validatePurgeInput(input: PurgeInput): void;
42
46
  export interface PurgeInput {
43
47
  olderThanMs?: number;
48
+ /** Restrict the sweep to one terminal status. Retention needs are tiered —
49
+ * succeeded rows are spent once their result is consumed, failed ones are
50
+ * worth keeping for diagnosis — and without this the shortest-lived tier
51
+ * sets the retention for every row. Absent means all terminal statuses. */
52
+ status?: TaskStatus;
53
+ /** Restrict the sweep to one task name. Absent means all names. */
54
+ name?: string;
44
55
  limit?: number;
45
56
  }
46
57
  export type Params = Record<string, unknown>;
@@ -109,6 +120,7 @@ export declare abstract class TaskStore {
109
120
  */
110
121
  private ownedWrite;
111
122
  private static one;
123
+ private static oneRef;
112
124
  /**
113
125
  * Bound how deep a queue may get before `submit` blocks. Off unless set.
114
126
  *
@@ -121,6 +133,12 @@ export declare abstract class TaskStore {
121
133
  submit(input: SubmitInput): Promise<Task>;
122
134
  get(taskId: string): Promise<Task | null>;
123
135
  getByKey(key: string): Promise<Task | null>;
136
+ /** The wait loop's probe: id + status alone, so polling a task with a large
137
+ * payload does not re-read and re-parse that payload on every beat. */
138
+ getStatus(taskId: string): Promise<TaskRef | null>;
139
+ /** getStatus, following a key instead of an id — re-resolved per call, so a
140
+ * `replace` moves the probe onto the new task. */
141
+ getStatusByKey(key: string): Promise<TaskRef | null>;
124
142
  list(input?: ListInput): Promise<Task[]>;
125
143
  cancel(taskId: string): Promise<Task | null>;
126
144
  retry(taskId: string, opts?: {
@@ -1,6 +1,6 @@
1
1
  import { newId } from "../ids.js";
2
2
  import { AlreadyExists, errorEnvelope, LostLease, ProtocolVersionMismatch, SerializationError, } from "../errors.js";
3
- import { rowToTask, STATUSES } from "../models.js";
3
+ import { isTerminalStatus, rowToRef, rowToTask, STATUSES, } from "../models.js";
4
4
  import { QueueDepthGate } from "../backpressure.js";
5
5
  const rejectMangled = function (_key, v) {
6
6
  if (typeof v === "number" && !Number.isFinite(v)) {
@@ -51,10 +51,43 @@ export function checkProtocolVersion(version) {
51
51
  // CONFLICTS is the canonical declaration; the type derives from it so the
52
52
  // runtime guard in submit() and the type can't drift apart (same pattern as
53
53
  // STATUSES/TaskStatus in models.ts).
54
- const CONFLICTS = ["reuse", "reject", "replace"];
54
+ const CONFLICTS = ["reuse", "reuse-succeeded", "reject", "replace"];
55
+ /**
56
+ * Whether a keyed submit's strategy accepts the task the key already points at.
57
+ *
58
+ * Both reuse strategies deduplicate work that is still in play — that is what a
59
+ * key is for, and the answer cannot depend on the outcome of a task that has no
60
+ * outcome yet. They differ only on what a *finished* task means: `reuse` treats
61
+ * the key as free again, while `reuse-succeeded` reads a succeeded task as a
62
+ * cached result. Neither ever hands back a failed or canceled one, which would
63
+ * poison the key for every later submit (see PROTOCOL.md "Key conflict").
64
+ */
65
+ function reusable(conflict, status) {
66
+ if (conflict === "replace")
67
+ return false;
68
+ if (!isTerminalStatus(status))
69
+ return true;
70
+ return conflict === "reuse-succeeded" && status === "succeeded";
71
+ }
55
72
  /** The queue a submit lands on when it names none. Owned here, where the
56
73
  * default is applied, so nothing above has to re-derive it. */
57
74
  export const DEFAULT_QUEUE = "default";
75
+ /** Validate a purge's inputs. Shared with RetentionSweeper, which fail-fasts at
76
+ * construction on the same rules an hourly sweep would otherwise only surface
77
+ * through its onError hook — one statement of the rules, two callers. */
78
+ export function validatePurgeInput(input) {
79
+ if (input.olderThanMs != null && (!Number.isFinite(input.olderThanMs) || input.olderThanMs < 0)) {
80
+ throw new Error(`olderThanMs must be >= 0, got ${input.olderThanMs}`);
81
+ }
82
+ if (input.limit != null && input.limit < 1) {
83
+ throw new Error(`limit must be >= 1, got ${input.limit}`);
84
+ }
85
+ // Terminal only: purge never deletes live work, so accepting `queued` here
86
+ // would be accepting a filter that silently matches nothing.
87
+ if (input.status != null && !isTerminalStatus(input.status)) {
88
+ throw new Error(`status must be terminal, got ${input.status}`);
89
+ }
90
+ }
58
91
  export const LEASE_EXPIRED_ERROR_JSON = dumpJson(errorEnvelope({
59
92
  type: "LeaseExpired",
60
93
  code: "lease_expired",
@@ -142,6 +175,9 @@ export class TaskStore {
142
175
  static one(rows) {
143
176
  return rows.length ? rowToTask(rows[0]) : null;
144
177
  }
178
+ static oneRef(rows) {
179
+ return rows.length ? rowToRef(rows[0]) : null;
180
+ }
145
181
  // ------------------------------------------------------------- client side
146
182
  /**
147
183
  * Bound how deep a queue may get before `submit` blocks. Off unless set.
@@ -207,12 +243,17 @@ export class TaskStore {
207
243
  // free after all, whatever the strategy.
208
244
  const current = (await fetch("get", { id: existing[0].task_id }))[0];
209
245
  if (current) {
210
- if (conflict === "reuse")
211
- return rowToTask(current);
212
246
  if (conflict === "reject")
213
247
  throw new AlreadyExists(key);
214
- // "replace": cancel the recorded task, then repoint the key below.
215
- await fetch("cancel", { id: existing[0].task_id });
248
+ if (reusable(conflict, current.status))
249
+ return rowToTask(current);
250
+ // The strategy declined the recorded task, so the key repoints to the
251
+ // fresh one inserted below. Cancel only what is still live: a terminal
252
+ // task has nothing to stop, and cancelling it would rewrite a settled
253
+ // row (and hand a `canceled` back to whoever is waiting on it).
254
+ if (!isTerminalStatus(current.status)) {
255
+ await fetch("cancel", { id: existing[0].task_id });
256
+ }
216
257
  }
217
258
  }
218
259
  const row = (await fetch("insert_task", ins))[0];
@@ -226,6 +267,16 @@ export class TaskStore {
226
267
  async getByKey(key) {
227
268
  return TaskStore.one(await this.fetch("get_by_key", { key }));
228
269
  }
270
+ /** The wait loop's probe: id + status alone, so polling a task with a large
271
+ * payload does not re-read and re-parse that payload on every beat. */
272
+ async getStatus(taskId) {
273
+ return TaskStore.oneRef(await this.fetch("get_status", { id: taskId }));
274
+ }
275
+ /** getStatus, following a key instead of an id — re-resolved per call, so a
276
+ * `replace` moves the probe onto the new task. */
277
+ async getStatusByKey(key) {
278
+ return TaskStore.oneRef(await this.fetch("get_status_by_key", { key }));
279
+ }
229
280
  async list(input = {}) {
230
281
  // Validate up front, like submit's conflict guard: a typo'd status otherwise
231
282
  // matches nothing and returns [] indistinguishably from "no such tasks".
@@ -280,14 +331,11 @@ export class TaskStore {
280
331
  * call it in a loop until it returns fewer than `limit`.
281
332
  */
282
333
  async purge(input = {}) {
283
- if (input.olderThanMs != null && input.olderThanMs < 0) {
284
- throw new Error(`olderThanMs must be >= 0, got ${input.olderThanMs}`);
285
- }
286
- if (input.limit != null && input.limit < 1) {
287
- throw new Error(`limit must be >= 1, got ${input.limit}`);
288
- }
334
+ validatePurgeInput(input);
289
335
  const rows = await this.fetch("purge", {
290
336
  older_than_ms: input.olderThanMs ?? 0,
337
+ status: input.status ?? null,
338
+ name: input.name ?? null,
291
339
  limit: input.limit ?? 1_000,
292
340
  });
293
341
  return rows.map((r) => r.id);
package/dist/wait.d.ts CHANGED
@@ -1,7 +1,17 @@
1
1
  import { type Task } from "./models.js";
2
2
  import type { TaskStore } from "./store/base.js";
3
+ export declare const DEFAULT_WAIT_TIMEOUT_MS = 30000;
3
4
  export declare const DEFAULT_POLL_MS = 100;
4
5
  export declare const MAX_POLL_MS = 500;
6
+ export interface PollOptions {
7
+ timeoutMs: number;
8
+ /** The first poll interval (default 100). */
9
+ pollMs?: number;
10
+ /** Ceiling the poll interval backs off to (default 500). Worth raising for a
11
+ * task known to take minutes — fewer reads — or lowering when shaving the
12
+ * average half-interval of completion-detection latency matters. */
13
+ maxPollMs?: number;
14
+ }
5
15
  /**
6
16
  * Grow the polling interval towards the ceiling.
7
17
  *
@@ -11,11 +21,21 @@ export declare const MAX_POLL_MS = 500;
11
21
  * Math.floor(1 * 1.5) === 1 would otherwise never grow past 1.
12
22
  */
13
23
  export declare function nextPollMs(current: number, maxMs: number): number;
14
- /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
15
- * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
16
- * it backs off towards `maxPollMs`. */
17
- export declare function pollWait(store: TaskStore, taskId: string, { timeoutMs, pollMs, maxPollMs, }: {
18
- timeoutMs: number;
19
- pollMs?: number;
20
- maxPollMs?: number;
21
- }): Promise<Task>;
24
+ /** Poll the task's status until terminal or timeout. Returns the terminal Task
25
+ * (any status). Throws TaskTimeout, leaving the task running. `pollMs` is the
26
+ * *first* interval; it backs off towards `maxPollMs`. */
27
+ export declare function pollWait(store: TaskStore, taskId: string, opts: PollOptions): Promise<Task>;
28
+ /**
29
+ * The same wait, following a key instead of an id.
30
+ *
31
+ * The key is re-resolved on every probe, because that is what a key means: a
32
+ * pointer to the task that is *current* under it. A `replace` landing mid-wait
33
+ * moves the wait onto the new task rather than reporting the cancellation of the
34
+ * old one, and a key that points at nothing yet is simply not finished — it
35
+ * polls until something appears, the same way waiting on an id that does not
36
+ * exist yet does.
37
+ *
38
+ * There is nothing to subscribe to before the key resolves, so those naps are
39
+ * plain sleeps; once it resolves, the store's push channel applies as usual.
40
+ */
41
+ export declare function pollWaitByKey(store: TaskStore, key: string, opts: PollOptions): Promise<Task>;
package/dist/wait.js CHANGED
@@ -1,9 +1,11 @@
1
1
  import { TaskTimeout } from "./errors.js";
2
2
  import { nowMs } from "./ids.js";
3
3
  import { isTerminal } from "./models.js";
4
+ export const DEFAULT_WAIT_TIMEOUT_MS = 30_000;
4
5
  export const DEFAULT_POLL_MS = 100;
5
6
  export const MAX_POLL_MS = 500;
6
7
  const GROWTH = 1.5;
8
+ const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
7
9
  /**
8
10
  * Grow the polling interval towards the ceiling.
9
11
  *
@@ -15,22 +17,59 @@ const GROWTH = 1.5;
15
17
  export function nextPollMs(current, maxMs) {
16
18
  return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
17
19
  }
18
- /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
19
- * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
20
- * it backs off towards `maxPollMs`. */
21
- export async function pollWait(store, taskId, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS, }) {
20
+ /**
21
+ * Poll `probe` until it reports a terminal status, then return the full task
22
+ * via `read`; or throw once the timeout elapses.
23
+ *
24
+ * The loop's repeated read is the status-only `probe` (see get_status.sql): a
25
+ * waiting caller asks nothing but "is it finished yet", and re-reading the whole
26
+ * row would drag the payload back — and re-parse it — on every beat for the life
27
+ * of the wait. The full row is read once, when the probe turns terminal or, on
28
+ * the timeout beat, for the error's snapshot. Between the probe and that read
29
+ * the row can vanish (purge) or the key repoint (`replace`); a read that comes
30
+ * back empty or non-terminal is simply not finished, and the loop keeps polling.
31
+ *
32
+ * `wake` is what the loop sleeps on between reads: a store with a push channel
33
+ * (Postgres) cuts it short when the task goes terminal, but the re-probe is the
34
+ * source of truth either way, so a plain sleep is always a correct answer.
35
+ */
36
+ async function poll(probe, read, wake, subject, key, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS }) {
22
37
  const deadline = nowMs() + timeoutMs;
23
38
  let interval = pollMs;
24
39
  for (;;) {
25
- const task = await store.get(taskId);
40
+ const ref = await probe();
41
+ const remaining = deadline - nowMs();
42
+ // The one full-read site: when the probe says finished, or on the timeout
43
+ // beat for the error's stuck-in-what-state snapshot. No ref means no row,
44
+ // so there is nothing for a read to add to either case.
45
+ const task = ref && (isTerminal(ref) || remaining <= 0) ? await read() : null;
26
46
  if (task && isTerminal(task))
27
47
  return task;
28
- const remaining = deadline - nowMs();
29
48
  if (remaining <= 0)
30
- throw new TaskTimeout(taskId, { timeoutMs, task });
31
- // A store with a push channel (Postgres) cuts the sleep short when the task
32
- // goes terminal; the re-get above stays the source of truth either way.
33
- await store.taskDoneWake(taskId, Math.min(interval, remaining));
49
+ throw new TaskTimeout(ref?.id ?? subject, { timeoutMs, task, key });
50
+ await wake(ref, Math.min(interval, remaining));
34
51
  interval = nextPollMs(interval, maxPollMs);
35
52
  }
36
53
  }
54
+ /** Poll the task's status until terminal or timeout. Returns the terminal Task
55
+ * (any status). Throws TaskTimeout, leaving the task running. `pollMs` is the
56
+ * *first* interval; it backs off towards `maxPollMs`. */
57
+ export function pollWait(store, taskId, opts) {
58
+ return poll(() => store.getStatus(taskId), () => store.get(taskId), (_ref, ms) => store.taskDoneWake(taskId, ms), taskId, null, opts);
59
+ }
60
+ /**
61
+ * The same wait, following a key instead of an id.
62
+ *
63
+ * The key is re-resolved on every probe, because that is what a key means: a
64
+ * pointer to the task that is *current* under it. A `replace` landing mid-wait
65
+ * moves the wait onto the new task rather than reporting the cancellation of the
66
+ * old one, and a key that points at nothing yet is simply not finished — it
67
+ * polls until something appears, the same way waiting on an id that does not
68
+ * exist yet does.
69
+ *
70
+ * There is nothing to subscribe to before the key resolves, so those naps are
71
+ * plain sleeps; once it resolves, the store's push channel applies as usual.
72
+ */
73
+ export function pollWaitByKey(store, key, opts) {
74
+ return poll(() => store.getStatusByKey(key), () => store.getByKey(key), (ref, ms) => (ref ? store.taskDoneWake(ref.id, ms) : sleep(ms)), key, key, opts);
75
+ }
package/dist/worker.js CHANGED
@@ -1,6 +1,6 @@
1
1
  import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
2
2
  import { TaskContext } from "./context.js";
3
- import { asEnvelope, errorEnvelope, exceptionEnvelope, LostLease, SerializationError, } from "./errors.js";
3
+ import { asEnvelope, errorEnvelope, EventLoopBlocked, exceptionEnvelope, LostLease, SerializationError, } from "./errors.js";
4
4
  import { newId } from "./ids.js";
5
5
  import { SQLiteStore } from "./store/sqlite.js";
6
6
  import { PostgresStore } from "./store/postgres.js";
@@ -703,6 +703,26 @@ export class Worker {
703
703
  let wake = null;
704
704
  // lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
705
705
  const interval = this.opts.heartbeatIntervalMs ?? Math.max(50, Math.floor(leaseMs / 3));
706
+ // When this heartbeat last ran. A loop cannot report its own absence — a
707
+ // handler that blocks for its whole attempt never lets the timer fire at all
708
+ // — so the check lives outside the loop and cancel() runs it too.
709
+ let lastBeatAt = Date.now();
710
+ /** Report a heartbeat that has not run for more than two intervals: a whole
711
+ * beat missed, which at lease/3 means the next such block loses the lease
712
+ * outright. Fires while the lease still holds — after it expires the only
713
+ * evidence is a task that ran twice, in two workers' logs, with no error in
714
+ * either. */
715
+ const checkBeat = () => {
716
+ const now = Date.now();
717
+ const lateMs = now - lastBeatAt - interval;
718
+ if (lateMs > interval) {
719
+ this.report(new EventLoopBlocked(lateMs, interval, leaseMs), {
720
+ phase: "execute",
721
+ taskId: ctxs[0].taskId,
722
+ });
723
+ }
724
+ lastBeatAt = now;
725
+ };
706
726
  const done = (async () => {
707
727
  while (active) {
708
728
  // Cancellable sleep: cancel() resolves this immediately and clears the
@@ -717,6 +737,7 @@ export class Worker {
717
737
  wake = null;
718
738
  if (!active)
719
739
  break;
740
+ checkBeat();
720
741
  const live = ctxs.filter((c) => !c.settled && !c.lostLease);
721
742
  if (!live.length)
722
743
  break;
@@ -746,6 +767,10 @@ export class Worker {
746
767
  active = false;
747
768
  if (wake)
748
769
  wake();
770
+ // The attempt is over, so this is the last chance to notice that the
771
+ // heartbeat never got one — the whole-attempt block, which is also the
772
+ // one where the lease is already gone.
773
+ checkBeat();
749
774
  },
750
775
  done,
751
776
  };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "cairnq",
3
- "version": "0.6.0",
3
+ "version": "0.8.0",
4
4
  "description": "SQLite-first, cross-language, storage-centered durable task runtime",
5
5
  "license": "MIT",
6
6
  "author": "Jannchie <jannchie@gmail.com>",
package/src/client.ts CHANGED
@@ -1,24 +1,35 @@
1
1
  import type { BackpressureOptions } from "./backpressure.js";
2
+ import { type RetentionOptions, RetentionSweeper } from "./retention.js";
2
3
  import { TaskCanceled, TaskFailed } from "./errors.js";
3
- import { isFailed, isSucceeded, type Task, type TaskStatus } from "./models.js";
4
+ import { isFailed, isSucceeded, type Task, type TaskRef, type TaskStatus } from "./models.js";
4
5
  import { SQLiteStore } from "./store/sqlite.js";
5
6
  import { PostgresStore } from "./store/postgres.js";
6
7
  import type { ListInput, PurgeInput, SubmitInput, TaskStore } from "./store/base.js";
7
8
  import { type TaskDef, taskName } from "./task.js";
8
- import { pollWait } from "./wait.js";
9
+ import { DEFAULT_WAIT_TIMEOUT_MS, type PollOptions, pollWait, pollWaitByKey } from "./wait.js";
9
10
 
10
11
  export type SubmitOptions = Omit<SubmitInput, "name" | "payload">;
11
- export interface CallOptions extends SubmitOptions {
12
+ /** The wait loop's knobs with the timeout optional (default 30s) — the public
13
+ * face of PollOptions, whose comments document each knob. */
14
+ export type WaitOptions = Partial<PollOptions>;
15
+ export interface CallOptions extends SubmitOptions, Omit<WaitOptions, "timeoutMs"> {
12
16
  waitTimeoutMs?: number;
13
- pollMs?: number;
14
17
  }
15
18
 
16
19
  /** Options this handle configures on the store it wraps, rather than the
17
20
  * store's own constructor arguments. */
18
- export type ClientOptions = Partial<BackpressureOptions>;
21
+ export type ClientOptions = Partial<BackpressureOptions> & {
22
+ /** Delete terminal tasks older than a cutoff, on a schedule, for as long as
23
+ * this handle is open. Off unless set — and off means rows accumulate forever,
24
+ * because nothing else in CairnQ removes them. */
25
+ retention?: RetentionOptions;
26
+ };
19
27
 
20
28
  /** API-side handle. Thin wrapper over a TaskStore + SDK-orchestrated wait/call. */
21
29
  export class CairnQ {
30
+ /** null unless `retention` was configured. */
31
+ private readonly sweeper: RetentionSweeper | null;
32
+
22
33
  constructor(
23
34
  private readonly _store: TaskStore,
24
35
  opts: ClientOptions = {},
@@ -28,6 +39,13 @@ export class CairnQ {
28
39
  if (opts.maxQueueDepth != null) {
29
40
  _store.useBackpressure(opts as BackpressureOptions);
30
41
  }
42
+ // Retention is the opposite case: it belongs to the handle, because a worker
43
+ // sharing the store must not also be deleting rows behind the API's back.
44
+ // Started here rather than in connect(), which is optional — every other
45
+ // path connects lazily, and retention that silently depends on an optional
46
+ // call is retention that silently does not happen.
47
+ this.sweeper = opts.retention ? new RetentionSweeper(_store, opts.retention) : null;
48
+ this.sweeper?.start();
31
49
  }
32
50
 
33
51
  static sqlite(path: string, opts: { busyTimeoutMs?: number } & ClientOptions = {}): CairnQ {
@@ -50,8 +68,11 @@ export class CairnQ {
50
68
  return this._store.connect();
51
69
  }
52
70
 
53
- close(): Promise<void> {
54
- return this._store.close();
71
+ /** Stop retention (waiting for a sweep in flight, so no purge outlives the
72
+ * store) and close the store. */
73
+ async close(): Promise<void> {
74
+ await this.sweeper?.stop();
75
+ await this._store.close();
55
76
  }
56
77
 
57
78
  /** Enqueue a task. With `maxQueueDepth` configured this blocks while the
@@ -80,6 +101,17 @@ export class CairnQ {
80
101
  return this._store.getByKey(key);
81
102
  }
82
103
 
104
+ /** The status-only probe wait polls on: id + status, no payload. Public for
105
+ * the same reason it exists — a dashboard or poller that only asks "is it
106
+ * finished yet" should not drag the payload back per ask. */
107
+ getStatus(taskId: string): Promise<TaskRef | null> {
108
+ return this._store.getStatus(taskId);
109
+ }
110
+
111
+ getStatusByKey(key: string): Promise<TaskRef | null> {
112
+ return this._store.getStatusByKey(key);
113
+ }
114
+
83
115
  list(input?: ListInput): Promise<Task[]> {
84
116
  return this._store.list(input);
85
117
  }
@@ -114,25 +146,43 @@ export class CairnQ {
114
146
  return this._store.stats();
115
147
  }
116
148
 
117
- wait(
118
- taskId: string,
119
- opts: { timeoutMs?: number; pollMs?: number } = {},
120
- ): Promise<Task> {
149
+ /** Wait for a task to finish. Resolves with the terminal Task (any status);
150
+ * throws TaskTimeout without stopping the task, so `wait(err.taskId)` picks the
151
+ * same wait back up — from another process, or after a longer deadline. */
152
+ wait(taskId: string, opts: WaitOptions = {}): Promise<Task> {
153
+ // `??`, not a spread default: a caller forwarding `timeoutMs: undefined`
154
+ // (call() does) must still get the default, and a spread would override it.
121
155
  return pollWait(this._store, taskId, {
122
- timeoutMs: opts.timeoutMs ?? 30_000,
123
- pollMs: opts.pollMs,
156
+ ...opts,
157
+ timeoutMs: opts.timeoutMs ?? DEFAULT_WAIT_TIMEOUT_MS,
158
+ });
159
+ }
160
+
161
+ /** Wait for whatever task the `key` currently points at — the cross-process
162
+ * form of picking a wait back up, when the id was never in hand or the process
163
+ * that held it is gone. Re-resolves the key on each poll, so a `replace`
164
+ * landing mid-wait moves the wait onto the new task, and a key with no task
165
+ * yet is waited for rather than rejected. */
166
+ waitByKey(key: string, opts: WaitOptions = {}): Promise<Task> {
167
+ return pollWaitByKey(this._store, key, {
168
+ ...opts,
169
+ timeoutMs: opts.timeoutMs ?? DEFAULT_WAIT_TIMEOUT_MS,
124
170
  });
125
171
  }
126
172
 
127
173
  /** submit + wait. Resolves with the result on success; rejects with
128
174
  * TaskFailed / TaskCanceled / TaskTimeout otherwise. Pass a TaskDef and the
129
- * resolved value is typed as its Result. */
175
+ * resolved value is typed as its Result.
176
+ *
177
+ * `waitTimeoutMs` bounds the wait, not the task: on timeout the task runs on,
178
+ * and `wait(err.taskId)` — or `waitByKey`, from a process that only has the
179
+ * key — resumes the wait rather than starting the work over. */
130
180
  async call(name: string, payload?: unknown, opts?: CallOptions): Promise<unknown>;
131
181
  async call<P, R>(task: TaskDef<P, R>, payload?: P, opts?: CallOptions): Promise<R>;
132
182
  async call(task: string | TaskDef, payload?: unknown, opts: CallOptions = {}): Promise<unknown> {
133
- const { waitTimeoutMs = 30_000, pollMs, ...submit } = opts;
183
+ const { waitTimeoutMs, pollMs, maxPollMs, ...submit } = opts;
134
184
  const created = await this.submit(taskName(task), payload, submit);
135
- const final = await pollWait(this._store, created.id, { timeoutMs: waitTimeoutMs, pollMs });
185
+ const final = await this.wait(created.id, { timeoutMs: waitTimeoutMs, pollMs, maxPollMs });
136
186
  if (isSucceeded(final)) return final.result;
137
187
  if (isFailed(final)) throw new TaskFailed(final.error);
138
188
  throw new TaskCanceled(final.id);