cairnq 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/wait.js CHANGED
@@ -4,6 +4,7 @@ import { isTerminal } from "./models.js";
4
4
  export const DEFAULT_POLL_MS = 100;
5
5
  export const MAX_POLL_MS = 500;
6
6
  const GROWTH = 1.5;
7
+ const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
7
8
  /**
8
9
  * Grow the polling interval towards the ceiling.
9
10
  *
@@ -15,22 +16,46 @@ const GROWTH = 1.5;
15
16
  export function nextPollMs(current, maxMs) {
16
17
  return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
17
18
  }
18
- /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
19
- * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
20
- * it backs off towards `maxPollMs`. */
21
- export async function pollWait(store, taskId, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS, }) {
19
+ /**
20
+ * Poll `read` until it yields a terminal task, or the timeout elapses.
21
+ *
22
+ * `wake` is what the loop sleeps on between reads: a store with a push channel
23
+ * (Postgres) cuts it short when the task goes terminal, but the re-read is the
24
+ * source of truth either way, so a plain sleep is always a correct answer.
25
+ */
26
+ async function poll(read, wake, subject, key, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS }) {
22
27
  const deadline = nowMs() + timeoutMs;
23
28
  let interval = pollMs;
24
29
  for (;;) {
25
- const task = await store.get(taskId);
30
+ const task = await read();
26
31
  if (task && isTerminal(task))
27
32
  return task;
28
33
  const remaining = deadline - nowMs();
29
34
  if (remaining <= 0)
30
- throw new TaskTimeout(taskId, { timeoutMs, task });
31
- // A store with a push channel (Postgres) cuts the sleep short when the task
32
- // goes terminal; the re-get above stays the source of truth either way.
33
- await store.taskDoneWake(taskId, Math.min(interval, remaining));
35
+ throw new TaskTimeout(task?.id ?? subject, { timeoutMs, task, key });
36
+ await wake(task, Math.min(interval, remaining));
34
37
  interval = nextPollMs(interval, maxPollMs);
35
38
  }
36
39
  }
40
+ /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
41
+ * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
42
+ * it backs off towards `maxPollMs`. */
43
+ export function pollWait(store, taskId, opts) {
44
+ return poll(() => store.get(taskId), (_task, ms) => store.taskDoneWake(taskId, ms), taskId, null, opts);
45
+ }
46
+ /**
47
+ * The same wait, following a key instead of an id.
48
+ *
49
+ * The key is re-resolved on every read, because that is what a key means: a
50
+ * pointer to the task that is *current* under it. A `replace` landing mid-wait
51
+ * moves the wait onto the new task rather than reporting the cancellation of the
52
+ * old one, and a key that points at nothing yet is simply not finished — it
53
+ * polls until something appears, the same way waiting on an id that does not
54
+ * exist yet does.
55
+ *
56
+ * There is nothing to subscribe to before the key resolves, so those naps are
57
+ * plain sleeps; once it resolves, the store's push channel applies as usual.
58
+ */
59
+ export function pollWaitByKey(store, key, opts) {
60
+ return poll(() => store.getByKey(key), (task, ms) => (task ? store.taskDoneWake(task.id, ms) : sleep(ms)), key, key, opts);
61
+ }
package/dist/worker.js CHANGED
@@ -1,6 +1,6 @@
1
1
  import { DEFAULT_RETRY_BACKOFF_MAX_MS, DEFAULT_RETRY_BACKOFF_MS, failDelayMs } from "./backoff.js";
2
2
  import { TaskContext } from "./context.js";
3
- import { asEnvelope, errorEnvelope, exceptionEnvelope, LostLease, SerializationError, } from "./errors.js";
3
+ import { asEnvelope, errorEnvelope, EventLoopBlocked, exceptionEnvelope, LostLease, SerializationError, } from "./errors.js";
4
4
  import { newId } from "./ids.js";
5
5
  import { SQLiteStore } from "./store/sqlite.js";
6
6
  import { PostgresStore } from "./store/postgres.js";
@@ -703,6 +703,26 @@ export class Worker {
703
703
  let wake = null;
704
704
  // lease/3 gives two beats of slack; the floor only matters for sub-150ms leases.
705
705
  const interval = this.opts.heartbeatIntervalMs ?? Math.max(50, Math.floor(leaseMs / 3));
706
+ // When this heartbeat last ran. A loop cannot report its own absence — a
707
+ // handler that blocks for its whole attempt never lets the timer fire at all
708
+ // — so the check lives outside the loop and cancel() runs it too.
709
+ let lastBeatAt = Date.now();
710
+ /** Report a heartbeat that has not run for more than two intervals: a whole
711
+ * beat missed, which at lease/3 means the next such block loses the lease
712
+ * outright. Fires while the lease still holds — after it expires the only
713
+ * evidence is a task that ran twice, in two workers' logs, with no error in
714
+ * either. */
715
+ const checkBeat = () => {
716
+ const now = Date.now();
717
+ const lateMs = now - lastBeatAt - interval;
718
+ if (lateMs > interval) {
719
+ this.report(new EventLoopBlocked(lateMs, interval, leaseMs), {
720
+ phase: "execute",
721
+ taskId: ctxs[0].taskId,
722
+ });
723
+ }
724
+ lastBeatAt = now;
725
+ };
706
726
  const done = (async () => {
707
727
  while (active) {
708
728
  // Cancellable sleep: cancel() resolves this immediately and clears the
@@ -717,6 +737,7 @@ export class Worker {
717
737
  wake = null;
718
738
  if (!active)
719
739
  break;
740
+ checkBeat();
720
741
  const live = ctxs.filter((c) => !c.settled && !c.lostLease);
721
742
  if (!live.length)
722
743
  break;
@@ -746,6 +767,10 @@ export class Worker {
746
767
  active = false;
747
768
  if (wake)
748
769
  wake();
770
+ // The attempt is over, so this is the last chance to notice that the
771
+ // heartbeat never got one — the whole-attempt block, which is also the
772
+ // one where the lease is already gone.
773
+ checkBeat();
749
774
  },
750
775
  done,
751
776
  };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "cairnq",
3
- "version": "0.6.0",
3
+ "version": "0.7.0",
4
4
  "description": "SQLite-first, cross-language, storage-centered durable task runtime",
5
5
  "license": "MIT",
6
6
  "author": "Jannchie <jannchie@gmail.com>",
package/src/client.ts CHANGED
@@ -1,11 +1,12 @@
1
1
  import type { BackpressureOptions } from "./backpressure.js";
2
+ import { type RetentionOptions, RetentionSweeper } from "./retention.js";
2
3
  import { TaskCanceled, TaskFailed } from "./errors.js";
3
4
  import { isFailed, isSucceeded, type Task, type TaskStatus } from "./models.js";
4
5
  import { SQLiteStore } from "./store/sqlite.js";
5
6
  import { PostgresStore } from "./store/postgres.js";
6
7
  import type { ListInput, PurgeInput, SubmitInput, TaskStore } from "./store/base.js";
7
8
  import { type TaskDef, taskName } from "./task.js";
8
- import { pollWait } from "./wait.js";
9
+ import { pollWait, pollWaitByKey } from "./wait.js";
9
10
 
10
11
  export type SubmitOptions = Omit<SubmitInput, "name" | "payload">;
11
12
  export interface CallOptions extends SubmitOptions {
@@ -15,10 +16,18 @@ export interface CallOptions extends SubmitOptions {
15
16
 
16
17
  /** Options this handle configures on the store it wraps, rather than the
17
18
  * store's own constructor arguments. */
18
- export type ClientOptions = Partial<BackpressureOptions>;
19
+ export type ClientOptions = Partial<BackpressureOptions> & {
20
+ /** Delete terminal tasks older than a cutoff, on a schedule, for as long as
21
+ * this handle is open. Off unless set — and off means rows accumulate forever,
22
+ * because nothing else in CairnQ removes them. */
23
+ retention?: RetentionOptions;
24
+ };
19
25
 
20
26
  /** API-side handle. Thin wrapper over a TaskStore + SDK-orchestrated wait/call. */
21
27
  export class CairnQ {
28
+ /** null unless `retention` was configured. */
29
+ private readonly sweeper: RetentionSweeper | null;
30
+
22
31
  constructor(
23
32
  private readonly _store: TaskStore,
24
33
  opts: ClientOptions = {},
@@ -28,6 +37,13 @@ export class CairnQ {
28
37
  if (opts.maxQueueDepth != null) {
29
38
  _store.useBackpressure(opts as BackpressureOptions);
30
39
  }
40
+ // Retention is the opposite case: it belongs to the handle, because a worker
41
+ // sharing the store must not also be deleting rows behind the API's back.
42
+ // Started here rather than in connect(), which is optional — every other
43
+ // path connects lazily, and retention that silently depends on an optional
44
+ // call is retention that silently does not happen.
45
+ this.sweeper = opts.retention ? new RetentionSweeper(_store, opts.retention) : null;
46
+ this.sweeper?.start();
31
47
  }
32
48
 
33
49
  static sqlite(path: string, opts: { busyTimeoutMs?: number } & ClientOptions = {}): CairnQ {
@@ -50,8 +66,11 @@ export class CairnQ {
50
66
  return this._store.connect();
51
67
  }
52
68
 
53
- close(): Promise<void> {
54
- return this._store.close();
69
+ /** Stop retention (waiting for a sweep in flight, so no purge outlives the
70
+ * store) and close the store. */
71
+ async close(): Promise<void> {
72
+ await this.sweeper?.stop();
73
+ await this._store.close();
55
74
  }
56
75
 
57
76
  /** Enqueue a task. With `maxQueueDepth` configured this blocks while the
@@ -114,6 +133,9 @@ export class CairnQ {
114
133
  return this._store.stats();
115
134
  }
116
135
 
136
+ /** Wait for a task to finish. Resolves with the terminal Task (any status);
137
+ * throws TaskTimeout without stopping the task, so `wait(err.taskId)` picks the
138
+ * same wait back up — from another process, or after a longer deadline. */
117
139
  wait(
118
140
  taskId: string,
119
141
  opts: { timeoutMs?: number; pollMs?: number } = {},
@@ -124,9 +146,25 @@ export class CairnQ {
124
146
  });
125
147
  }
126
148
 
149
+ /** Wait for whatever task the `key` currently points at — the cross-process
150
+ * form of picking a wait back up, when the id was never in hand or the process
151
+ * that held it is gone. Re-resolves the key on each poll, so a `replace`
152
+ * landing mid-wait moves the wait onto the new task, and a key with no task
153
+ * yet is waited for rather than rejected. */
154
+ waitByKey(key: string, opts: { timeoutMs?: number; pollMs?: number } = {}): Promise<Task> {
155
+ return pollWaitByKey(this._store, key, {
156
+ timeoutMs: opts.timeoutMs ?? 30_000,
157
+ pollMs: opts.pollMs,
158
+ });
159
+ }
160
+
127
161
  /** submit + wait. Resolves with the result on success; rejects with
128
162
  * TaskFailed / TaskCanceled / TaskTimeout otherwise. Pass a TaskDef and the
129
- * resolved value is typed as its Result. */
163
+ * resolved value is typed as its Result.
164
+ *
165
+ * `waitTimeoutMs` bounds the wait, not the task: on timeout the task runs on,
166
+ * and `wait(err.taskId)` — or `waitByKey`, from a process that only has the
167
+ * key — resumes the wait rather than starting the work over. */
130
168
  async call(name: string, payload?: unknown, opts?: CallOptions): Promise<unknown>;
131
169
  async call<P, R>(task: TaskDef<P, R>, payload?: P, opts?: CallOptions): Promise<R>;
132
170
  async call(task: string | TaskDef, payload?: unknown, opts: CallOptions = {}): Promise<unknown> {
package/src/errors.ts CHANGED
@@ -77,8 +77,12 @@ export class QueueFull extends CairnQError {
77
77
  * observed. No worker running, no handler for the name, wrong queue, and two
78
78
  * processes on different database files all look identical from the API side —
79
79
  * queued, never claimed — so that case names the likely causes. */
80
- function timeoutDetail(task: Task | null): string {
81
- if (!task) return "task not found — wrong database file, or already purged?";
80
+ function timeoutDetail(task: Task | null, key: string | null): string {
81
+ if (!task) {
82
+ return key === null
83
+ ? "task not found — wrong database file, or already purged?"
84
+ : "no task under this key — never submitted, or already purged?";
85
+ }
82
86
  if (isQueued(task)) {
83
87
  const delayMs = task.run_at_ms - nowMs();
84
88
  if (task.attempt === 0 && delayMs <= 0) {
@@ -94,23 +98,33 @@ function timeoutDetail(task: Task | null): string {
94
98
  return `still running (attempt ${task.attempt}/${task.max_attempts})`;
95
99
  }
96
100
 
97
- /** wait/call did not reach a terminal status in time. The task keeps running.
98
- * `task` is the last snapshot wait() observed (null if get() found nothing), and
99
- * the message says what state it was stuck in — a queued-never-claimed task is
100
- * the classic first-run failure (no worker, no handler, wrong queue or file). */
101
+ /** wait/call did not reach a terminal status in time. The task keeps running, so
102
+ * `taskId` is the handle for picking the wait back up — `wait(err.taskId)`
103
+ * re-attaches to the same task from anywhere that can reach the store. `task` is
104
+ * the last snapshot wait() observed (null if the lookup found nothing), and the
105
+ * message says what state it was stuck in — a queued-never-claimed task is the
106
+ * classic first-run failure (no worker, no handler, wrong queue or file).
107
+ *
108
+ * `key` is set when the wait watched a key rather than an id; `taskId` is then
109
+ * the task the key pointed at, or the key itself when it pointed at nothing —
110
+ * there was no id to report. */
101
111
  export class TaskTimeout extends CairnQError {
102
112
  readonly task: Task | null;
113
+ readonly key: string | null;
103
114
  constructor(
104
115
  public taskId: string,
105
- opts: { timeoutMs?: number; task?: Task | null } = {},
116
+ opts: { timeoutMs?: number; task?: Task | null; key?: string | null } = {},
106
117
  ) {
118
+ const key = opts.key ?? null;
119
+ const subject = key === null ? `task ${taskId}` : `key ${key}`;
107
120
  super(
108
121
  opts.timeoutMs == null
109
- ? `task ${taskId} did not finish in time`
110
- : `task ${taskId} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null)}`,
122
+ ? `${subject} did not finish in time`
123
+ : `${subject} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null, key)}`,
111
124
  );
112
125
  this.name = "TaskTimeout";
113
126
  this.task = opts.task ?? null;
127
+ this.key = key;
114
128
  }
115
129
  }
116
130
 
@@ -146,6 +160,36 @@ export class TaskCanceled extends CairnQError {
146
160
  }
147
161
  }
148
162
 
163
+ /**
164
+ * A heartbeat beat came back later than its own interval allowed.
165
+ *
166
+ * The heartbeat shares the event loop with the handlers whose leases it renews,
167
+ * so a handler that blocks the loop stops the renewal with it: the lease expires,
168
+ * the task is recovered and redelivered, and a second worker starts computing
169
+ * what the first is still computing — one task, billed twice, with no error
170
+ * anywhere. Nothing inside the blocked handler can observe that, which is why it
171
+ * is reported through `onError` alongside the other things the run loop survived.
172
+ *
173
+ * The cause is always synchronous work in a handler: a tight loop, a large
174
+ * JSON.parse, a `*Sync` filesystem or crypto call. Node has one loop and no way
175
+ * to preempt it — move the work to a worker thread, a child process, or an async
176
+ * API that yields.
177
+ */
178
+ export class EventLoopBlocked extends CairnQError {
179
+ constructor(
180
+ readonly lateMs: number,
181
+ readonly intervalMs: number,
182
+ readonly leaseMs: number,
183
+ ) {
184
+ super(
185
+ `heartbeat beat was ${lateMs}ms late (interval ${intervalMs}ms, lease ${leaseMs}ms): ` +
186
+ `the event loop was blocked long enough to miss a beat. Synchronous work in a ` +
187
+ `handler starves lease renewal — move it off the loop.`,
188
+ );
189
+ this.name = "EventLoopBlocked";
190
+ }
191
+ }
192
+
149
193
  /** A worker write affected 0 rows: the lease expired and was reclaimed. */
150
194
  export class LostLease extends CairnQError {
151
195
  constructor(public taskId: string) {
package/src/index.ts CHANGED
@@ -2,6 +2,8 @@ export { CairnQ } from "./client.js";
2
2
  export type { CallOptions, ClientOptions, SubmitOptions } from "./client.js";
3
3
  export { QueueDepthGate } from "./backpressure.js";
4
4
  export type { BackpressureOptions, QueueDepthLimit } from "./backpressure.js";
5
+ export { RetentionSweeper } from "./retention.js";
6
+ export type { RetentionOptions } from "./retention.js";
5
7
  export { Worker } from "./worker.js";
6
8
  export type { BatchHandler, Handler, TypedHandler, WorkerOptions } from "./worker.js";
7
9
  export { TaskContext } from "./context.js";
@@ -32,6 +34,7 @@ export {
32
34
  TaskCanceled,
33
35
  TaskError,
34
36
  LostLease,
37
+ EventLoopBlocked,
35
38
  ProtocolVersionMismatch,
36
39
  SerializationError,
37
40
  } from "./errors.js";
@@ -0,0 +1,136 @@
1
+ import type { PurgeInput, TaskStore } from "./store/base.js";
2
+
3
+ /** Sweep every hour unless asked otherwise — often enough that a queue with a
4
+ * day of retention never carries more than an hour of extra rows, rare enough
5
+ * that the sweep is invisible next to the task traffic. */
6
+ const DEFAULT_INTERVAL_MS = 3_600_000;
7
+ /** Rows per purge statement. The same bound `purge` defaults to: big enough that
8
+ * a backlog drains in few statements, small enough that each is a short write. */
9
+ const DEFAULT_LIMIT = 1_000;
10
+
11
+ export interface RetentionOptions {
12
+ /**
13
+ * How long a terminal task is kept after it finished. Required: there is no
14
+ * safe default for how long someone else's results stay readable.
15
+ */
16
+ olderThanMs: number;
17
+ /** Time between sweeps. Default 3_600_000 (one hour). */
18
+ intervalMs?: number;
19
+ /** Rows deleted per statement while draining. Default 1_000. */
20
+ limit?: number;
21
+ /**
22
+ * Called for a sweep that threw. The next sweep runs on schedule regardless —
23
+ * a purge that failed because the database was busy is not a reason to stop
24
+ * retaining — so without this a store quietly stops being swept. Must not throw.
25
+ */
26
+ onError?: (err: unknown) => void;
27
+ }
28
+
29
+ /**
30
+ * Deletes terminal tasks on a schedule, for as long as the handle is open.
31
+ *
32
+ * `purge` exists because nothing else in CairnQ removes rows, and a queue whose
33
+ * payloads carry real data — an image, a document, a batch of embeddings — turns
34
+ * that into a disk leak measured in gigabytes per backfill. Every deployment
35
+ * that runs longer than a demo needs the sweep; leaving it to an external
36
+ * scheduler means the leak is the default and remembering is the opt-in.
37
+ *
38
+ * It sweeps in bounded batches with a yield between them, so draining a backlog
39
+ * that accumulated while nothing was sweeping stays a sequence of short writes
40
+ * rather than one long one — on SQLite that matters, since a long write holds
41
+ * the single write lock against every producer and worker on the file.
42
+ */
43
+ export class RetentionSweeper {
44
+ /** Whether the scheduled loop is running. */
45
+ private active = false;
46
+ /** Set by stop(), so a drain in progress can cut itself short too. */
47
+ private stopping = false;
48
+ /** Resolves the current sleep early, so stop() need not wait out an interval. */
49
+ private wake: (() => void) | null = null;
50
+ /** The loop itself, awaited by stop() so no purge outlives the store. */
51
+ private loop: Promise<void> | null = null;
52
+ private readonly intervalMs: number;
53
+ private readonly purgeInput: PurgeInput;
54
+
55
+ constructor(
56
+ private readonly store: TaskStore,
57
+ private readonly opts: RetentionOptions,
58
+ ) {
59
+ if (!Number.isFinite(opts.olderThanMs) || opts.olderThanMs < 0) {
60
+ throw new Error(`retention.olderThanMs must be >= 0, got ${opts.olderThanMs}`);
61
+ }
62
+ this.intervalMs = opts.intervalMs ?? DEFAULT_INTERVAL_MS;
63
+ if (!Number.isFinite(this.intervalMs) || this.intervalMs < 1) {
64
+ throw new Error(`retention.intervalMs must be >= 1, got ${this.intervalMs}`);
65
+ }
66
+ this.purgeInput = { olderThanMs: opts.olderThanMs, limit: opts.limit ?? DEFAULT_LIMIT };
67
+ }
68
+
69
+ start(): void {
70
+ if (this.active) return;
71
+ this.active = true;
72
+ this.stopping = false;
73
+ this.loop = this.run();
74
+ }
75
+
76
+ /** Stop sweeping and wait for the sweep in flight, if any. */
77
+ async stop(): Promise<void> {
78
+ this.stopping = true;
79
+ this.active = false;
80
+ this.wake?.();
81
+ await this.loop;
82
+ this.loop = null;
83
+ }
84
+
85
+ private async run(): Promise<void> {
86
+ // Sleep first: a process that restarts often would otherwise purge on every
87
+ // boot, which is a write burst exactly when the store is busiest.
88
+ while (!this.stopping) {
89
+ await this.sleep(this.intervalMs);
90
+ if (this.stopping) return;
91
+ try {
92
+ await this.sweep();
93
+ } catch (err) {
94
+ try {
95
+ this.opts.onError?.(err);
96
+ } catch {
97
+ // A reporting hook must never take the sweep down with it — the same
98
+ // rule the worker's onError follows.
99
+ }
100
+ }
101
+ }
102
+ }
103
+
104
+ /**
105
+ * Delete everything past the cutoff now, in bounded batches, and return how
106
+ * many rows went. The scheduled loop calls this; call it directly to drain on
107
+ * demand — after a backfill, or from a maintenance command.
108
+ */
109
+ async sweep(): Promise<number> {
110
+ const limit = this.purgeInput.limit as number;
111
+ let deleted = 0;
112
+ for (;;) {
113
+ const ids = await this.store.purge(this.purgeInput);
114
+ deleted += ids.length;
115
+ if (ids.length < limit || this.stopping) return deleted;
116
+ // Hand the loop back between batches: a large drain must not starve the
117
+ // submits and claims sharing this process.
118
+ await this.sleep(0);
119
+ }
120
+ }
121
+
122
+ /** Sleep, interruptible by stop(). Unref'd: retention is housekeeping, and a
123
+ * pending sweep must never be the reason a process refuses to exit. */
124
+ private sleep(ms: number): Promise<void> {
125
+ return new Promise<void>((resolve) => {
126
+ const timer = setTimeout(resolve, ms);
127
+ timer.unref?.();
128
+ this.wake = () => {
129
+ clearTimeout(timer);
130
+ resolve();
131
+ };
132
+ }).finally(() => {
133
+ this.wake = null;
134
+ });
135
+ }
136
+ }
package/src/store/base.ts CHANGED
@@ -6,7 +6,7 @@ import {
6
6
  ProtocolVersionMismatch,
7
7
  SerializationError,
8
8
  } from "../errors.js";
9
- import { rowToTask, STATUSES, type Task, type TaskStatus } from "../models.js";
9
+ import { rowToTask, STATUSES, TERMINAL, type Task, type TaskStatus } from "../models.js";
10
10
  import { type BackpressureOptions, QueueDepthGate } from "../backpressure.js";
11
11
 
12
12
  const rejectMangled = function (this: unknown, _key: string, v: unknown): unknown {
@@ -62,9 +62,25 @@ export function checkProtocolVersion(version: number): void {
62
62
  // CONFLICTS is the canonical declaration; the type derives from it so the
63
63
  // runtime guard in submit() and the type can't drift apart (same pattern as
64
64
  // STATUSES/TaskStatus in models.ts).
65
- const CONFLICTS = ["reuse", "reject", "replace"] as const;
65
+ const CONFLICTS = ["reuse", "reuse-succeeded", "reject", "replace"] as const;
66
66
  export type Conflict = (typeof CONFLICTS)[number];
67
67
 
68
+ /**
69
+ * Whether a keyed submit's strategy accepts the task the key already points at.
70
+ *
71
+ * Both reuse strategies deduplicate work that is still in play — that is what a
72
+ * key is for, and the answer cannot depend on the outcome of a task that has no
73
+ * outcome yet. They differ only on what a *finished* task means: `reuse` treats
74
+ * the key as free again, while `reuse-succeeded` reads a succeeded task as a
75
+ * cached result. Neither ever hands back a failed or canceled one, which would
76
+ * poison the key for every later submit (see PROTOCOL.md "Key conflict").
77
+ */
78
+ function reusable(conflict: Conflict, status: TaskStatus): boolean {
79
+ if (conflict === "replace") return false;
80
+ if (!TERMINAL.includes(status)) return true;
81
+ return conflict === "reuse-succeeded" && status === "succeeded";
82
+ }
83
+
68
84
  /** The queue a submit lands on when it names none. Owned here, where the
69
85
  * default is applied, so nothing above has to re-derive it. */
70
86
  export const DEFAULT_QUEUE = "default";
@@ -283,10 +299,15 @@ export abstract class TaskStore {
283
299
  // free after all, whatever the strategy.
284
300
  const current = (await fetch("get", { id: existing[0].task_id }))[0];
285
301
  if (current) {
286
- if (conflict === "reuse") return rowToTask(current);
287
302
  if (conflict === "reject") throw new AlreadyExists(key);
288
- // "replace": cancel the recorded task, then repoint the key below.
289
- await fetch("cancel", { id: existing[0].task_id });
303
+ if (reusable(conflict, current.status as TaskStatus)) return rowToTask(current);
304
+ // The strategy declined the recorded task, so the key repoints to the
305
+ // fresh one inserted below. Cancel only what is still live: a terminal
306
+ // task has nothing to stop, and cancelling it would rewrite a settled
307
+ // row (and hand a `canceled` back to whoever is waiting on it).
308
+ if (!TERMINAL.includes(current.status as TaskStatus)) {
309
+ await fetch("cancel", { id: existing[0].task_id });
310
+ }
290
311
  }
291
312
  }
292
313
  const row = (await fetch("insert_task", ins))[0];
package/src/wait.ts CHANGED
@@ -7,6 +7,14 @@ export const DEFAULT_POLL_MS = 100;
7
7
  export const MAX_POLL_MS = 500;
8
8
  const GROWTH = 1.5;
9
9
 
10
+ const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
11
+
12
+ export interface PollOptions {
13
+ timeoutMs: number;
14
+ pollMs?: number;
15
+ maxPollMs?: number;
16
+ }
17
+
10
18
  /**
11
19
  * Grow the polling interval towards the ceiling.
12
20
  *
@@ -19,28 +27,64 @@ export function nextPollMs(current: number, maxMs: number): number {
19
27
  return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
20
28
  }
21
29
 
22
- /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
23
- * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
24
- * it backs off towards `maxPollMs`. */
25
- export async function pollWait(
26
- store: TaskStore,
27
- taskId: string,
28
- {
29
- timeoutMs,
30
- pollMs = DEFAULT_POLL_MS,
31
- maxPollMs = MAX_POLL_MS,
32
- }: { timeoutMs: number; pollMs?: number; maxPollMs?: number },
30
+ /**
31
+ * Poll `read` until it yields a terminal task, or the timeout elapses.
32
+ *
33
+ * `wake` is what the loop sleeps on between reads: a store with a push channel
34
+ * (Postgres) cuts it short when the task goes terminal, but the re-read is the
35
+ * source of truth either way, so a plain sleep is always a correct answer.
36
+ */
37
+ async function poll(
38
+ read: () => Promise<Task | null>,
39
+ wake: (task: Task | null, ms: number) => Promise<void>,
40
+ subject: string,
41
+ key: string | null,
42
+ { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS }: PollOptions,
33
43
  ): Promise<Task> {
34
44
  const deadline = nowMs() + timeoutMs;
35
45
  let interval = pollMs;
36
46
  for (;;) {
37
- const task = await store.get(taskId);
47
+ const task = await read();
38
48
  if (task && isTerminal(task)) return task;
39
49
  const remaining = deadline - nowMs();
40
- if (remaining <= 0) throw new TaskTimeout(taskId, { timeoutMs, task });
41
- // A store with a push channel (Postgres) cuts the sleep short when the task
42
- // goes terminal; the re-get above stays the source of truth either way.
43
- await store.taskDoneWake(taskId, Math.min(interval, remaining));
50
+ if (remaining <= 0) throw new TaskTimeout(task?.id ?? subject, { timeoutMs, task, key });
51
+ await wake(task, Math.min(interval, remaining));
44
52
  interval = nextPollMs(interval, maxPollMs);
45
53
  }
46
54
  }
55
+
56
+ /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
57
+ * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
58
+ * it backs off towards `maxPollMs`. */
59
+ export function pollWait(store: TaskStore, taskId: string, opts: PollOptions): Promise<Task> {
60
+ return poll(
61
+ () => store.get(taskId),
62
+ (_task, ms) => store.taskDoneWake(taskId, ms),
63
+ taskId,
64
+ null,
65
+ opts,
66
+ );
67
+ }
68
+
69
+ /**
70
+ * The same wait, following a key instead of an id.
71
+ *
72
+ * The key is re-resolved on every read, because that is what a key means: a
73
+ * pointer to the task that is *current* under it. A `replace` landing mid-wait
74
+ * moves the wait onto the new task rather than reporting the cancellation of the
75
+ * old one, and a key that points at nothing yet is simply not finished — it
76
+ * polls until something appears, the same way waiting on an id that does not
77
+ * exist yet does.
78
+ *
79
+ * There is nothing to subscribe to before the key resolves, so those naps are
80
+ * plain sleeps; once it resolves, the store's push channel applies as usual.
81
+ */
82
+ export function pollWaitByKey(store: TaskStore, key: string, opts: PollOptions): Promise<Task> {
83
+ return poll(
84
+ () => store.getByKey(key),
85
+ (task, ms) => (task ? store.taskDoneWake(task.id, ms) : sleep(ms)),
86
+ key,
87
+ key,
88
+ opts,
89
+ );
90
+ }