cairnq 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/README.md +28 -0
  2. package/dist/_protocol/migrations/postgres/0002_purge_index.sql +6 -0
  3. package/dist/_protocol/migrations/postgres/0003_notify.sql +38 -0
  4. package/dist/_protocol/migrations/sqlite/0002_purge_index.sql +6 -0
  5. package/dist/_protocol/sql/postgres/claim.sql +18 -5
  6. package/dist/_protocol/sql/postgres/fail.sql +28 -8
  7. package/dist/_protocol/sql/postgres/insert_task.sql +6 -3
  8. package/dist/_protocol/sql/postgres/list.sql +3 -1
  9. package/dist/_protocol/sql/postgres/lock_key.sql +9 -0
  10. package/dist/_protocol/sql/postgres/progress.sql +4 -3
  11. package/dist/_protocol/sql/postgres/protocol_version.sql +4 -0
  12. package/dist/_protocol/sql/postgres/purge.sql +25 -0
  13. package/dist/_protocol/sql/postgres/recover_leases.sql +38 -14
  14. package/dist/_protocol/sql/postgres/retry.sql +3 -0
  15. package/dist/_protocol/sql/postgres/stats.sql +8 -0
  16. package/dist/_protocol/sql/sqlite/claim.sql +13 -2
  17. package/dist/_protocol/sql/sqlite/claimable_probe.sql +6 -2
  18. package/dist/_protocol/sql/sqlite/fail.sql +30 -8
  19. package/dist/_protocol/sql/sqlite/list.sql +3 -1
  20. package/dist/_protocol/sql/sqlite/lock_key.sql +5 -0
  21. package/dist/_protocol/sql/sqlite/progress.sql +6 -2
  22. package/dist/_protocol/sql/sqlite/protocol_version.sql +4 -0
  23. package/dist/_protocol/sql/sqlite/purge.sql +18 -0
  24. package/dist/_protocol/sql/sqlite/recover_leases.sql +25 -7
  25. package/dist/_protocol/sql/sqlite/retry.sql +3 -0
  26. package/dist/_protocol/sql/sqlite/stats.sql +8 -0
  27. package/dist/client.d.ts +10 -2
  28. package/dist/client.js +12 -0
  29. package/dist/context.d.ts +17 -1
  30. package/dist/context.js +60 -6
  31. package/dist/errors.d.ts +18 -2
  32. package/dist/errors.js +49 -3
  33. package/dist/index.d.ts +3 -2
  34. package/dist/index.js +2 -1
  35. package/dist/sql.js +16 -9
  36. package/dist/store/base.d.ts +107 -9
  37. package/dist/store/base.js +370 -1
  38. package/dist/store/postgres.d.ts +62 -63
  39. package/dist/store/postgres.js +245 -222
  40. package/dist/store/sqlite.d.ts +34 -59
  41. package/dist/store/sqlite.js +200 -232
  42. package/dist/wait.d.ts +15 -2
  43. package/dist/wait.js +23 -5
  44. package/dist/worker.d.ts +53 -1
  45. package/dist/worker.js +202 -42
  46. package/package.json +9 -2
  47. package/src/client.ts +16 -2
  48. package/src/context.ts +70 -13
  49. package/src/errors.ts +59 -4
  50. package/src/index.ts +3 -1
  51. package/src/sql.ts +15 -8
  52. package/src/store/base.ts +430 -27
  53. package/src/store/postgres.ts +243 -267
  54. package/src/store/sqlite.ts +211 -265
  55. package/src/wait.ts +28 -5
  56. package/src/worker.ts +242 -42
@@ -1,38 +1,105 @@
1
1
  import { mkdirSync } from "node:fs";
2
- import { dirname } from "node:path";
2
+ import { dirname, resolve } from "node:path";
3
3
  import Database from "better-sqlite3";
4
- import { newId, nowMs } from "../ids.js";
5
- import { AlreadyExists, errorEnvelope, LostLease, ProtocolVersionMismatch } from "../errors.js";
6
- import { rowToTask } from "../models.js";
4
+ import { nowMs } from "../ids.js";
7
5
  import { loadMigrations, loadStatements } from "../sql.js";
8
- const SUPPORTED_PROTOCOL_MAJOR = 1;
9
- const LEASE_EXPIRED_ERROR = errorEnvelope({
10
- type: "LeaseExpired",
11
- code: "lease_expired",
12
- message: "task lease expired and max attempts reached",
13
- retryable: false,
14
- });
15
- // Serialized once: it's an immutable constant bound on every claim that finds work.
16
- const LEASE_EXPIRED_ERROR_JSON = JSON.stringify(LEASE_EXPIRED_ERROR);
6
+ import { checkProtocolVersion, statementParams, TaskStore, } from "./base.js";
7
+ const WAL_RETRY_DELAY_MS = 50;
8
+ const WAL_RETRY_BUDGET_MS = 5_000;
9
+ /** Sleep without yielding — the whole open path is synchronous already. */
10
+ function sleepSync(ms) {
11
+ Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
12
+ }
13
+ /** Whether this path names an in-memory database rather than a file. */
14
+ function isMemory(path) {
15
+ return path === ":memory:" || path.includes("mode=memory");
16
+ }
17
+ /**
18
+ * Serializes every SQLiteStore on one database file, process-wide.
19
+ *
20
+ * better-sqlite3 is synchronous, and a transaction holds SQLite's write lock
21
+ * across `await`s (the callback seam is shared with Postgres, so it is async). A
22
+ * second connection in this process then blocks the only thread waiting for that
23
+ * lock, and the holder can never reach COMMIT — reaching it needs the thread the
24
+ * waiter is sitting on. busy_timeout cannot break that inversion, being one
25
+ * thread; the wait just burns the timeout and throws "database is locked". So the
26
+ * two must not overlap at all.
27
+ *
28
+ * Keyed by database, not by store: what the lock protects is the file. Across
29
+ * processes there is no inversion (the holder keeps its own thread) and
30
+ * busy_timeout still applies. An in-memory database is private to one connection
31
+ * and gets a key of its own.
32
+ */
33
+ const fileLocks = new Map();
34
+ let memoryDbSeq = 0;
35
+ /**
36
+ * Put the database in WAL mode, waiting out a concurrent cold start.
37
+ *
38
+ * journal_mode is a persistent property of the file, so only the first connection
39
+ * to a new database actually switches it — and that switch needs an exclusive
40
+ * lock. `busy_timeout` does not cover it: SQLite returns SQLITE_BUSY for a
41
+ * journal_mode change rather than invoking the busy handler, so several processes
42
+ * opening the same new database at once would otherwise get an instant "database
43
+ * is locked". Retry briefly instead; the window is only as long as one other
44
+ * opener's switch.
45
+ *
46
+ * Callers must skip in-memory databases: those report journal_mode = "memory" and
47
+ * can never be WAL, so waiting for one is waiting for something that will not
48
+ * happen.
49
+ */
50
+ function enableWal(db) {
51
+ const deadline = Date.now() + WAL_RETRY_BUDGET_MS;
52
+ for (;;) {
53
+ try {
54
+ const rows = db.pragma("journal_mode = WAL");
55
+ if (rows[0]?.journal_mode?.toLowerCase() === "wal")
56
+ return;
57
+ }
58
+ catch (err) {
59
+ const message = String(err.message ?? err);
60
+ if (!/locked|busy/i.test(message))
61
+ throw err;
62
+ }
63
+ if (Date.now() >= deadline) {
64
+ throw new Error("could not switch the database to WAL mode: it stayed locked by another connection");
65
+ }
66
+ sleepSync(WAL_RETRY_DELAY_MS);
67
+ }
68
+ }
17
69
  /**
18
- * SQLiteStore — better-sqlite3 backend executing the shared cairnq-protocol SQL.
70
+ * SQLiteStore — the SQLite dialect of the shared cairnq-protocol SQL.
71
+ *
72
+ * Everything protocol-shaped lives in TaskStore; this file is only what SQLite
73
+ * does differently: better-sqlite3's synchronous driver, BEGIN IMMEDIATE
74
+ * transactions, a read-only probe in front of the write lock, and time supplied
75
+ * by the SDK (`:now_ms`) rather than by the database.
19
76
  *
20
- * The driver is synchronous, which suits SQLite's single writer: claim is one
21
- * short transaction, the handler runs outside any transaction, and
22
- * progress/heartbeat/succeed/fail are each their own short write. JS being
23
- * single-threaded means sync DB calls never interleave. Cross-process contention
24
- * (deployment mode B) is absorbed by busy_timeout.
77
+ * The driver being synchronous suits SQLite's single writer: claim is one short
78
+ * transaction, the handler runs outside any transaction, and
79
+ * progress/heartbeat/succeed/fail are each their own short write. Cross-process
80
+ * contention is absorbed by busy_timeout.
25
81
  */
26
- export class SQLiteStore {
82
+ export class SQLiteStore extends TaskStore {
27
83
  path;
28
84
  opts;
29
85
  db = null;
30
86
  stmts = {};
31
87
  statements;
88
+ /** This store's entry in `fileLocks` — see there for why it is per-database. */
89
+ lockKey;
32
90
  constructor(path, opts = {}) {
91
+ super();
33
92
  this.path = path;
34
93
  this.opts = opts;
35
94
  this.statements = loadStatements("sqlite");
95
+ // Only a bare ":memory:" is guaranteed private to its connection, so only
96
+ // it gets a lock of its own. A "mode=memory" URI stays path-keyed: with
97
+ // cache=shared it names ONE shared database, and on a build without URI
98
+ // filenames it is a literal file — in both cases two stores on that string
99
+ // must share a lock. Over-serializing a private URI-memory database is
100
+ // harmless; skipping the lock on a shared one is the deadlock this map
101
+ // exists to prevent.
102
+ this.lockKey = path === ":memory:" ? `memory#${memoryDbSeq++}` : resolve(path);
36
103
  }
37
104
  async connect() {
38
105
  this.ensure();
@@ -47,251 +114,152 @@ export class SQLiteStore {
47
114
  ensure() {
48
115
  if (this.db)
49
116
  return this.db;
50
- if (this.path !== ":memory:")
117
+ const memory = isMemory(this.path);
118
+ if (!memory)
51
119
  mkdirSync(dirname(this.path), { recursive: true });
52
120
  const db = new Database(this.path);
53
- db.pragma("journal_mode = WAL");
54
- db.pragma("foreign_keys = ON");
121
+ // busy_timeout first, so every later statement waits out contention instead
122
+ // of failing instantly.
55
123
  db.pragma(`busy_timeout = ${this.opts.busyTimeoutMs ?? 5000}`);
124
+ // WAL exists so several processes can share one file. An in-memory database
125
+ // is private to this connection, so there is nothing to share or wait for.
126
+ if (!memory)
127
+ enableWal(db);
128
+ db.pragma("foreign_keys = ON");
56
129
  this.applyMigrations(db);
57
130
  for (const [name, sql] of Object.entries(this.statements)) {
58
131
  this.stmts[name] = db.prepare(sql);
59
132
  }
60
133
  this.db = db;
61
- this.checkVersion();
134
+ checkProtocolVersion(this.readProtocolVersion());
62
135
  return db;
63
136
  }
64
137
  applyMigrations(db) {
65
138
  db.exec("create table if not exists cairnq_migrations " +
66
139
  "(name text primary key, applied_at_ms integer not null)");
67
- const applied = new Set(db.prepare("select name from cairnq_migrations").all().map((r) => r.name));
68
- // `or ignore`: another process may apply the same migration concurrently on a
69
- // fresh shared db (mode B cold start). Migrations are idempotent.
70
- const insert = db.prepare("insert or ignore into cairnq_migrations (name, applied_at_ms) values (?, ?)");
140
+ const isApplied = db.prepare("select 1 from cairnq_migrations where name = ?");
141
+ const insert = db.prepare("insert into cairnq_migrations (name, applied_at_ms) values (?, ?)");
71
142
  for (const { name, sql } of loadMigrations("sqlite")) {
72
- if (applied.has(name))
73
- continue;
143
+ // Check and apply under one write lock. Two processes cold-starting on a
144
+ // shared database would otherwise both see a migration as unapplied and
145
+ // both run it — harmless for the idempotent ones, not for a future ALTER.
146
+ // `immediate` takes the write lock up front; the loser sees it applied.
74
147
  db.transaction(() => {
148
+ if (isApplied.get(name))
149
+ return;
75
150
  db.exec(sql);
76
151
  insert.run(name, nowMs());
77
- })();
78
- }
79
- }
80
- checkVersion() {
81
- const version = this.readProtocolVersion();
82
- if (version !== SUPPORTED_PROTOCOL_MAJOR) {
83
- throw new ProtocolVersionMismatch(`storage protocol_version=${version}, SDK supports ${SUPPORTED_PROTOCOL_MAJOR}`);
152
+ }).immediate();
84
153
  }
85
154
  }
86
155
  readProtocolVersion() {
87
- const row = this.db
88
- .prepare("select value from cairnq_meta where key = 'protocol_version'")
89
- .get();
90
- return row ? Number(row.value) : 0;
156
+ const rows = this.runNow("protocol_version", {});
157
+ return rows.length ? Number(rows[0].value) : 0;
91
158
  }
92
159
  async protocolVersion() {
93
160
  this.ensure();
94
- return this.readProtocolVersion();
161
+ // Under the store lock: this public read must not slip a statement into
162
+ // another operation's open transaction on the shared connection.
163
+ return this.withLock(() => this.readProtocolVersion());
95
164
  }
96
- all(name, params) {
97
- return this.stmts[name].all(params);
98
- }
99
- run(name, params) {
100
- this.stmts[name].run(params);
101
- }
102
- // An ownership-checked worker write (heartbeat/progress/succeed/complete/fail).
103
- // Each statement's WHERE pins worker_id + a live lease, so 0 rows back means the
104
- // lease was lost — every such write reports it the same way.
105
- ownedWrite(name, taskId, params) {
106
- const rows = this.all(name, params);
107
- if (!rows.length)
108
- throw new LostLease(taskId);
109
- return rowToTask(rows[0]);
110
- }
111
- // ------------------------------------------------------------- client side
112
- async submit(input) {
113
- this.ensure();
165
+ // ------------------------------------------------------------ dialect seam
166
+ /**
167
+ * Adapt the dialect-neutral parameters to what this statement binds.
168
+ *
169
+ * SQLite statements carry no DB clock, so every absolute `*_ms` is derived here
170
+ * from one `now`, and booleans cross as 0/1. The result is narrowed to the
171
+ * names the SQL actually uses, which is what makes it safe for a caller to pass
172
+ * one superset of parameters for both dialects.
173
+ *
174
+ * Each derivation writes a name Postgres does not use (`lease_until_ms` from
175
+ * `lease_ms`, and so on), so a statement binds one or the other, never both —
176
+ * which is why the derived values can be computed unconditionally and left for
177
+ * the narrowing step to discard.
178
+ */
179
+ bind(sql, params) {
114
180
  const now = nowMs();
115
- const id = newId("task");
116
- const ins = {
117
- id,
118
- name: input.name,
119
- queue: input.queue ?? "default",
120
- payload: JSON.stringify(input.payload ?? {}),
121
- metadata: JSON.stringify(input.metadata ?? {}),
122
- max_attempts: input.maxAttempts ?? 3,
123
- priority: input.priority ?? 0,
124
- run_at_ms: now + (input.runAtDelayMs ?? 0),
125
- parent_id: input.parentId ?? null,
126
- root_id: input.rootId ?? id,
127
- correlation_id: input.correlationId ?? null,
128
- now_ms: now,
129
- };
130
- const key = input.key ?? null;
131
- const conflict = input.conflict ?? "reuse";
132
- const txn = this.db.transaction(() => {
133
- if (key === null)
134
- return this.all("insert_task", ins)[0];
135
- const existing = this.all("get_key", { key });
136
- if (existing.length) {
137
- const exId = existing[0].task_id;
138
- if (conflict === "reuse")
139
- return this.all("get", { id: exId })[0];
140
- if (conflict === "reject")
141
- throw new AlreadyExists(key);
142
- if (conflict === "replace") {
143
- this.all("cancel", { id: exId, now_ms: now });
144
- const row = this.all("insert_task", ins)[0];
145
- this.run("upsert_key", { key, task_id: id, now_ms: now });
146
- return row;
147
- }
148
- throw new Error(`unknown conflict strategy: ${conflict}`);
181
+ const bound = {};
182
+ for (const name of statementParams(sql)) {
183
+ switch (name) {
184
+ case "now_ms":
185
+ bound[name] = now;
186
+ break;
187
+ case "lease_until_ms":
188
+ bound[name] = now + params.lease_ms;
189
+ break;
190
+ case "run_at_ms":
191
+ bound[name] = now + params.delay_ms;
192
+ break;
193
+ case "before_ms":
194
+ bound[name] = now - params.older_than_ms;
195
+ break;
196
+ case "queues":
197
+ bound[name] = JSON.stringify(params.queues);
198
+ break;
199
+ case "names":
200
+ // json_each needs a JSON array; null stays null so the SQL's
201
+ // `:names is null` arm means "no filter".
202
+ bound[name] = params.names == null ? null : JSON.stringify(params.names);
203
+ break;
204
+ case "retryable":
205
+ case "reset_attempt":
206
+ bound[name] = params[name] ? 1 : 0;
207
+ break;
208
+ default:
209
+ bound[name] = params[name];
149
210
  }
150
- const row = this.all("insert_task", ins)[0];
151
- this.run("upsert_key", { key, task_id: id, now_ms: now });
152
- return row;
153
- });
154
- return rowToTask(txn.immediate());
155
- }
156
- async get(taskId) {
157
- this.ensure();
158
- const rows = this.all("get", { id: taskId });
159
- return rows.length ? rowToTask(rows[0]) : null;
160
- }
161
- async getByKey(key) {
162
- this.ensure();
163
- const rows = this.all("get_by_key", { key });
164
- return rows.length ? rowToTask(rows[0]) : null;
165
- }
166
- async list(input = {}) {
167
- this.ensure();
168
- const rows = this.all("list", {
169
- status: input.status ?? null,
170
- queue: input.queue ?? null,
171
- name: input.name ?? null,
172
- root_id: input.rootId ?? null,
173
- correlation_id: input.correlationId ?? null,
174
- limit: input.limit ?? 100,
175
- offset: input.offset ?? 0,
176
- });
177
- return rows.map(rowToTask);
178
- }
179
- async cancel(taskId) {
180
- this.ensure();
181
- const rows = this.all("cancel", { id: taskId, now_ms: nowMs() });
182
- return rows.length ? rowToTask(rows[0]) : null;
183
- }
184
- async cancelByKey(key) {
185
- this.ensure();
186
- const txn = this.db.transaction(() => {
187
- const existing = this.all("get_key", { key });
188
- if (!existing.length)
189
- return null;
190
- const rows = this.all("cancel", { id: existing[0].task_id, now_ms: nowMs() });
191
- return rows.length ? rows[0] : null;
192
- });
193
- const row = txn.immediate();
194
- return row ? rowToTask(row) : null;
195
- }
196
- async retry(taskId, opts = {}) {
197
- this.ensure();
198
- const rows = this.all("retry", {
199
- id: taskId,
200
- now_ms: nowMs(),
201
- reset_attempt: opts.resetAttempt ? 1 : 0,
202
- });
203
- return rows.length ? rowToTask(rows[0]) : null;
204
- }
205
- async retryByKey(key, opts = {}) {
206
- this.ensure();
207
- const txn = this.db.transaction(() => {
208
- const existing = this.all("get_key", { key });
209
- if (!existing.length)
210
- return null;
211
- const rows = this.all("retry", {
212
- id: existing[0].task_id,
213
- now_ms: nowMs(),
214
- reset_attempt: opts.resetAttempt ? 1 : 0,
215
- });
216
- return rows.length ? rows[0] : null;
217
- });
218
- const row = txn.immediate();
219
- return row ? rowToTask(row) : null;
211
+ }
212
+ return bound;
220
213
  }
221
- // ------------------------------------------------------------- worker side
222
- async claim(input) {
223
- this.ensure();
224
- const now = nowMs();
225
- const queues = JSON.stringify(input.queues);
226
- const leaseMs = input.leaseMs ?? 30_000;
227
- const limit = input.limit ?? 1;
228
- // Read-only probe first: skip the write lock entirely when idle.
229
- const probe = this.all("claimable_probe", { queues, now_ms: now })[0];
230
- if (!probe || !probe.has_work)
214
+ runNow(name, params) {
215
+ const stmt = this.stmts[name];
216
+ const bound = this.bind(this.statements[name], params);
217
+ // Nearly every protocol statement ends in RETURNING; upsert_key does not, and
218
+ // better-sqlite3 refuses .all() on a statement that yields no rows.
219
+ if (!stmt.reader) {
220
+ stmt.run(bound);
231
221
  return [];
232
- const txn = this.db.transaction(() => {
233
- this.all("recover_leases", {
234
- now_ms: now,
235
- lease_expired_error: LEASE_EXPIRED_ERROR_JSON,
236
- });
237
- return this.all("claim", {
238
- queues,
239
- now_ms: now,
240
- worker_id: input.workerId,
241
- lease_until_ms: now + leaseMs,
242
- limit,
243
- });
244
- });
245
- return txn.immediate().map(rowToTask);
246
- }
247
- async heartbeat(input) {
248
- this.ensure();
249
- const now = nowMs();
250
- return this.ownedWrite("heartbeat", input.taskId, {
251
- id: input.taskId,
252
- worker_id: input.workerId,
253
- now_ms: now,
254
- lease_until_ms: now + (input.leaseMs ?? 30_000),
255
- });
222
+ }
223
+ return stmt.all(bound);
256
224
  }
257
- async progress(input) {
258
- this.ensure();
259
- return this.ownedWrite("progress", input.taskId, {
260
- id: input.taskId,
261
- worker_id: input.workerId,
262
- now_ms: nowMs(),
263
- progress: input.progress,
264
- message: input.message,
265
- });
225
+ /** Serialize an operation against every other operation on this database. */
226
+ withLock(fn) {
227
+ const previous = fileLocks.get(this.lockKey) ?? Promise.resolve();
228
+ const run = previous.then(fn, fn);
229
+ fileLocks.set(this.lockKey, run.then(() => undefined, () => undefined));
230
+ return run;
266
231
  }
267
- async succeed(input) {
232
+ async fetch(name, params) {
268
233
  this.ensure();
269
- return this.ownedWrite("succeed", input.taskId, {
270
- id: input.taskId,
271
- worker_id: input.workerId,
272
- now_ms: nowMs(),
273
- result: input.result == null ? null : JSON.stringify(input.result),
274
- message: null,
275
- });
234
+ return this.withLock(() => this.runNow(name, params));
276
235
  }
277
- async complete(input) {
278
- this.ensure();
279
- return this.ownedWrite("complete", input.taskId, {
280
- id: input.taskId,
281
- worker_id: input.workerId,
282
- now_ms: nowMs(),
283
- result: input.result == null ? null : JSON.stringify(input.result),
236
+ async tx(fn) {
237
+ const db = this.ensure();
238
+ // BEGIN IMMEDIATE by hand rather than db.transaction(): the callback is async
239
+ // (the seam is shared with Postgres), and better-sqlite3's wrapper only takes
240
+ // a synchronous one. The lock above makes the manual version safe.
241
+ return this.withLock(async () => {
242
+ db.exec("BEGIN IMMEDIATE");
243
+ try {
244
+ const out = await fn(async (name, params) => this.runNow(name, params));
245
+ db.exec("COMMIT");
246
+ return out;
247
+ }
248
+ catch (err) {
249
+ try {
250
+ db.exec("ROLLBACK");
251
+ }
252
+ catch {
253
+ // Already rolled back by SQLite (e.g. a constraint abort).
254
+ }
255
+ throw err;
256
+ }
284
257
  });
285
258
  }
286
- async fail(input) {
287
- this.ensure();
288
- return this.ownedWrite("fail", input.taskId, {
289
- id: input.taskId,
290
- worker_id: input.workerId,
291
- now_ms: nowMs(),
292
- error: JSON.stringify(input.error ?? {}),
293
- retryable: input.retryable === false ? 0 : 1,
294
- delay_ms: input.delayMs ?? 0,
295
- });
259
+ async hasClaimableWork(params) {
260
+ // Read-only probe first: an idle worker never takes SQLite's single write
261
+ // lock, so idle workers don't serialize against each other.
262
+ const rows = await this.fetch("claimable_probe", params);
263
+ return Boolean(rows[0]?.has_work);
296
264
  }
297
265
  }
package/dist/wait.d.ts CHANGED
@@ -1,8 +1,21 @@
1
1
  import { type Task } from "./models.js";
2
2
  import type { TaskStore } from "./store/base.js";
3
+ export declare const DEFAULT_POLL_MS = 100;
4
+ export declare const MAX_POLL_MS = 500;
5
+ /**
6
+ * Grow the polling interval towards the ceiling.
7
+ *
8
+ * wait() has no idea whether the task takes 50ms or an hour. Starting tight keeps
9
+ * short tasks snappy; growing keeps a long wait from costing a read every 100ms
10
+ * for its whole duration. The +1 keeps truncation from pinning tiny intervals:
11
+ * Math.floor(1 * 1.5) === 1 would otherwise never grow past 1.
12
+ */
13
+ export declare function nextPollMs(current: number, maxMs: number): number;
3
14
  /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
4
- * Throws TaskTimeout, leaving the task running. */
5
- export declare function pollWait(store: TaskStore, taskId: string, { timeoutMs, pollMs }: {
15
+ * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
16
+ * it backs off towards `maxPollMs`. */
17
+ export declare function pollWait(store: TaskStore, taskId: string, { timeoutMs, pollMs, maxPollMs, }: {
6
18
  timeoutMs: number;
7
19
  pollMs?: number;
20
+ maxPollMs?: number;
8
21
  }): Promise<Task>;
package/dist/wait.js CHANGED
@@ -1,18 +1,36 @@
1
1
  import { TaskTimeout } from "./errors.js";
2
2
  import { nowMs } from "./ids.js";
3
3
  import { isTerminal } from "./models.js";
4
- const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
4
+ export const DEFAULT_POLL_MS = 100;
5
+ export const MAX_POLL_MS = 500;
6
+ const GROWTH = 1.5;
7
+ /**
8
+ * Grow the polling interval towards the ceiling.
9
+ *
10
+ * wait() has no idea whether the task takes 50ms or an hour. Starting tight keeps
11
+ * short tasks snappy; growing keeps a long wait from costing a read every 100ms
12
+ * for its whole duration. The +1 keeps truncation from pinning tiny intervals:
13
+ * Math.floor(1 * 1.5) === 1 would otherwise never grow past 1.
14
+ */
15
+ export function nextPollMs(current, maxMs) {
16
+ return Math.min(maxMs, Math.max(current + 1, Math.floor(current * GROWTH)));
17
+ }
5
18
  /** Poll get() until terminal or timeout. Returns the terminal Task (any status).
6
- * Throws TaskTimeout, leaving the task running. */
7
- export async function pollWait(store, taskId, { timeoutMs, pollMs = 150 }) {
19
+ * Throws TaskTimeout, leaving the task running. `pollMs` is the *first* interval;
20
+ * it backs off towards `maxPollMs`. */
21
+ export async function pollWait(store, taskId, { timeoutMs, pollMs = DEFAULT_POLL_MS, maxPollMs = MAX_POLL_MS, }) {
8
22
  const deadline = nowMs() + timeoutMs;
23
+ let interval = pollMs;
9
24
  for (;;) {
10
25
  const task = await store.get(taskId);
11
26
  if (task && isTerminal(task))
12
27
  return task;
13
28
  const remaining = deadline - nowMs();
14
29
  if (remaining <= 0)
15
- throw new TaskTimeout(taskId);
16
- await sleep(Math.min(pollMs, remaining));
30
+ throw new TaskTimeout(taskId, { timeoutMs, task });
31
+ // A store with a push channel (Postgres) cuts the sleep short when the task
32
+ // goes terminal; the re-get above stays the source of truth either way.
33
+ await store.taskDoneWake(taskId, Math.min(interval, remaining));
34
+ interval = nextPollMs(interval, maxPollMs);
17
35
  }
18
36
  }
package/dist/worker.d.ts CHANGED
@@ -4,13 +4,40 @@ import { type TaskDef } from "./task.js";
4
4
  export type Handler = (ctx: TaskContext, payload: any) => unknown | Promise<unknown>;
5
5
  /** Handler typed against a TaskDef<P, R>: payload is P, the return is R. */
6
6
  export type TypedHandler<P, R> = (ctx: TaskContext, payload: P) => R | Promise<R>;
7
+ /** Where an error the worker recovered from came from. */
8
+ export type ErrorPhase = "claim" | "execute";
7
9
  export interface WorkerOptions {
8
10
  concurrency?: number;
9
11
  leaseMs?: number;
10
12
  heartbeatIntervalMs?: number;
11
13
  pollIntervalMs?: number;
12
14
  claimBatch?: number;
15
+ /** Base delay before re-running a failed attempt; doubles per attempt. 0 disables. */
16
+ retryBackoffMs?: number;
17
+ /** Ceiling for the doubling. */
18
+ retryBackoffMaxMs?: number;
19
+ /**
20
+ * Wall-clock ceiling for one attempt. The heartbeat renews the lease for as
21
+ * long as a handler runs, so a hung handler would otherwise hold its task
22
+ * `running` (and its concurrency slot) forever — cancel can't help,
23
+ * cooperative checks need a live handler. On expiry the worker abandons the
24
+ * attempt (ctx.signal aborts, further ctx writes throw LostLease) and records
25
+ * a retryable `handler_timeout` failure. Unset disables the ceiling.
26
+ */
27
+ maxRunMs?: number;
28
+ /**
29
+ * Called for errors the worker survived — a claim that threw, a store write
30
+ * that failed while finalizing a task. Without it these are silent: the run
31
+ * loop carries on either way, so this is the only place an operator learns a
32
+ * worker is limping. Must not throw.
33
+ */
34
+ onError?: (err: unknown, info: {
35
+ phase: ErrorPhase;
36
+ taskId?: string;
37
+ }) => void;
13
38
  }
39
+ /** Exponential backoff for the next attempt of a task that just failed. */
40
+ export declare function retryDelayMs(attempt: number, baseMs: number, maxMs: number): number;
14
41
  export declare class Worker {
15
42
  private readonly store;
16
43
  private readonly queues;
@@ -18,7 +45,8 @@ export declare class Worker {
18
45
  private readonly handlers;
19
46
  private readonly workerId;
20
47
  private stopped;
21
- private stopResolvers;
48
+ private stopWake;
49
+ private readonly stopped$;
22
50
  private ownsStore;
23
51
  constructor(store: TaskStore, queues: string[], opts?: WorkerOptions);
24
52
  static sqlite(path: string, opts?: WorkerOptions & {
@@ -39,9 +67,11 @@ export declare class Worker {
39
67
  /** Close the underlying store connection. Call after run() returns. */
40
68
  close(): Promise<void>;
41
69
  private closeIfOwned;
70
+ private report;
42
71
  run(opts?: {
43
72
  concurrency?: number;
44
73
  }): Promise<void>;
74
+ private loop;
45
75
  /** Blocking-style entry point for a standalone worker process: run until
46
76
  * SIGINT/SIGTERM, then close the store. Use this at a script's top level;
47
77
  * use run() / background() when you manage the event loop yourself. */
@@ -52,9 +82,31 @@ export declare class Worker {
52
82
  background<T>(fn: () => Promise<T>, opts?: {
53
83
  concurrency?: number;
54
84
  }): Promise<T>;
85
+ /**
86
+ * Run one task to completion. Never rejects: a task-level failure is reported
87
+ * through onError and the loop moves on. (It used to reject into a promise
88
+ * nobody awaited — an unhandled rejection that took the process down.)
89
+ */
55
90
  private execute;
91
+ /**
92
+ * Run one attempt, bounded by maxRunMs when set. On timeout the attempt is
93
+ * abandoned: the context is flagged lease-lost first (ctx.signal aborts, and
94
+ * a handler that keeps running can never write again — see
95
+ * TaskContext.owned), then the still-pending promise is left to settle on
96
+ * its own, its outcome discarded. The caller records the handler_timeout
97
+ * failure; lease recovery is NOT involved, so redelivery is immediate.
98
+ */
99
+ private attempt;
56
100
  private startHeartbeat;
57
101
  private safeFail;
102
+ /**
103
+ * The empty-poll sleep. A store with a push channel (Postgres LISTEN/NOTIFY)
104
+ * cuts it short when a task on this worker's queues becomes claimable;
105
+ * stop() interrupts it either way, and sleepOrStop bounds it at `ms` so the
106
+ * poll fallback — which also drives lease recovery — never stretches.
107
+ */
108
+ private idle;
58
109
  private sleepOrStop;
110
+ /** Take SIGINT/SIGTERM for the duration of serve(). Returns the undo. */
59
111
  private installSignals;
60
112
  }