cairnq 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/README.md +28 -0
  2. package/dist/_protocol/migrations/postgres/0001_init.sql +3 -1
  3. package/dist/_protocol/migrations/postgres/0002_purge_index.sql +6 -0
  4. package/dist/_protocol/migrations/postgres/0003_notify.sql +38 -0
  5. package/dist/_protocol/migrations/postgres/0004_lease_index.sql +16 -0
  6. package/dist/_protocol/migrations/postgres/0005_clear_terminal_lease.sql +17 -0
  7. package/dist/_protocol/migrations/sqlite/0001_init.sql +3 -1
  8. package/dist/_protocol/migrations/sqlite/0002_purge_index.sql +6 -0
  9. package/dist/_protocol/migrations/sqlite/0004_lease_index.sql +22 -0
  10. package/dist/_protocol/migrations/sqlite/0005_clear_terminal_lease.sql +17 -0
  11. package/dist/_protocol/sql/postgres/claim.sql +18 -5
  12. package/dist/_protocol/sql/postgres/claim_one_queue.sql +35 -0
  13. package/dist/_protocol/sql/postgres/complete.sql +3 -0
  14. package/dist/_protocol/sql/postgres/fail.sql +30 -8
  15. package/dist/_protocol/sql/postgres/insert_task.sql +6 -3
  16. package/dist/_protocol/sql/postgres/list.sql +3 -1
  17. package/dist/_protocol/sql/postgres/lock_key.sql +9 -0
  18. package/dist/_protocol/sql/postgres/progress.sql +4 -3
  19. package/dist/_protocol/sql/postgres/protocol_version.sql +4 -0
  20. package/dist/_protocol/sql/postgres/purge.sql +25 -0
  21. package/dist/_protocol/sql/postgres/recover_leases.sql +49 -14
  22. package/dist/_protocol/sql/postgres/retry.sql +3 -0
  23. package/dist/_protocol/sql/postgres/stats.sql +8 -0
  24. package/dist/_protocol/sql/postgres/succeed.sql +4 -0
  25. package/dist/_protocol/sql/sqlite/claim.sql +13 -2
  26. package/dist/_protocol/sql/sqlite/claim_one_queue.sql +36 -0
  27. package/dist/_protocol/sql/sqlite/claimable_probe.sql +6 -2
  28. package/dist/_protocol/sql/sqlite/complete.sql +3 -0
  29. package/dist/_protocol/sql/sqlite/fail.sql +32 -8
  30. package/dist/_protocol/sql/sqlite/list.sql +3 -1
  31. package/dist/_protocol/sql/sqlite/lock_key.sql +5 -0
  32. package/dist/_protocol/sql/sqlite/progress.sql +6 -2
  33. package/dist/_protocol/sql/sqlite/protocol_version.sql +4 -0
  34. package/dist/_protocol/sql/sqlite/purge.sql +18 -0
  35. package/dist/_protocol/sql/sqlite/recover_leases.sql +25 -7
  36. package/dist/_protocol/sql/sqlite/retry.sql +3 -0
  37. package/dist/_protocol/sql/sqlite/stats.sql +8 -0
  38. package/dist/_protocol/sql/sqlite/succeed.sql +4 -0
  39. package/dist/client.d.ts +10 -2
  40. package/dist/client.js +12 -0
  41. package/dist/context.d.ts +17 -1
  42. package/dist/context.js +60 -6
  43. package/dist/errors.d.ts +18 -2
  44. package/dist/errors.js +49 -3
  45. package/dist/index.d.ts +3 -2
  46. package/dist/index.js +2 -1
  47. package/dist/sql.js +16 -9
  48. package/dist/store/base.d.ts +114 -9
  49. package/dist/store/base.js +376 -1
  50. package/dist/store/postgres.d.ts +62 -63
  51. package/dist/store/postgres.js +245 -222
  52. package/dist/store/sqlite.d.ts +83 -60
  53. package/dist/store/sqlite.js +370 -234
  54. package/dist/wait.d.ts +15 -2
  55. package/dist/wait.js +23 -5
  56. package/dist/worker.d.ts +53 -1
  57. package/dist/worker.js +202 -42
  58. package/package.json +9 -2
  59. package/src/client.ts +16 -2
  60. package/src/context.ts +70 -13
  61. package/src/errors.ts +59 -4
  62. package/src/index.ts +3 -1
  63. package/src/sql.ts +15 -8
  64. package/src/store/base.ts +443 -27
  65. package/src/store/postgres.ts +243 -267
  66. package/src/store/sqlite.ts +378 -263
  67. package/src/wait.ts +28 -5
  68. package/src/worker.ts +242 -42
@@ -1,38 +1,188 @@
1
1
  import { mkdirSync } from "node:fs";
2
- import { dirname } from "node:path";
2
+ import { dirname, resolve } from "node:path";
3
3
  import Database from "better-sqlite3";
4
- import { newId, nowMs } from "../ids.js";
5
- import { AlreadyExists, errorEnvelope, LostLease, ProtocolVersionMismatch } from "../errors.js";
6
- import { rowToTask } from "../models.js";
4
+ import { nowMs } from "../ids.js";
7
5
  import { loadMigrations, loadStatements } from "../sql.js";
8
- const SUPPORTED_PROTOCOL_MAJOR = 1;
9
- const LEASE_EXPIRED_ERROR = errorEnvelope({
10
- type: "LeaseExpired",
11
- code: "lease_expired",
12
- message: "task lease expired and max attempts reached",
13
- retryable: false,
14
- });
15
- // Serialized once: it's an immutable constant bound on every claim that finds work.
16
- const LEASE_EXPIRED_ERROR_JSON = JSON.stringify(LEASE_EXPIRED_ERROR);
6
+ import { checkProtocolVersion, statementParams, TaskStore, } from "./base.js";
7
+ const WAL_RETRY_DELAY_MS = 50;
8
+ const WAL_RETRY_BUDGET_MS = 5_000;
9
+ const BUSY_RETRY_BASE_MS = 1;
10
+ const BUSY_RETRY_MAX_DELAY_MS = 50;
17
11
  /**
18
- * SQLiteStore — better-sqlite3 backend executing the shared cairnq-protocol SQL.
12
+ * How often a live connection revisits its planner statistics.
19
13
  *
20
- * The driver is synchronous, which suits SQLite's single writer: claim is one
21
- * short transaction, the handler runs outside any transaction, and
22
- * progress/heartbeat/succeed/fail are each their own short write. JS being
23
- * single-threaded means sync DB calls never interleave. Cross-process contention
24
- * (deployment mode B) is absorbed by busy_timeout.
14
+ * Bounds how long the planner can work from a stale table shape; a minute is
15
+ * arbitrary but small next to the days a worker holds its connection. It does not
16
+ * set how often an ANALYZE actually runs — SQLite decides that itself, and only
17
+ * when a table has outgrown its statistics by roughly 24x, so a shorter interval
18
+ * costs more no-ops (a few microseconds each) rather than more analyzing.
25
19
  */
26
- export class SQLiteStore {
20
+ const STATS_REFRESH_INTERVAL_MS = 60_000;
21
+ /** Sleep without yielding — the whole open path is synchronous already. */
22
+ function sleepSync(ms) {
23
+ Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
24
+ }
25
+ /** Sleep by yielding to the event loop — the point of the busy retry loop. */
26
+ function sleep(ms) {
27
+ return new Promise((resolve) => setTimeout(resolve, ms));
28
+ }
29
+ /**
30
+ * Whether this error is SQLite refusing to wait for the write lock.
31
+ *
32
+ * Prefix match: the code carries detail suffixes (`SQLITE_BUSY_SNAPSHOT`). Only
33
+ * SQLITE_BUSY qualifies — SQLITE_LOCKED is same-connection table contention,
34
+ * which the per-file lock prevents and a retry could not resolve anyway.
35
+ */
36
+ function isBusy(err) {
37
+ if (!err || typeof err !== "object")
38
+ return false;
39
+ const code = err.code;
40
+ return typeof code === "string" && code.startsWith("SQLITE_BUSY");
41
+ }
42
+ /** Whether this path names an in-memory database rather than a file. */
43
+ function isMemory(path) {
44
+ return path === ":memory:" || path.includes("mode=memory");
45
+ }
46
+ /**
47
+ * Whether cairnq_tasks has been analyzed at all.
48
+ *
49
+ * Two steps because sqlite_stat1 does not exist until something runs ANALYZE, and
50
+ * querying a missing table is an error rather than an empty result.
51
+ */
52
+ function hasStatistics(db) {
53
+ const table = db
54
+ .prepare("select 1 from sqlite_master where type = 'table' and name = 'sqlite_stat1'")
55
+ .get();
56
+ if (!table)
57
+ return false;
58
+ return Boolean(db.prepare("select 1 from sqlite_stat1 where tbl = 'cairnq_tasks'").get());
59
+ }
60
+ /**
61
+ * Bring cairnq_tasks' statistics up to date, cheaply enough to call on a timer.
62
+ *
63
+ * Without them the planner misreads `status = 'running'` as a large fraction of the
64
+ * table and passes over the partial cairnq_tasks_lease_idx that lease recovery is
65
+ * indexed for.
66
+ *
67
+ * The explicit bootstrap is not redundant with `PRAGMA optimize`. Before SQLite
68
+ * 3.46 the pragma skips a table that has no sqlite_stat1 entry entirely — no mask
69
+ * changes that, verified on 3.45.1 — so on those builds it can never produce the
70
+ * *first* statistics, and the index stays unused for the life of the database.
71
+ * Distro Pythons link exactly those builds (Ubuntu 24.04 ships 3.45.1), while
72
+ * better-sqlite3 bundles its own newer one, so this is also what keeps the two SDKs
73
+ * behaving alike rather than by luck of packaging.
74
+ *
75
+ * Once an entry exists, every version's pragma applies its own growth heuristic,
76
+ * which is the part worth deferring to: it is a few microseconds when there is
77
+ * nothing to do, where a bare ANALYZE would rescan the table every time.
78
+ */
79
+ function refreshStatistics(db) {
80
+ if (hasStatistics(db))
81
+ db.pragma("optimize");
82
+ // Scoped to the one table whose shape the planner gets wrong; the key and meta
83
+ // tables are read by primary key, where statistics change nothing. A database
84
+ // this one shares with the caller's own tables is left alone.
85
+ else
86
+ db.exec("ANALYZE cairnq_tasks");
87
+ }
88
+ /**
89
+ * Serializes every SQLiteStore on one database file, process-wide.
90
+ *
91
+ * better-sqlite3 is synchronous, and a transaction holds SQLite's write lock
92
+ * across `await`s (the callback seam is shared with Postgres, so it is async). Two
93
+ * connections in this process would then contend for that lock the expensive way:
94
+ * every loser spends SQLITE_BUSY retries and backoff on a holder it could simply
95
+ * have queued behind.
96
+ *
97
+ * Keyed by database, not by store: what the lock protects is the file. Across
98
+ * processes there is nothing to serialize from here — each holder has its own
99
+ * thread, and `withLock`'s retry absorbs that contention. An in-memory database is
100
+ * private to one connection and gets a key of its own.
101
+ */
102
+ const fileLocks = new Map();
103
+ let memoryDbSeq = 0;
104
+ /**
105
+ * Put the database in WAL mode, waiting out a concurrent cold start.
106
+ *
107
+ * journal_mode is a persistent property of the file, so only the first connection
108
+ * to a new database actually switches it — and that switch needs an exclusive
109
+ * lock. `busy_timeout` does not cover it: SQLite returns SQLITE_BUSY for a
110
+ * journal_mode change rather than invoking the busy handler, so several processes
111
+ * opening the same new database at once would otherwise get an instant "database
112
+ * is locked". Retry briefly instead; the window is only as long as one other
113
+ * opener's switch.
114
+ *
115
+ * Callers must skip in-memory databases: those report journal_mode = "memory" and
116
+ * can never be WAL, so waiting for one is waiting for something that will not
117
+ * happen.
118
+ */
119
+ function enableWal(db) {
120
+ const deadline = Date.now() + WAL_RETRY_BUDGET_MS;
121
+ for (;;) {
122
+ try {
123
+ const rows = db.pragma("journal_mode = WAL");
124
+ if (rows[0]?.journal_mode?.toLowerCase() === "wal")
125
+ return;
126
+ }
127
+ catch (err) {
128
+ if (!isBusy(err))
129
+ throw err;
130
+ }
131
+ if (Date.now() >= deadline) {
132
+ throw new Error("could not switch the database to WAL mode: it stayed locked by another connection");
133
+ }
134
+ sleepSync(WAL_RETRY_DELAY_MS);
135
+ }
136
+ }
137
+ /**
138
+ * SQLiteStore — the SQLite dialect of the shared cairnq-protocol SQL.
139
+ *
140
+ * Everything protocol-shaped lives in TaskStore; this file is only what SQLite
141
+ * does differently: better-sqlite3's synchronous driver, BEGIN IMMEDIATE
142
+ * transactions, a read-only probe in front of the write lock, and time supplied
143
+ * by the SDK (`:now_ms`) rather than by the database.
144
+ *
145
+ * The driver being synchronous suits SQLite's single writer: claim is one short
146
+ * transaction, the handler runs outside any transaction, and
147
+ * progress/heartbeat/succeed/fail are each their own short write.
148
+ *
149
+ * Cross-process contention is absorbed by retrying in JavaScript, not by
150
+ * busy_timeout. The two cost the same wait but not the same blocking: a nonzero
151
+ * busy_timeout waits *inside* the synchronous driver, so a caller that loses the
152
+ * write lock stalls this process's event loop for up to the whole timeout — the
153
+ * P99 of an HTTP server that submits tasks. Executing a statement takes
154
+ * microseconds; waiting for a lock takes milliseconds to seconds, and only the
155
+ * second part needs to happen off the thread. So busy_timeout goes to 0 (fail
156
+ * immediately) and the wait becomes an awaited backoff, which the event loop runs
157
+ * through. The budget is the same either way — `busyTimeoutMs`.
158
+ *
159
+ * The open path keeps a real busy_timeout: it is synchronous by nature (WAL
160
+ * switch, migrations) and happens once, under the caller's `connect()`.
161
+ */
162
+ export class SQLiteStore extends TaskStore {
27
163
  path;
28
- opts;
29
164
  db = null;
30
165
  stmts = {};
31
166
  statements;
167
+ /** This store's entry in `fileLocks` — see there for why it is per-database. */
168
+ lockKey;
169
+ /** How long a single operation may keep retrying a lost write lock. */
170
+ busyBudgetMs;
171
+ /** When this connection may next revisit its planner statistics. */
172
+ nextStatsRefreshAt = 0;
32
173
  constructor(path, opts = {}) {
174
+ super();
33
175
  this.path = path;
34
- this.opts = opts;
176
+ this.busyBudgetMs = opts.busyTimeoutMs ?? 5000;
35
177
  this.statements = loadStatements("sqlite");
178
+ // Only a bare ":memory:" is guaranteed private to its connection, so only
179
+ // it gets a lock of its own. A "mode=memory" URI stays path-keyed: with
180
+ // cache=shared it names ONE shared database, and on a build without URI
181
+ // filenames it is a literal file — in both cases two stores on that string
182
+ // must share a lock. Over-serializing a private URI-memory database is
183
+ // harmless; skipping the lock on a shared one is the deadlock this map
184
+ // exists to prevent.
185
+ this.lockKey = path === ":memory:" ? `memory#${memoryDbSeq++}` : resolve(path);
36
186
  }
37
187
  async connect() {
38
188
  this.ensure();
@@ -47,251 +197,237 @@ export class SQLiteStore {
47
197
  ensure() {
48
198
  if (this.db)
49
199
  return this.db;
50
- if (this.path !== ":memory:")
200
+ const memory = isMemory(this.path);
201
+ if (!memory)
51
202
  mkdirSync(dirname(this.path), { recursive: true });
52
203
  const db = new Database(this.path);
53
- db.pragma("journal_mode = WAL");
204
+ // Only the synchronous part of the open path gets a real busy_timeout: the WAL
205
+ // switch and the migrations cannot await a retry. See the class comment.
206
+ db.pragma(`busy_timeout = ${this.busyBudgetMs}`);
207
+ // WAL exists so several processes can share one file. An in-memory database
208
+ // is private to this connection, so there is nothing to share or wait for.
209
+ if (!memory)
210
+ enableWal(db);
54
211
  db.pragma("foreign_keys = ON");
55
- db.pragma(`busy_timeout = ${this.opts.busyTimeoutMs ?? 5000}`);
56
212
  this.applyMigrations(db);
213
+ // Everything past here either awaits its retry or is optional, so stop blocking.
214
+ db.pragma("busy_timeout = 0");
215
+ // Give the query planner statistics (see refreshStatistics), repeated on a timer
216
+ // from here on (see maybeRefreshStatistics).
217
+ try {
218
+ refreshStatistics(db);
219
+ }
220
+ catch (err) {
221
+ // Statistics are an optimization, never correctness, so losing them to a
222
+ // concurrent writer must not fail the open — the next one gets another
223
+ // chance. Anything else is a real fault and belongs to the caller.
224
+ if (!isBusy(err))
225
+ throw err;
226
+ }
227
+ this.nextStatsRefreshAt = Date.now() + STATS_REFRESH_INTERVAL_MS;
57
228
  for (const [name, sql] of Object.entries(this.statements)) {
58
229
  this.stmts[name] = db.prepare(sql);
59
230
  }
60
231
  this.db = db;
61
- this.checkVersion();
232
+ checkProtocolVersion(this.readProtocolVersion());
62
233
  return db;
63
234
  }
64
235
  applyMigrations(db) {
65
236
  db.exec("create table if not exists cairnq_migrations " +
66
237
  "(name text primary key, applied_at_ms integer not null)");
67
- const applied = new Set(db.prepare("select name from cairnq_migrations").all().map((r) => r.name));
68
- // `or ignore`: another process may apply the same migration concurrently on a
69
- // fresh shared db (mode B cold start). Migrations are idempotent.
70
- const insert = db.prepare("insert or ignore into cairnq_migrations (name, applied_at_ms) values (?, ?)");
238
+ const isApplied = db.prepare("select 1 from cairnq_migrations where name = ?");
239
+ const insert = db.prepare("insert into cairnq_migrations (name, applied_at_ms) values (?, ?)");
71
240
  for (const { name, sql } of loadMigrations("sqlite")) {
72
- if (applied.has(name))
73
- continue;
241
+ // Check and apply under one write lock. Two processes cold-starting on a
242
+ // shared database would otherwise both see a migration as unapplied and
243
+ // both run it — harmless for the idempotent ones, not for a future ALTER.
244
+ // `immediate` takes the write lock up front; the loser sees it applied.
74
245
  db.transaction(() => {
246
+ if (isApplied.get(name))
247
+ return;
75
248
  db.exec(sql);
76
249
  insert.run(name, nowMs());
77
- })();
78
- }
79
- }
80
- checkVersion() {
81
- const version = this.readProtocolVersion();
82
- if (version !== SUPPORTED_PROTOCOL_MAJOR) {
83
- throw new ProtocolVersionMismatch(`storage protocol_version=${version}, SDK supports ${SUPPORTED_PROTOCOL_MAJOR}`);
250
+ }).immediate();
84
251
  }
85
252
  }
86
253
  readProtocolVersion() {
87
- const row = this.db
88
- .prepare("select value from cairnq_meta where key = 'protocol_version'")
89
- .get();
90
- return row ? Number(row.value) : 0;
254
+ const rows = this.runNow("protocol_version", {});
255
+ return rows.length ? Number(rows[0].value) : 0;
91
256
  }
92
257
  async protocolVersion() {
93
258
  this.ensure();
94
- return this.readProtocolVersion();
95
- }
96
- all(name, params) {
97
- return this.stmts[name].all(params);
98
- }
99
- run(name, params) {
100
- this.stmts[name].run(params);
259
+ // Under the store lock: this public read must not slip a statement into
260
+ // another operation's open transaction on the shared connection.
261
+ return this.withLock(() => this.readProtocolVersion());
101
262
  }
102
- // An ownership-checked worker write (heartbeat/progress/succeed/complete/fail).
103
- // Each statement's WHERE pins worker_id + a live lease, so 0 rows back means the
104
- // lease was lost — every such write reports it the same way.
105
- ownedWrite(name, taskId, params) {
106
- const rows = this.all(name, params);
107
- if (!rows.length)
108
- throw new LostLease(taskId);
109
- return rowToTask(rows[0]);
110
- }
111
- // ------------------------------------------------------------- client side
112
- async submit(input) {
113
- this.ensure();
263
+ // ------------------------------------------------------------ dialect seam
264
+ /**
265
+ * Adapt the dialect-neutral parameters to what this statement binds.
266
+ *
267
+ * SQLite statements carry no DB clock, so every absolute `*_ms` is derived here
268
+ * from one `now`, and booleans cross as 0/1. The result is narrowed to the
269
+ * names the SQL actually uses, which is what makes it safe for a caller to pass
270
+ * one superset of parameters for both dialects.
271
+ *
272
+ * Each derivation writes a name Postgres does not use (`lease_until_ms` from
273
+ * `lease_ms`, and so on), so a statement binds one or the other, never both —
274
+ * which is why the derived values can be computed unconditionally and left for
275
+ * the narrowing step to discard.
276
+ */
277
+ bind(sql, params) {
114
278
  const now = nowMs();
115
- const id = newId("task");
116
- const ins = {
117
- id,
118
- name: input.name,
119
- queue: input.queue ?? "default",
120
- payload: JSON.stringify(input.payload ?? {}),
121
- metadata: JSON.stringify(input.metadata ?? {}),
122
- max_attempts: input.maxAttempts ?? 3,
123
- priority: input.priority ?? 0,
124
- run_at_ms: now + (input.runAtDelayMs ?? 0),
125
- parent_id: input.parentId ?? null,
126
- root_id: input.rootId ?? id,
127
- correlation_id: input.correlationId ?? null,
128
- now_ms: now,
129
- };
130
- const key = input.key ?? null;
131
- const conflict = input.conflict ?? "reuse";
132
- const txn = this.db.transaction(() => {
133
- if (key === null)
134
- return this.all("insert_task", ins)[0];
135
- const existing = this.all("get_key", { key });
136
- if (existing.length) {
137
- const exId = existing[0].task_id;
138
- if (conflict === "reuse")
139
- return this.all("get", { id: exId })[0];
140
- if (conflict === "reject")
141
- throw new AlreadyExists(key);
142
- if (conflict === "replace") {
143
- this.all("cancel", { id: exId, now_ms: now });
144
- const row = this.all("insert_task", ins)[0];
145
- this.run("upsert_key", { key, task_id: id, now_ms: now });
146
- return row;
147
- }
148
- throw new Error(`unknown conflict strategy: ${conflict}`);
279
+ const bound = {};
280
+ for (const name of statementParams(sql)) {
281
+ switch (name) {
282
+ case "now_ms":
283
+ bound[name] = now;
284
+ break;
285
+ case "lease_until_ms":
286
+ bound[name] = now + params.lease_ms;
287
+ break;
288
+ case "run_at_ms":
289
+ bound[name] = now + params.delay_ms;
290
+ break;
291
+ case "before_ms":
292
+ bound[name] = now - params.older_than_ms;
293
+ break;
294
+ case "queues":
295
+ bound[name] = JSON.stringify(params.queues);
296
+ break;
297
+ case "names":
298
+ // json_each needs a JSON array; null stays null so the SQL's
299
+ // `:names is null` arm means "no filter".
300
+ bound[name] = params.names == null ? null : JSON.stringify(params.names);
301
+ break;
302
+ case "retryable":
303
+ case "reset_attempt":
304
+ bound[name] = params[name] ? 1 : 0;
305
+ break;
306
+ default:
307
+ bound[name] = params[name];
149
308
  }
150
- const row = this.all("insert_task", ins)[0];
151
- this.run("upsert_key", { key, task_id: id, now_ms: now });
152
- return row;
153
- });
154
- return rowToTask(txn.immediate());
155
- }
156
- async get(taskId) {
157
- this.ensure();
158
- const rows = this.all("get", { id: taskId });
159
- return rows.length ? rowToTask(rows[0]) : null;
160
- }
161
- async getByKey(key) {
162
- this.ensure();
163
- const rows = this.all("get_by_key", { key });
164
- return rows.length ? rowToTask(rows[0]) : null;
165
- }
166
- async list(input = {}) {
167
- this.ensure();
168
- const rows = this.all("list", {
169
- status: input.status ?? null,
170
- queue: input.queue ?? null,
171
- name: input.name ?? null,
172
- root_id: input.rootId ?? null,
173
- correlation_id: input.correlationId ?? null,
174
- limit: input.limit ?? 100,
175
- offset: input.offset ?? 0,
176
- });
177
- return rows.map(rowToTask);
178
- }
179
- async cancel(taskId) {
180
- this.ensure();
181
- const rows = this.all("cancel", { id: taskId, now_ms: nowMs() });
182
- return rows.length ? rowToTask(rows[0]) : null;
183
- }
184
- async cancelByKey(key) {
185
- this.ensure();
186
- const txn = this.db.transaction(() => {
187
- const existing = this.all("get_key", { key });
188
- if (!existing.length)
189
- return null;
190
- const rows = this.all("cancel", { id: existing[0].task_id, now_ms: nowMs() });
191
- return rows.length ? rows[0] : null;
192
- });
193
- const row = txn.immediate();
194
- return row ? rowToTask(row) : null;
195
- }
196
- async retry(taskId, opts = {}) {
197
- this.ensure();
198
- const rows = this.all("retry", {
199
- id: taskId,
200
- now_ms: nowMs(),
201
- reset_attempt: opts.resetAttempt ? 1 : 0,
202
- });
203
- return rows.length ? rowToTask(rows[0]) : null;
204
- }
205
- async retryByKey(key, opts = {}) {
206
- this.ensure();
207
- const txn = this.db.transaction(() => {
208
- const existing = this.all("get_key", { key });
209
- if (!existing.length)
210
- return null;
211
- const rows = this.all("retry", {
212
- id: existing[0].task_id,
213
- now_ms: nowMs(),
214
- reset_attempt: opts.resetAttempt ? 1 : 0,
215
- });
216
- return rows.length ? rows[0] : null;
217
- });
218
- const row = txn.immediate();
219
- return row ? rowToTask(row) : null;
309
+ }
310
+ return bound;
220
311
  }
221
- // ------------------------------------------------------------- worker side
222
- async claim(input) {
223
- this.ensure();
224
- const now = nowMs();
225
- const queues = JSON.stringify(input.queues);
226
- const leaseMs = input.leaseMs ?? 30_000;
227
- const limit = input.limit ?? 1;
228
- // Read-only probe first: skip the write lock entirely when idle.
229
- const probe = this.all("claimable_probe", { queues, now_ms: now })[0];
230
- if (!probe || !probe.has_work)
312
+ runNow(name, params) {
313
+ const stmt = this.stmts[name];
314
+ const bound = this.bind(this.statements[name], params);
315
+ // Nearly every protocol statement ends in RETURNING; upsert_key does not, and
316
+ // better-sqlite3 refuses .all() on a statement that yields no rows.
317
+ if (!stmt.reader) {
318
+ stmt.run(bound);
231
319
  return [];
232
- const txn = this.db.transaction(() => {
233
- this.all("recover_leases", {
234
- now_ms: now,
235
- lease_expired_error: LEASE_EXPIRED_ERROR_JSON,
236
- });
237
- return this.all("claim", {
238
- queues,
239
- now_ms: now,
240
- worker_id: input.workerId,
241
- lease_until_ms: now + leaseMs,
242
- limit,
243
- });
244
- });
245
- return txn.immediate().map(rowToTask);
320
+ }
321
+ return stmt.all(bound);
246
322
  }
247
- async heartbeat(input) {
248
- this.ensure();
249
- const now = nowMs();
250
- return this.ownedWrite("heartbeat", input.taskId, {
251
- id: input.taskId,
252
- worker_id: input.workerId,
253
- now_ms: now,
254
- lease_until_ms: now + (input.leaseMs ?? 30_000),
255
- });
323
+ /** Queue an operation behind every other operation on this database. */
324
+ enqueue(fn) {
325
+ const previous = fileLocks.get(this.lockKey) ?? Promise.resolve();
326
+ const run = previous.then(fn, fn);
327
+ fileLocks.set(this.lockKey, run.then(() => undefined, () => undefined));
328
+ return run;
256
329
  }
257
- async progress(input) {
258
- this.ensure();
259
- return this.ownedWrite("progress", input.taskId, {
260
- id: input.taskId,
261
- worker_id: input.workerId,
262
- now_ms: nowMs(),
263
- progress: input.progress,
264
- message: input.message,
265
- });
330
+ /**
331
+ * Serialize an operation against this database, waiting out a lost write lock on
332
+ * a jittered backoff. Replaces busy_timeout's synchronous wait (see the class
333
+ * comment); on exhausting the budget the original SQLITE_BUSY surfaces, which is
334
+ * what a nonzero busy_timeout would have thrown too.
335
+ *
336
+ * Each attempt re-queues rather than backing off while holding its turn: the
337
+ * contention left to retry is cross-process, and under WAL a *reader* never sees
338
+ * SQLITE_BUSY at all — so sleeping in place would stall this process's reads
339
+ * (including the worker's own poll) on a lock they were never waiting for.
340
+ *
341
+ * Retrying is safe because an attempt is one statement, or one transaction that
342
+ * has already rolled back: nothing partially applied survives it. `fn` may
343
+ * therefore run more than once and must not carry effects of its own — the
344
+ * callers in TaskStore build their ids and payloads before opening one.
345
+ */
346
+ async withLock(fn) {
347
+ const deadline = Date.now() + this.busyBudgetMs;
348
+ let delay = BUSY_RETRY_BASE_MS;
349
+ for (;;) {
350
+ try {
351
+ return await this.enqueue(fn);
352
+ }
353
+ catch (err) {
354
+ if (!isBusy(err) || Date.now() >= deadline)
355
+ throw err;
356
+ // Jitter so several losers don't wake together and collide again.
357
+ await sleep(delay * (0.5 + Math.random()));
358
+ delay = Math.min(delay * 2, BUSY_RETRY_MAX_DELAY_MS);
359
+ }
360
+ }
266
361
  }
267
- async succeed(input) {
268
- this.ensure();
269
- return this.ownedWrite("succeed", input.taskId, {
270
- id: input.taskId,
271
- worker_id: input.workerId,
272
- now_ms: nowMs(),
273
- result: input.result == null ? null : JSON.stringify(input.result),
274
- message: null,
275
- });
362
+ /**
363
+ * Revisit this connection's planner statistics, at most once per
364
+ * STATS_REFRESH_INTERVAL_MS.
365
+ *
366
+ * A connection lives for days, and the statements were prepared against whatever
367
+ * the table looked like when it opened — a worker started against an empty
368
+ * database plans as if it were still empty however large the backlog grows. The
369
+ * prepared statements do pick the refreshed plans up: ANALYZE bumps the schema
370
+ * cookie, so SQLite silently re-prepares them on next use. That is what makes
371
+ * this worth doing rather than a restart-only concern.
372
+ *
373
+ * Queued rather than run under `withLock`: statistics are best-effort, so losing
374
+ * the write lock to another process should cost nothing — skip and let the next
375
+ * interval try, instead of spending an operation's whole retry budget on them.
376
+ */
377
+ async maybeRefreshStatistics(db) {
378
+ const now = Date.now();
379
+ if (now < this.nextStatsRefreshAt)
380
+ return;
381
+ // Claim the slot before running, not after: otherwise a burst of concurrent
382
+ // operations all see it due and queue an ANALYZE apiece.
383
+ this.nextStatsRefreshAt = now + STATS_REFRESH_INTERVAL_MS;
384
+ try {
385
+ await this.enqueue(() => refreshStatistics(db));
386
+ }
387
+ catch (err) {
388
+ if (!isBusy(err))
389
+ throw err;
390
+ }
276
391
  }
277
- async complete(input) {
278
- this.ensure();
279
- return this.ownedWrite("complete", input.taskId, {
280
- id: input.taskId,
281
- worker_id: input.workerId,
282
- now_ms: nowMs(),
283
- result: input.result == null ? null : JSON.stringify(input.result),
284
- });
392
+ async fetch(name, params) {
393
+ const db = this.ensure();
394
+ await this.maybeRefreshStatistics(db);
395
+ return this.withLock(() => this.runNow(name, params));
285
396
  }
286
- async fail(input) {
287
- this.ensure();
288
- return this.ownedWrite("fail", input.taskId, {
289
- id: input.taskId,
290
- worker_id: input.workerId,
291
- now_ms: nowMs(),
292
- error: JSON.stringify(input.error ?? {}),
293
- retryable: input.retryable === false ? 0 : 1,
294
- delay_ms: input.delayMs ?? 0,
397
+ async tx(fn) {
398
+ const db = this.ensure();
399
+ await this.maybeRefreshStatistics(db);
400
+ // BEGIN IMMEDIATE by hand rather than db.transaction(): the callback is async
401
+ // (the seam is shared with Postgres), and better-sqlite3's wrapper only takes
402
+ // a synchronous one. The lock above makes the manual version safe.
403
+ return this.withLock(async () => {
404
+ // With busy_timeout at 0 this is where a lost write lock surfaces, and it
405
+ // fails before the transaction exists — so the retry re-runs `fn` cleanly.
406
+ db.exec("BEGIN IMMEDIATE");
407
+ try {
408
+ const out = await fn(async (name, params) => this.runNow(name, params));
409
+ db.exec("COMMIT");
410
+ return out;
411
+ }
412
+ catch (err) {
413
+ // Nothing to roll back when BEGIN was what failed — the common case under
414
+ // contention — or when SQLite already did it (a constraint abort).
415
+ if (db.inTransaction) {
416
+ try {
417
+ db.exec("ROLLBACK");
418
+ }
419
+ catch {
420
+ // Raced with SQLite's own rollback; the transaction is gone either way.
421
+ }
422
+ }
423
+ throw err;
424
+ }
295
425
  });
296
426
  }
427
+ async hasClaimableWork(params) {
428
+ // Read-only probe first: an idle worker never takes SQLite's single write
429
+ // lock, so idle workers don't serialize against each other.
430
+ const rows = await this.fetch("claimable_probe", params);
431
+ return Boolean(rows[0]?.has_work);
432
+ }
297
433
  }