cairnq 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/README.md +28 -0
  2. package/dist/_protocol/migrations/postgres/0001_init.sql +3 -1
  3. package/dist/_protocol/migrations/postgres/0002_purge_index.sql +6 -0
  4. package/dist/_protocol/migrations/postgres/0003_notify.sql +38 -0
  5. package/dist/_protocol/migrations/postgres/0004_lease_index.sql +16 -0
  6. package/dist/_protocol/migrations/postgres/0005_clear_terminal_lease.sql +17 -0
  7. package/dist/_protocol/migrations/sqlite/0001_init.sql +3 -1
  8. package/dist/_protocol/migrations/sqlite/0002_purge_index.sql +6 -0
  9. package/dist/_protocol/migrations/sqlite/0004_lease_index.sql +22 -0
  10. package/dist/_protocol/migrations/sqlite/0005_clear_terminal_lease.sql +17 -0
  11. package/dist/_protocol/sql/postgres/claim.sql +18 -5
  12. package/dist/_protocol/sql/postgres/claim_one_queue.sql +35 -0
  13. package/dist/_protocol/sql/postgres/complete.sql +3 -0
  14. package/dist/_protocol/sql/postgres/fail.sql +30 -8
  15. package/dist/_protocol/sql/postgres/insert_task.sql +6 -3
  16. package/dist/_protocol/sql/postgres/list.sql +3 -1
  17. package/dist/_protocol/sql/postgres/lock_key.sql +9 -0
  18. package/dist/_protocol/sql/postgres/progress.sql +4 -3
  19. package/dist/_protocol/sql/postgres/protocol_version.sql +4 -0
  20. package/dist/_protocol/sql/postgres/purge.sql +25 -0
  21. package/dist/_protocol/sql/postgres/recover_leases.sql +49 -14
  22. package/dist/_protocol/sql/postgres/retry.sql +3 -0
  23. package/dist/_protocol/sql/postgres/stats.sql +8 -0
  24. package/dist/_protocol/sql/postgres/succeed.sql +4 -0
  25. package/dist/_protocol/sql/sqlite/claim.sql +13 -2
  26. package/dist/_protocol/sql/sqlite/claim_one_queue.sql +36 -0
  27. package/dist/_protocol/sql/sqlite/claimable_probe.sql +6 -2
  28. package/dist/_protocol/sql/sqlite/complete.sql +3 -0
  29. package/dist/_protocol/sql/sqlite/fail.sql +32 -8
  30. package/dist/_protocol/sql/sqlite/list.sql +3 -1
  31. package/dist/_protocol/sql/sqlite/lock_key.sql +5 -0
  32. package/dist/_protocol/sql/sqlite/progress.sql +6 -2
  33. package/dist/_protocol/sql/sqlite/protocol_version.sql +4 -0
  34. package/dist/_protocol/sql/sqlite/purge.sql +18 -0
  35. package/dist/_protocol/sql/sqlite/recover_leases.sql +25 -7
  36. package/dist/_protocol/sql/sqlite/retry.sql +3 -0
  37. package/dist/_protocol/sql/sqlite/stats.sql +8 -0
  38. package/dist/_protocol/sql/sqlite/succeed.sql +4 -0
  39. package/dist/client.d.ts +10 -2
  40. package/dist/client.js +12 -0
  41. package/dist/context.d.ts +17 -1
  42. package/dist/context.js +60 -6
  43. package/dist/errors.d.ts +18 -2
  44. package/dist/errors.js +49 -3
  45. package/dist/index.d.ts +3 -2
  46. package/dist/index.js +2 -1
  47. package/dist/sql.js +16 -9
  48. package/dist/store/base.d.ts +114 -9
  49. package/dist/store/base.js +376 -1
  50. package/dist/store/postgres.d.ts +62 -63
  51. package/dist/store/postgres.js +245 -222
  52. package/dist/store/sqlite.d.ts +83 -60
  53. package/dist/store/sqlite.js +370 -234
  54. package/dist/wait.d.ts +15 -2
  55. package/dist/wait.js +23 -5
  56. package/dist/worker.d.ts +53 -1
  57. package/dist/worker.js +202 -42
  58. package/package.json +9 -2
  59. package/src/client.ts +16 -2
  60. package/src/context.ts +70 -13
  61. package/src/errors.ts +59 -4
  62. package/src/index.ts +3 -1
  63. package/src/sql.ts +15 -8
  64. package/src/store/base.ts +443 -27
  65. package/src/store/postgres.ts +243 -267
  66. package/src/store/sqlite.ts +378 -263
  67. package/src/wait.ts +28 -5
  68. package/src/worker.ts +242 -42
@@ -1,47 +1,210 @@
1
1
  import { mkdirSync } from "node:fs";
2
- import { dirname } from "node:path";
2
+ import { dirname, resolve } from "node:path";
3
3
 
4
4
  import Database from "better-sqlite3";
5
5
 
6
- import { newId, nowMs } from "../ids.js";
7
- import { AlreadyExists, errorEnvelope, LostLease, ProtocolVersionMismatch } from "../errors.js";
8
- import { rowToTask, type Task } from "../models.js";
6
+ import { nowMs } from "../ids.js";
9
7
  import { loadMigrations, loadStatements } from "../sql.js";
10
- import type { ListInput, SubmitInput, TaskStore } from "./base.js";
8
+ import {
9
+ checkProtocolVersion,
10
+ type Fetch,
11
+ type Params,
12
+ statementParams,
13
+ TaskStore,
14
+ } from "./base.js";
11
15
 
12
16
  type DB = Database.Database;
13
17
  type Stmt = Database.Statement;
14
18
 
15
- const SUPPORTED_PROTOCOL_MAJOR = 1;
19
+ const WAL_RETRY_DELAY_MS = 50;
20
+ const WAL_RETRY_BUDGET_MS = 5_000;
16
21
 
17
- const LEASE_EXPIRED_ERROR = errorEnvelope({
18
- type: "LeaseExpired",
19
- code: "lease_expired",
20
- message: "task lease expired and max attempts reached",
21
- retryable: false,
22
- });
23
- // Serialized once: it's an immutable constant bound on every claim that finds work.
24
- const LEASE_EXPIRED_ERROR_JSON = JSON.stringify(LEASE_EXPIRED_ERROR);
22
+ const BUSY_RETRY_BASE_MS = 1;
23
+ const BUSY_RETRY_MAX_DELAY_MS = 50;
25
24
 
26
25
  /**
27
- * SQLiteStore — better-sqlite3 backend executing the shared cairnq-protocol SQL.
26
+ * How often a live connection revisits its planner statistics.
28
27
  *
29
- * The driver is synchronous, which suits SQLite's single writer: claim is one
30
- * short transaction, the handler runs outside any transaction, and
31
- * progress/heartbeat/succeed/fail are each their own short write. JS being
32
- * single-threaded means sync DB calls never interleave. Cross-process contention
33
- * (deployment mode B) is absorbed by busy_timeout.
28
+ * Bounds how long the planner can work from a stale table shape; a minute is
29
+ * arbitrary but small next to the days a worker holds its connection. It does not
30
+ * set how often an ANALYZE actually runs — SQLite decides that itself, and only
31
+ * when a table has outgrown its statistics by roughly 24x, so a shorter interval
32
+ * costs more no-ops (a few microseconds each) rather than more analyzing.
34
33
  */
35
- export class SQLiteStore implements TaskStore {
34
+ const STATS_REFRESH_INTERVAL_MS = 60_000;
35
+
36
+ /** Sleep without yielding — the whole open path is synchronous already. */
37
+ function sleepSync(ms: number): void {
38
+ Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
39
+ }
40
+
41
+ /** Sleep by yielding to the event loop — the point of the busy retry loop. */
42
+ function sleep(ms: number): Promise<void> {
43
+ return new Promise((resolve) => setTimeout(resolve, ms));
44
+ }
45
+
46
+ /**
47
+ * Whether this error is SQLite refusing to wait for the write lock.
48
+ *
49
+ * Prefix match: the code carries detail suffixes (`SQLITE_BUSY_SNAPSHOT`). Only
50
+ * SQLITE_BUSY qualifies — SQLITE_LOCKED is same-connection table contention,
51
+ * which the per-file lock prevents and a retry could not resolve anyway.
52
+ */
53
+ function isBusy(err: unknown): boolean {
54
+ if (!err || typeof err !== "object") return false;
55
+ const code = (err as { code?: unknown }).code;
56
+ return typeof code === "string" && code.startsWith("SQLITE_BUSY");
57
+ }
58
+
59
+ /** Whether this path names an in-memory database rather than a file. */
60
+ function isMemory(path: string): boolean {
61
+ return path === ":memory:" || path.includes("mode=memory");
62
+ }
63
+
64
+ /**
65
+ * Whether cairnq_tasks has been analyzed at all.
66
+ *
67
+ * Two steps because sqlite_stat1 does not exist until something runs ANALYZE, and
68
+ * querying a missing table is an error rather than an empty result.
69
+ */
70
+ function hasStatistics(db: DB): boolean {
71
+ const table = db
72
+ .prepare("select 1 from sqlite_master where type = 'table' and name = 'sqlite_stat1'")
73
+ .get();
74
+ if (!table) return false;
75
+ return Boolean(
76
+ db.prepare("select 1 from sqlite_stat1 where tbl = 'cairnq_tasks'").get(),
77
+ );
78
+ }
79
+
80
+ /**
81
+ * Bring cairnq_tasks' statistics up to date, cheaply enough to call on a timer.
82
+ *
83
+ * Without them the planner misreads `status = 'running'` as a large fraction of the
84
+ * table and passes over the partial cairnq_tasks_lease_idx that lease recovery is
85
+ * indexed for.
86
+ *
87
+ * The explicit bootstrap is not redundant with `PRAGMA optimize`. Before SQLite
88
+ * 3.46 the pragma skips a table that has no sqlite_stat1 entry entirely — no mask
89
+ * changes that, verified on 3.45.1 — so on those builds it can never produce the
90
+ * *first* statistics, and the index stays unused for the life of the database.
91
+ * Distro Pythons link exactly those builds (Ubuntu 24.04 ships 3.45.1), while
92
+ * better-sqlite3 bundles its own newer one, so this is also what keeps the two SDKs
93
+ * behaving alike rather than by luck of packaging.
94
+ *
95
+ * Once an entry exists, every version's pragma applies its own growth heuristic,
96
+ * which is the part worth deferring to: it is a few microseconds when there is
97
+ * nothing to do, where a bare ANALYZE would rescan the table every time.
98
+ */
99
+ function refreshStatistics(db: DB): void {
100
+ if (hasStatistics(db)) db.pragma("optimize");
101
+ // Scoped to the one table whose shape the planner gets wrong; the key and meta
102
+ // tables are read by primary key, where statistics change nothing. A database
103
+ // this one shares with the caller's own tables is left alone.
104
+ else db.exec("ANALYZE cairnq_tasks");
105
+ }
106
+
107
+ /**
108
+ * Serializes every SQLiteStore on one database file, process-wide.
109
+ *
110
+ * better-sqlite3 is synchronous, and a transaction holds SQLite's write lock
111
+ * across `await`s (the callback seam is shared with Postgres, so it is async). Two
112
+ * connections in this process would then contend for that lock the expensive way:
113
+ * every loser spends SQLITE_BUSY retries and backoff on a holder it could simply
114
+ * have queued behind.
115
+ *
116
+ * Keyed by database, not by store: what the lock protects is the file. Across
117
+ * processes there is nothing to serialize from here — each holder has its own
118
+ * thread, and `withLock`'s retry absorbs that contention. An in-memory database is
119
+ * private to one connection and gets a key of its own.
120
+ */
121
+ const fileLocks = new Map<string, Promise<unknown>>();
122
+ let memoryDbSeq = 0;
123
+
124
+ /**
125
+ * Put the database in WAL mode, waiting out a concurrent cold start.
126
+ *
127
+ * journal_mode is a persistent property of the file, so only the first connection
128
+ * to a new database actually switches it — and that switch needs an exclusive
129
+ * lock. `busy_timeout` does not cover it: SQLite returns SQLITE_BUSY for a
130
+ * journal_mode change rather than invoking the busy handler, so several processes
131
+ * opening the same new database at once would otherwise get an instant "database
132
+ * is locked". Retry briefly instead; the window is only as long as one other
133
+ * opener's switch.
134
+ *
135
+ * Callers must skip in-memory databases: those report journal_mode = "memory" and
136
+ * can never be WAL, so waiting for one is waiting for something that will not
137
+ * happen.
138
+ */
139
+ function enableWal(db: DB): void {
140
+ const deadline = Date.now() + WAL_RETRY_BUDGET_MS;
141
+ for (;;) {
142
+ try {
143
+ const rows = db.pragma("journal_mode = WAL") as { journal_mode?: string }[];
144
+ if (rows[0]?.journal_mode?.toLowerCase() === "wal") return;
145
+ } catch (err) {
146
+ if (!isBusy(err)) throw err;
147
+ }
148
+ if (Date.now() >= deadline) {
149
+ throw new Error(
150
+ "could not switch the database to WAL mode: it stayed locked by another connection",
151
+ );
152
+ }
153
+ sleepSync(WAL_RETRY_DELAY_MS);
154
+ }
155
+ }
156
+
157
+ /**
158
+ * SQLiteStore — the SQLite dialect of the shared cairnq-protocol SQL.
159
+ *
160
+ * Everything protocol-shaped lives in TaskStore; this file is only what SQLite
161
+ * does differently: better-sqlite3's synchronous driver, BEGIN IMMEDIATE
162
+ * transactions, a read-only probe in front of the write lock, and time supplied
163
+ * by the SDK (`:now_ms`) rather than by the database.
164
+ *
165
+ * The driver being synchronous suits SQLite's single writer: claim is one short
166
+ * transaction, the handler runs outside any transaction, and
167
+ * progress/heartbeat/succeed/fail are each their own short write.
168
+ *
169
+ * Cross-process contention is absorbed by retrying in JavaScript, not by
170
+ * busy_timeout. The two cost the same wait but not the same blocking: a nonzero
171
+ * busy_timeout waits *inside* the synchronous driver, so a caller that loses the
172
+ * write lock stalls this process's event loop for up to the whole timeout — the
173
+ * P99 of an HTTP server that submits tasks. Executing a statement takes
174
+ * microseconds; waiting for a lock takes milliseconds to seconds, and only the
175
+ * second part needs to happen off the thread. So busy_timeout goes to 0 (fail
176
+ * immediately) and the wait becomes an awaited backoff, which the event loop runs
177
+ * through. The budget is the same either way — `busyTimeoutMs`.
178
+ *
179
+ * The open path keeps a real busy_timeout: it is synchronous by nature (WAL
180
+ * switch, migrations) and happens once, under the caller's `connect()`.
181
+ */
182
+ export class SQLiteStore extends TaskStore {
36
183
  private db: DB | null = null;
37
184
  private stmts: Record<string, Stmt> = {};
38
185
  private readonly statements: Record<string, string>;
186
+ /** This store's entry in `fileLocks` — see there for why it is per-database. */
187
+ private readonly lockKey: string;
188
+ /** How long a single operation may keep retrying a lost write lock. */
189
+ private readonly busyBudgetMs: number;
190
+ /** When this connection may next revisit its planner statistics. */
191
+ private nextStatsRefreshAt = 0;
39
192
 
40
193
  constructor(
41
194
  private readonly path: string,
42
- private readonly opts: { busyTimeoutMs?: number } = {},
195
+ opts: { busyTimeoutMs?: number } = {},
43
196
  ) {
197
+ super();
198
+ this.busyBudgetMs = opts.busyTimeoutMs ?? 5000;
44
199
  this.statements = loadStatements("sqlite");
200
+ // Only a bare ":memory:" is guaranteed private to its connection, so only
201
+ // it gets a lock of its own. A "mode=memory" URI stays path-keyed: with
202
+ // cache=shared it names ONE shared database, and on a build without URI
203
+ // filenames it is a literal file — in both cases two stores on that string
204
+ // must share a lock. Over-serializing a private URI-memory database is
205
+ // harmless; skipping the lock on a shared one is the deadlock this map
206
+ // exists to prevent.
207
+ this.lockKey = path === ":memory:" ? `memory#${memoryDbSeq++}` : resolve(path);
45
208
  }
46
209
 
47
210
  async connect(): Promise<void> {
@@ -58,17 +221,35 @@ export class SQLiteStore implements TaskStore {
58
221
 
59
222
  private ensure(): DB {
60
223
  if (this.db) return this.db;
61
- if (this.path !== ":memory:") mkdirSync(dirname(this.path), { recursive: true });
224
+ const memory = isMemory(this.path);
225
+ if (!memory) mkdirSync(dirname(this.path), { recursive: true });
62
226
  const db = new Database(this.path);
63
- db.pragma("journal_mode = WAL");
227
+ // Only the synchronous part of the open path gets a real busy_timeout: the WAL
228
+ // switch and the migrations cannot await a retry. See the class comment.
229
+ db.pragma(`busy_timeout = ${this.busyBudgetMs}`);
230
+ // WAL exists so several processes can share one file. An in-memory database
231
+ // is private to this connection, so there is nothing to share or wait for.
232
+ if (!memory) enableWal(db);
64
233
  db.pragma("foreign_keys = ON");
65
- db.pragma(`busy_timeout = ${this.opts.busyTimeoutMs ?? 5000}`);
66
234
  this.applyMigrations(db);
235
+ // Everything past here either awaits its retry or is optional, so stop blocking.
236
+ db.pragma("busy_timeout = 0");
237
+ // Give the query planner statistics (see refreshStatistics), repeated on a timer
238
+ // from here on (see maybeRefreshStatistics).
239
+ try {
240
+ refreshStatistics(db);
241
+ } catch (err) {
242
+ // Statistics are an optimization, never correctness, so losing them to a
243
+ // concurrent writer must not fail the open — the next one gets another
244
+ // chance. Anything else is a real fault and belongs to the caller.
245
+ if (!isBusy(err)) throw err;
246
+ }
247
+ this.nextStatsRefreshAt = Date.now() + STATS_REFRESH_INTERVAL_MS;
67
248
  for (const [name, sql] of Object.entries(this.statements)) {
68
249
  this.stmts[name] = db.prepare(sql);
69
250
  }
70
251
  this.db = db;
71
- this.checkVersion();
252
+ checkProtocolVersion(this.readProtocolVersion());
72
253
  return db;
73
254
  }
74
255
 
@@ -77,275 +258,209 @@ export class SQLiteStore implements TaskStore {
77
258
  "create table if not exists cairnq_migrations " +
78
259
  "(name text primary key, applied_at_ms integer not null)",
79
260
  );
80
- const applied = new Set(
81
- (db.prepare("select name from cairnq_migrations").all() as { name: string }[]).map(
82
- (r) => r.name,
83
- ),
84
- );
85
- // `or ignore`: another process may apply the same migration concurrently on a
86
- // fresh shared db (mode B cold start). Migrations are idempotent.
261
+ const isApplied = db.prepare("select 1 from cairnq_migrations where name = ?");
87
262
  const insert = db.prepare(
88
- "insert or ignore into cairnq_migrations (name, applied_at_ms) values (?, ?)",
263
+ "insert into cairnq_migrations (name, applied_at_ms) values (?, ?)",
89
264
  );
90
265
  for (const { name, sql } of loadMigrations("sqlite")) {
91
- if (applied.has(name)) continue;
266
+ // Check and apply under one write lock. Two processes cold-starting on a
267
+ // shared database would otherwise both see a migration as unapplied and
268
+ // both run it — harmless for the idempotent ones, not for a future ALTER.
269
+ // `immediate` takes the write lock up front; the loser sees it applied.
92
270
  db.transaction(() => {
271
+ if (isApplied.get(name)) return;
93
272
  db.exec(sql);
94
273
  insert.run(name, nowMs());
95
- })();
96
- }
97
- }
98
-
99
- private checkVersion(): void {
100
- const version = this.readProtocolVersion();
101
- if (version !== SUPPORTED_PROTOCOL_MAJOR) {
102
- throw new ProtocolVersionMismatch(
103
- `storage protocol_version=${version}, SDK supports ${SUPPORTED_PROTOCOL_MAJOR}`,
104
- );
274
+ }).immediate();
105
275
  }
106
276
  }
107
277
 
108
278
  private readProtocolVersion(): number {
109
- const row = this.db!
110
- .prepare("select value from cairnq_meta where key = 'protocol_version'")
111
- .get() as { value: string } | undefined;
112
- return row ? Number(row.value) : 0;
279
+ const rows = this.runNow("protocol_version", {});
280
+ return rows.length ? Number(rows[0].value) : 0;
113
281
  }
114
282
 
115
283
  async protocolVersion(): Promise<number> {
116
284
  this.ensure();
117
- return this.readProtocolVersion();
118
- }
119
-
120
- private all(name: string, params: Record<string, unknown>): any[] {
121
- return this.stmts[name].all(params) as any[];
285
+ // Under the store lock: this public read must not slip a statement into
286
+ // another operation's open transaction on the shared connection.
287
+ return this.withLock(() => this.readProtocolVersion());
122
288
  }
123
289
 
124
- private run(name: string, params: Record<string, unknown>): void {
125
- this.stmts[name].run(params);
126
- }
127
-
128
- // An ownership-checked worker write (heartbeat/progress/succeed/complete/fail).
129
- // Each statement's WHERE pins worker_id + a live lease, so 0 rows back means the
130
- // lease was lost — every such write reports it the same way.
131
- private ownedWrite(name: string, taskId: string, params: Record<string, unknown>): Task {
132
- const rows = this.all(name, params);
133
- if (!rows.length) throw new LostLease(taskId);
134
- return rowToTask(rows[0]);
135
- }
136
-
137
- // ------------------------------------------------------------- client side
138
- async submit(input: SubmitInput): Promise<Task> {
139
- this.ensure();
290
+ // ------------------------------------------------------------ dialect seam
291
+ /**
292
+ * Adapt the dialect-neutral parameters to what this statement binds.
293
+ *
294
+ * SQLite statements carry no DB clock, so every absolute `*_ms` is derived here
295
+ * from one `now`, and booleans cross as 0/1. The result is narrowed to the
296
+ * names the SQL actually uses, which is what makes it safe for a caller to pass
297
+ * one superset of parameters for both dialects.
298
+ *
299
+ * Each derivation writes a name Postgres does not use (`lease_until_ms` from
300
+ * `lease_ms`, and so on), so a statement binds one or the other, never both —
301
+ * which is why the derived values can be computed unconditionally and left for
302
+ * the narrowing step to discard.
303
+ */
304
+ private bind(sql: string, params: Params): Params {
140
305
  const now = nowMs();
141
- const id = newId("task");
142
- const ins = {
143
- id,
144
- name: input.name,
145
- queue: input.queue ?? "default",
146
- payload: JSON.stringify(input.payload ?? {}),
147
- metadata: JSON.stringify(input.metadata ?? {}),
148
- max_attempts: input.maxAttempts ?? 3,
149
- priority: input.priority ?? 0,
150
- run_at_ms: now + (input.runAtDelayMs ?? 0),
151
- parent_id: input.parentId ?? null,
152
- root_id: input.rootId ?? id,
153
- correlation_id: input.correlationId ?? null,
154
- now_ms: now,
155
- };
156
- const key = input.key ?? null;
157
- const conflict = input.conflict ?? "reuse";
158
-
159
- const txn = this.db!.transaction(() => {
160
- if (key === null) return this.all("insert_task", ins)[0];
161
- const existing = this.all("get_key", { key }) as { task_id: string }[];
162
- if (existing.length) {
163
- const exId = existing[0].task_id;
164
- if (conflict === "reuse") return this.all("get", { id: exId })[0];
165
- if (conflict === "reject") throw new AlreadyExists(key);
166
- if (conflict === "replace") {
167
- this.all("cancel", { id: exId, now_ms: now });
168
- const row = this.all("insert_task", ins)[0];
169
- this.run("upsert_key", { key, task_id: id, now_ms: now });
170
- return row;
171
- }
172
- throw new Error(`unknown conflict strategy: ${conflict}`);
306
+ const bound: Params = {};
307
+ for (const name of statementParams(sql)) {
308
+ switch (name) {
309
+ case "now_ms":
310
+ bound[name] = now;
311
+ break;
312
+ case "lease_until_ms":
313
+ bound[name] = now + (params.lease_ms as number);
314
+ break;
315
+ case "run_at_ms":
316
+ bound[name] = now + (params.delay_ms as number);
317
+ break;
318
+ case "before_ms":
319
+ bound[name] = now - (params.older_than_ms as number);
320
+ break;
321
+ case "queues":
322
+ bound[name] = JSON.stringify(params.queues);
323
+ break;
324
+ case "names":
325
+ // json_each needs a JSON array; null stays null so the SQL's
326
+ // `:names is null` arm means "no filter".
327
+ bound[name] = params.names == null ? null : JSON.stringify(params.names);
328
+ break;
329
+ case "retryable":
330
+ case "reset_attempt":
331
+ bound[name] = params[name] ? 1 : 0;
332
+ break;
333
+ default:
334
+ bound[name] = params[name];
173
335
  }
174
- const row = this.all("insert_task", ins)[0];
175
- this.run("upsert_key", { key, task_id: id, now_ms: now });
176
- return row;
177
- });
178
- return rowToTask(txn.immediate());
179
- }
180
-
181
- async get(taskId: string): Promise<Task | null> {
182
- this.ensure();
183
- const rows = this.all("get", { id: taskId });
184
- return rows.length ? rowToTask(rows[0]) : null;
185
- }
186
-
187
- async getByKey(key: string): Promise<Task | null> {
188
- this.ensure();
189
- const rows = this.all("get_by_key", { key });
190
- return rows.length ? rowToTask(rows[0]) : null;
191
- }
192
-
193
- async list(input: ListInput = {}): Promise<Task[]> {
194
- this.ensure();
195
- const rows = this.all("list", {
196
- status: input.status ?? null,
197
- queue: input.queue ?? null,
198
- name: input.name ?? null,
199
- root_id: input.rootId ?? null,
200
- correlation_id: input.correlationId ?? null,
201
- limit: input.limit ?? 100,
202
- offset: input.offset ?? 0,
203
- });
204
- return rows.map(rowToTask);
205
- }
206
-
207
- async cancel(taskId: string): Promise<Task | null> {
208
- this.ensure();
209
- const rows = this.all("cancel", { id: taskId, now_ms: nowMs() });
210
- return rows.length ? rowToTask(rows[0]) : null;
211
- }
212
-
213
- async cancelByKey(key: string): Promise<Task | null> {
214
- this.ensure();
215
- const txn = this.db!.transaction(() => {
216
- const existing = this.all("get_key", { key }) as { task_id: string }[];
217
- if (!existing.length) return null;
218
- const rows = this.all("cancel", { id: existing[0].task_id, now_ms: nowMs() });
219
- return rows.length ? rows[0] : null;
220
- });
221
- const row = txn.immediate();
222
- return row ? rowToTask(row) : null;
223
- }
224
-
225
- async retry(taskId: string, opts: { resetAttempt?: boolean } = {}): Promise<Task | null> {
226
- this.ensure();
227
- const rows = this.all("retry", {
228
- id: taskId,
229
- now_ms: nowMs(),
230
- reset_attempt: opts.resetAttempt ? 1 : 0,
231
- });
232
- return rows.length ? rowToTask(rows[0]) : null;
336
+ }
337
+ return bound;
233
338
  }
234
339
 
235
- async retryByKey(key: string, opts: { resetAttempt?: boolean } = {}): Promise<Task | null> {
236
- this.ensure();
237
- const txn = this.db!.transaction(() => {
238
- const existing = this.all("get_key", { key }) as { task_id: string }[];
239
- if (!existing.length) return null;
240
- const rows = this.all("retry", {
241
- id: existing[0].task_id,
242
- now_ms: nowMs(),
243
- reset_attempt: opts.resetAttempt ? 1 : 0,
244
- });
245
- return rows.length ? rows[0] : null;
246
- });
247
- const row = txn.immediate();
248
- return row ? rowToTask(row) : null;
340
+ private runNow(name: string, params: Params): any[] {
341
+ const stmt = this.stmts[name];
342
+ const bound = this.bind(this.statements[name], params);
343
+ // Nearly every protocol statement ends in RETURNING; upsert_key does not, and
344
+ // better-sqlite3 refuses .all() on a statement that yields no rows.
345
+ if (!stmt.reader) {
346
+ stmt.run(bound);
347
+ return [];
348
+ }
349
+ return stmt.all(bound) as any[];
249
350
  }
250
351
 
251
- // ------------------------------------------------------------- worker side
252
- async claim(input: {
253
- queues: string[];
254
- workerId: string;
255
- leaseMs?: number;
256
- limit?: number;
257
- }): Promise<Task[]> {
258
- this.ensure();
259
- const now = nowMs();
260
- const queues = JSON.stringify(input.queues);
261
- const leaseMs = input.leaseMs ?? 30_000;
262
- const limit = input.limit ?? 1;
263
- // Read-only probe first: skip the write lock entirely when idle.
264
- const probe = this.all("claimable_probe", { queues, now_ms: now })[0] as { has_work: number };
265
- if (!probe || !probe.has_work) return [];
266
- const txn = this.db!.transaction(() => {
267
- this.all("recover_leases", {
268
- now_ms: now,
269
- lease_expired_error: LEASE_EXPIRED_ERROR_JSON,
270
- });
271
- return this.all("claim", {
272
- queues,
273
- now_ms: now,
274
- worker_id: input.workerId,
275
- lease_until_ms: now + leaseMs,
276
- limit,
277
- });
278
- });
279
- return (txn.immediate() as any[]).map(rowToTask);
352
+ /** Queue an operation behind every other operation on this database. */
353
+ private enqueue<T>(fn: () => T | Promise<T>): Promise<T> {
354
+ const previous = fileLocks.get(this.lockKey) ?? Promise.resolve();
355
+ const run = previous.then(fn, fn) as Promise<T>;
356
+ fileLocks.set(
357
+ this.lockKey,
358
+ run.then(
359
+ () => undefined,
360
+ () => undefined,
361
+ ),
362
+ );
363
+ return run;
280
364
  }
281
365
 
282
- async heartbeat(input: {
283
- taskId: string;
284
- workerId: string;
285
- leaseMs?: number;
286
- }): Promise<Task> {
287
- this.ensure();
288
- const now = nowMs();
289
- return this.ownedWrite("heartbeat", input.taskId, {
290
- id: input.taskId,
291
- worker_id: input.workerId,
292
- now_ms: now,
293
- lease_until_ms: now + (input.leaseMs ?? 30_000),
294
- });
366
+ /**
367
+ * Serialize an operation against this database, waiting out a lost write lock on
368
+ * a jittered backoff. Replaces busy_timeout's synchronous wait (see the class
369
+ * comment); on exhausting the budget the original SQLITE_BUSY surfaces, which is
370
+ * what a nonzero busy_timeout would have thrown too.
371
+ *
372
+ * Each attempt re-queues rather than backing off while holding its turn: the
373
+ * contention left to retry is cross-process, and under WAL a *reader* never sees
374
+ * SQLITE_BUSY at all — so sleeping in place would stall this process's reads
375
+ * (including the worker's own poll) on a lock they were never waiting for.
376
+ *
377
+ * Retrying is safe because an attempt is one statement, or one transaction that
378
+ * has already rolled back: nothing partially applied survives it. `fn` may
379
+ * therefore run more than once and must not carry effects of its own — the
380
+ * callers in TaskStore build their ids and payloads before opening one.
381
+ */
382
+ private async withLock<T>(fn: () => T | Promise<T>): Promise<T> {
383
+ const deadline = Date.now() + this.busyBudgetMs;
384
+ let delay = BUSY_RETRY_BASE_MS;
385
+ for (;;) {
386
+ try {
387
+ return await this.enqueue(fn);
388
+ } catch (err) {
389
+ if (!isBusy(err) || Date.now() >= deadline) throw err;
390
+ // Jitter so several losers don't wake together and collide again.
391
+ await sleep(delay * (0.5 + Math.random()));
392
+ delay = Math.min(delay * 2, BUSY_RETRY_MAX_DELAY_MS);
393
+ }
394
+ }
295
395
  }
296
396
 
297
- async progress(input: {
298
- taskId: string;
299
- workerId: string;
300
- progress: number | null;
301
- message: string | null;
302
- }): Promise<Task> {
303
- this.ensure();
304
- return this.ownedWrite("progress", input.taskId, {
305
- id: input.taskId,
306
- worker_id: input.workerId,
307
- now_ms: nowMs(),
308
- progress: input.progress,
309
- message: input.message,
310
- });
397
+ /**
398
+ * Revisit this connection's planner statistics, at most once per
399
+ * STATS_REFRESH_INTERVAL_MS.
400
+ *
401
+ * A connection lives for days, and the statements were prepared against whatever
402
+ * the table looked like when it opened — a worker started against an empty
403
+ * database plans as if it were still empty however large the backlog grows. The
404
+ * prepared statements do pick the refreshed plans up: ANALYZE bumps the schema
405
+ * cookie, so SQLite silently re-prepares them on next use. That is what makes
406
+ * this worth doing rather than a restart-only concern.
407
+ *
408
+ * Queued rather than run under `withLock`: statistics are best-effort, so losing
409
+ * the write lock to another process should cost nothing — skip and let the next
410
+ * interval try, instead of spending an operation's whole retry budget on them.
411
+ */
412
+ private async maybeRefreshStatistics(db: DB): Promise<void> {
413
+ const now = Date.now();
414
+ if (now < this.nextStatsRefreshAt) return;
415
+ // Claim the slot before running, not after: otherwise a burst of concurrent
416
+ // operations all see it due and queue an ANALYZE apiece.
417
+ this.nextStatsRefreshAt = now + STATS_REFRESH_INTERVAL_MS;
418
+ try {
419
+ await this.enqueue(() => refreshStatistics(db));
420
+ } catch (err) {
421
+ if (!isBusy(err)) throw err;
422
+ }
311
423
  }
312
424
 
313
- async succeed(input: { taskId: string; workerId: string; result: unknown }): Promise<Task> {
314
- this.ensure();
315
- return this.ownedWrite("succeed", input.taskId, {
316
- id: input.taskId,
317
- worker_id: input.workerId,
318
- now_ms: nowMs(),
319
- result: input.result == null ? null : JSON.stringify(input.result),
320
- message: null,
321
- });
425
+ protected async fetch(name: string, params: Params): Promise<any[]> {
426
+ const db = this.ensure();
427
+ await this.maybeRefreshStatistics(db);
428
+ return this.withLock(() => this.runNow(name, params));
322
429
  }
323
430
 
324
- async complete(input: { taskId: string; workerId: string; result: unknown }): Promise<Task> {
325
- this.ensure();
326
- return this.ownedWrite("complete", input.taskId, {
327
- id: input.taskId,
328
- worker_id: input.workerId,
329
- now_ms: nowMs(),
330
- result: input.result == null ? null : JSON.stringify(input.result),
431
+ protected async tx<T>(fn: (fetch: Fetch) => Promise<T>): Promise<T> {
432
+ const db = this.ensure();
433
+ await this.maybeRefreshStatistics(db);
434
+ // BEGIN IMMEDIATE by hand rather than db.transaction(): the callback is async
435
+ // (the seam is shared with Postgres), and better-sqlite3's wrapper only takes
436
+ // a synchronous one. The lock above makes the manual version safe.
437
+ return this.withLock(async () => {
438
+ // With busy_timeout at 0 this is where a lost write lock surfaces, and it
439
+ // fails before the transaction exists — so the retry re-runs `fn` cleanly.
440
+ db.exec("BEGIN IMMEDIATE");
441
+ try {
442
+ const out = await fn(async (name, params) => this.runNow(name, params));
443
+ db.exec("COMMIT");
444
+ return out;
445
+ } catch (err) {
446
+ // Nothing to roll back when BEGIN was what failed — the common case under
447
+ // contention — or when SQLite already did it (a constraint abort).
448
+ if (db.inTransaction) {
449
+ try {
450
+ db.exec("ROLLBACK");
451
+ } catch {
452
+ // Raced with SQLite's own rollback; the transaction is gone either way.
453
+ }
454
+ }
455
+ throw err;
456
+ }
331
457
  });
332
458
  }
333
459
 
334
- async fail(input: {
335
- taskId: string;
336
- workerId: string;
337
- error: unknown;
338
- retryable?: boolean;
339
- delayMs?: number;
340
- }): Promise<Task> {
341
- this.ensure();
342
- return this.ownedWrite("fail", input.taskId, {
343
- id: input.taskId,
344
- worker_id: input.workerId,
345
- now_ms: nowMs(),
346
- error: JSON.stringify(input.error ?? {}),
347
- retryable: input.retryable === false ? 0 : 1,
348
- delay_ms: input.delayMs ?? 0,
349
- });
460
+ protected async hasClaimableWork(params: Params): Promise<boolean> {
461
+ // Read-only probe first: an idle worker never takes SQLite's single write
462
+ // lock, so idle workers don't serialize against each other.
463
+ const rows = await this.fetch("claimable_probe", params);
464
+ return Boolean(rows[0]?.has_work);
350
465
  }
351
466
  }