cairnq 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +3 -2
  2. package/dist/_protocol/migrations/postgres/0006_claim_name_index.sql +25 -0
  3. package/dist/_protocol/migrations/sqlite/0006_claim_name_index.sql +26 -0
  4. package/dist/_protocol/sql/postgres/claim_one_name.sql +37 -0
  5. package/dist/_protocol/sql/postgres/claim_one_queue_one_name.sql +34 -0
  6. package/dist/_protocol/sql/postgres/heartbeat_batch.sql +24 -0
  7. package/dist/_protocol/sql/postgres/queue_depth.sql +22 -0
  8. package/dist/_protocol/sql/sqlite/claim_one_name.sql +36 -0
  9. package/dist/_protocol/sql/sqlite/claim_one_queue_one_name.sql +31 -0
  10. package/dist/_protocol/sql/sqlite/heartbeat_batch.sql +29 -0
  11. package/dist/_protocol/sql/sqlite/queue_depth.sql +26 -0
  12. package/dist/backoff.d.ts +31 -0
  13. package/dist/backoff.js +40 -0
  14. package/dist/backpressure.d.ts +59 -0
  15. package/dist/backpressure.js +122 -0
  16. package/dist/client.d.ts +16 -3
  17. package/dist/client.js +19 -5
  18. package/dist/context.d.ts +48 -2
  19. package/dist/context.js +101 -10
  20. package/dist/errors.d.ts +37 -0
  21. package/dist/errors.js +60 -0
  22. package/dist/index.d.ts +7 -3
  23. package/dist/index.js +2 -1
  24. package/dist/store/base.d.ts +73 -0
  25. package/dist/store/base.js +124 -14
  26. package/dist/store/sqlite.d.ts +35 -0
  27. package/dist/store/sqlite.js +163 -3
  28. package/dist/worker.d.ts +238 -13
  29. package/dist/worker.js +512 -120
  30. package/package.json +2 -1
  31. package/src/backoff.ts +53 -0
  32. package/src/backpressure.ts +140 -0
  33. package/src/client.ts +33 -5
  34. package/src/context.ts +116 -9
  35. package/src/errors.ts +66 -0
  36. package/src/index.ts +7 -2
  37. package/src/store/base.ts +136 -13
  38. package/src/store/sqlite.ts +168 -2
  39. package/src/worker.ts +671 -132
package/src/store/base.ts CHANGED
@@ -7,6 +7,7 @@ import {
7
7
  SerializationError,
8
8
  } from "../errors.js";
9
9
  import { rowToTask, STATUSES, type Task, type TaskStatus } from "../models.js";
10
+ import { type BackpressureOptions, QueueDepthGate } from "../backpressure.js";
10
11
 
11
12
  const rejectMangled = function (this: unknown, _key: string, v: unknown): unknown {
12
13
  if (typeof v === "number" && !Number.isFinite(v)) {
@@ -64,6 +65,10 @@ export function checkProtocolVersion(version: number): void {
64
65
  const CONFLICTS = ["reuse", "reject", "replace"] as const;
65
66
  export type Conflict = (typeof CONFLICTS)[number];
66
67
 
68
+ /** The queue a submit lands on when it names none. Owned here, where the
69
+ * default is applied, so nothing above has to re-derive it. */
70
+ export const DEFAULT_QUEUE = "default";
71
+
67
72
  export interface SubmitInput {
68
73
  name: string;
69
74
  payload: unknown;
@@ -149,6 +154,9 @@ export function statementParams(sql: string): readonly string[] {
149
154
  * behavior; the shared SQL already stops them from drifting in wording.
150
155
  */
151
156
  export abstract class TaskStore {
157
+ /** Set by useBackpressure; null means submit is ungated. */
158
+ private gate: QueueDepthGate | null = null;
159
+
152
160
  // ------------------------------------------------------------ dialect seam
153
161
  abstract connect(): Promise<void>;
154
162
  abstract close(): Promise<void>;
@@ -211,12 +219,24 @@ export abstract class TaskStore {
211
219
  }
212
220
 
213
221
  // ------------------------------------------------------------- client side
222
+ /**
223
+ * Bound how deep a queue may get before `submit` blocks. Off unless set.
224
+ *
225
+ * It hangs here rather than on `CairnQ` because the store is the one choke
226
+ * point every submit passes through — a handler spawning children via
227
+ * `TaskContext.submit` is the shape most likely to outrun its workers, and
228
+ * gating only the client would leave exactly that path unbounded.
229
+ */
230
+ useBackpressure(opts: BackpressureOptions): void {
231
+ this.gate = new QueueDepthGate(this, opts);
232
+ }
233
+
214
234
  async submit(input: SubmitInput): Promise<Task> {
215
235
  const id = newId("task");
216
236
  const ins: Params = {
217
237
  id,
218
238
  name: input.name,
219
- queue: input.queue ?? "default",
239
+ queue: input.queue ?? DEFAULT_QUEUE,
220
240
  payload: dumpJson(input.payload ?? {}),
221
241
  metadata: dumpJson(input.metadata ?? {}),
222
242
  max_attempts: input.maxAttempts ?? 3,
@@ -243,6 +263,10 @@ export abstract class TaskStore {
243
263
  if (input.runAtDelayMs != null && input.runAtDelayMs < 0) {
244
264
  throw new Error(`runAtDelayMs must be >= 0, got ${input.runAtDelayMs}`);
245
265
  }
266
+ // After validation and before the first write: bad arguments should fail
267
+ // now, not after waiting out a full queue. Reads the resolved queue, so the
268
+ // gate cannot throttle one queue while the row lands on another.
269
+ if (this.gate) await this.gate.acquire(ins.queue as string);
246
270
  if (key === null) return rowToTask((await this.fetch("insert_task", ins))[0]);
247
271
 
248
272
  // A key makes submit a read-then-write, so it has to be one transaction —
@@ -370,6 +394,22 @@ export abstract class TaskStore {
370
394
  return out;
371
395
  }
372
396
 
397
+ /**
398
+ * How many more tasks fit on `queue` under `maxDepth` — 0 once it is full.
399
+ *
400
+ * The cheap half of backpressure: bounded at `maxDepth` index entries, unlike
401
+ * `stats()`, which aggregates the whole table (terminal rows included) and so
402
+ * costs more the longer a database has been running. Use it directly to shed
403
+ * load or shape a producer; `QueueDepthGate` builds the blocking form on top.
404
+ */
405
+ async queueDepth(queue: string, maxDepth: number): Promise<number> {
406
+ if (!Number.isInteger(maxDepth) || maxDepth < 0) {
407
+ throw new Error(`maxDepth must be a non-negative integer, got ${maxDepth}`);
408
+ }
409
+ const rows = await this.fetch("queue_depth", { queue, max_depth: maxDepth });
410
+ return Number(rows[0]?.headroom ?? 0);
411
+ }
412
+
373
413
  // ------------------------------------------------------------- worker side
374
414
  /**
375
415
  * Take up to `limit` claimable tasks. `names` restricts the claim to task names
@@ -385,26 +425,82 @@ export abstract class TaskStore {
385
425
  limit?: number;
386
426
  names?: string[];
387
427
  }): Promise<Task[]> {
388
- // One queue is the common case and gets its own statement: a list-valued queue
389
- // filter cannot be read in claim order, so the planner sorts every claimable
390
- // row to take LIMIT of them, and claim's cost grows with the queued backlog
391
- // while it holds the claim transaction. See claim_one_queue.sql.
428
+ const names = input.names ?? null;
429
+ const claimed = await this.claimSession(
430
+ { queues: input.queues, workerId: input.workerId, leaseMs: input.leaseMs, names },
431
+ (claim) => claim(names, input.limit ?? 1),
432
+ );
433
+ return claimed ?? [];
434
+ }
435
+
436
+ /**
437
+ * Open one claim transaction and let the caller draw from it repeatedly.
438
+ *
439
+ * The transaction is what has to live here: the read-only probe that keeps an
440
+ * idle worker off SQLite's single write lock, the `recover_leases` whose
441
+ * reclaimed leases must be visible to the claims that follow and to nobody in
442
+ * between, and the write lock itself. *What* gets claimed under it is the
443
+ * caller's business — a worker drawing a separate quota per task name is
444
+ * scheduling policy, and this layer has no vocabulary for the "handler call"
445
+ * that policy is denominated in. It knows queues, names, limits and rows.
446
+ *
447
+ * `plan` is handed a `claim(names, limit)` it may call any number of times,
448
+ * each a separate statement under the same lock and the same recovery, and
449
+ * each free to size itself from what the previous one returned. That feedback
450
+ * is the reason this is a callback rather than a list of quotas: a caller
451
+ * dividing a budget up front has to guess, and every share handed to a name
452
+ * with nothing queued is a slot left idle until the next poll.
453
+ *
454
+ * `plan` runs with the write lock held, so it must await nothing but that
455
+ * callback.
456
+ *
457
+ * `names` is the union `plan` might ask for — the probe and the recovery are
458
+ * filtered by it. Returns undefined when the probe finds nothing claimable, in
459
+ * which case `plan` never runs and no transaction is opened.
460
+ */
461
+ async claimSession<T>(
462
+ input: { queues: string[]; workerId: string; leaseMs?: number; names: string[] | null },
463
+ plan: (claim: (names: string[] | null, limit: number) => Promise<Task[]>) => Promise<T>,
464
+ ): Promise<T | undefined> {
465
+ // A list-valued filter cannot be read in claim order, so the planner sorts
466
+ // every claimable row to take LIMIT of them and the claim's cost grows with
467
+ // the backlog while it holds the transaction. Both filters therefore have an
468
+ // equality form, picked per draw: one queue is the common deployment, and one
469
+ // name is every per-name quota. See claim_one_queue.sql and claim_one_name.sql.
392
470
  const oneQueue = input.queues.length === 1;
393
- const params: Params = {
471
+ const base: Params = {
394
472
  queues: input.queues,
395
473
  queue: oneQueue ? input.queues[0] : null,
396
- names: input.names ?? null,
474
+ names: input.names,
475
+ name: null,
397
476
  worker_id: input.workerId,
398
477
  lease_ms: input.leaseMs ?? 30_000,
399
- limit: input.limit ?? 1,
478
+ limit: 1,
400
479
  lease_expired_error: LEASE_EXPIRED_ERROR_JSON,
401
480
  };
402
- if (!(await this.hasClaimableWork(params))) return [];
403
- // Recovery must share the claim's transaction: a lease reclaimed here has to
404
- // be visible to the claim that follows, and to nobody in between.
481
+ if (!(await this.hasClaimableWork(base))) return undefined;
405
482
  return this.tx(async (fetch) => {
406
- await fetch("recover_leases", params);
407
- return (await fetch(oneQueue ? "claim_one_queue" : "claim", params)).map(rowToTask);
483
+ await fetch("recover_leases", base);
484
+ return plan(async (names, limit) => {
485
+ // A draw asking for nothing, or filtered to no names, claims nothing —
486
+ // answer it here rather than spending a statement to learn that.
487
+ if (limit <= 0 || names?.length === 0) return [];
488
+ const oneName = names?.length === 1;
489
+ const statement = oneName
490
+ ? oneQueue
491
+ ? "claim_one_queue_one_name"
492
+ : "claim_one_name"
493
+ : oneQueue
494
+ ? "claim_one_queue"
495
+ : "claim";
496
+ const rows = await fetch(statement, {
497
+ ...base,
498
+ names,
499
+ name: oneName ? names![0] : null,
500
+ limit,
501
+ });
502
+ return rows.map(rowToTask);
503
+ });
408
504
  });
409
505
  }
410
506
 
@@ -416,6 +512,33 @@ export abstract class TaskStore {
416
512
  });
417
513
  }
418
514
 
515
+ /**
516
+ * Renew several leases in one statement. Returns `taskId -> cancel requested`
517
+ * for the tasks this worker still holds.
518
+ *
519
+ * Deliberately not an ownedWrite: ownership is per task here, so there is no
520
+ * single answer to "did it work". A task **absent** from the result lost its
521
+ * lease, and the caller decides what that means for that one task rather than
522
+ * failing the whole beat.
523
+ *
524
+ * It returns flags rather than Tasks because nothing downstream needs a task:
525
+ * the caller renews leases and observes cancellation, and whole rows would drag
526
+ * every payload back on every beat for the life of the call.
527
+ */
528
+ async heartbeatBatch(input: {
529
+ taskIds: string[];
530
+ workerId: string;
531
+ leaseMs?: number;
532
+ }): Promise<Map<string, boolean>> {
533
+ if (!input.taskIds.length) return new Map();
534
+ const rows = await this.fetch("heartbeat_batch", {
535
+ ids: input.taskIds,
536
+ worker_id: input.workerId,
537
+ lease_ms: input.leaseMs ?? 30_000,
538
+ });
539
+ return new Map(rows.map((r) => [r.id as string, r.cancel_requested_at_ms != null]));
540
+ }
541
+
419
542
  async progress(input: {
420
543
  taskId: string;
421
544
  workerId: string;
@@ -7,6 +7,7 @@ import { nowMs } from "../ids.js";
7
7
  import { loadMigrations, loadStatements } from "../sql.js";
8
8
  import {
9
9
  checkProtocolVersion,
10
+ COMMENT,
10
11
  type Fetch,
11
12
  type Params,
12
13
  statementParams,
@@ -61,6 +62,34 @@ function isMemory(path: string): boolean {
61
62
  return path === ":memory:" || path.includes("mode=memory");
62
63
  }
63
64
 
65
+ /**
66
+ * Whether this statement writes, and so belongs in a group commit.
67
+ *
68
+ * Read from the SQL rather than from a list of statement names, which would be a
69
+ * second place to remember when the protocol gains a statement. Every protocol
70
+ * statement is a single top-level `select`, `insert`, `update` or `delete`.
71
+ *
72
+ * Reads must stay out of the batch: `claimable_probe` exists precisely so an idle
73
+ * worker never takes SQLite's write lock, and a BEGIN IMMEDIATE around it would
74
+ * hand that back.
75
+ */
76
+ function isWriteStatement(sql: string): boolean {
77
+ return !/^\s*select/i.test(sql.replace(COMMENT, ""));
78
+ }
79
+
80
+ /**
81
+ * One write waiting for its turn on the shared connection.
82
+ *
83
+ * The rows go back to the caller that asked for them, so a batch resolves each
84
+ * member with its own result rather than a merged one.
85
+ */
86
+ interface Pending {
87
+ name: string;
88
+ params: Params;
89
+ resolve(rows: any[]): void;
90
+ reject(err: unknown): void;
91
+ }
92
+
64
93
  /**
65
94
  * Whether cairnq_tasks has been analyzed at all.
66
95
  *
@@ -189,6 +218,12 @@ export class SQLiteStore extends TaskStore {
189
218
  private readonly busyBudgetMs: number;
190
219
  /** When this connection may next revisit its planner statistics. */
191
220
  private nextStatsRefreshAt = 0;
221
+ /** Which statements are writes — see isWriteStatement. */
222
+ private readonly writes: Record<string, boolean>;
223
+ /** Writes waiting to be group-committed — see flush(). */
224
+ private pending: Pending[] = [];
225
+ /** Whether a flusher is already queued to drain `pending`. */
226
+ private flushing = false;
192
227
 
193
228
  constructor(
194
229
  private readonly path: string,
@@ -197,6 +232,9 @@ export class SQLiteStore extends TaskStore {
197
232
  super();
198
233
  this.busyBudgetMs = opts.busyTimeoutMs ?? 5000;
199
234
  this.statements = loadStatements("sqlite");
235
+ this.writes = Object.fromEntries(
236
+ Object.entries(this.statements).map(([name, sql]) => [name, isWriteStatement(sql)]),
237
+ );
200
238
  // Only a bare ":memory:" is guaranteed private to its connection, so only
201
239
  // it gets a lock of its own. A "mode=memory" URI stays path-keyed: with
202
240
  // cache=shared it names ONE shared database, and on a build without URI
@@ -319,7 +357,10 @@ export class SQLiteStore extends TaskStore {
319
357
  bound[name] = now - (params.older_than_ms as number);
320
358
  break;
321
359
  case "queues":
322
- bound[name] = JSON.stringify(params.queues);
360
+ case "ids":
361
+ // json_each needs a JSON array. Postgres binds the array itself as
362
+ // text[], so only this dialect encodes.
363
+ bound[name] = JSON.stringify(params[name]);
323
364
  break;
324
365
  case "names":
325
366
  // json_each needs a JSON array; null stays null so the SQL's
@@ -422,10 +463,135 @@ export class SQLiteStore extends TaskStore {
422
463
  }
423
464
  }
424
465
 
466
+ /**
467
+ * Group commit: one transaction for every write already waiting on the lock.
468
+ *
469
+ * A write costs microseconds to execute and a transaction costs a WAL commit, so
470
+ * N concurrent writes spend nearly all their time on N commits they could have
471
+ * shared. Measured at 200 finalizes: 80µs each one-transaction-apiece against
472
+ * 10µs each in one transaction (`bench/sweep` sweep B).
473
+ *
474
+ * Nothing waits to form a batch — a flusher takes whatever arrived while the
475
+ * previous one held the lock, so this trades no latency for the throughput. What
476
+ * it does trade is atomicity: two callers' writes now land together or not at
477
+ * all. Under at-least-once that is not observable (a lost batch is a
478
+ * redelivery), and it is why every member is resolved only after COMMIT.
479
+ */
480
+ private flush(db: DB): void {
481
+ // One writer waiting is the uncontended case, and it stays exactly as cheap as
482
+ // before: wrapping a single statement in BEGIN/COMMIT would add two statements
483
+ // to every write on an idle store.
484
+ if (this.pending.length === 1) {
485
+ const only = this.pending[0];
486
+ let rows: any[];
487
+ try {
488
+ rows = this.runNow(only.name, only.params);
489
+ } catch (err) {
490
+ // Leave it pending on a lost write lock: withLock re-runs this flusher.
491
+ if (isBusy(err)) throw err;
492
+ this.pending.shift();
493
+ only.reject(err);
494
+ return;
495
+ }
496
+ this.pending.shift();
497
+ only.resolve(rows);
498
+ return;
499
+ }
500
+
501
+ // BEGIN before consuming, so a lost write lock leaves the batch where the
502
+ // retry will find it — with anything that arrived meanwhile.
503
+ db.exec("BEGIN IMMEDIATE");
504
+ const batch = this.pending;
505
+ this.pending = [];
506
+ const out: { rows?: any[]; err?: unknown }[] = [];
507
+ try {
508
+ for (const w of batch) {
509
+ try {
510
+ out.push({ rows: this.runNow(w.name, w.params) });
511
+ } catch (err) {
512
+ // A statement error aborts that statement, not the transaction, so the
513
+ // rest of the batch is still good and this one waiter carries the error.
514
+ // If SQLite tore the transaction down instead, nothing in it survived
515
+ // and every member has to hear about it.
516
+ if (!db.inTransaction) throw err;
517
+ out.push({ err });
518
+ }
519
+ }
520
+ db.exec("COMMIT");
521
+ } catch (err) {
522
+ if (db.inTransaction) {
523
+ try {
524
+ db.exec("ROLLBACK");
525
+ } catch {
526
+ // Raced with SQLite's own rollback; the transaction is gone either way.
527
+ }
528
+ }
529
+ if (isBusy(err)) {
530
+ // Back to the head of the queue, ahead of later arrivals, so the retry
531
+ // preserves the order the writes were issued in.
532
+ this.pending = batch.concat(this.pending);
533
+ throw err;
534
+ }
535
+ for (const w of batch) w.reject(err);
536
+ return;
537
+ }
538
+ // Only now: before COMMIT a rollback could still take the write back, and a
539
+ // caller holding its row would have observed a write that never happened.
540
+ for (let i = 0; i < batch.length; i++) {
541
+ // Presence, not truthiness — a thrown value is not guaranteed to be one.
542
+ if ("err" in out[i]) batch[i].reject(out[i].err);
543
+ else batch[i].resolve(out[i].rows!);
544
+ }
545
+ }
546
+
547
+ /**
548
+ * Make sure some flusher is draining `pending`, without ever running two.
549
+ *
550
+ * The flusher loops instead of re-arming itself per batch. A caller that awaits
551
+ * its writes one at a time resumes and issues the next one *before* the flusher
552
+ * gets its turn back, so re-arming would cost that write an extra trip through
553
+ * the lock queue — measured as ~2x on sequential writes, which is most of them.
554
+ * Looping picks it up in the same session for free.
555
+ *
556
+ * The exit is safe because the last `pending` check and clearing the flag happen
557
+ * in one synchronous step: a write that arrives before it keeps the loop going,
558
+ * and one that arrives after sees the flag down and starts a new flusher.
559
+ */
560
+ private scheduleFlush(db: DB): void {
561
+ if (this.flushing) return;
562
+ this.flushing = true;
563
+ void (async () => {
564
+ try {
565
+ while (this.pending.length) {
566
+ try {
567
+ await this.withLock(() => this.flush(db));
568
+ } catch (err) {
569
+ // flush only throws on a lost write lock, and only after putting its
570
+ // batch back — so reaching here means withLock spent the whole budget
571
+ // and those writes are still queued with nobody else coming for them.
572
+ // Anything that arrived behind them is failed with the same error
573
+ // rather than left hanging: this store cannot write at all right now,
574
+ // which is what a lone write would have been told too.
575
+ const stranded = this.pending;
576
+ this.pending = [];
577
+ for (const w of stranded) w.reject(err);
578
+ }
579
+ }
580
+ } finally {
581
+ this.flushing = false;
582
+ }
583
+ })();
584
+ }
585
+
425
586
  protected async fetch(name: string, params: Params): Promise<any[]> {
426
587
  const db = this.ensure();
427
588
  await this.maybeRefreshStatistics(db);
428
- return this.withLock(() => this.runNow(name, params));
589
+ // Reads keep their own turn on the lock — see isWriteStatement.
590
+ if (!this.writes[name]) return this.withLock(() => this.runNow(name, params));
591
+ return new Promise<any[]>((resolve, reject) => {
592
+ this.pending.push({ name, params, resolve, reject });
593
+ this.scheduleFlush(db);
594
+ });
429
595
  }
430
596
 
431
597
  protected async tx<T>(fn: (fetch: Fetch) => Promise<T>): Promise<T> {