cairnq 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/README.md +28 -0
  2. package/dist/_protocol/migrations/postgres/0001_init.sql +3 -1
  3. package/dist/_protocol/migrations/postgres/0002_purge_index.sql +6 -0
  4. package/dist/_protocol/migrations/postgres/0003_notify.sql +38 -0
  5. package/dist/_protocol/migrations/postgres/0004_lease_index.sql +16 -0
  6. package/dist/_protocol/migrations/postgres/0005_clear_terminal_lease.sql +17 -0
  7. package/dist/_protocol/migrations/sqlite/0001_init.sql +3 -1
  8. package/dist/_protocol/migrations/sqlite/0002_purge_index.sql +6 -0
  9. package/dist/_protocol/migrations/sqlite/0004_lease_index.sql +22 -0
  10. package/dist/_protocol/migrations/sqlite/0005_clear_terminal_lease.sql +17 -0
  11. package/dist/_protocol/sql/postgres/claim.sql +18 -5
  12. package/dist/_protocol/sql/postgres/claim_one_queue.sql +35 -0
  13. package/dist/_protocol/sql/postgres/complete.sql +3 -0
  14. package/dist/_protocol/sql/postgres/fail.sql +30 -8
  15. package/dist/_protocol/sql/postgres/insert_task.sql +6 -3
  16. package/dist/_protocol/sql/postgres/list.sql +3 -1
  17. package/dist/_protocol/sql/postgres/lock_key.sql +9 -0
  18. package/dist/_protocol/sql/postgres/progress.sql +4 -3
  19. package/dist/_protocol/sql/postgres/protocol_version.sql +4 -0
  20. package/dist/_protocol/sql/postgres/purge.sql +25 -0
  21. package/dist/_protocol/sql/postgres/recover_leases.sql +49 -14
  22. package/dist/_protocol/sql/postgres/retry.sql +3 -0
  23. package/dist/_protocol/sql/postgres/stats.sql +8 -0
  24. package/dist/_protocol/sql/postgres/succeed.sql +4 -0
  25. package/dist/_protocol/sql/sqlite/claim.sql +13 -2
  26. package/dist/_protocol/sql/sqlite/claim_one_queue.sql +36 -0
  27. package/dist/_protocol/sql/sqlite/claimable_probe.sql +6 -2
  28. package/dist/_protocol/sql/sqlite/complete.sql +3 -0
  29. package/dist/_protocol/sql/sqlite/fail.sql +32 -8
  30. package/dist/_protocol/sql/sqlite/list.sql +3 -1
  31. package/dist/_protocol/sql/sqlite/lock_key.sql +5 -0
  32. package/dist/_protocol/sql/sqlite/progress.sql +6 -2
  33. package/dist/_protocol/sql/sqlite/protocol_version.sql +4 -0
  34. package/dist/_protocol/sql/sqlite/purge.sql +18 -0
  35. package/dist/_protocol/sql/sqlite/recover_leases.sql +25 -7
  36. package/dist/_protocol/sql/sqlite/retry.sql +3 -0
  37. package/dist/_protocol/sql/sqlite/stats.sql +8 -0
  38. package/dist/_protocol/sql/sqlite/succeed.sql +4 -0
  39. package/dist/client.d.ts +10 -2
  40. package/dist/client.js +12 -0
  41. package/dist/context.d.ts +17 -1
  42. package/dist/context.js +60 -6
  43. package/dist/errors.d.ts +18 -2
  44. package/dist/errors.js +49 -3
  45. package/dist/index.d.ts +3 -2
  46. package/dist/index.js +2 -1
  47. package/dist/sql.js +16 -9
  48. package/dist/store/base.d.ts +114 -9
  49. package/dist/store/base.js +376 -1
  50. package/dist/store/postgres.d.ts +62 -63
  51. package/dist/store/postgres.js +245 -222
  52. package/dist/store/sqlite.d.ts +83 -60
  53. package/dist/store/sqlite.js +370 -234
  54. package/dist/wait.d.ts +15 -2
  55. package/dist/wait.js +23 -5
  56. package/dist/worker.d.ts +53 -1
  57. package/dist/worker.js +202 -42
  58. package/package.json +9 -2
  59. package/src/client.ts +16 -2
  60. package/src/context.ts +70 -13
  61. package/src/errors.ts +59 -4
  62. package/src/index.ts +3 -1
  63. package/src/sql.ts +15 -8
  64. package/src/store/base.ts +443 -27
  65. package/src/store/postgres.ts +243 -267
  66. package/src/store/sqlite.ts +378 -263
  67. package/src/wait.ts +28 -5
  68. package/src/worker.ts +242 -42
@@ -1,16 +1,40 @@
1
- -- Fail a task. Ownership-checked. Single CASE-based statement handles both
2
- -- branches atomically: retryable && attempt < max_attempts -> requeue with
3
- -- backoff (run_at = now + delay_ms); otherwise -> terminal 'failed'.
1
+ -- Fail a task. Ownership-checked. One CASE-based statement decides all three
2
+ -- outcomes atomically:
3
+ -- 1. a cancel was requested while it ran -> terminal 'canceled'. Cancel wins,
4
+ -- exactly as in complete.sql: a task the user cancelled must never be
5
+ -- redelivered, whether the attempt ended in a return or in an exception.
6
+ -- 2. retryable && attempt < max_attempts -> requeue with backoff
7
+ -- (run_at = now + delay_ms).
8
+ -- 3. otherwise -> terminal 'failed'.
9
+ -- The error envelope is recorded on every branch, so a canceled-while-failing
10
+ -- task still carries why its last attempt failed. progress/message describe the
11
+ -- attempt in flight, so only the requeue branch clears them: a terminal record
12
+ -- keeps how far the last attempt got, a re-queued one must not advertise a dead
13
+ -- attempt's progress bar until the next attempt overwrites it.
4
14
  -- :retryable is 0/1. :error is a JSON envelope text.
5
15
  -- params: id, worker_id, now_ms, error, retryable, delay_ms
6
16
  update cairnq_tasks
7
17
  set
8
- status = case when :retryable = 1 and attempt < max_attempts then 'queued' else 'failed' end,
18
+ status = case
19
+ when cancel_requested_at_ms is not null then 'canceled'
20
+ when :retryable = 1 and attempt < max_attempts then 'queued'
21
+ else 'failed'
22
+ end,
9
23
  error = :error,
10
- worker_id = case when :retryable = 1 and attempt < max_attempts then null else worker_id end,
11
- lease_until_ms = case when :retryable = 1 and attempt < max_attempts then null else lease_until_ms end,
12
- run_at_ms = case when :retryable = 1 and attempt < max_attempts then :now_ms + :delay_ms else run_at_ms end,
13
- completed_at_ms = case when :retryable = 1 and attempt < max_attempts then null else :now_ms end,
24
+ worker_id = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
25
+ then null else worker_id end,
26
+ -- Unconditional, unlike worker_id above: all three branches end the attempt
27
+ -- that held the lease — two terminally, one to wait for redelivery — and none
28
+ -- of them leaves anyone owning it (see succeed.sql).
29
+ lease_until_ms = null,
30
+ run_at_ms = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
31
+ then :now_ms + :delay_ms else run_at_ms end,
32
+ completed_at_ms = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
33
+ then null else :now_ms end,
34
+ progress = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
35
+ then null else progress end,
36
+ message = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
37
+ then null else message end,
14
38
  updated_at_ms = :now_ms
15
39
  where id = :id
16
40
  and status = 'running'
@@ -7,5 +7,7 @@ where (:status is null or status = :status)
7
7
  and (:name is null or name = :name)
8
8
  and (:root_id is null or root_id = :root_id)
9
9
  and (:correlation_id is null or correlation_id = :correlation_id)
10
- order by created_at_ms desc
10
+ -- id breaks created_at_ms ties, as in claim.sql: without it, paginating with
11
+ -- offset across same-millisecond rows could repeat or skip a task.
12
+ order by created_at_ms desc, id desc
11
13
  limit :limit offset :offset;
@@ -0,0 +1,5 @@
1
+ -- No-op on SQLite: BEGIN IMMEDIATE already serializes every keyed transaction
2
+ -- on the database's single write lock, so there is nothing further to lock.
3
+ -- Exists so the shared TaskStore logic can take the key lock unconditionally;
4
+ -- see the postgres dialect for the real one.
5
+ select 1 as locked;
@@ -1,8 +1,12 @@
1
1
  -- Update progress/message. Ownership-checked. Does not change status.
2
2
  -- params: id, worker_id, now_ms, progress, message
3
- -- message is coalesced so progress(value) without a message keeps the prior one.
3
+ -- Both fields are coalesced, symmetrically: progress(value) keeps the prior
4
+ -- message, progress(null, message) keeps the prior fraction. Passing null means
5
+ -- "leave this alone", never "clear it".
4
6
  update cairnq_tasks
5
- set progress = :progress, message = coalesce(:message, message), updated_at_ms = :now_ms
7
+ set progress = coalesce(:progress, progress),
8
+ message = coalesce(:message, message),
9
+ updated_at_ms = :now_ms
6
10
  where id = :id
7
11
  and status = 'running'
8
12
  and worker_id = :worker_id
@@ -0,0 +1,4 @@
1
+ -- The storage's protocol major, from cairnq_meta (written by the migrations).
2
+ -- Returns no row on a database from before the first migration.
3
+ -- params: (none)
4
+ select value from cairnq_meta where key = 'protocol_version';
@@ -0,0 +1,18 @@
1
+ -- Retention: delete terminal tasks that completed before a cutoff. Nothing else
2
+ -- ever removes rows, so without this a long-lived database only grows.
3
+ -- Bounded by :limit so a large backlog is drained in short transactions instead
4
+ -- of one long write that blocks every other writer. The key pointer of a purged
5
+ -- task goes with it via cairnq_task_keys' ON DELETE CASCADE.
6
+ -- The LIMIT lives in a subquery: plain `delete ... limit` needs a non-default
7
+ -- SQLite build option.
8
+ -- params: before_ms, limit
9
+ delete from cairnq_tasks
10
+ where id in (
11
+ select id from cairnq_tasks
12
+ where status in ('succeeded', 'failed', 'canceled')
13
+ and completed_at_ms is not null
14
+ and completed_at_ms < :before_ms
15
+ order by completed_at_ms asc
16
+ limit :limit
17
+ )
18
+ returning id;
@@ -1,15 +1,33 @@
1
1
  -- Reclaim tasks whose lease expired. Run inside the same write transaction as
2
- -- claim, just before it. attempt < max_attempts -> back to 'queued' for redelivery;
3
- -- otherwise -> 'failed' with a lease-expired error envelope.
2
+ -- claim, just before it. Three outcomes, mirroring fail.sql:
3
+ -- 1. a cancel was requested before the worker died -> terminal 'canceled'
4
+ -- (a cancelled task must never be redelivered by the crash path either);
5
+ -- 2. attempt < max_attempts -> back to 'queued' for redelivery;
6
+ -- 3. otherwise -> terminal 'failed' with a lease-expired error envelope.
4
7
  -- params: now_ms, lease_expired_error (JSON envelope text)
5
8
  update cairnq_tasks
6
9
  set
7
- status = case when attempt < max_attempts then 'queued' else 'failed' end,
8
- worker_id = case when attempt < max_attempts then null else worker_id end,
10
+ status = case
11
+ when cancel_requested_at_ms is not null then 'canceled'
12
+ when attempt < max_attempts then 'queued'
13
+ else 'failed'
14
+ end,
15
+ worker_id = case when cancel_requested_at_ms is null and attempt < max_attempts
16
+ then null else worker_id end,
9
17
  lease_until_ms = null,
10
- run_at_ms = case when attempt < max_attempts then :now_ms else run_at_ms end,
11
- error = case when attempt >= max_attempts then :lease_expired_error else error end,
12
- completed_at_ms = case when attempt >= max_attempts then :now_ms else completed_at_ms end,
18
+ run_at_ms = case when cancel_requested_at_ms is null and attempt < max_attempts
19
+ then :now_ms else run_at_ms end,
20
+ -- Only the failed branch records lease expiry: a canceled task did not fail.
21
+ error = case when cancel_requested_at_ms is null and attempt >= max_attempts
22
+ then :lease_expired_error else error end,
23
+ completed_at_ms = case when cancel_requested_at_ms is null and attempt < max_attempts
24
+ then completed_at_ms else :now_ms end,
25
+ -- Only the requeue branch clears them: they describe the dead attempt, and a
26
+ -- task waiting to be redelivered must not report its progress bar.
27
+ progress = case when cancel_requested_at_ms is null and attempt < max_attempts
28
+ then null else progress end,
29
+ message = case when cancel_requested_at_ms is null and attempt < max_attempts
30
+ then null else message end,
13
31
  updated_at_ms = :now_ms
14
32
  where status = 'running'
15
33
  and lease_until_ms is not null
@@ -10,6 +10,9 @@ set
10
10
  run_at_ms = :now_ms,
11
11
  cancel_requested_at_ms = null,
12
12
  completed_at_ms = null,
13
+ -- The previous attempt's progress bar dies with the attempt.
14
+ progress = null,
15
+ message = null,
13
16
  attempt = case when :reset_attempt = 1 then 0 else attempt end,
14
17
  updated_at_ms = :now_ms
15
18
  where id = :id and status in ('failed', 'canceled')
@@ -0,0 +1,8 @@
1
+ -- Queue depth at a glance: task counts grouped by queue and status. Read-only.
2
+ -- A queue appears only while it has rows — terminal tasks count until purge
3
+ -- removes them. The SDK zero-fills the statuses a queue has no rows in.
4
+ -- params: (none)
5
+ select queue, status, count(*) as count
6
+ from cairnq_tasks
7
+ group by queue, status
8
+ order by queue asc, status asc;
@@ -6,6 +6,10 @@ set
6
6
  result = :result,
7
7
  progress = 1.0,
8
8
  message = coalesce(:message, message),
9
+ -- A lease describes an attempt in flight; this one just ended. worker_id is
10
+ -- what carries the audit trail (who ran it), so nothing is lost by clearing
11
+ -- it, and the terminal-lease invariant holds — see PROTOCOL.md §Lease model.
12
+ lease_until_ms = null,
9
13
  completed_at_ms = :now_ms,
10
14
  updated_at_ms = :now_ms
11
15
  where id = :id
package/dist/client.d.ts CHANGED
@@ -1,5 +1,5 @@
1
- import { type Task } from "./models.js";
2
- import type { ListInput, SubmitInput, TaskStore } from "./store/base.js";
1
+ import { type Task, type TaskStatus } from "./models.js";
2
+ import type { ListInput, PurgeInput, SubmitInput, TaskStore } from "./store/base.js";
3
3
  import { type TaskDef } from "./task.js";
4
4
  export type SubmitOptions = Omit<SubmitInput, "name" | "payload">;
5
5
  export interface CallOptions extends SubmitOptions {
@@ -34,6 +34,14 @@ export declare class CairnQ {
34
34
  retryByKey(key: string, opts?: {
35
35
  resetAttempt?: boolean;
36
36
  }): Promise<Task | null>;
37
+ /** Delete terminal tasks that finished more than `olderThanMs` ago and return
38
+ * their ids. Nothing else in CairnQ removes rows, so a long-lived database
39
+ * needs this on a schedule. Each call is bounded by `limit` to keep the write
40
+ * short; loop until it returns fewer than `limit`. */
41
+ purge(input?: PurgeInput): Promise<string[]>;
42
+ /** Task counts per queue, keyed by status and zero-filled across all statuses
43
+ * — `(await stats()).default.queued` is the backlog of a queue. */
44
+ stats(): Promise<Record<string, Record<TaskStatus, number>>>;
37
45
  wait(taskId: string, opts?: {
38
46
  timeoutMs?: number;
39
47
  pollMs?: number;
package/dist/client.js CHANGED
@@ -51,6 +51,18 @@ export class CairnQ {
51
51
  retryByKey(key, opts) {
52
52
  return this._store.retryByKey(key, opts);
53
53
  }
54
+ /** Delete terminal tasks that finished more than `olderThanMs` ago and return
55
+ * their ids. Nothing else in CairnQ removes rows, so a long-lived database
56
+ * needs this on a schedule. Each call is bounded by `limit` to keep the write
57
+ * short; loop until it returns fewer than `limit`. */
58
+ purge(input) {
59
+ return this._store.purge(input);
60
+ }
61
+ /** Task counts per queue, keyed by status and zero-filled across all statuses
62
+ * — `(await stats()).default.queued` is the backlog of a queue. */
63
+ stats() {
64
+ return this._store.stats();
65
+ }
54
66
  wait(taskId, opts = {}) {
55
67
  return pollWait(this._store, taskId, {
56
68
  timeoutMs: opts.timeoutMs ?? 30_000,
package/dist/context.d.ts CHANGED
@@ -8,6 +8,9 @@ export declare class TaskContext {
8
8
  private readonly task;
9
9
  readonly workerId: string;
10
10
  private readonly leaseMs;
11
+ private readonly abort;
12
+ private leaseLost;
13
+ private cancelSeen;
11
14
  constructor(store: TaskStore, task: Task, workerId: string, leaseMs: number);
12
15
  get taskId(): string;
13
16
  get name(): string;
@@ -17,9 +20,22 @@ export declare class TaskContext {
17
20
  get rootId(): string | null;
18
21
  get correlationId(): string | null;
19
22
  get payload(): any;
23
+ /**
24
+ * True once this worker has lost the task's lease — it expired and another
25
+ * worker reclaimed it. Nothing this handler writes will be recorded any more
26
+ * and the task is already running elsewhere, so a long handler should check
27
+ * this (or `signal`) and bail out instead of continuing to do side effects.
28
+ */
29
+ get lostLease(): boolean;
30
+ /** Aborts when the lease is lost. Pass it to fetch / any AbortSignal-aware API. */
31
+ get signal(): AbortSignal;
32
+ /** @internal Called by the worker when an owned write reports a lost lease. */
33
+ markLeaseLost(): void;
34
+ private observe;
35
+ private owned;
20
36
  progress(value: number | null, message?: string | null): Promise<Task>;
21
37
  heartbeat(): Promise<Task>;
22
- /** Cooperative cancel check. */
38
+ /** Cooperative cancel check. Free once a heartbeat has already seen the flag. */
23
39
  canceled(): Promise<boolean>;
24
40
  /** Submit a child task; parent/root/correlation are wired automatically. */
25
41
  submit(name: string, payload?: unknown, opts?: SubmitOptions): Promise<Task>;
package/dist/context.js CHANGED
@@ -1,3 +1,4 @@
1
+ import { LostLease } from "./errors.js";
1
2
  import { cancelRequested } from "./models.js";
2
3
  import { taskName } from "./task.js";
3
4
  import { pollWait } from "./wait.js";
@@ -7,6 +8,11 @@ export class TaskContext {
7
8
  task;
8
9
  workerId;
9
10
  leaseMs;
11
+ abort = new AbortController();
12
+ leaseLost = false;
13
+ // Cancellation is monotonic: once the DB has told us a cancel was requested it
14
+ // can't be taken back, so canceled() can answer from this without a re-read.
15
+ cancelSeen = false;
10
16
  constructor(store, task, workerId, leaseMs) {
11
17
  this.store = store;
12
18
  this.task = task;
@@ -37,27 +43,75 @@ export class TaskContext {
37
43
  get payload() {
38
44
  return this.task.payload;
39
45
  }
46
+ /**
47
+ * True once this worker has lost the task's lease — it expired and another
48
+ * worker reclaimed it. Nothing this handler writes will be recorded any more
49
+ * and the task is already running elsewhere, so a long handler should check
50
+ * this (or `signal`) and bail out instead of continuing to do side effects.
51
+ */
52
+ get lostLease() {
53
+ return this.leaseLost;
54
+ }
55
+ /** Aborts when the lease is lost. Pass it to fetch / any AbortSignal-aware API. */
56
+ get signal() {
57
+ return this.abort.signal;
58
+ }
59
+ /** @internal Called by the worker when an owned write reports a lost lease. */
60
+ markLeaseLost() {
61
+ if (this.leaseLost)
62
+ return;
63
+ this.leaseLost = true;
64
+ this.abort.abort(new LostLease(this.task.id));
65
+ }
66
+ // Every owned write returns the current row, so cancellation and lease loss
67
+ // ride along on writes the handler was making anyway.
68
+ observe(task) {
69
+ if (cancelRequested(task))
70
+ this.cancelSeen = true;
71
+ return task;
72
+ }
73
+ async owned(write) {
74
+ // Short-circuit once the lease is known lost: nothing this context writes
75
+ // may be recorded any more. Locally, not just via the store's ownership
76
+ // check — after an abandoned (timed-out) attempt the same worker may
77
+ // re-claim this task under the same workerId, and a zombie handler's write
78
+ // would then pass ownership against the NEW attempt.
79
+ if (this.leaseLost)
80
+ throw new LostLease(this.task.id);
81
+ try {
82
+ return this.observe(await write());
83
+ }
84
+ catch (err) {
85
+ if (err instanceof LostLease)
86
+ this.markLeaseLost();
87
+ throw err;
88
+ }
89
+ }
40
90
  async progress(value, message = null) {
41
- return this.store.progress({
91
+ return this.owned(() => this.store.progress({
42
92
  taskId: this.task.id,
43
93
  workerId: this.workerId,
44
94
  progress: value,
45
95
  message,
46
- });
96
+ }));
47
97
  }
48
98
  async heartbeat() {
49
- return this.store.heartbeat({
99
+ return this.owned(() => this.store.heartbeat({
50
100
  taskId: this.task.id,
51
101
  workerId: this.workerId,
52
102
  leaseMs: this.leaseMs,
53
- });
103
+ }));
54
104
  }
55
- /** Cooperative cancel check. */
105
+ /** Cooperative cancel check. Free once a heartbeat has already seen the flag. */
56
106
  async canceled() {
107
+ if (this.cancelSeen)
108
+ return true;
57
109
  const t = await this.store.get(this.task.id);
58
110
  if (!t)
59
111
  return true;
60
- return cancelRequested(t) || t.status === "canceled";
112
+ if (cancelRequested(t))
113
+ this.cancelSeen = true;
114
+ return this.cancelSeen || t.status === "canceled";
61
115
  }
62
116
  async submit(task, payload, opts = {}) {
63
117
  return this.store.submit({
package/dist/errors.d.ts CHANGED
@@ -1,3 +1,4 @@
1
+ import { type Task } from "./models.js";
1
2
  /** The single shape of the JSON error envelope (see PROTOCOL.md). Everything that
2
3
  * records an error — a handler exception, a missing handler, lease expiry, a thrown
3
4
  * TaskError — builds it here, so the contract's fields live in one place. */
@@ -9,15 +10,23 @@ export declare function errorEnvelope(e: {
9
10
  details?: Record<string, unknown>;
10
11
  }): Record<string, unknown>;
11
12
  export declare class CairnQError extends Error {
13
+ constructor(message?: string);
12
14
  }
13
15
  export declare class AlreadyExists extends CairnQError {
14
16
  key: string;
15
17
  constructor(key: string);
16
18
  }
17
- /** wait/call did not reach a terminal status in time. The task keeps running. */
19
+ /** wait/call did not reach a terminal status in time. The task keeps running.
20
+ * `task` is the last snapshot wait() observed (null if get() found nothing), and
21
+ * the message says what state it was stuck in — a queued-never-claimed task is
22
+ * the classic first-run failure (no worker, no handler, wrong queue or file). */
18
23
  export declare class TaskTimeout extends CairnQError {
19
24
  taskId: string;
20
- constructor(taskId: string);
25
+ readonly task: Task | null;
26
+ constructor(taskId: string, opts?: {
27
+ timeoutMs?: number;
28
+ task?: Task | null;
29
+ });
21
30
  }
22
31
  /** A waited-on task ended in `failed`. The envelope's fields are unpacked onto the
23
32
  * error — read `e.code` / `e.message` / `e.retryable` / `e.details` instead of
@@ -42,6 +51,13 @@ export declare class LostLease extends CairnQError {
42
51
  export declare class ProtocolVersionMismatch extends CairnQError {
43
52
  constructor(message: string);
44
53
  }
54
+ /** A value could not be encoded for a protocol JSON column (non-finite number,
55
+ * BigInt, circular structure, …). Raised at the boundary — submit rejects with
56
+ * it, and a worker records a handler result that triggers it as a permanent
57
+ * `unserializable_result` failure. The Python SDK raises the same named error. */
58
+ export declare class SerializationError extends CairnQError {
59
+ constructor(message: string);
60
+ }
45
61
  /** Throw inside a handler to control how the failure is recorded. Defaults to
46
62
  * non-retryable so deterministic errors fail fast instead of burning retries.
47
63
  * Any other thrown value is treated as retryable. */
package/dist/errors.js CHANGED
@@ -1,3 +1,5 @@
1
+ import { nowMs } from "./ids.js";
2
+ import { cancelRequested, isQueued } from "./models.js";
1
3
  /** The single shape of the JSON error envelope (see PROTOCOL.md). Everything that
2
4
  * records an error — a handler exception, a missing handler, lease expiry, a thrown
3
5
  * TaskError — builds it here, so the contract's fields live in one place. */
@@ -11,6 +13,13 @@ export function errorEnvelope(e) {
11
13
  };
12
14
  }
13
15
  export class CairnQError extends Error {
16
+ constructor(message) {
17
+ super(message);
18
+ // Subclasses each set their own; without this a bare CairnQError reports
19
+ // "Error", and `err.name` is how callers (and the conformance runner) tell
20
+ // one apart from another.
21
+ this.name = "CairnQError";
22
+ }
14
23
  }
15
24
  export class AlreadyExists extends CairnQError {
16
25
  key;
@@ -20,13 +29,40 @@ export class AlreadyExists extends CairnQError {
20
29
  this.name = "AlreadyExists";
21
30
  }
22
31
  }
23
- /** wait/call did not reach a terminal status in time. The task keeps running. */
32
+ /** One line of "why hasn't this finished" from the last snapshot wait()
33
+ * observed. No worker running, no handler for the name, wrong queue, and two
34
+ * processes on different database files all look identical from the API side —
35
+ * queued, never claimed — so that case names the likely causes. */
36
+ function timeoutDetail(task) {
37
+ if (!task)
38
+ return "task not found — wrong database file, or already purged?";
39
+ if (isQueued(task)) {
40
+ const delayMs = task.run_at_ms - nowMs();
41
+ if (task.attempt === 0 && delayMs <= 0) {
42
+ return (`never claimed by a worker — is a worker running with a handler for ` +
43
+ `'${task.name}' on queue '${task.queue}', against this same database?`);
44
+ }
45
+ const next = delayMs > 0 ? `, next run in ~${delayMs}ms` : "";
46
+ return `still queued (attempt ${task.attempt}/${task.max_attempts})${next}`;
47
+ }
48
+ if (cancelRequested(task))
49
+ return "cancel requested, waiting for the handler to observe it";
50
+ return `still running (attempt ${task.attempt}/${task.max_attempts})`;
51
+ }
52
+ /** wait/call did not reach a terminal status in time. The task keeps running.
53
+ * `task` is the last snapshot wait() observed (null if get() found nothing), and
54
+ * the message says what state it was stuck in — a queued-never-claimed task is
55
+ * the classic first-run failure (no worker, no handler, wrong queue or file). */
24
56
  export class TaskTimeout extends CairnQError {
25
57
  taskId;
26
- constructor(taskId) {
27
- super(`task ${taskId} did not finish in time`);
58
+ task;
59
+ constructor(taskId, opts = {}) {
60
+ super(opts.timeoutMs == null
61
+ ? `task ${taskId} did not finish in time`
62
+ : `task ${taskId} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null)}`);
28
63
  this.taskId = taskId;
29
64
  this.name = "TaskTimeout";
65
+ this.task = opts.task ?? null;
30
66
  }
31
67
  }
32
68
  /** A waited-on task ended in `failed`. The envelope's fields are unpacked onto the
@@ -72,6 +108,16 @@ export class ProtocolVersionMismatch extends CairnQError {
72
108
  this.name = "ProtocolVersionMismatch";
73
109
  }
74
110
  }
111
+ /** A value could not be encoded for a protocol JSON column (non-finite number,
112
+ * BigInt, circular structure, …). Raised at the boundary — submit rejects with
113
+ * it, and a worker records a handler result that triggers it as a permanent
114
+ * `unserializable_result` failure. The Python SDK raises the same named error. */
115
+ export class SerializationError extends CairnQError {
116
+ constructor(message) {
117
+ super(message);
118
+ this.name = "SerializationError";
119
+ }
120
+ }
75
121
  /** Throw inside a handler to control how the failure is recorded. Defaults to
76
122
  * non-retryable so deterministic errors fail fast instead of burning retries.
77
123
  * Any other thrown value is treated as retryable. */
package/dist/index.d.ts CHANGED
@@ -7,7 +7,8 @@ export { defineTask } from "./task.js";
7
7
  export type { TaskDef } from "./task.js";
8
8
  export { SQLiteStore } from "./store/sqlite.js";
9
9
  export { PostgresStore } from "./store/postgres.js";
10
- export type { ListInput, SubmitInput, TaskStore, Conflict } from "./store/base.js";
10
+ export { TaskStore } from "./store/base.js";
11
+ export type { ListInput, PurgeInput, SubmitInput, Conflict } from "./store/base.js";
11
12
  export type { Task, TaskStatus } from "./models.js";
12
13
  export { STATUSES, isTerminal, cancelRequested, isQueued, isRunning, isSucceeded, isFailed, isCanceled, } from "./models.js";
13
- export { CairnQError, AlreadyExists, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, } from "./errors.js";
14
+ export { CairnQError, AlreadyExists, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, SerializationError, } from "./errors.js";
package/dist/index.js CHANGED
@@ -4,5 +4,6 @@ export { TaskContext } from "./context.js";
4
4
  export { defineTask } from "./task.js";
5
5
  export { SQLiteStore } from "./store/sqlite.js";
6
6
  export { PostgresStore } from "./store/postgres.js";
7
+ export { TaskStore } from "./store/base.js";
7
8
  export { STATUSES, isTerminal, cancelRequested, isQueued, isRunning, isSucceeded, isFailed, isCanceled, } from "./models.js";
8
- export { CairnQError, AlreadyExists, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, } from "./errors.js";
9
+ export { CairnQError, AlreadyExists, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, SerializationError, } from "./errors.js";
package/dist/sql.js CHANGED
@@ -1,19 +1,23 @@
1
1
  import { existsSync, readdirSync, readFileSync } from "node:fs";
2
2
  import { dirname, join } from "node:path";
3
3
  import { fileURLToPath } from "node:url";
4
- // Locate the shared cairnq-protocol dir. Resolution: $CAIRNQ_PROTOCOL_DIR ->
5
- // vendored `_protocol/` next to this module -> walk up to `cairnq-protocol/`
6
- // (monorepo dev). Both SDKs load the SAME .sql strings (zero-drift guarantee).
7
- // The dir is laid out per-dialect (sql/<dialect>/*.sql, migrations/<dialect>/*.sql)
8
- // so a second backend (Postgres) slots in beside sqlite; `dialect` picks the subtree.
4
+ // Locate the shared cairnq-protocol dir. Resolution: $CAIRNQ_PROTOCOL_DIR -> walk
5
+ // up to `cairnq-protocol/` (monorepo dev) -> vendored `_protocol/` next to this
6
+ // module (written at publish time). Both SDKs load the SAME .sql strings
7
+ // (zero-drift guarantee). The dir is laid out per-dialect (sql/<dialect>/*.sql,
8
+ // migrations/<dialect>/*.sql) so a second backend (Postgres) slots in beside
9
+ // sqlite; `dialect` picks the subtree.
10
+ //
11
+ // The source tree wins over the vendored copy on purpose: vendoring is a publish
12
+ // step that also runs locally, and a stale `_protocol/` shadowing the canonical
13
+ // SQL means edits to cairnq-protocol/ are silently not under test. An installed
14
+ // package has no repo above it, so it falls through to the vendored copy.
9
15
  export function findProtocolRoot() {
10
16
  const env = process.env.CAIRNQ_PROTOCOL_DIR;
11
17
  if (env)
12
18
  return env;
13
- let dir = dirname(fileURLToPath(import.meta.url));
14
- const vendored = join(dir, "_protocol");
15
- if (existsSync(join(vendored, "sql")))
16
- return vendored;
19
+ const start = dirname(fileURLToPath(import.meta.url));
20
+ let dir = start;
17
21
  for (let i = 0; i < 10; i++) {
18
22
  const candidate = join(dir, "cairnq-protocol");
19
23
  if (existsSync(join(candidate, "sql")))
@@ -23,6 +27,9 @@ export function findProtocolRoot() {
23
27
  break;
24
28
  dir = parent;
25
29
  }
30
+ const vendored = join(start, "_protocol");
31
+ if (existsSync(join(vendored, "sql")))
32
+ return vendored;
26
33
  throw new Error("cannot locate cairnq-protocol; set CAIRNQ_PROTOCOL_DIR");
27
34
  }
28
35
  export function loadStatements(dialect = "sqlite", root = findProtocolRoot()) {