cairnq 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -0
- package/dist/_protocol/migrations/postgres/0001_init.sql +3 -1
- package/dist/_protocol/migrations/postgres/0002_purge_index.sql +6 -0
- package/dist/_protocol/migrations/postgres/0003_notify.sql +38 -0
- package/dist/_protocol/migrations/postgres/0004_lease_index.sql +16 -0
- package/dist/_protocol/migrations/postgres/0005_clear_terminal_lease.sql +17 -0
- package/dist/_protocol/migrations/sqlite/0001_init.sql +3 -1
- package/dist/_protocol/migrations/sqlite/0002_purge_index.sql +6 -0
- package/dist/_protocol/migrations/sqlite/0004_lease_index.sql +22 -0
- package/dist/_protocol/migrations/sqlite/0005_clear_terminal_lease.sql +17 -0
- package/dist/_protocol/sql/postgres/claim.sql +18 -5
- package/dist/_protocol/sql/postgres/claim_one_queue.sql +35 -0
- package/dist/_protocol/sql/postgres/complete.sql +3 -0
- package/dist/_protocol/sql/postgres/fail.sql +30 -8
- package/dist/_protocol/sql/postgres/insert_task.sql +6 -3
- package/dist/_protocol/sql/postgres/list.sql +3 -1
- package/dist/_protocol/sql/postgres/lock_key.sql +9 -0
- package/dist/_protocol/sql/postgres/progress.sql +4 -3
- package/dist/_protocol/sql/postgres/protocol_version.sql +4 -0
- package/dist/_protocol/sql/postgres/purge.sql +25 -0
- package/dist/_protocol/sql/postgres/recover_leases.sql +49 -14
- package/dist/_protocol/sql/postgres/retry.sql +3 -0
- package/dist/_protocol/sql/postgres/stats.sql +8 -0
- package/dist/_protocol/sql/postgres/succeed.sql +4 -0
- package/dist/_protocol/sql/sqlite/claim.sql +13 -2
- package/dist/_protocol/sql/sqlite/claim_one_queue.sql +36 -0
- package/dist/_protocol/sql/sqlite/claimable_probe.sql +6 -2
- package/dist/_protocol/sql/sqlite/complete.sql +3 -0
- package/dist/_protocol/sql/sqlite/fail.sql +32 -8
- package/dist/_protocol/sql/sqlite/list.sql +3 -1
- package/dist/_protocol/sql/sqlite/lock_key.sql +5 -0
- package/dist/_protocol/sql/sqlite/progress.sql +6 -2
- package/dist/_protocol/sql/sqlite/protocol_version.sql +4 -0
- package/dist/_protocol/sql/sqlite/purge.sql +18 -0
- package/dist/_protocol/sql/sqlite/recover_leases.sql +25 -7
- package/dist/_protocol/sql/sqlite/retry.sql +3 -0
- package/dist/_protocol/sql/sqlite/stats.sql +8 -0
- package/dist/_protocol/sql/sqlite/succeed.sql +4 -0
- package/dist/client.d.ts +10 -2
- package/dist/client.js +12 -0
- package/dist/context.d.ts +17 -1
- package/dist/context.js +60 -6
- package/dist/errors.d.ts +18 -2
- package/dist/errors.js +49 -3
- package/dist/index.d.ts +3 -2
- package/dist/index.js +2 -1
- package/dist/sql.js +16 -9
- package/dist/store/base.d.ts +114 -9
- package/dist/store/base.js +376 -1
- package/dist/store/postgres.d.ts +62 -63
- package/dist/store/postgres.js +245 -222
- package/dist/store/sqlite.d.ts +83 -60
- package/dist/store/sqlite.js +370 -234
- package/dist/wait.d.ts +15 -2
- package/dist/wait.js +23 -5
- package/dist/worker.d.ts +53 -1
- package/dist/worker.js +202 -42
- package/package.json +9 -2
- package/src/client.ts +16 -2
- package/src/context.ts +70 -13
- package/src/errors.ts +59 -4
- package/src/index.ts +3 -1
- package/src/sql.ts +15 -8
- package/src/store/base.ts +443 -27
- package/src/store/postgres.ts +243 -267
- package/src/store/sqlite.ts +378 -263
- package/src/wait.ts +28 -5
- package/src/worker.ts +242 -42
|
@@ -1,16 +1,40 @@
|
|
|
1
|
-
-- Fail a task. Ownership-checked.
|
|
2
|
-
--
|
|
3
|
-
--
|
|
1
|
+
-- Fail a task. Ownership-checked. One CASE-based statement decides all three
|
|
2
|
+
-- outcomes atomically:
|
|
3
|
+
-- 1. a cancel was requested while it ran -> terminal 'canceled'. Cancel wins,
|
|
4
|
+
-- exactly as in complete.sql: a task the user cancelled must never be
|
|
5
|
+
-- redelivered, whether the attempt ended in a return or in an exception.
|
|
6
|
+
-- 2. retryable && attempt < max_attempts -> requeue with backoff
|
|
7
|
+
-- (run_at = now + delay_ms).
|
|
8
|
+
-- 3. otherwise -> terminal 'failed'.
|
|
9
|
+
-- The error envelope is recorded on every branch, so a canceled-while-failing
|
|
10
|
+
-- task still carries why its last attempt failed. progress/message describe the
|
|
11
|
+
-- attempt in flight, so only the requeue branch clears them: a terminal record
|
|
12
|
+
-- keeps how far the last attempt got, a re-queued one must not advertise a dead
|
|
13
|
+
-- attempt's progress bar until the next attempt overwrites it.
|
|
4
14
|
-- :retryable is 0/1. :error is a JSON envelope text.
|
|
5
15
|
-- params: id, worker_id, now_ms, error, retryable, delay_ms
|
|
6
16
|
update cairnq_tasks
|
|
7
17
|
set
|
|
8
|
-
status = case
|
|
18
|
+
status = case
|
|
19
|
+
when cancel_requested_at_ms is not null then 'canceled'
|
|
20
|
+
when :retryable = 1 and attempt < max_attempts then 'queued'
|
|
21
|
+
else 'failed'
|
|
22
|
+
end,
|
|
9
23
|
error = :error,
|
|
10
|
-
worker_id = case when :retryable = 1 and attempt < max_attempts
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
24
|
+
worker_id = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
|
|
25
|
+
then null else worker_id end,
|
|
26
|
+
-- Unconditional, unlike worker_id above: all three branches end the attempt
|
|
27
|
+
-- that held the lease — two terminally, one to wait for redelivery — and none
|
|
28
|
+
-- of them leaves anyone owning it (see succeed.sql).
|
|
29
|
+
lease_until_ms = null,
|
|
30
|
+
run_at_ms = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
|
|
31
|
+
then :now_ms + :delay_ms else run_at_ms end,
|
|
32
|
+
completed_at_ms = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
|
|
33
|
+
then null else :now_ms end,
|
|
34
|
+
progress = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
|
|
35
|
+
then null else progress end,
|
|
36
|
+
message = case when cancel_requested_at_ms is null and :retryable = 1 and attempt < max_attempts
|
|
37
|
+
then null else message end,
|
|
14
38
|
updated_at_ms = :now_ms
|
|
15
39
|
where id = :id
|
|
16
40
|
and status = 'running'
|
|
@@ -7,5 +7,7 @@ where (:status is null or status = :status)
|
|
|
7
7
|
and (:name is null or name = :name)
|
|
8
8
|
and (:root_id is null or root_id = :root_id)
|
|
9
9
|
and (:correlation_id is null or correlation_id = :correlation_id)
|
|
10
|
-
|
|
10
|
+
-- id breaks created_at_ms ties, as in claim.sql: without it, paginating with
|
|
11
|
+
-- offset across same-millisecond rows could repeat or skip a task.
|
|
12
|
+
order by created_at_ms desc, id desc
|
|
11
13
|
limit :limit offset :offset;
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
-- No-op on SQLite: BEGIN IMMEDIATE already serializes every keyed transaction
|
|
2
|
+
-- on the database's single write lock, so there is nothing further to lock.
|
|
3
|
+
-- Exists so the shared TaskStore logic can take the key lock unconditionally;
|
|
4
|
+
-- see the postgres dialect for the real one.
|
|
5
|
+
select 1 as locked;
|
|
@@ -1,8 +1,12 @@
|
|
|
1
1
|
-- Update progress/message. Ownership-checked. Does not change status.
|
|
2
2
|
-- params: id, worker_id, now_ms, progress, message
|
|
3
|
-
--
|
|
3
|
+
-- Both fields are coalesced, symmetrically: progress(value) keeps the prior
|
|
4
|
+
-- message, progress(null, message) keeps the prior fraction. Passing null means
|
|
5
|
+
-- "leave this alone", never "clear it".
|
|
4
6
|
update cairnq_tasks
|
|
5
|
-
set progress =
|
|
7
|
+
set progress = coalesce(:progress, progress),
|
|
8
|
+
message = coalesce(:message, message),
|
|
9
|
+
updated_at_ms = :now_ms
|
|
6
10
|
where id = :id
|
|
7
11
|
and status = 'running'
|
|
8
12
|
and worker_id = :worker_id
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
-- Retention: delete terminal tasks that completed before a cutoff. Nothing else
|
|
2
|
+
-- ever removes rows, so without this a long-lived database only grows.
|
|
3
|
+
-- Bounded by :limit so a large backlog is drained in short transactions instead
|
|
4
|
+
-- of one long write that blocks every other writer. The key pointer of a purged
|
|
5
|
+
-- task goes with it via cairnq_task_keys' ON DELETE CASCADE.
|
|
6
|
+
-- The LIMIT lives in a subquery: plain `delete ... limit` needs a non-default
|
|
7
|
+
-- SQLite build option.
|
|
8
|
+
-- params: before_ms, limit
|
|
9
|
+
delete from cairnq_tasks
|
|
10
|
+
where id in (
|
|
11
|
+
select id from cairnq_tasks
|
|
12
|
+
where status in ('succeeded', 'failed', 'canceled')
|
|
13
|
+
and completed_at_ms is not null
|
|
14
|
+
and completed_at_ms < :before_ms
|
|
15
|
+
order by completed_at_ms asc
|
|
16
|
+
limit :limit
|
|
17
|
+
)
|
|
18
|
+
returning id;
|
|
@@ -1,15 +1,33 @@
|
|
|
1
1
|
-- Reclaim tasks whose lease expired. Run inside the same write transaction as
|
|
2
|
-
-- claim, just before it.
|
|
3
|
-
--
|
|
2
|
+
-- claim, just before it. Three outcomes, mirroring fail.sql:
|
|
3
|
+
-- 1. a cancel was requested before the worker died -> terminal 'canceled'
|
|
4
|
+
-- (a cancelled task must never be redelivered by the crash path either);
|
|
5
|
+
-- 2. attempt < max_attempts -> back to 'queued' for redelivery;
|
|
6
|
+
-- 3. otherwise -> terminal 'failed' with a lease-expired error envelope.
|
|
4
7
|
-- params: now_ms, lease_expired_error (JSON envelope text)
|
|
5
8
|
update cairnq_tasks
|
|
6
9
|
set
|
|
7
|
-
status = case
|
|
8
|
-
|
|
10
|
+
status = case
|
|
11
|
+
when cancel_requested_at_ms is not null then 'canceled'
|
|
12
|
+
when attempt < max_attempts then 'queued'
|
|
13
|
+
else 'failed'
|
|
14
|
+
end,
|
|
15
|
+
worker_id = case when cancel_requested_at_ms is null and attempt < max_attempts
|
|
16
|
+
then null else worker_id end,
|
|
9
17
|
lease_until_ms = null,
|
|
10
|
-
run_at_ms = case when
|
|
11
|
-
|
|
12
|
-
|
|
18
|
+
run_at_ms = case when cancel_requested_at_ms is null and attempt < max_attempts
|
|
19
|
+
then :now_ms else run_at_ms end,
|
|
20
|
+
-- Only the failed branch records lease expiry: a canceled task did not fail.
|
|
21
|
+
error = case when cancel_requested_at_ms is null and attempt >= max_attempts
|
|
22
|
+
then :lease_expired_error else error end,
|
|
23
|
+
completed_at_ms = case when cancel_requested_at_ms is null and attempt < max_attempts
|
|
24
|
+
then completed_at_ms else :now_ms end,
|
|
25
|
+
-- Only the requeue branch clears them: they describe the dead attempt, and a
|
|
26
|
+
-- task waiting to be redelivered must not report its progress bar.
|
|
27
|
+
progress = case when cancel_requested_at_ms is null and attempt < max_attempts
|
|
28
|
+
then null else progress end,
|
|
29
|
+
message = case when cancel_requested_at_ms is null and attempt < max_attempts
|
|
30
|
+
then null else message end,
|
|
13
31
|
updated_at_ms = :now_ms
|
|
14
32
|
where status = 'running'
|
|
15
33
|
and lease_until_ms is not null
|
|
@@ -10,6 +10,9 @@ set
|
|
|
10
10
|
run_at_ms = :now_ms,
|
|
11
11
|
cancel_requested_at_ms = null,
|
|
12
12
|
completed_at_ms = null,
|
|
13
|
+
-- The previous attempt's progress bar dies with the attempt.
|
|
14
|
+
progress = null,
|
|
15
|
+
message = null,
|
|
13
16
|
attempt = case when :reset_attempt = 1 then 0 else attempt end,
|
|
14
17
|
updated_at_ms = :now_ms
|
|
15
18
|
where id = :id and status in ('failed', 'canceled')
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
-- Queue depth at a glance: task counts grouped by queue and status. Read-only.
|
|
2
|
+
-- A queue appears only while it has rows — terminal tasks count until purge
|
|
3
|
+
-- removes them. The SDK zero-fills the statuses a queue has no rows in.
|
|
4
|
+
-- params: (none)
|
|
5
|
+
select queue, status, count(*) as count
|
|
6
|
+
from cairnq_tasks
|
|
7
|
+
group by queue, status
|
|
8
|
+
order by queue asc, status asc;
|
|
@@ -6,6 +6,10 @@ set
|
|
|
6
6
|
result = :result,
|
|
7
7
|
progress = 1.0,
|
|
8
8
|
message = coalesce(:message, message),
|
|
9
|
+
-- A lease describes an attempt in flight; this one just ended. worker_id is
|
|
10
|
+
-- what carries the audit trail (who ran it), so nothing is lost by clearing
|
|
11
|
+
-- it, and the terminal-lease invariant holds — see PROTOCOL.md §Lease model.
|
|
12
|
+
lease_until_ms = null,
|
|
9
13
|
completed_at_ms = :now_ms,
|
|
10
14
|
updated_at_ms = :now_ms
|
|
11
15
|
where id = :id
|
package/dist/client.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { type Task } from "./models.js";
|
|
2
|
-
import type { ListInput, SubmitInput, TaskStore } from "./store/base.js";
|
|
1
|
+
import { type Task, type TaskStatus } from "./models.js";
|
|
2
|
+
import type { ListInput, PurgeInput, SubmitInput, TaskStore } from "./store/base.js";
|
|
3
3
|
import { type TaskDef } from "./task.js";
|
|
4
4
|
export type SubmitOptions = Omit<SubmitInput, "name" | "payload">;
|
|
5
5
|
export interface CallOptions extends SubmitOptions {
|
|
@@ -34,6 +34,14 @@ export declare class CairnQ {
|
|
|
34
34
|
retryByKey(key: string, opts?: {
|
|
35
35
|
resetAttempt?: boolean;
|
|
36
36
|
}): Promise<Task | null>;
|
|
37
|
+
/** Delete terminal tasks that finished more than `olderThanMs` ago and return
|
|
38
|
+
* their ids. Nothing else in CairnQ removes rows, so a long-lived database
|
|
39
|
+
* needs this on a schedule. Each call is bounded by `limit` to keep the write
|
|
40
|
+
* short; loop until it returns fewer than `limit`. */
|
|
41
|
+
purge(input?: PurgeInput): Promise<string[]>;
|
|
42
|
+
/** Task counts per queue, keyed by status and zero-filled across all statuses
|
|
43
|
+
* — `(await stats()).default.queued` is the backlog of a queue. */
|
|
44
|
+
stats(): Promise<Record<string, Record<TaskStatus, number>>>;
|
|
37
45
|
wait(taskId: string, opts?: {
|
|
38
46
|
timeoutMs?: number;
|
|
39
47
|
pollMs?: number;
|
package/dist/client.js
CHANGED
|
@@ -51,6 +51,18 @@ export class CairnQ {
|
|
|
51
51
|
retryByKey(key, opts) {
|
|
52
52
|
return this._store.retryByKey(key, opts);
|
|
53
53
|
}
|
|
54
|
+
/** Delete terminal tasks that finished more than `olderThanMs` ago and return
|
|
55
|
+
* their ids. Nothing else in CairnQ removes rows, so a long-lived database
|
|
56
|
+
* needs this on a schedule. Each call is bounded by `limit` to keep the write
|
|
57
|
+
* short; loop until it returns fewer than `limit`. */
|
|
58
|
+
purge(input) {
|
|
59
|
+
return this._store.purge(input);
|
|
60
|
+
}
|
|
61
|
+
/** Task counts per queue, keyed by status and zero-filled across all statuses
|
|
62
|
+
* — `(await stats()).default.queued` is the backlog of a queue. */
|
|
63
|
+
stats() {
|
|
64
|
+
return this._store.stats();
|
|
65
|
+
}
|
|
54
66
|
wait(taskId, opts = {}) {
|
|
55
67
|
return pollWait(this._store, taskId, {
|
|
56
68
|
timeoutMs: opts.timeoutMs ?? 30_000,
|
package/dist/context.d.ts
CHANGED
|
@@ -8,6 +8,9 @@ export declare class TaskContext {
|
|
|
8
8
|
private readonly task;
|
|
9
9
|
readonly workerId: string;
|
|
10
10
|
private readonly leaseMs;
|
|
11
|
+
private readonly abort;
|
|
12
|
+
private leaseLost;
|
|
13
|
+
private cancelSeen;
|
|
11
14
|
constructor(store: TaskStore, task: Task, workerId: string, leaseMs: number);
|
|
12
15
|
get taskId(): string;
|
|
13
16
|
get name(): string;
|
|
@@ -17,9 +20,22 @@ export declare class TaskContext {
|
|
|
17
20
|
get rootId(): string | null;
|
|
18
21
|
get correlationId(): string | null;
|
|
19
22
|
get payload(): any;
|
|
23
|
+
/**
|
|
24
|
+
* True once this worker has lost the task's lease — it expired and another
|
|
25
|
+
* worker reclaimed it. Nothing this handler writes will be recorded any more
|
|
26
|
+
* and the task is already running elsewhere, so a long handler should check
|
|
27
|
+
* this (or `signal`) and bail out instead of continuing to do side effects.
|
|
28
|
+
*/
|
|
29
|
+
get lostLease(): boolean;
|
|
30
|
+
/** Aborts when the lease is lost. Pass it to fetch / any AbortSignal-aware API. */
|
|
31
|
+
get signal(): AbortSignal;
|
|
32
|
+
/** @internal Called by the worker when an owned write reports a lost lease. */
|
|
33
|
+
markLeaseLost(): void;
|
|
34
|
+
private observe;
|
|
35
|
+
private owned;
|
|
20
36
|
progress(value: number | null, message?: string | null): Promise<Task>;
|
|
21
37
|
heartbeat(): Promise<Task>;
|
|
22
|
-
/** Cooperative cancel check. */
|
|
38
|
+
/** Cooperative cancel check. Free once a heartbeat has already seen the flag. */
|
|
23
39
|
canceled(): Promise<boolean>;
|
|
24
40
|
/** Submit a child task; parent/root/correlation are wired automatically. */
|
|
25
41
|
submit(name: string, payload?: unknown, opts?: SubmitOptions): Promise<Task>;
|
package/dist/context.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { LostLease } from "./errors.js";
|
|
1
2
|
import { cancelRequested } from "./models.js";
|
|
2
3
|
import { taskName } from "./task.js";
|
|
3
4
|
import { pollWait } from "./wait.js";
|
|
@@ -7,6 +8,11 @@ export class TaskContext {
|
|
|
7
8
|
task;
|
|
8
9
|
workerId;
|
|
9
10
|
leaseMs;
|
|
11
|
+
abort = new AbortController();
|
|
12
|
+
leaseLost = false;
|
|
13
|
+
// Cancellation is monotonic: once the DB has told us a cancel was requested it
|
|
14
|
+
// can't be taken back, so canceled() can answer from this without a re-read.
|
|
15
|
+
cancelSeen = false;
|
|
10
16
|
constructor(store, task, workerId, leaseMs) {
|
|
11
17
|
this.store = store;
|
|
12
18
|
this.task = task;
|
|
@@ -37,27 +43,75 @@ export class TaskContext {
|
|
|
37
43
|
get payload() {
|
|
38
44
|
return this.task.payload;
|
|
39
45
|
}
|
|
46
|
+
/**
|
|
47
|
+
* True once this worker has lost the task's lease — it expired and another
|
|
48
|
+
* worker reclaimed it. Nothing this handler writes will be recorded any more
|
|
49
|
+
* and the task is already running elsewhere, so a long handler should check
|
|
50
|
+
* this (or `signal`) and bail out instead of continuing to do side effects.
|
|
51
|
+
*/
|
|
52
|
+
get lostLease() {
|
|
53
|
+
return this.leaseLost;
|
|
54
|
+
}
|
|
55
|
+
/** Aborts when the lease is lost. Pass it to fetch / any AbortSignal-aware API. */
|
|
56
|
+
get signal() {
|
|
57
|
+
return this.abort.signal;
|
|
58
|
+
}
|
|
59
|
+
/** @internal Called by the worker when an owned write reports a lost lease. */
|
|
60
|
+
markLeaseLost() {
|
|
61
|
+
if (this.leaseLost)
|
|
62
|
+
return;
|
|
63
|
+
this.leaseLost = true;
|
|
64
|
+
this.abort.abort(new LostLease(this.task.id));
|
|
65
|
+
}
|
|
66
|
+
// Every owned write returns the current row, so cancellation and lease loss
|
|
67
|
+
// ride along on writes the handler was making anyway.
|
|
68
|
+
observe(task) {
|
|
69
|
+
if (cancelRequested(task))
|
|
70
|
+
this.cancelSeen = true;
|
|
71
|
+
return task;
|
|
72
|
+
}
|
|
73
|
+
async owned(write) {
|
|
74
|
+
// Short-circuit once the lease is known lost: nothing this context writes
|
|
75
|
+
// may be recorded any more. Locally, not just via the store's ownership
|
|
76
|
+
// check — after an abandoned (timed-out) attempt the same worker may
|
|
77
|
+
// re-claim this task under the same workerId, and a zombie handler's write
|
|
78
|
+
// would then pass ownership against the NEW attempt.
|
|
79
|
+
if (this.leaseLost)
|
|
80
|
+
throw new LostLease(this.task.id);
|
|
81
|
+
try {
|
|
82
|
+
return this.observe(await write());
|
|
83
|
+
}
|
|
84
|
+
catch (err) {
|
|
85
|
+
if (err instanceof LostLease)
|
|
86
|
+
this.markLeaseLost();
|
|
87
|
+
throw err;
|
|
88
|
+
}
|
|
89
|
+
}
|
|
40
90
|
async progress(value, message = null) {
|
|
41
|
-
return this.store.progress({
|
|
91
|
+
return this.owned(() => this.store.progress({
|
|
42
92
|
taskId: this.task.id,
|
|
43
93
|
workerId: this.workerId,
|
|
44
94
|
progress: value,
|
|
45
95
|
message,
|
|
46
|
-
});
|
|
96
|
+
}));
|
|
47
97
|
}
|
|
48
98
|
async heartbeat() {
|
|
49
|
-
return this.store.heartbeat({
|
|
99
|
+
return this.owned(() => this.store.heartbeat({
|
|
50
100
|
taskId: this.task.id,
|
|
51
101
|
workerId: this.workerId,
|
|
52
102
|
leaseMs: this.leaseMs,
|
|
53
|
-
});
|
|
103
|
+
}));
|
|
54
104
|
}
|
|
55
|
-
/** Cooperative cancel check. */
|
|
105
|
+
/** Cooperative cancel check. Free once a heartbeat has already seen the flag. */
|
|
56
106
|
async canceled() {
|
|
107
|
+
if (this.cancelSeen)
|
|
108
|
+
return true;
|
|
57
109
|
const t = await this.store.get(this.task.id);
|
|
58
110
|
if (!t)
|
|
59
111
|
return true;
|
|
60
|
-
|
|
112
|
+
if (cancelRequested(t))
|
|
113
|
+
this.cancelSeen = true;
|
|
114
|
+
return this.cancelSeen || t.status === "canceled";
|
|
61
115
|
}
|
|
62
116
|
async submit(task, payload, opts = {}) {
|
|
63
117
|
return this.store.submit({
|
package/dist/errors.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { type Task } from "./models.js";
|
|
1
2
|
/** The single shape of the JSON error envelope (see PROTOCOL.md). Everything that
|
|
2
3
|
* records an error — a handler exception, a missing handler, lease expiry, a thrown
|
|
3
4
|
* TaskError — builds it here, so the contract's fields live in one place. */
|
|
@@ -9,15 +10,23 @@ export declare function errorEnvelope(e: {
|
|
|
9
10
|
details?: Record<string, unknown>;
|
|
10
11
|
}): Record<string, unknown>;
|
|
11
12
|
export declare class CairnQError extends Error {
|
|
13
|
+
constructor(message?: string);
|
|
12
14
|
}
|
|
13
15
|
export declare class AlreadyExists extends CairnQError {
|
|
14
16
|
key: string;
|
|
15
17
|
constructor(key: string);
|
|
16
18
|
}
|
|
17
|
-
/** wait/call did not reach a terminal status in time. The task keeps running.
|
|
19
|
+
/** wait/call did not reach a terminal status in time. The task keeps running.
|
|
20
|
+
* `task` is the last snapshot wait() observed (null if get() found nothing), and
|
|
21
|
+
* the message says what state it was stuck in — a queued-never-claimed task is
|
|
22
|
+
* the classic first-run failure (no worker, no handler, wrong queue or file). */
|
|
18
23
|
export declare class TaskTimeout extends CairnQError {
|
|
19
24
|
taskId: string;
|
|
20
|
-
|
|
25
|
+
readonly task: Task | null;
|
|
26
|
+
constructor(taskId: string, opts?: {
|
|
27
|
+
timeoutMs?: number;
|
|
28
|
+
task?: Task | null;
|
|
29
|
+
});
|
|
21
30
|
}
|
|
22
31
|
/** A waited-on task ended in `failed`. The envelope's fields are unpacked onto the
|
|
23
32
|
* error — read `e.code` / `e.message` / `e.retryable` / `e.details` instead of
|
|
@@ -42,6 +51,13 @@ export declare class LostLease extends CairnQError {
|
|
|
42
51
|
export declare class ProtocolVersionMismatch extends CairnQError {
|
|
43
52
|
constructor(message: string);
|
|
44
53
|
}
|
|
54
|
+
/** A value could not be encoded for a protocol JSON column (non-finite number,
|
|
55
|
+
* BigInt, circular structure, …). Raised at the boundary — submit rejects with
|
|
56
|
+
* it, and a worker records a handler result that triggers it as a permanent
|
|
57
|
+
* `unserializable_result` failure. The Python SDK raises the same named error. */
|
|
58
|
+
export declare class SerializationError extends CairnQError {
|
|
59
|
+
constructor(message: string);
|
|
60
|
+
}
|
|
45
61
|
/** Throw inside a handler to control how the failure is recorded. Defaults to
|
|
46
62
|
* non-retryable so deterministic errors fail fast instead of burning retries.
|
|
47
63
|
* Any other thrown value is treated as retryable. */
|
package/dist/errors.js
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { nowMs } from "./ids.js";
|
|
2
|
+
import { cancelRequested, isQueued } from "./models.js";
|
|
1
3
|
/** The single shape of the JSON error envelope (see PROTOCOL.md). Everything that
|
|
2
4
|
* records an error — a handler exception, a missing handler, lease expiry, a thrown
|
|
3
5
|
* TaskError — builds it here, so the contract's fields live in one place. */
|
|
@@ -11,6 +13,13 @@ export function errorEnvelope(e) {
|
|
|
11
13
|
};
|
|
12
14
|
}
|
|
13
15
|
export class CairnQError extends Error {
|
|
16
|
+
constructor(message) {
|
|
17
|
+
super(message);
|
|
18
|
+
// Subclasses each set their own; without this a bare CairnQError reports
|
|
19
|
+
// "Error", and `err.name` is how callers (and the conformance runner) tell
|
|
20
|
+
// one apart from another.
|
|
21
|
+
this.name = "CairnQError";
|
|
22
|
+
}
|
|
14
23
|
}
|
|
15
24
|
export class AlreadyExists extends CairnQError {
|
|
16
25
|
key;
|
|
@@ -20,13 +29,40 @@ export class AlreadyExists extends CairnQError {
|
|
|
20
29
|
this.name = "AlreadyExists";
|
|
21
30
|
}
|
|
22
31
|
}
|
|
23
|
-
/**
|
|
32
|
+
/** One line of "why hasn't this finished" from the last snapshot wait()
|
|
33
|
+
* observed. No worker running, no handler for the name, wrong queue, and two
|
|
34
|
+
* processes on different database files all look identical from the API side —
|
|
35
|
+
* queued, never claimed — so that case names the likely causes. */
|
|
36
|
+
function timeoutDetail(task) {
|
|
37
|
+
if (!task)
|
|
38
|
+
return "task not found — wrong database file, or already purged?";
|
|
39
|
+
if (isQueued(task)) {
|
|
40
|
+
const delayMs = task.run_at_ms - nowMs();
|
|
41
|
+
if (task.attempt === 0 && delayMs <= 0) {
|
|
42
|
+
return (`never claimed by a worker — is a worker running with a handler for ` +
|
|
43
|
+
`'${task.name}' on queue '${task.queue}', against this same database?`);
|
|
44
|
+
}
|
|
45
|
+
const next = delayMs > 0 ? `, next run in ~${delayMs}ms` : "";
|
|
46
|
+
return `still queued (attempt ${task.attempt}/${task.max_attempts})${next}`;
|
|
47
|
+
}
|
|
48
|
+
if (cancelRequested(task))
|
|
49
|
+
return "cancel requested, waiting for the handler to observe it";
|
|
50
|
+
return `still running (attempt ${task.attempt}/${task.max_attempts})`;
|
|
51
|
+
}
|
|
52
|
+
/** wait/call did not reach a terminal status in time. The task keeps running.
|
|
53
|
+
* `task` is the last snapshot wait() observed (null if get() found nothing), and
|
|
54
|
+
* the message says what state it was stuck in — a queued-never-claimed task is
|
|
55
|
+
* the classic first-run failure (no worker, no handler, wrong queue or file). */
|
|
24
56
|
export class TaskTimeout extends CairnQError {
|
|
25
57
|
taskId;
|
|
26
|
-
|
|
27
|
-
|
|
58
|
+
task;
|
|
59
|
+
constructor(taskId, opts = {}) {
|
|
60
|
+
super(opts.timeoutMs == null
|
|
61
|
+
? `task ${taskId} did not finish in time`
|
|
62
|
+
: `task ${taskId} did not finish within ${opts.timeoutMs}ms: ${timeoutDetail(opts.task ?? null)}`);
|
|
28
63
|
this.taskId = taskId;
|
|
29
64
|
this.name = "TaskTimeout";
|
|
65
|
+
this.task = opts.task ?? null;
|
|
30
66
|
}
|
|
31
67
|
}
|
|
32
68
|
/** A waited-on task ended in `failed`. The envelope's fields are unpacked onto the
|
|
@@ -72,6 +108,16 @@ export class ProtocolVersionMismatch extends CairnQError {
|
|
|
72
108
|
this.name = "ProtocolVersionMismatch";
|
|
73
109
|
}
|
|
74
110
|
}
|
|
111
|
+
/** A value could not be encoded for a protocol JSON column (non-finite number,
|
|
112
|
+
* BigInt, circular structure, …). Raised at the boundary — submit rejects with
|
|
113
|
+
* it, and a worker records a handler result that triggers it as a permanent
|
|
114
|
+
* `unserializable_result` failure. The Python SDK raises the same named error. */
|
|
115
|
+
export class SerializationError extends CairnQError {
|
|
116
|
+
constructor(message) {
|
|
117
|
+
super(message);
|
|
118
|
+
this.name = "SerializationError";
|
|
119
|
+
}
|
|
120
|
+
}
|
|
75
121
|
/** Throw inside a handler to control how the failure is recorded. Defaults to
|
|
76
122
|
* non-retryable so deterministic errors fail fast instead of burning retries.
|
|
77
123
|
* Any other thrown value is treated as retryable. */
|
package/dist/index.d.ts
CHANGED
|
@@ -7,7 +7,8 @@ export { defineTask } from "./task.js";
|
|
|
7
7
|
export type { TaskDef } from "./task.js";
|
|
8
8
|
export { SQLiteStore } from "./store/sqlite.js";
|
|
9
9
|
export { PostgresStore } from "./store/postgres.js";
|
|
10
|
-
export
|
|
10
|
+
export { TaskStore } from "./store/base.js";
|
|
11
|
+
export type { ListInput, PurgeInput, SubmitInput, Conflict } from "./store/base.js";
|
|
11
12
|
export type { Task, TaskStatus } from "./models.js";
|
|
12
13
|
export { STATUSES, isTerminal, cancelRequested, isQueued, isRunning, isSucceeded, isFailed, isCanceled, } from "./models.js";
|
|
13
|
-
export { CairnQError, AlreadyExists, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, } from "./errors.js";
|
|
14
|
+
export { CairnQError, AlreadyExists, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, SerializationError, } from "./errors.js";
|
package/dist/index.js
CHANGED
|
@@ -4,5 +4,6 @@ export { TaskContext } from "./context.js";
|
|
|
4
4
|
export { defineTask } from "./task.js";
|
|
5
5
|
export { SQLiteStore } from "./store/sqlite.js";
|
|
6
6
|
export { PostgresStore } from "./store/postgres.js";
|
|
7
|
+
export { TaskStore } from "./store/base.js";
|
|
7
8
|
export { STATUSES, isTerminal, cancelRequested, isQueued, isRunning, isSucceeded, isFailed, isCanceled, } from "./models.js";
|
|
8
|
-
export { CairnQError, AlreadyExists, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, } from "./errors.js";
|
|
9
|
+
export { CairnQError, AlreadyExists, TaskTimeout, TaskFailed, TaskCanceled, TaskError, LostLease, ProtocolVersionMismatch, SerializationError, } from "./errors.js";
|
package/dist/sql.js
CHANGED
|
@@ -1,19 +1,23 @@
|
|
|
1
1
|
import { existsSync, readdirSync, readFileSync } from "node:fs";
|
|
2
2
|
import { dirname, join } from "node:path";
|
|
3
3
|
import { fileURLToPath } from "node:url";
|
|
4
|
-
// Locate the shared cairnq-protocol dir. Resolution: $CAIRNQ_PROTOCOL_DIR ->
|
|
5
|
-
//
|
|
6
|
-
// (
|
|
7
|
-
// The dir is laid out per-dialect (sql/<dialect>/*.sql,
|
|
8
|
-
// so a second backend (Postgres) slots in beside
|
|
4
|
+
// Locate the shared cairnq-protocol dir. Resolution: $CAIRNQ_PROTOCOL_DIR -> walk
|
|
5
|
+
// up to `cairnq-protocol/` (monorepo dev) -> vendored `_protocol/` next to this
|
|
6
|
+
// module (written at publish time). Both SDKs load the SAME .sql strings
|
|
7
|
+
// (zero-drift guarantee). The dir is laid out per-dialect (sql/<dialect>/*.sql,
|
|
8
|
+
// migrations/<dialect>/*.sql) so a second backend (Postgres) slots in beside
|
|
9
|
+
// sqlite; `dialect` picks the subtree.
|
|
10
|
+
//
|
|
11
|
+
// The source tree wins over the vendored copy on purpose: vendoring is a publish
|
|
12
|
+
// step that also runs locally, and a stale `_protocol/` shadowing the canonical
|
|
13
|
+
// SQL means edits to cairnq-protocol/ are silently not under test. An installed
|
|
14
|
+
// package has no repo above it, so it falls through to the vendored copy.
|
|
9
15
|
export function findProtocolRoot() {
|
|
10
16
|
const env = process.env.CAIRNQ_PROTOCOL_DIR;
|
|
11
17
|
if (env)
|
|
12
18
|
return env;
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
if (existsSync(join(vendored, "sql")))
|
|
16
|
-
return vendored;
|
|
19
|
+
const start = dirname(fileURLToPath(import.meta.url));
|
|
20
|
+
let dir = start;
|
|
17
21
|
for (let i = 0; i < 10; i++) {
|
|
18
22
|
const candidate = join(dir, "cairnq-protocol");
|
|
19
23
|
if (existsSync(join(candidate, "sql")))
|
|
@@ -23,6 +27,9 @@ export function findProtocolRoot() {
|
|
|
23
27
|
break;
|
|
24
28
|
dir = parent;
|
|
25
29
|
}
|
|
30
|
+
const vendored = join(start, "_protocol");
|
|
31
|
+
if (existsSync(join(vendored, "sql")))
|
|
32
|
+
return vendored;
|
|
26
33
|
throw new Error("cannot locate cairnq-protocol; set CAIRNQ_PROTOCOL_DIR");
|
|
27
34
|
}
|
|
28
35
|
export function loadStatements(dialect = "sqlite", root = findProtocolRoot()) {
|