@ultimat3/jobs 1.2.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +567 -0
- package/README.md +355 -16
- package/package.json +7 -5
- package/src/backfill-gate.ts +97 -0
- package/src/backfill-inspect.ts +73 -0
- package/src/backfill-ledger.ts +183 -0
- package/src/backfill-pass.ts +276 -0
- package/src/backfill-pending.ts +131 -0
- package/src/backfill-rate.ts +109 -0
- package/src/backfill-registry.ts +108 -0
- package/src/backfill-scope.ts +70 -0
- package/src/backfill.ts +213 -0
- package/src/driver-memory.ts +61 -9
- package/src/driver-nats.ts +2 -1
- package/src/driver-pg-ddl.ts +180 -0
- package/src/driver-pg-rows.ts +123 -0
- package/src/driver-pg-sql.ts +251 -55
- package/src/driver-pg.ts +138 -92
- package/src/driver-redis.ts +2 -1
- package/src/driver.ts +91 -7
- package/src/errors.ts +290 -4
- package/src/events-pg.ts +121 -0
- package/src/events.ts +7 -1
- package/src/execute.ts +289 -0
- package/src/heartbeat.ts +146 -0
- package/src/index.ts +121 -27
- package/src/inspect.ts +43 -2
- package/src/job.ts +72 -2
- package/src/leases.ts +90 -0
- package/src/limits.ts +0 -0
- package/src/metrics.ts +35 -0
- package/src/outbox-pg.ts +137 -0
- package/src/outbox.ts +115 -53
- package/src/register.ts +1 -1
- package/src/retry.ts +6 -1
- package/src/run-signal.ts +50 -0
- package/src/scheduler-pg.ts +103 -0
- package/src/scheduler.ts +159 -245
- package/src/steps.ts +155 -31
- package/src/task.ts +229 -0
- package/src/tenant.ts +61 -0
- package/src/worker-fleet-slots.ts +124 -0
- package/src/worker-run.ts +132 -0
- package/src/worker.ts +207 -190
package/src/inspect.ts
CHANGED
|
@@ -2,13 +2,15 @@
|
|
|
2
2
|
// object so `x jobs ... --json` and the MCP tool share one shape — an agent debugging a stuck
|
|
3
3
|
// queue reads exactly what the dashboard renders.
|
|
4
4
|
|
|
5
|
+
import type { BackfillProgress } from './backfill-inspect';
|
|
6
|
+
import { backfillForRun } from './backfill-inspect';
|
|
5
7
|
import type { JobDriver, JobFilter, JobRecord, QueueStats } from './driver';
|
|
6
|
-
import { JobsNotImplementedError } from './errors';
|
|
8
|
+
import { CancelUnsupportedError, JobNotCancellableError, JobsNotImplementedError } from './errors';
|
|
7
9
|
import { registeredJobs } from './job';
|
|
8
10
|
import { retrySchedule } from './retry';
|
|
9
11
|
import type { Scheduler } from './scheduler';
|
|
10
|
-
import { registeredTasks } from './scheduler';
|
|
11
12
|
import type { StepRecord } from './steps';
|
|
13
|
+
import { registeredTasks } from './task';
|
|
12
14
|
|
|
13
15
|
export interface QueueDepthReport {
|
|
14
16
|
readonly driver: string;
|
|
@@ -67,9 +69,19 @@ export interface JobTrace {
|
|
|
67
69
|
readonly runAt: string;
|
|
68
70
|
readonly lastError: string | null;
|
|
69
71
|
readonly tenantId: string | null;
|
|
72
|
+
/** W3C `traceparent` of the request that queued it — paste it into the trace viewer. */
|
|
73
|
+
readonly traceparent: string | null;
|
|
74
|
+
/** Actor id of whoever asked. Audit only: the body ran with system authority. */
|
|
75
|
+
readonly enqueuedBy: string | null;
|
|
70
76
|
readonly steps: readonly StepTrace[];
|
|
71
77
|
/** Remaining retry delays in ms, jitter excluded. */
|
|
72
78
|
readonly retryDelaysMs: readonly number[];
|
|
79
|
+
/**
|
|
80
|
+
* The `x_backfills` row this run wrote, when the job is a `backfill()`. `null` for every other
|
|
81
|
+
* job and for a driver with no ledger — a step trace says which batch is next, and this says how
|
|
82
|
+
* many rows the pass has actually put behind it.
|
|
83
|
+
*/
|
|
84
|
+
readonly backfill: BackfillProgress | null;
|
|
73
85
|
}
|
|
74
86
|
|
|
75
87
|
const iso = (ms: number | undefined): string | null =>
|
|
@@ -103,6 +115,7 @@ export async function inspectJob(driver: JobDriver, jobId: string): Promise<JobT
|
|
|
103
115
|
if (record === undefined) return undefined;
|
|
104
116
|
const steps = await driver.steps.list(record.runId);
|
|
105
117
|
const handle = registeredJobs().find((candidate) => candidate.name === record.name);
|
|
118
|
+
const backfill = await backfillForRun(driver, record.runId);
|
|
106
119
|
return {
|
|
107
120
|
id: record.id,
|
|
108
121
|
name: record.name,
|
|
@@ -115,8 +128,11 @@ export async function inspectJob(driver: JobDriver, jobId: string): Promise<JobT
|
|
|
115
128
|
runAt: new Date(record.runAt).toISOString(),
|
|
116
129
|
lastError: record.lastError ?? null,
|
|
117
130
|
tenantId: record.tenantId ?? null,
|
|
131
|
+
traceparent: record.traceparent ?? null,
|
|
132
|
+
enqueuedBy: record.enqueuedBy ?? null,
|
|
118
133
|
steps: steps.map(toStepTrace),
|
|
119
134
|
retryDelaysMs: handle === undefined ? [] : [...retrySchedule(handle.retry)],
|
|
135
|
+
backfill: backfill ?? null,
|
|
120
136
|
};
|
|
121
137
|
}
|
|
122
138
|
|
|
@@ -169,6 +185,31 @@ export async function retryFromStep(
|
|
|
169
185
|
return inspectJob(driver, jobId);
|
|
170
186
|
}
|
|
171
187
|
|
|
188
|
+
/**
|
|
189
|
+
* Stop a job from outside — the surface `x jobs cancel <id>` binds to. The hard half was already
|
|
190
|
+
* built: `execute.ts` cancels the attempt and `steps.ts` fences every write behind that signal.
|
|
191
|
+
* What was missing was anything that could TRIGGER it, so a runaway backfill against production
|
|
192
|
+
* left two options: scale the worker to zero (stopping every job) or `UPDATE x_jobs` by hand,
|
|
193
|
+
* which the worker's next ack overwrote because `SQL_ACK` had no state guard.
|
|
194
|
+
*
|
|
195
|
+
* Refuses loudly rather than answering "nothing happened": an operator cancelling a 40M-row sweep
|
|
196
|
+
* has to know whether they stopped it or missed it.
|
|
197
|
+
*/
|
|
198
|
+
export async function cancelJob(
|
|
199
|
+
driver: JobDriver,
|
|
200
|
+
jobId: string,
|
|
201
|
+
reason?: string,
|
|
202
|
+
): Promise<JobTrace | undefined> {
|
|
203
|
+
const introspect = requireIntrospection(driver);
|
|
204
|
+
if (introspect.cancel === undefined) throw new CancelUnsupportedError({ driver: driver.name });
|
|
205
|
+
const record = await introspect.cancel(jobId, reason);
|
|
206
|
+
if (record === undefined) {
|
|
207
|
+
const current = await introspect.job(jobId);
|
|
208
|
+
throw new JobNotCancellableError({ jobId, state: current?.state ?? 'missing' });
|
|
209
|
+
}
|
|
210
|
+
return inspectJob(driver, jobId);
|
|
211
|
+
}
|
|
212
|
+
|
|
172
213
|
export interface JobsManifest {
|
|
173
214
|
readonly jobs: readonly {
|
|
174
215
|
readonly name: string;
|
package/src/job.ts
CHANGED
|
@@ -24,10 +24,18 @@ import { jobsFacade } from './outbox';
|
|
|
24
24
|
import type { RetryPolicy } from './retry';
|
|
25
25
|
import { DEFAULT_RETRY } from './retry';
|
|
26
26
|
import type { StepApi } from './steps';
|
|
27
|
+
import type { JobTenant } from './tenant';
|
|
28
|
+
import { assertJobTenant, jobTenantFor } from './tenant';
|
|
27
29
|
|
|
28
30
|
export interface JobRunArgs<I> {
|
|
29
31
|
readonly input: I;
|
|
30
32
|
readonly step: StepApi;
|
|
33
|
+
/**
|
|
34
|
+
* `ctx.signal` aborts when this attempt's `timeout` passes — the same seam an action reads, so
|
|
35
|
+
* `throwIfAborted(ctx)` and `fetch(url, { signal: ctx.signal })` work here unchanged. Past it
|
|
36
|
+
* the run belongs to whoever claims it next and `step.run` refuses to write, so a loop that
|
|
37
|
+
* never checks it is a body running beside its own retry.
|
|
38
|
+
*/
|
|
31
39
|
readonly ctx: Ctx;
|
|
32
40
|
/** 1-based. Assume at-least-once: never branch on `attempt === 1` for correctness. */
|
|
33
41
|
readonly attempt: number;
|
|
@@ -44,9 +52,32 @@ export interface JobDefinition<I> {
|
|
|
44
52
|
readonly input: StandardSchemaV1<unknown, I>;
|
|
45
53
|
/** REQUIRED. See the file header — this is the whole point. */
|
|
46
54
|
readonly idempotencyKey: (input: I) => string;
|
|
55
|
+
/**
|
|
56
|
+
* REQUIRED, and the org this job's body runs under. `tenant: (input) => input.orgId` derives it
|
|
57
|
+
* from the payload; `tenant: 'none'` says this job belongs to no tenant, and then every
|
|
58
|
+
* tenant-scoped read inside it fails closed with `X_TENANCY_ACTOR_ORG_REQUIRED`.
|
|
59
|
+
*
|
|
60
|
+
* There is no default, because both candidates are wrong. Until this field existed the worker ran
|
|
61
|
+
* a body with no ambient context at all, so `@ultimat3/entity`'s tenant guard — which derives
|
|
62
|
+
* from `tryUseContext()` and not from the ctx it is handed — read no actor, added no predicate
|
|
63
|
+
* and accepted a caller-named `orgId` unchecked: the same write refused over HTTP as
|
|
64
|
+
* `X_TENANCY_ACTOR_MISMATCH` was ACCEPTED through the job surface. A boot-supplied service actor
|
|
65
|
+
* would close that with ONE identity for every job, which is a cross-tenant read waiting for the
|
|
66
|
+
* first job that takes an org id in its input. So the job declares it, per job, from its own
|
|
67
|
+
* payload — the value an author already had to pass anyway.
|
|
68
|
+
*/
|
|
69
|
+
readonly tenant: JobTenant<I>;
|
|
47
70
|
readonly retry: RetryPolicy;
|
|
48
71
|
readonly queue?: string;
|
|
49
|
-
/**
|
|
72
|
+
/**
|
|
73
|
+
* Max in-flight runs of THIS job across the fleet. Omit for the queue-wide cap.
|
|
74
|
+
*
|
|
75
|
+
* Enforced by `JobDriver.leases` — a row every replica can see — and NOT by `limits.ts`, which
|
|
76
|
+
* counts one process's heap. A driver with no lease store cannot hold this cap, so
|
|
77
|
+
* `createWorker().start()` refuses to boot rather than let it pass silently
|
|
78
|
+
* (`X_JOB_CONCURRENCY_UNENFORCEABLE`): this field was declared, documented and in the manifest
|
|
79
|
+
* while nothing read it, which is exactly what axiom 3 exists to prevent.
|
|
80
|
+
*/
|
|
50
81
|
readonly concurrency?: number;
|
|
51
82
|
readonly timeout?: DurationInput;
|
|
52
83
|
run(args: JobRunArgs<I>): Promise<unknown>;
|
|
@@ -58,6 +89,18 @@ export interface JobDefinition<I> {
|
|
|
58
89
|
*/
|
|
59
90
|
export interface JobActor {
|
|
60
91
|
readonly orgId?: string | undefined;
|
|
92
|
+
/**
|
|
93
|
+
* The actor's id, recorded as `enqueuedBy` — ATTRIBUTION, never authority.
|
|
94
|
+
*
|
|
95
|
+
* The framework picks one answer to "whose permissions does a job run with" (axiom 1) and it is
|
|
96
|
+
* this: a job body runs with SYSTEM authority and this is an audit column. Impersonating the
|
|
97
|
+
* enqueuer at claim time is the defensible alternative and is rejected for one reason — a job
|
|
98
|
+
* that sleeps three days, or dead-letters and is retried next quarter, would then act as
|
|
99
|
+
* somebody whose role, org membership or employment has changed since. `02-primitives.md`
|
|
100
|
+
* already frames a job as server-authoritative work. A job that must act FOR a user takes that
|
|
101
|
+
* user's id in its input and re-authorises it in the body, where the check is visible in review.
|
|
102
|
+
*/
|
|
103
|
+
readonly id?: string | undefined;
|
|
61
104
|
}
|
|
62
105
|
|
|
63
106
|
/**
|
|
@@ -74,6 +117,13 @@ export interface JobHandle<I = unknown> {
|
|
|
74
117
|
readonly input: StandardSchemaV1<unknown, I>;
|
|
75
118
|
parse(raw: unknown): I;
|
|
76
119
|
idempotencyKeyFor(input: I): string;
|
|
120
|
+
/**
|
|
121
|
+
* The org THIS payload's run acts under — `undefined` for `tenant: 'none'`. A method and never a
|
|
122
|
+
* `readonly tenant: JobTenant<I>` field: a function-typed property is contravariant in its
|
|
123
|
+
* parameter, so `JobHandle<OrgInput>` would stop being assignable to `AnyJobHandle` and the
|
|
124
|
+
* registry, the worker and a task's enqueue list could no longer hold heterogeneous handles.
|
|
125
|
+
*/
|
|
126
|
+
tenantFor(input: I): string | undefined;
|
|
77
127
|
run(args: JobRunArgs<I>): Promise<unknown>;
|
|
78
128
|
/**
|
|
79
129
|
* Put this job on the queue. Joins the caller's transaction when the app installed the
|
|
@@ -112,15 +162,26 @@ export function job<I>(definition: JobDefinition<I>): JobHandle<I> {
|
|
|
112
162
|
anonymous += 1;
|
|
113
163
|
const name = definition.name ?? `anonymous-job-${anonymous}`;
|
|
114
164
|
|
|
115
|
-
// Runtime
|
|
165
|
+
// Runtime backstops for generated code and JS callers; TS already forbids omitting either.
|
|
116
166
|
if (typeof definition.idempotencyKey !== 'function') {
|
|
117
167
|
throw new IdempotencyRequiredError({ job: name });
|
|
118
168
|
}
|
|
169
|
+
assertJobTenant(name, definition.tenant);
|
|
119
170
|
assert(
|
|
120
171
|
definition.retry.attempts >= 1,
|
|
121
172
|
`job "${name}" needs retry.attempts >= 1, got ${String(definition.retry.attempts)}`,
|
|
122
173
|
`set retry: { attempts: 1 } or higher on job("${name}") — 0 attempts means the job is never executed at all, not that it never retries`,
|
|
123
174
|
);
|
|
175
|
+
// `concurrency: 0` is not "no cap" — it is a fleet slot table that grants nothing.
|
|
176
|
+
// `createFleetSlots.acquire` reads `limit === undefined` as uncapped, so a declared `0` reaches
|
|
177
|
+
// `leases.acquire(key, 0, …)`, answers `false` forever with no log line, and the job is
|
|
178
|
+
// permanently unrunnable. Refused where it is written, the way `createPacer` refuses `rate: 0`.
|
|
179
|
+
assert(
|
|
180
|
+
definition.concurrency === undefined ||
|
|
181
|
+
(Number.isInteger(definition.concurrency) && definition.concurrency >= 1),
|
|
182
|
+
`job "${name}" declares concurrency ${String(definition.concurrency)}, which no worker can ever fill`,
|
|
183
|
+
`set a whole concurrency of 1 or more on job("${name}"), or omit the field for no cap at all`,
|
|
184
|
+
);
|
|
124
185
|
|
|
125
186
|
const handle: JobHandle<I> = {
|
|
126
187
|
kind: 'job',
|
|
@@ -142,6 +203,9 @@ export function job<I>(definition: JobDefinition<I>): JobHandle<I> {
|
|
|
142
203
|
);
|
|
143
204
|
return key;
|
|
144
205
|
},
|
|
206
|
+
tenantFor(input: I): string | undefined {
|
|
207
|
+
return jobTenantFor(name, definition.tenant, input);
|
|
208
|
+
},
|
|
145
209
|
run(args: JobRunArgs<I>): Promise<unknown> {
|
|
146
210
|
return definition.run(args);
|
|
147
211
|
},
|
|
@@ -152,9 +216,15 @@ export function job<I>(definition: JobDefinition<I>): JobHandle<I> {
|
|
|
152
216
|
const tenantId = options.tenantId ?? tenantFor(actor);
|
|
153
217
|
// `NO_TENANT` is the limiter's own bucket for an absent tenant, so leaving the column
|
|
154
218
|
// empty is the same limit and one less fake org id on the row.
|
|
219
|
+
//
|
|
220
|
+
// `enqueuedBy` is the actor's id and NOTHING ELSE crosses: the body runs with system
|
|
221
|
+
// authority, so this is an audit column, not a principal. See `JobActor.id` for why the
|
|
222
|
+
// framework chose attribution over impersonation.
|
|
223
|
+
const enqueuedBy = options.enqueuedBy ?? actor?.id;
|
|
155
224
|
return handle.enqueue(input, {
|
|
156
225
|
...options,
|
|
157
226
|
...(tenantId === NO_TENANT ? {} : { tenantId }),
|
|
227
|
+
...(enqueuedBy === undefined ? {} : { enqueuedBy }),
|
|
158
228
|
});
|
|
159
229
|
},
|
|
160
230
|
// Reads `handle`, never the captured `name`: `nameJobs()` rebinds the property in place.
|
package/src/leases.ts
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
// Fleet-wide slot counting: the seam that makes `job.concurrency` true. `limits.ts` counts slots
|
|
2
|
+
// in ONE process's heap, so `perTenant: 2` on twenty pods is forty concurrent runs — the number a
|
|
3
|
+
// downstream API rate-limits you for. A lease is a row somewhere every replica can see, held for
|
|
4
|
+
// a TTL and renewed by the same heartbeat that renews the visibility lease, so a killed worker
|
|
5
|
+
// gives its slot back by expiry rather than by cleanup nobody runs.
|
|
6
|
+
|
|
7
|
+
import type { Clock } from '@ultimat3/core';
|
|
8
|
+
import { systemClock } from '@ultimat3/core';
|
|
9
|
+
import { nowMs } from './clock';
|
|
10
|
+
|
|
11
|
+
/** A granted slot. `slot` plus `holder` is what renew and release are addressed by. */
|
|
12
|
+
export interface HeldLease {
|
|
13
|
+
readonly key: string;
|
|
14
|
+
readonly slot: number;
|
|
15
|
+
readonly holder: string;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export interface LeaseStore {
|
|
19
|
+
/**
|
|
20
|
+
* Take a slot under `limit` for `key`, or answer `undefined`. Never over-grants; under
|
|
21
|
+
* contention it may refuse a slot that is genuinely free, which costs one poll interval.
|
|
22
|
+
*/
|
|
23
|
+
acquire(
|
|
24
|
+
key: string,
|
|
25
|
+
limit: number,
|
|
26
|
+
ttlMs: number,
|
|
27
|
+
holder: string,
|
|
28
|
+
): Promise<HeldLease | undefined>;
|
|
29
|
+
/** Push the expiry out. `false` means the slot is no longer this holder's. */
|
|
30
|
+
renew(lease: HeldLease, ttlMs: number): Promise<boolean>;
|
|
31
|
+
release(lease: HeldLease): Promise<void>;
|
|
32
|
+
/** Live slots for `key`. Diagnostics only — never the acquire decision, which must be atomic. */
|
|
33
|
+
held(key: string): Promise<number>;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export interface MemoryLeaseStoreOptions {
|
|
37
|
+
readonly clock?: Clock;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* The `x dev` / test implementation. One heap, so it is not a fleet gate — it exists so the
|
|
42
|
+
* memory driver enforces `concurrency` with the same code path the pg driver does, and so the
|
|
43
|
+
* "a second worker is refused" test is a real test rather than a pg-only one.
|
|
44
|
+
*/
|
|
45
|
+
export function createMemoryLeaseStore(options: MemoryLeaseStoreOptions = {}): LeaseStore {
|
|
46
|
+
const clock = options.clock ?? systemClock;
|
|
47
|
+
const slots = new Map<string, Map<number, { holder: string; expiresAt: number }>>();
|
|
48
|
+
|
|
49
|
+
const live = (key: string): Map<number, { holder: string; expiresAt: number }> => {
|
|
50
|
+
const at = nowMs(clock);
|
|
51
|
+
const held = slots.get(key) ?? new Map();
|
|
52
|
+
for (const [slot, entry] of held) if (entry.expiresAt <= at) held.delete(slot);
|
|
53
|
+
slots.set(key, held);
|
|
54
|
+
return held;
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
return {
|
|
58
|
+
acquire(key, limit, ttlMs, holder) {
|
|
59
|
+
const held = live(key);
|
|
60
|
+
for (let slot = 0; slot < limit; slot += 1) {
|
|
61
|
+
if (held.has(slot)) continue;
|
|
62
|
+
held.set(slot, { holder, expiresAt: nowMs(clock) + ttlMs });
|
|
63
|
+
return Promise.resolve({ key, slot, holder });
|
|
64
|
+
}
|
|
65
|
+
return Promise.resolve(undefined);
|
|
66
|
+
},
|
|
67
|
+
renew(lease, ttlMs) {
|
|
68
|
+
const entry = live(lease.key).get(lease.slot);
|
|
69
|
+
if (entry === undefined || entry.holder !== lease.holder) return Promise.resolve(false);
|
|
70
|
+
entry.expiresAt = nowMs(clock) + ttlMs;
|
|
71
|
+
return Promise.resolve(true);
|
|
72
|
+
},
|
|
73
|
+
release(lease) {
|
|
74
|
+
const held = slots.get(lease.key);
|
|
75
|
+
const entry = held?.get(lease.slot);
|
|
76
|
+
// Only the holder gives it back: releasing a slot the TTL already handed to someone else
|
|
77
|
+
// would let two runs share it, which is the whole failure this store exists to prevent.
|
|
78
|
+
if (entry !== undefined && entry.holder === lease.holder) held?.delete(lease.slot);
|
|
79
|
+
return Promise.resolve();
|
|
80
|
+
},
|
|
81
|
+
held(key) {
|
|
82
|
+
return Promise.resolve(live(key).size);
|
|
83
|
+
},
|
|
84
|
+
};
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/** The lease key for a job's own fleet-wide cap. One shape, so pg and memory agree on it. */
|
|
88
|
+
export function jobLeaseKey(jobName: string): string {
|
|
89
|
+
return `job:${jobName}`;
|
|
90
|
+
}
|
package/src/limits.ts
CHANGED
|
Binary file
|
package/src/metrics.ts
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
// The two queue gauges an alert can actually be written against. `queue_depth` and `jobs_total`
|
|
2
|
+
// live in `@ultimat3/core`'s `runtime-metrics.ts` because the deploy chart scales on them; these
|
|
3
|
+
// two are not scaling signals, they are the ones every queue team pages on, and neither existed:
|
|
4
|
+
//
|
|
5
|
+
// "page if the oldest job in payments is older than 5 minutes" — no series carried
|
|
6
|
+
// `oldestReadyMs`, which `QueueStats` computes, `inspectQueues` renders and nothing published.
|
|
7
|
+
// `queue_depth` cannot tell 10 jobs stuck for an hour from 10 enqueued a second ago.
|
|
8
|
+
// "page if the dead-letter queue is not empty" — `jobs_total{outcome="dead"}` is a COUNTER, so
|
|
9
|
+
// a DLQ that filled overnight and stopped growing has a flat rate and alerts on nothing.
|
|
10
|
+
//
|
|
11
|
+
// Declared here rather than in core because they are this package's facts and core is another
|
|
12
|
+
// package's file; if they move next to `queueDepth` later, these two functions are the only
|
|
13
|
+
// callers to redirect.
|
|
14
|
+
|
|
15
|
+
import type { Gauge } from '@ultimat3/core';
|
|
16
|
+
import { gauge } from '@ultimat3/core';
|
|
17
|
+
|
|
18
|
+
/** Seconds and not milliseconds: every Prometheus duration is seconds, and the alert is `> 300`. */
|
|
19
|
+
export const queueOldestReady: Gauge = gauge('queue_oldest_ready_seconds', {
|
|
20
|
+
unit: 's',
|
|
21
|
+
description: 'Age of the oldest claimable job, by queue — 0 when the queue is empty',
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
export const queueDeadJobs: Gauge = gauge('queue_dead_jobs', {
|
|
25
|
+
unit: '{job}',
|
|
26
|
+
description: 'Jobs sitting in the dead-letter state, by queue',
|
|
27
|
+
});
|
|
28
|
+
|
|
29
|
+
export function recordQueueOldestReady(queue: string, oldestReadyMs: number): void {
|
|
30
|
+
queueOldestReady.record(oldestReadyMs / 1000, { queue });
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function recordQueueDeadJobs(queue: string, dead: number): void {
|
|
34
|
+
queueDeadJobs.record(dead, { queue });
|
|
35
|
+
}
|
package/src/outbox-pg.ts
ADDED
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
// The `x_outbox` implementation of `OutboxStore` — the half of the transactional outbox that
|
|
2
|
+
// makes it transactional. `stage()` runs on the CALLER'S OWN connection, so the queue row is in
|
|
3
|
+
// the same Postgres transaction as the business rows and commits or vanishes with them. The
|
|
4
|
+
// memory store hangs rows off the `Tx` object and is correct only inside one process; this is
|
|
5
|
+
// what a deployment installs.
|
|
6
|
+
//
|
|
7
|
+
// `txExecutor` is injected rather than resolved here for the reason `createPgDriver` takes a
|
|
8
|
+
// `PgExecutor`: this package holds no `@ultimat3/db` dependency, and "which connection is this
|
|
9
|
+
// `Tx` on" is a question only boot can answer. Boot has `currentTx()` — a `DbTx` IS a client on
|
|
10
|
+
// the transaction's connection — so the wiring is one line there and no tier crossing here.
|
|
11
|
+
|
|
12
|
+
import type { Clock } from '@ultimat3/core';
|
|
13
|
+
import type { Tx } from '@ultimat3/entity';
|
|
14
|
+
import { nowMs } from './clock';
|
|
15
|
+
import type { PgExecutor } from './driver-pg';
|
|
16
|
+
import {
|
|
17
|
+
SQL_OUTBOX_CLAIM,
|
|
18
|
+
SQL_OUTBOX_MARK_PUBLISHED,
|
|
19
|
+
SQL_OUTBOX_PENDING_COUNT,
|
|
20
|
+
SQL_OUTBOX_STAGE,
|
|
21
|
+
} from './driver-pg-sql';
|
|
22
|
+
import type { OutboxRecord, OutboxStore } from './outbox';
|
|
23
|
+
|
|
24
|
+
interface OutboxRow {
|
|
25
|
+
readonly id: string;
|
|
26
|
+
readonly job: string;
|
|
27
|
+
readonly queue: string;
|
|
28
|
+
readonly input: unknown;
|
|
29
|
+
readonly idempotency_key: string;
|
|
30
|
+
readonly max_attempts: number;
|
|
31
|
+
readonly run_at: number | string;
|
|
32
|
+
readonly staged_at: number | string;
|
|
33
|
+
readonly tenant_id: string | null;
|
|
34
|
+
readonly traceparent: string | null;
|
|
35
|
+
readonly enqueued_by: string | null;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export interface PgOutboxOptions {
|
|
39
|
+
/**
|
|
40
|
+
* The pooled executor the RELAY uses: `claim`, `markPublished` and `pendingCount` all run after
|
|
41
|
+
* the caller's transaction is gone, so they must not be bound to it.
|
|
42
|
+
*/
|
|
43
|
+
readonly executor: PgExecutor;
|
|
44
|
+
/**
|
|
45
|
+
* The executor bound to `tx`'s own connection. Boot supplies it; without it `stage()` would
|
|
46
|
+
* write on a second connection and the outbox would guarantee nothing at all.
|
|
47
|
+
*/
|
|
48
|
+
readonly txExecutor: (tx: Tx) => PgExecutor;
|
|
49
|
+
readonly clock?: Clock;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function toRecord(row: OutboxRow): OutboxRecord {
|
|
53
|
+
return {
|
|
54
|
+
id: row.id,
|
|
55
|
+
job: row.job,
|
|
56
|
+
queue: row.queue,
|
|
57
|
+
input: row.input,
|
|
58
|
+
idempotencyKey: row.idempotency_key,
|
|
59
|
+
maxAttempts: Number(row.max_attempts),
|
|
60
|
+
runAt: Number(row.run_at),
|
|
61
|
+
stagedAt: Number(row.staged_at),
|
|
62
|
+
...(row.tenant_id === null ? {} : { tenantId: row.tenant_id }),
|
|
63
|
+
...(row.traceparent === null ? {} : { traceparent: row.traceparent }),
|
|
64
|
+
...(row.enqueued_by === null ? {} : { enqueuedBy: row.enqueued_by }),
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export function createPgOutboxStore(options: PgOutboxOptions): OutboxStore {
|
|
69
|
+
// What each open transaction has staged, for `commit()`'s return value only. Never the source
|
|
70
|
+
// of truth — that is the row, and the row's fate is the transaction's. A WeakMap so a `Tx` that
|
|
71
|
+
// is neither committed nor rolled back (a process killed mid-request) leaves nothing behind.
|
|
72
|
+
const staged = new WeakMap<object, OutboxRecord[]>();
|
|
73
|
+
const key = (tx: Tx): object => tx as unknown as object;
|
|
74
|
+
|
|
75
|
+
return {
|
|
76
|
+
async stage(tx, record) {
|
|
77
|
+
await options
|
|
78
|
+
.txExecutor(tx)
|
|
79
|
+
.query(SQL_OUTBOX_STAGE, [
|
|
80
|
+
record.id,
|
|
81
|
+
record.job,
|
|
82
|
+
record.queue,
|
|
83
|
+
JSON.stringify(record.input ?? null),
|
|
84
|
+
record.idempotencyKey,
|
|
85
|
+
record.maxAttempts,
|
|
86
|
+
record.runAt,
|
|
87
|
+
record.stagedAt,
|
|
88
|
+
record.tenantId ?? null,
|
|
89
|
+
record.traceparent ?? null,
|
|
90
|
+
record.enqueuedBy ?? null,
|
|
91
|
+
]);
|
|
92
|
+
const bucket = staged.get(key(tx)) ?? [];
|
|
93
|
+
bucket.push(record);
|
|
94
|
+
staged.set(key(tx), bucket);
|
|
95
|
+
},
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* Nothing to do but report. The rows became visible when Postgres committed them — there is
|
|
99
|
+
* no second write here, and that absence IS the guarantee: a commit hook that had to run
|
|
100
|
+
* would be one more thing between the business rows and the job row.
|
|
101
|
+
*/
|
|
102
|
+
commit(tx) {
|
|
103
|
+
const bucket = staged.get(key(tx)) ?? [];
|
|
104
|
+
staged.delete(key(tx));
|
|
105
|
+
return Promise.resolve(bucket);
|
|
106
|
+
},
|
|
107
|
+
|
|
108
|
+
/** Also nothing: the ROLLBACK already took the rows. Only the bookkeeping is ours to drop. */
|
|
109
|
+
rollback(tx) {
|
|
110
|
+
staged.delete(key(tx));
|
|
111
|
+
return Promise.resolve();
|
|
112
|
+
},
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* `for update skip locked` is in the statement, and under autocommit its row locks last only
|
|
116
|
+
* for that statement — so two relays can hand the same row to `enqueue`. That is the
|
|
117
|
+
* at-least-once the whole design already assumes and the idempotency key already collapses;
|
|
118
|
+
* what the clause buys is that two relays running side by side do not serialise on each other.
|
|
119
|
+
*/
|
|
120
|
+
async claim(limit) {
|
|
121
|
+
const rows = await options.executor.query<OutboxRow>(SQL_OUTBOX_CLAIM, [limit]);
|
|
122
|
+
return rows.map(toRecord);
|
|
123
|
+
},
|
|
124
|
+
|
|
125
|
+
async markPublished(id, at) {
|
|
126
|
+
await options.executor.query(SQL_OUTBOX_MARK_PUBLISHED, [id, at || nowMs(options.clock)]);
|
|
127
|
+
},
|
|
128
|
+
|
|
129
|
+
async pendingCount() {
|
|
130
|
+
const rows = await options.executor.query<{ pending: number | string }>(
|
|
131
|
+
SQL_OUTBOX_PENDING_COUNT,
|
|
132
|
+
[],
|
|
133
|
+
);
|
|
134
|
+
return Number(rows[0]?.pending ?? 0);
|
|
135
|
+
},
|
|
136
|
+
};
|
|
137
|
+
}
|