@10x-media/jobs 0.1.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/LICENSE +21 -0
- package/README.md +351 -0
- package/dist/deriveJobStatus-CM_sCsgm.js +72 -0
- package/dist/deriveJobStatus-CM_sCsgm.js.map +1 -0
- package/dist/exports/client.d.ts +49 -0
- package/dist/exports/client.js +573 -0
- package/dist/exports/client.js.map +1 -0
- package/dist/exports/i18n.d.ts +47 -0
- package/dist/exports/i18n.js +3 -0
- package/dist/exports/rsc.d.ts +19 -0
- package/dist/exports/rsc.js +66 -0
- package/dist/exports/rsc.js.map +1 -0
- package/dist/exports/types.d.ts +2 -0
- package/dist/exports/types.js +1 -0
- package/dist/index-CYsbNe4q.d.ts +530 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.js +1622 -0
- package/dist/index.js.map +1 -0
- package/dist/jobStatusMeta-BLBSPUcE.js +41 -0
- package/dist/jobStatusMeta-BLBSPUcE.js.map +1 -0
- package/dist/keys-BYs8GTJy.js +41 -0
- package/dist/keys-BYs8GTJy.js.map +1 -0
- package/dist/server-JsvYf5vy.js +13 -0
- package/dist/server-JsvYf5vy.js.map +1 -0
- package/dist/translations-KyQ3962W.js +61 -0
- package/dist/translations-KyQ3962W.js.map +1 -0
- package/package.json +110 -0
package/dist/index.js
ADDED
|
@@ -0,0 +1,1622 @@
|
|
|
1
|
+
import { t as keys } from "./keys-BYs8GTJy.js";
|
|
2
|
+
import { n as labelForKey } from "./server-JsvYf5vy.js";
|
|
3
|
+
import { t as translations } from "./translations-KyQ3962W.js";
|
|
4
|
+
import { t as deriveJobStatus } from "./deriveJobStatus-CM_sCsgm.js";
|
|
5
|
+
import { ValidationError, definePlugin, getCurrentDate } from "payload";
|
|
6
|
+
import { deepMergeSimple } from "payload/shared";
|
|
7
|
+
import { hostname } from "node:os";
|
|
8
|
+
//#region src/plugin/resolve.ts
|
|
9
|
+
/** Apply an `Override`, falling back to `defaults` when none was provided. */
|
|
10
|
+
const resolve = (override, defaults) => {
|
|
11
|
+
if (override === void 0) return defaults;
|
|
12
|
+
return typeof override === "function" ? override(defaults) : override;
|
|
13
|
+
};
|
|
14
|
+
//#endregion
|
|
15
|
+
//#region src/plugin/registerJobsEnhancements.ts
|
|
16
|
+
const STATUS_CELL = "@10x-media/jobs/client#JobStatusCell";
|
|
17
|
+
const STATUS_HEADER = "@10x-media/jobs/client#JobStatusHeader";
|
|
18
|
+
const RELATIVE_CELL = "@10x-media/jobs/client#RelativeTimeCell";
|
|
19
|
+
const ERROR_PANEL = "@10x-media/jobs/client#JobErrorPanel";
|
|
20
|
+
const LOG_TIMELINE = "@10x-media/jobs/client#JobLogTimeline";
|
|
21
|
+
const DOC_DESCRIPTION = "@10x-media/jobs/client#JobDocDescription";
|
|
22
|
+
const HEALTH_BAR = "@10x-media/jobs/rsc#JobsHealthBar";
|
|
23
|
+
/** Stored field that titles the document: the workflow or task the job runs. */
|
|
24
|
+
const TITLE_FIELD = {
|
|
25
|
+
name: "jobTitle",
|
|
26
|
+
type: "text",
|
|
27
|
+
admin: {
|
|
28
|
+
disableListColumn: true,
|
|
29
|
+
hidden: true
|
|
30
|
+
}
|
|
31
|
+
};
|
|
32
|
+
/**
|
|
33
|
+
* Keep `jobTitle` in sync with the job's workflow or task on every write, falling
|
|
34
|
+
* back to the existing doc so partial state updates never clobber it.
|
|
35
|
+
*/
|
|
36
|
+
const setJobTitle = ({ data, originalDoc }) => {
|
|
37
|
+
const source = {
|
|
38
|
+
...originalDoc ?? {},
|
|
39
|
+
...data
|
|
40
|
+
};
|
|
41
|
+
const workflow = typeof source.workflowSlug === "string" ? source.workflowSlug : void 0;
|
|
42
|
+
const task = typeof source.taskSlug === "string" ? source.taskSlug : void 0;
|
|
43
|
+
const title = workflow || task;
|
|
44
|
+
return title ? {
|
|
45
|
+
...data,
|
|
46
|
+
jobTitle: title
|
|
47
|
+
} : data;
|
|
48
|
+
};
|
|
49
|
+
/** Our default document Field component for specific job fields. */
|
|
50
|
+
const FIELD_COMPONENTS = {
|
|
51
|
+
error: ERROR_PANEL,
|
|
52
|
+
log: LOG_TIMELINE
|
|
53
|
+
};
|
|
54
|
+
const DEFAULT_JOBS_COLUMNS = [
|
|
55
|
+
"workflowSlug",
|
|
56
|
+
"status",
|
|
57
|
+
"queue",
|
|
58
|
+
"totalTried",
|
|
59
|
+
"updatedAt"
|
|
60
|
+
];
|
|
61
|
+
/** Friendlier labels for a few of Payload's default job fields. */
|
|
62
|
+
const FIELD_LABELS = {
|
|
63
|
+
totalTried: labelForKey(keys.fieldAttempts),
|
|
64
|
+
workflowSlug: labelForKey(keys.fieldWorkflow)
|
|
65
|
+
};
|
|
66
|
+
/**
|
|
67
|
+
* Payload appends `createdAt`/`updatedAt` after overrides run, but only when they
|
|
68
|
+
* are absent, so we inject our own to attach the relative-time cell to them.
|
|
69
|
+
*/
|
|
70
|
+
const TIMESTAMP_FIELDS = [{
|
|
71
|
+
name: "createdAt",
|
|
72
|
+
type: "date",
|
|
73
|
+
label: labelForKey(keys.fieldCreated),
|
|
74
|
+
index: true,
|
|
75
|
+
admin: {
|
|
76
|
+
disableBulkEdit: true,
|
|
77
|
+
hidden: true
|
|
78
|
+
}
|
|
79
|
+
}, {
|
|
80
|
+
name: "updatedAt",
|
|
81
|
+
type: "date",
|
|
82
|
+
label: labelForKey(keys.fieldUpdated),
|
|
83
|
+
index: true,
|
|
84
|
+
admin: {
|
|
85
|
+
disableBulkEdit: true,
|
|
86
|
+
hidden: true
|
|
87
|
+
}
|
|
88
|
+
}];
|
|
89
|
+
const fieldName = (field) => "name" in field && typeof field.name === "string" ? field.name : void 0;
|
|
90
|
+
const setCell = (field, cell) => ({
|
|
91
|
+
...field,
|
|
92
|
+
admin: {
|
|
93
|
+
...field.admin,
|
|
94
|
+
components: {
|
|
95
|
+
...field.admin?.components,
|
|
96
|
+
Cell: cell
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
});
|
|
100
|
+
/**
|
|
101
|
+
* Resolve a field's list cell: an explicit `cells` override wins (`false` keeps
|
|
102
|
+
* Payload's default), otherwise date fields get the relative-time cell.
|
|
103
|
+
*/
|
|
104
|
+
const resolveCell = (field, cells) => {
|
|
105
|
+
const name = fieldName(field);
|
|
106
|
+
const override = name ? cells?.[name] : void 0;
|
|
107
|
+
if (override === false) return field;
|
|
108
|
+
if (override) return setCell(field, override);
|
|
109
|
+
if (field.type === "date") return setCell(field, RELATIVE_CELL);
|
|
110
|
+
return field;
|
|
111
|
+
};
|
|
112
|
+
/** Execution-state fields surfaced in the header; hidden from the form, kept as data/columns. */
|
|
113
|
+
const HIDE_ALWAYS = new Set([
|
|
114
|
+
"completedAt",
|
|
115
|
+
"totalTried",
|
|
116
|
+
"hasError",
|
|
117
|
+
"processing",
|
|
118
|
+
"taskStatus"
|
|
119
|
+
]);
|
|
120
|
+
/** Inputs shown only when creating a job; hidden once it exists. */
|
|
121
|
+
const HIDE_ON_EDIT = new Set(["waitUntil"]);
|
|
122
|
+
/** Execution output: read-only in the admin (admin-only, so the runner can still write it). */
|
|
123
|
+
const READONLY_OUTPUTS = new Set(["error", "log"]);
|
|
124
|
+
/** Config inputs: editable on Create, read-only on edit; kept in the sidebar. */
|
|
125
|
+
const READONLY_INPUTS = new Set([
|
|
126
|
+
"input",
|
|
127
|
+
"workflowSlug",
|
|
128
|
+
"taskSlug",
|
|
129
|
+
"queue"
|
|
130
|
+
]);
|
|
131
|
+
/**
|
|
132
|
+
* Lock a field for the read-only-record model. Runner-written fields use
|
|
133
|
+
* admin-only props (`hidden`/`readOnly`) so the jobs runner is never blocked;
|
|
134
|
+
* inputs use field `access.update` so they stay editable on Create but lock once
|
|
135
|
+
* the job exists.
|
|
136
|
+
*/
|
|
137
|
+
const lockField = (field, name) => {
|
|
138
|
+
if (HIDE_ALWAYS.has(name)) return {
|
|
139
|
+
...field,
|
|
140
|
+
admin: {
|
|
141
|
+
...field.admin,
|
|
142
|
+
hidden: true
|
|
143
|
+
}
|
|
144
|
+
};
|
|
145
|
+
if (HIDE_ON_EDIT.has(name)) return {
|
|
146
|
+
...field,
|
|
147
|
+
admin: {
|
|
148
|
+
...field.admin,
|
|
149
|
+
condition: (_data, _siblingData, { operation }) => operation === "create"
|
|
150
|
+
}
|
|
151
|
+
};
|
|
152
|
+
if (READONLY_OUTPUTS.has(name)) return {
|
|
153
|
+
...field,
|
|
154
|
+
admin: {
|
|
155
|
+
...field.admin,
|
|
156
|
+
readOnly: true
|
|
157
|
+
}
|
|
158
|
+
};
|
|
159
|
+
if (READONLY_INPUTS.has(name)) {
|
|
160
|
+
const access = "access" in field && field.access ? field.access : {};
|
|
161
|
+
return {
|
|
162
|
+
...field,
|
|
163
|
+
access: {
|
|
164
|
+
...access,
|
|
165
|
+
update: () => false
|
|
166
|
+
}
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
return field;
|
|
170
|
+
};
|
|
171
|
+
/** Recursively relabel fields, apply cell overrides, and lock the record, descending into tabs. */
|
|
172
|
+
const enhanceFields = (fields, cells, lockRecord) => fields.map((field) => {
|
|
173
|
+
let next = field;
|
|
174
|
+
const name = fieldName(field);
|
|
175
|
+
if (name && FIELD_LABELS[name]) next = {
|
|
176
|
+
...field,
|
|
177
|
+
label: FIELD_LABELS[name]
|
|
178
|
+
};
|
|
179
|
+
next = resolveCell(next, cells);
|
|
180
|
+
if (lockRecord && name) next = lockField(next, name);
|
|
181
|
+
if (name && FIELD_COMPONENTS[name]) next = {
|
|
182
|
+
...next,
|
|
183
|
+
admin: {
|
|
184
|
+
...next.admin,
|
|
185
|
+
components: {
|
|
186
|
+
...next.admin?.components,
|
|
187
|
+
Field: FIELD_COMPONENTS[name]
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
};
|
|
191
|
+
if ("tabs" in next) next = {
|
|
192
|
+
...next,
|
|
193
|
+
tabs: next.tabs.map((tab) => ({
|
|
194
|
+
...tab,
|
|
195
|
+
fields: enhanceFields(tab.fields, cells, lockRecord)
|
|
196
|
+
}))
|
|
197
|
+
};
|
|
198
|
+
else if ("fields" in next) next = {
|
|
199
|
+
...next,
|
|
200
|
+
fields: enhanceFields(next.fields, cells, lockRecord)
|
|
201
|
+
};
|
|
202
|
+
return next;
|
|
203
|
+
});
|
|
204
|
+
/** Build the derived Status column, honoring the `status` override (`false` removes it). */
|
|
205
|
+
const buildStatusField = (status) => {
|
|
206
|
+
if (status === false) return null;
|
|
207
|
+
return {
|
|
208
|
+
name: "status",
|
|
209
|
+
type: "ui",
|
|
210
|
+
label: labelForKey(keys.fieldStatus),
|
|
211
|
+
admin: { components: {
|
|
212
|
+
Cell: status ?? STATUS_CELL,
|
|
213
|
+
Field: STATUS_HEADER
|
|
214
|
+
} }
|
|
215
|
+
};
|
|
216
|
+
};
|
|
217
|
+
/**
|
|
218
|
+
* Enhance Payload's built-in `payload-jobs` collection through its sanctioned
|
|
219
|
+
* `jobs.jobsCollectionOverrides` seam. Composes with a host-provided override
|
|
220
|
+
* (ours layer on top of theirs) so neither clobbers the other, and every piece
|
|
221
|
+
* is replaceable through `options` (`status`, `cells`, `beforeListTable`).
|
|
222
|
+
*/
|
|
223
|
+
const registerJobsEnhancements = (config, options) => {
|
|
224
|
+
const existing = config.jobs?.jobsCollectionOverrides;
|
|
225
|
+
const enhance = ({ defaultJobsCollection }) => {
|
|
226
|
+
const base = existing ? existing({ defaultJobsCollection }) : defaultJobsCollection;
|
|
227
|
+
const existingNames = new Set(base.fields.flatMap((field) => {
|
|
228
|
+
const name = fieldName(field);
|
|
229
|
+
return name ? [name] : [];
|
|
230
|
+
}));
|
|
231
|
+
const timestamps = TIMESTAMP_FIELDS.filter((field) => {
|
|
232
|
+
const name = fieldName(field);
|
|
233
|
+
return name ? !existingNames.has(name) : false;
|
|
234
|
+
}).map((field) => resolveCell(field, options.cells));
|
|
235
|
+
const statusField = buildStatusField(options.status);
|
|
236
|
+
const lockRecord = options.readOnlyRecord !== false;
|
|
237
|
+
const hostBeforeTable = base.admin?.components?.beforeListTable ?? [];
|
|
238
|
+
const ourBeforeTable = options.beforeListTable === false ? [] : options.beforeListTable ?? [{
|
|
239
|
+
path: HEALTH_BAR,
|
|
240
|
+
serverProps: { cap: options.healthBarCap }
|
|
241
|
+
}];
|
|
242
|
+
return {
|
|
243
|
+
...base,
|
|
244
|
+
admin: {
|
|
245
|
+
...base.admin,
|
|
246
|
+
hidden: options.hidden ?? false,
|
|
247
|
+
useAsTitle: "jobTitle",
|
|
248
|
+
defaultColumns: resolve(options.defaultColumns, DEFAULT_JOBS_COLUMNS),
|
|
249
|
+
components: {
|
|
250
|
+
...base.admin?.components,
|
|
251
|
+
Description: DOC_DESCRIPTION,
|
|
252
|
+
beforeListTable: [...hostBeforeTable, ...ourBeforeTable]
|
|
253
|
+
}
|
|
254
|
+
},
|
|
255
|
+
hooks: {
|
|
256
|
+
...base.hooks,
|
|
257
|
+
beforeChange: [...base.hooks?.beforeChange ?? [], setJobTitle]
|
|
258
|
+
},
|
|
259
|
+
fields: [
|
|
260
|
+
...statusField ? [statusField] : [],
|
|
261
|
+
TITLE_FIELD,
|
|
262
|
+
...enhanceFields(base.fields, options.cells, lockRecord),
|
|
263
|
+
...timestamps
|
|
264
|
+
]
|
|
265
|
+
};
|
|
266
|
+
};
|
|
267
|
+
config.jobs = {
|
|
268
|
+
...config.jobs,
|
|
269
|
+
jobsCollectionOverrides: enhance
|
|
270
|
+
};
|
|
271
|
+
};
|
|
272
|
+
//#endregion
|
|
273
|
+
//#region src/plugin/registerTranslations.ts
|
|
274
|
+
/**
|
|
275
|
+
* Merge this plugin's translations into the host config. A host value wins on
|
|
276
|
+
* conflict (`deepMergeSimple` lets the second argument override), so projects can
|
|
277
|
+
* override any string.
|
|
278
|
+
*/
|
|
279
|
+
const registerTranslations = (config) => {
|
|
280
|
+
config.i18n ??= {};
|
|
281
|
+
config.i18n.translations = deepMergeSimple(translations, config.i18n.translations ?? {});
|
|
282
|
+
};
|
|
283
|
+
//#endregion
|
|
284
|
+
//#region src/queueControl/access.ts
|
|
285
|
+
/** The default endpoint guard: logged-in users only (safer than Payload's open default). */
|
|
286
|
+
const loggedInAccess = ({ req }) => Boolean(req.user);
|
|
287
|
+
/**
|
|
288
|
+
* An access checker for serverless cron triggers: a logged-in user passes, otherwise
|
|
289
|
+
* the request must carry `Authorization: Bearer ${process.env[envVar]}`. Payload ships
|
|
290
|
+
* no built-in secret check, so this implements the canonical Vercel-cron pattern.
|
|
291
|
+
*/
|
|
292
|
+
const cronSecretAccess = (options = {}) => {
|
|
293
|
+
const envVar = options.envVar ?? "CRON_SECRET";
|
|
294
|
+
return ({ req }) => {
|
|
295
|
+
if (req.user) return true;
|
|
296
|
+
const secret = process.env[envVar];
|
|
297
|
+
if (!secret) return false;
|
|
298
|
+
return req.headers.get("authorization") === `Bearer ${secret}`;
|
|
299
|
+
};
|
|
300
|
+
};
|
|
301
|
+
//#endregion
|
|
302
|
+
//#region src/queueControl/options.ts
|
|
303
|
+
/** Resolve queue-control options to a fully-defaulted object, or `null` when disabled. */
|
|
304
|
+
const resolveQueueControlOptions = (options) => {
|
|
305
|
+
if (options === void 0 || options === false) return null;
|
|
306
|
+
return {
|
|
307
|
+
access: options.access ?? loggedInAccess,
|
|
308
|
+
queues: options.queues ?? ["default"]
|
|
309
|
+
};
|
|
310
|
+
};
|
|
311
|
+
//#endregion
|
|
312
|
+
//#region src/reliability/leaseLogic.ts
|
|
313
|
+
/** The expiry instant for a lease acquired/renewed at `now` for `ttlMs`. */
|
|
314
|
+
const leaseExpiry = (now, ttlMs) => new Date(now.getTime() + ttlMs);
|
|
315
|
+
//#endregion
|
|
316
|
+
//#region src/reliability/jobLeaseStore.mongo.ts
|
|
317
|
+
const JOBS_SLUG$4 = "payload-jobs";
|
|
318
|
+
/** Reach the raw Mongoose model the adapter exposes on `db.collections[slug]`. */
|
|
319
|
+
const model$1 = (payload) => {
|
|
320
|
+
const found = payload.db.collections[JOBS_SLUG$4];
|
|
321
|
+
if (!found) throw new Error(`@10x-media/jobs: missing Mongoose model for "${JOBS_SLUG$4}"`);
|
|
322
|
+
return found;
|
|
323
|
+
};
|
|
324
|
+
/** The stale-orphan predicate shared by requeue and dead-letter. */
|
|
325
|
+
const stale = (now, fallbackMs) => ({
|
|
326
|
+
hasError: { $ne: true },
|
|
327
|
+
processing: true,
|
|
328
|
+
$or: [{ leaseExpiresAt: { $lt: now } }, {
|
|
329
|
+
leaseExpiresAt: null,
|
|
330
|
+
updatedAt: { $lt: new Date(now.getTime() - fallbackMs) }
|
|
331
|
+
}]
|
|
332
|
+
});
|
|
333
|
+
/** Single-document `findOneAndUpdate` is an atomic compare-and-set on MongoDB. */
|
|
334
|
+
const createMongoJobLeaseStore = (payload) => {
|
|
335
|
+
const m = model$1(payload);
|
|
336
|
+
return {
|
|
337
|
+
deadLetter: async ({ error, fallbackMs, jobId, now }) => {
|
|
338
|
+
return { ok: await m.findOneAndUpdate({
|
|
339
|
+
_id: jobId,
|
|
340
|
+
...stale(now, fallbackMs)
|
|
341
|
+
}, {
|
|
342
|
+
$inc: { fenceToken: 1 },
|
|
343
|
+
$set: {
|
|
344
|
+
claimedBy: null,
|
|
345
|
+
error,
|
|
346
|
+
hasError: true,
|
|
347
|
+
leaseExpiresAt: null,
|
|
348
|
+
processing: false
|
|
349
|
+
}
|
|
350
|
+
}, { new: true }) !== null };
|
|
351
|
+
},
|
|
352
|
+
read: async (jobId) => {
|
|
353
|
+
const doc = await m.findOne({ _id: jobId }).lean();
|
|
354
|
+
if (doc === null) return null;
|
|
355
|
+
return {
|
|
356
|
+
claimedBy: doc.claimedBy ?? null,
|
|
357
|
+
fenceToken: doc.fenceToken ?? 0,
|
|
358
|
+
leaseExpiresAt: doc.leaseExpiresAt ? new Date(doc.leaseExpiresAt) : null,
|
|
359
|
+
processing: doc.processing === true,
|
|
360
|
+
recoveryAttempts: doc.recoveryAttempts ?? 0,
|
|
361
|
+
updatedAt: doc.updatedAt ? new Date(doc.updatedAt) : /* @__PURE__ */ new Date(0)
|
|
362
|
+
};
|
|
363
|
+
},
|
|
364
|
+
renew: async (jobId, fenceToken, ttlMs, now) => {
|
|
365
|
+
return { ok: await m.findOneAndUpdate({
|
|
366
|
+
_id: jobId,
|
|
367
|
+
fenceToken
|
|
368
|
+
}, { $set: { leaseExpiresAt: leaseExpiry(now, ttlMs) } }, { new: true }) !== null };
|
|
369
|
+
},
|
|
370
|
+
releaseAllClaims: async (owner) => {
|
|
371
|
+
return { released: (await m.updateMany({
|
|
372
|
+
claimedBy: owner,
|
|
373
|
+
processing: true
|
|
374
|
+
}, {
|
|
375
|
+
$inc: {
|
|
376
|
+
fenceToken: 1,
|
|
377
|
+
recoveryAttempts: 1
|
|
378
|
+
},
|
|
379
|
+
$set: {
|
|
380
|
+
claimedBy: null,
|
|
381
|
+
leaseExpiresAt: null,
|
|
382
|
+
processing: false,
|
|
383
|
+
waitUntil: null
|
|
384
|
+
}
|
|
385
|
+
})).modifiedCount ?? 0 };
|
|
386
|
+
},
|
|
387
|
+
requeue: async (jobId, now, fallbackMs) => {
|
|
388
|
+
return { ok: await m.findOneAndUpdate({
|
|
389
|
+
_id: jobId,
|
|
390
|
+
...stale(now, fallbackMs)
|
|
391
|
+
}, {
|
|
392
|
+
$inc: {
|
|
393
|
+
fenceToken: 1,
|
|
394
|
+
recoveryAttempts: 1
|
|
395
|
+
},
|
|
396
|
+
$set: {
|
|
397
|
+
claimedBy: null,
|
|
398
|
+
leaseExpiresAt: null,
|
|
399
|
+
processing: false,
|
|
400
|
+
waitUntil: null
|
|
401
|
+
}
|
|
402
|
+
}, { new: true }) !== null };
|
|
403
|
+
},
|
|
404
|
+
stampClaim: async (jobId, owner, ttlMs, now) => {
|
|
405
|
+
const doc = await m.findOneAndUpdate({
|
|
406
|
+
_id: jobId,
|
|
407
|
+
processing: true
|
|
408
|
+
}, {
|
|
409
|
+
$inc: { fenceToken: 1 },
|
|
410
|
+
$set: {
|
|
411
|
+
claimedBy: owner,
|
|
412
|
+
leaseExpiresAt: leaseExpiry(now, ttlMs),
|
|
413
|
+
startedAt: now
|
|
414
|
+
}
|
|
415
|
+
}, { new: true });
|
|
416
|
+
return {
|
|
417
|
+
fenceToken: doc?.fenceToken ?? 0,
|
|
418
|
+
ok: doc !== null
|
|
419
|
+
};
|
|
420
|
+
}
|
|
421
|
+
};
|
|
422
|
+
};
|
|
423
|
+
//#endregion
|
|
424
|
+
//#region src/reliability/jobLeaseStore.postgres.ts
|
|
425
|
+
const JOBS_SLUG$3 = "payload-jobs";
|
|
426
|
+
/** Payload tables for kebab slugs are the slug with hyphens replaced by underscores. */
|
|
427
|
+
const tableName$1 = (payload) => {
|
|
428
|
+
const defaultTable = JOBS_SLUG$3.replace(/-/g, "_");
|
|
429
|
+
const map = payload.db.tableNameMap;
|
|
430
|
+
return (map && typeof map.get === "function" ? map.get(defaultTable) : void 0) ?? defaultTable;
|
|
431
|
+
};
|
|
432
|
+
const pool$1 = (payload) => {
|
|
433
|
+
const found = payload.db.pool;
|
|
434
|
+
if (!found) throw new Error(`@10x-media/jobs: missing pg pool on the "${payload.db.name}" adapter`);
|
|
435
|
+
return found;
|
|
436
|
+
};
|
|
437
|
+
/** Single-statement `UPDATE ... WHERE <guard> RETURNING` is atomic under READ COMMITTED. */
|
|
438
|
+
const createPostgresJobLeaseStore = (payload) => {
|
|
439
|
+
const table = tableName$1(payload);
|
|
440
|
+
const db = pool$1(payload);
|
|
441
|
+
const staleGuard = "processing = true AND has_error IS NOT TRUE AND (lease_expires_at < $2 OR (lease_expires_at IS NULL AND updated_at < $3))";
|
|
442
|
+
return {
|
|
443
|
+
deadLetter: async ({ error, fallbackMs, jobId, now }) => {
|
|
444
|
+
return { ok: (await db.query(`UPDATE ${table}
|
|
445
|
+
SET processing = false, has_error = true, error = $4::jsonb,
|
|
446
|
+
lease_expires_at = NULL, claimed_by = NULL, fence_token = COALESCE(fence_token, 0) + 1
|
|
447
|
+
WHERE id = $1 AND ${staleGuard}
|
|
448
|
+
RETURNING id`, [
|
|
449
|
+
jobId,
|
|
450
|
+
now,
|
|
451
|
+
new Date(now.getTime() - fallbackMs),
|
|
452
|
+
JSON.stringify(error)
|
|
453
|
+
])).rowCount === 1 };
|
|
454
|
+
},
|
|
455
|
+
read: async (jobId) => {
|
|
456
|
+
const row = (await db.query(`SELECT processing, lease_expires_at, claimed_by, fence_token, recovery_attempts, updated_at
|
|
457
|
+
FROM ${table} WHERE id = $1`, [jobId])).rows[0];
|
|
458
|
+
if (row === void 0) return null;
|
|
459
|
+
return {
|
|
460
|
+
claimedBy: row.claimed_by ?? null,
|
|
461
|
+
fenceToken: Number(row.fence_token ?? 0),
|
|
462
|
+
leaseExpiresAt: row.lease_expires_at ? new Date(row.lease_expires_at) : null,
|
|
463
|
+
processing: row.processing === true,
|
|
464
|
+
recoveryAttempts: Number(row.recovery_attempts ?? 0),
|
|
465
|
+
updatedAt: row.updated_at ? new Date(row.updated_at) : /* @__PURE__ */ new Date(0)
|
|
466
|
+
};
|
|
467
|
+
},
|
|
468
|
+
renew: async (jobId, fenceToken, ttlMs, now) => {
|
|
469
|
+
return { ok: (await db.query(`UPDATE ${table} SET lease_expires_at = $1
|
|
470
|
+
WHERE id = $2 AND fence_token = $3
|
|
471
|
+
RETURNING id`, [
|
|
472
|
+
leaseExpiry(now, ttlMs),
|
|
473
|
+
jobId,
|
|
474
|
+
fenceToken
|
|
475
|
+
])).rowCount === 1 };
|
|
476
|
+
},
|
|
477
|
+
releaseAllClaims: async (owner) => {
|
|
478
|
+
return { released: (await db.query(`UPDATE ${table}
|
|
479
|
+
SET processing = false, claimed_by = NULL, lease_expires_at = NULL, wait_until = NULL,
|
|
480
|
+
recovery_attempts = COALESCE(recovery_attempts, 0) + 1,
|
|
481
|
+
fence_token = COALESCE(fence_token, 0) + 1
|
|
482
|
+
WHERE claimed_by = $1 AND processing = true
|
|
483
|
+
RETURNING id`, [owner])).rowCount ?? 0 };
|
|
484
|
+
},
|
|
485
|
+
requeue: async (jobId, now, fallbackMs) => {
|
|
486
|
+
return { ok: (await db.query(`UPDATE ${table}
|
|
487
|
+
SET processing = false, lease_expires_at = NULL, claimed_by = NULL, wait_until = NULL,
|
|
488
|
+
recovery_attempts = COALESCE(recovery_attempts, 0) + 1,
|
|
489
|
+
fence_token = COALESCE(fence_token, 0) + 1
|
|
490
|
+
WHERE id = $1 AND ${staleGuard}
|
|
491
|
+
RETURNING id`, [
|
|
492
|
+
jobId,
|
|
493
|
+
now,
|
|
494
|
+
new Date(now.getTime() - fallbackMs)
|
|
495
|
+
])).rowCount === 1 };
|
|
496
|
+
},
|
|
497
|
+
stampClaim: async (jobId, owner, ttlMs, now) => {
|
|
498
|
+
const res = await db.query(`UPDATE ${table}
|
|
499
|
+
SET started_at = $1, claimed_by = $2, lease_expires_at = $3,
|
|
500
|
+
fence_token = COALESCE(fence_token, 0) + 1
|
|
501
|
+
WHERE id = $4 AND processing = true
|
|
502
|
+
RETURNING fence_token`, [
|
|
503
|
+
now,
|
|
504
|
+
owner,
|
|
505
|
+
leaseExpiry(now, ttlMs),
|
|
506
|
+
jobId
|
|
507
|
+
]);
|
|
508
|
+
const ok = res.rowCount === 1;
|
|
509
|
+
return {
|
|
510
|
+
fenceToken: ok ? Number(res.rows[0]?.fence_token) : 0,
|
|
511
|
+
ok
|
|
512
|
+
};
|
|
513
|
+
}
|
|
514
|
+
};
|
|
515
|
+
};
|
|
516
|
+
//#endregion
|
|
517
|
+
//#region src/reliability/jobLeaseStore.ts
|
|
518
|
+
/** Build the job-lease store for the running adapter. Throws for an unsupported adapter. */
|
|
519
|
+
const createJobLeaseStore = (payload) => {
|
|
520
|
+
if (payload.db.name === "mongoose") return createMongoJobLeaseStore(payload);
|
|
521
|
+
if (payload.db.name === "postgres") return createPostgresJobLeaseStore(payload);
|
|
522
|
+
throw new Error(`@10x-media/jobs reliability does not support db adapter "${payload.db.name}"`);
|
|
523
|
+
};
|
|
524
|
+
//#endregion
|
|
525
|
+
//#region src/reliability/leaseMode.ts
|
|
526
|
+
/**
|
|
527
|
+
* Heartbeat mode renews a running job's lease from a side timer. Serverless mode
|
|
528
|
+
* never renews (the function is hard-killed at maxDuration with no SIGTERM), so the
|
|
529
|
+
* stamp encodes the platform limit and the sweeper recovers from it.
|
|
530
|
+
*/
|
|
531
|
+
const isHeartbeatMode = (options) => options.serverlessMaxDurationMs === null;
|
|
532
|
+
/**
|
|
533
|
+
* The lease TTL stamped when a worker claims a job, and the grace used to age out a
|
|
534
|
+
* claimed-but-never-stamped (null lease) job by its updatedAt. In serverless mode
|
|
535
|
+
* this is the platform hard-kill duration; otherwise the configured job lease TTL.
|
|
536
|
+
*/
|
|
537
|
+
const initialLeaseTtlMs = (options) => options.serverlessMaxDurationMs ?? options.jobLeaseTtlMs;
|
|
538
|
+
//#endregion
|
|
539
|
+
//#region src/reliability/recoveryDecision.ts
|
|
540
|
+
/**
|
|
541
|
+
* Requeue an orphaned job while it is below the recovery cap; dead-letter once it
|
|
542
|
+
* reaches the cap, to stop a poison job from thrashing the queue. `recoveryAttempts`
|
|
543
|
+
* is the count before this pass (a requeue increments it). `maxRecoveries` of 0
|
|
544
|
+
* dead-letters on the first orphan.
|
|
545
|
+
*/
|
|
546
|
+
const decideRecovery = (recoveryAttempts, maxRecoveries) => recoveryAttempts < maxRecoveries ? "requeue" : "deadLetter";
|
|
547
|
+
//#endregion
|
|
548
|
+
//#region src/reliability/sweeper.ts
|
|
549
|
+
const JOBS_SLUG$2 = "payload-jobs";
|
|
550
|
+
/** The dead-letter error payload written when a job reaches the recovery cap. */
|
|
551
|
+
const deadLetterError = (attempts) => ({
|
|
552
|
+
cancelled: false,
|
|
553
|
+
message: `@10x-media/jobs sweeper: dead-lettered after reaching the recovery cap (${attempts} attempts)`,
|
|
554
|
+
recovered: false
|
|
555
|
+
});
|
|
556
|
+
/**
|
|
557
|
+
* One sweep pass: find stale `processing: true` orphans and either requeue them
|
|
558
|
+
* (below the recovery cap) or dead-letter them (at the cap), each through a fenced
|
|
559
|
+
* conditional write that re-checks staleness, so a job that renewed between the find
|
|
560
|
+
* and the write is skipped and two overlapping sweepers never both reclaim the same
|
|
561
|
+
* orphan. Gated by leadership: callers pass `isLeader` from the Plan 1 sweeper lease.
|
|
562
|
+
*/
|
|
563
|
+
const runSweep = async (args) => {
|
|
564
|
+
const { isLeader = true, limit = 100, options, payload } = args;
|
|
565
|
+
const result = {
|
|
566
|
+
deadLettered: 0,
|
|
567
|
+
requeued: 0,
|
|
568
|
+
scanned: 0
|
|
569
|
+
};
|
|
570
|
+
if (!isLeader) return result;
|
|
571
|
+
const now = args.now ?? getCurrentDate();
|
|
572
|
+
const store = args.store ?? createJobLeaseStore(payload);
|
|
573
|
+
const fallbackMs = initialLeaseTtlMs(options);
|
|
574
|
+
const cutoff = new Date(now.getTime() - fallbackMs).toISOString();
|
|
575
|
+
const where = { and: [
|
|
576
|
+
{ processing: { equals: true } },
|
|
577
|
+
{ completedAt: { exists: false } },
|
|
578
|
+
{ hasError: { not_equals: true } },
|
|
579
|
+
{ or: [{ leaseExpiresAt: { less_than: now.toISOString() } }, { and: [{ leaseExpiresAt: { exists: false } }, { updatedAt: { less_than: cutoff } }] }] }
|
|
580
|
+
] };
|
|
581
|
+
const find = payload.find;
|
|
582
|
+
const { docs } = await find({
|
|
583
|
+
collection: JOBS_SLUG$2,
|
|
584
|
+
depth: 0,
|
|
585
|
+
limit,
|
|
586
|
+
where
|
|
587
|
+
});
|
|
588
|
+
result.scanned = docs.length;
|
|
589
|
+
for (const doc of docs) {
|
|
590
|
+
const attempts = typeof doc.recoveryAttempts === "number" ? doc.recoveryAttempts : 0;
|
|
591
|
+
if (decideRecovery(attempts, options.maxRecoveries) === "requeue") {
|
|
592
|
+
if ((await store.requeue(doc.id, now, fallbackMs)).ok) result.requeued += 1;
|
|
593
|
+
} else if ((await store.deadLetter({
|
|
594
|
+
error: deadLetterError(attempts),
|
|
595
|
+
fallbackMs,
|
|
596
|
+
jobId: doc.id,
|
|
597
|
+
now
|
|
598
|
+
})).ok) result.deadLettered += 1;
|
|
599
|
+
}
|
|
600
|
+
return result;
|
|
601
|
+
};
|
|
602
|
+
//#endregion
|
|
603
|
+
//#region src/queueControl/pauseState.ts
|
|
604
|
+
/** The default (nothing paused) state. */
|
|
605
|
+
const emptyPauseState = () => ({
|
|
606
|
+
global: false,
|
|
607
|
+
queues: []
|
|
608
|
+
});
|
|
609
|
+
/** Pause everything (no queue) or a single queue (idempotent). */
|
|
610
|
+
const applyPause = (state, queue) => {
|
|
611
|
+
if (queue === void 0) return {
|
|
612
|
+
...state,
|
|
613
|
+
global: true
|
|
614
|
+
};
|
|
615
|
+
return state.queues.includes(queue) ? state : {
|
|
616
|
+
...state,
|
|
617
|
+
queues: [...state.queues, queue]
|
|
618
|
+
};
|
|
619
|
+
};
|
|
620
|
+
/** Resume everything (no queue, clears the global flag) or a single queue. */
|
|
621
|
+
const applyResume = (state, queue) => {
|
|
622
|
+
if (queue === void 0) return {
|
|
623
|
+
...state,
|
|
624
|
+
global: false
|
|
625
|
+
};
|
|
626
|
+
return {
|
|
627
|
+
...state,
|
|
628
|
+
queues: state.queues.filter((q) => q !== queue)
|
|
629
|
+
};
|
|
630
|
+
};
|
|
631
|
+
/** Whether `queue` is currently paused (a global pause covers every queue). */
|
|
632
|
+
const isPaused = (state, queue) => state.global || state.queues.includes(queue);
|
|
633
|
+
//#endregion
|
|
634
|
+
//#region src/queueControl/pauseStore.ts
|
|
635
|
+
const PAUSE_KEY = "@10x-media/jobs:pause-state";
|
|
636
|
+
/**
|
|
637
|
+
* Build the pause store over `payload.kv` (always available, durable, cluster-wide).
|
|
638
|
+
* Pause and resume read-modify-write the single state value; this is last-writer-wins
|
|
639
|
+
* (kv has no atomic compare-and-set), which is acceptable for rare admin actions.
|
|
640
|
+
*/
|
|
641
|
+
const createPauseStore = (payload) => {
|
|
642
|
+
const getState = async () => await payload.kv.get(PAUSE_KEY) ?? emptyPauseState();
|
|
643
|
+
return {
|
|
644
|
+
getState,
|
|
645
|
+
isPaused: async (queue) => isPaused(await getState(), queue),
|
|
646
|
+
pause: async (queue) => {
|
|
647
|
+
await payload.kv.set(PAUSE_KEY, applyPause(await getState(), queue));
|
|
648
|
+
},
|
|
649
|
+
resume: async (queue) => {
|
|
650
|
+
await payload.kv.set(PAUSE_KEY, applyResume(await getState(), queue));
|
|
651
|
+
}
|
|
652
|
+
};
|
|
653
|
+
};
|
|
654
|
+
//#endregion
|
|
655
|
+
//#region src/queueControl/queueHealth.ts
|
|
656
|
+
const JOBS_SLUG$1 = "payload-jobs";
|
|
657
|
+
const STATS_SLUG = "payload-jobs-stats";
|
|
658
|
+
const queueClause = (queue) => queue ? [{ queue: { equals: queue } }] : [];
|
|
659
|
+
const pendingWhere = (queue) => ({ and: [
|
|
660
|
+
{ completedAt: { exists: false } },
|
|
661
|
+
{ hasError: { not_equals: true } },
|
|
662
|
+
{ processing: { equals: false } },
|
|
663
|
+
...queueClause(queue)
|
|
664
|
+
] });
|
|
665
|
+
const processingWhere = (queue) => ({ and: [{ processing: { equals: true } }, ...queueClause(queue)] });
|
|
666
|
+
const failedWhere = (queue) => ({ and: [{ hasError: { equals: true } }, ...queueClause(queue)] });
|
|
667
|
+
const recoveredWhere = (queue) => ({ and: [{ recoveryAttempts: { greater_than: 0 } }, ...queueClause(queue)] });
|
|
668
|
+
const readStats = async (payload) => {
|
|
669
|
+
const db = payload.db;
|
|
670
|
+
try {
|
|
671
|
+
return await db.findGlobal({ slug: STATS_SLUG });
|
|
672
|
+
} catch {
|
|
673
|
+
return null;
|
|
674
|
+
}
|
|
675
|
+
};
|
|
676
|
+
const lastScheduledRunFor = (stats, queue) => {
|
|
677
|
+
const entry = stats?.stats?.scheduledRuns?.queues?.[queue];
|
|
678
|
+
if (!entry) return null;
|
|
679
|
+
const runs = [...Object.values(entry.tasks ?? {}), ...Object.values(entry.workflows ?? {})].map((r) => r.lastScheduledRun).filter((r) => typeof r === "string");
|
|
680
|
+
return runs.length > 0 ? runs.sort().at(-1) ?? null : null;
|
|
681
|
+
};
|
|
682
|
+
/**
|
|
683
|
+
* Aggregate queue health via `payload.count` per state (pending, processing, failed,
|
|
684
|
+
* and, when reliability is on, recovered), plus the oldest-pending age and the last
|
|
685
|
+
* scheduled run per queue. Reuses Payload's own run-selector predicates. The stats
|
|
686
|
+
* global is read through `payload.db.findGlobal` inside a try/catch because that call
|
|
687
|
+
* throws when the global does not exist (no schedule has run).
|
|
688
|
+
*/
|
|
689
|
+
const getQueueHealth = async (payload, options = {}) => {
|
|
690
|
+
const queues = options.queues ?? ["default"];
|
|
691
|
+
const includeRecovered = options.includeRecovered ?? false;
|
|
692
|
+
const count = payload.count;
|
|
693
|
+
const find = payload.find;
|
|
694
|
+
const countWhere = async (where) => (await count({
|
|
695
|
+
collection: JOBS_SLUG$1,
|
|
696
|
+
where
|
|
697
|
+
})).totalDocs;
|
|
698
|
+
const totals = {
|
|
699
|
+
failed: await countWhere(failedWhere()),
|
|
700
|
+
pending: await countWhere(pendingWhere()),
|
|
701
|
+
processing: await countWhere(processingWhere()),
|
|
702
|
+
recovered: includeRecovered ? await countWhere(recoveredWhere()) : 0
|
|
703
|
+
};
|
|
704
|
+
const oldestCreated = (await find({
|
|
705
|
+
collection: JOBS_SLUG$1,
|
|
706
|
+
depth: 0,
|
|
707
|
+
limit: 1,
|
|
708
|
+
sort: "createdAt",
|
|
709
|
+
where: pendingWhere()
|
|
710
|
+
})).docs[0]?.createdAt;
|
|
711
|
+
const nowMs = (options.now ?? /* @__PURE__ */ new Date()).getTime();
|
|
712
|
+
const oldestPendingAgeMs = oldestCreated ? nowMs - new Date(oldestCreated).getTime() : null;
|
|
713
|
+
const stats = await readStats(payload);
|
|
714
|
+
const perQueue = [];
|
|
715
|
+
for (const queue of queues) perQueue.push({
|
|
716
|
+
failed: await countWhere(failedWhere(queue)),
|
|
717
|
+
lastScheduledRun: lastScheduledRunFor(stats, queue),
|
|
718
|
+
pending: await countWhere(pendingWhere(queue)),
|
|
719
|
+
processing: await countWhere(processingWhere(queue)),
|
|
720
|
+
queue,
|
|
721
|
+
recovered: includeRecovered ? await countWhere(recoveredWhere(queue)) : 0
|
|
722
|
+
});
|
|
723
|
+
return {
|
|
724
|
+
oldestPendingAgeMs,
|
|
725
|
+
queues: perQueue,
|
|
726
|
+
totals
|
|
727
|
+
};
|
|
728
|
+
};
|
|
729
|
+
//#endregion
|
|
730
|
+
//#region src/queueControl/runTargets.ts
|
|
731
|
+
/**
|
|
732
|
+
* Compute the run targets for one cycle, given the worker's configured queues (or
|
|
733
|
+
* undefined for all queues) and the current pause state. A global pause yields no
|
|
734
|
+
* targets. A specific queue list drops the paused queues. All-queues running excludes
|
|
735
|
+
* paused queues via `not_in` paired with `allQueues: true` (the exclusion requires
|
|
736
|
+
* allQueues, because Payload's built-in single-queue filter only applies otherwise).
|
|
737
|
+
*/
|
|
738
|
+
const runTargetsForPause = (queues, state) => {
|
|
739
|
+
if (state.global) return [];
|
|
740
|
+
if (queues && queues.length > 0) return queues.filter((queue) => !state.queues.includes(queue)).map((queue) => ({ queue }));
|
|
741
|
+
if (state.queues.length > 0) return [{
|
|
742
|
+
allQueues: true,
|
|
743
|
+
where: { queue: { not_in: state.queues } }
|
|
744
|
+
}];
|
|
745
|
+
return [{ allQueues: true }];
|
|
746
|
+
};
|
|
747
|
+
//#endregion
|
|
748
|
+
//#region src/queueControl/endpoints.ts
|
|
749
|
+
const unauthorized = () => Response.json({ message: "Unauthorized" }, { status: 401 });
|
|
750
|
+
/** GET queue health, grouped by queue. */
|
|
751
|
+
const statusEndpoint = (deps) => ({
|
|
752
|
+
handler: async (req) => {
|
|
753
|
+
if (!await deps.access({ req })) return unauthorized();
|
|
754
|
+
const report = await getQueueHealth(req.payload, {
|
|
755
|
+
includeRecovered: deps.reliability !== null,
|
|
756
|
+
queues: deps.queues
|
|
757
|
+
});
|
|
758
|
+
return Response.json(report, { status: 200 });
|
|
759
|
+
},
|
|
760
|
+
method: "get",
|
|
761
|
+
path: "/queue-status"
|
|
762
|
+
});
|
|
763
|
+
/** GET a hardened, pause-aware run. Mirrors the native run params. */
|
|
764
|
+
const runControlEndpoint = (deps) => ({
|
|
765
|
+
handler: async (req) => {
|
|
766
|
+
if (!await deps.access({ req })) return unauthorized();
|
|
767
|
+
const query = req.query;
|
|
768
|
+
const limit = query.limit ? Number(query.limit) : void 0;
|
|
769
|
+
const silent = query.silent === "true";
|
|
770
|
+
const state = await createPauseStore(req.payload).getState();
|
|
771
|
+
const targets = runTargetsForPause(query.allQueues === "true" ? void 0 : [query.queue ?? "default"], state);
|
|
772
|
+
if (query.disableScheduling !== "true") await req.payload.jobs.handleSchedules({ allQueues: true });
|
|
773
|
+
const results = [];
|
|
774
|
+
for (const target of targets) results.push(await req.payload.jobs.run({
|
|
775
|
+
...target,
|
|
776
|
+
...limit !== void 0 ? { limit } : {},
|
|
777
|
+
silent
|
|
778
|
+
}));
|
|
779
|
+
return Response.json({
|
|
780
|
+
paused: state,
|
|
781
|
+
ran: targets.length,
|
|
782
|
+
results
|
|
783
|
+
}, { status: 200 });
|
|
784
|
+
},
|
|
785
|
+
method: "get",
|
|
786
|
+
path: "/queue-run"
|
|
787
|
+
});
|
|
788
|
+
/** GET a one-shot sweep for serverless (one cron invocation, so no leader election). */
|
|
789
|
+
const sweepEndpoint = (deps) => ({
|
|
790
|
+
handler: async (req) => {
|
|
791
|
+
if (!await deps.access({ req })) return unauthorized();
|
|
792
|
+
if (!deps.reliability) return Response.json({ message: "reliability is not enabled; the sweeper is unavailable" }, { status: 400 });
|
|
793
|
+
const result = await runSweep({
|
|
794
|
+
isLeader: true,
|
|
795
|
+
options: deps.reliability,
|
|
796
|
+
payload: req.payload,
|
|
797
|
+
store: createJobLeaseStore(req.payload)
|
|
798
|
+
});
|
|
799
|
+
return Response.json(result, { status: 200 });
|
|
800
|
+
},
|
|
801
|
+
method: "get",
|
|
802
|
+
path: "/queue-sweep"
|
|
803
|
+
});
|
|
804
|
+
//#endregion
|
|
805
|
+
//#region src/queueControl/registerQueueControl.ts
|
|
806
|
+
/**
|
|
807
|
+
* Register the queue-control layer: harden the native run endpoint by setting
|
|
808
|
+
* `jobs.access.run` to the configured checker, and add the status, hardened-run, and
|
|
809
|
+
* sweep endpoints to `payload-jobs` through the same `jobsCollectionOverrides` seam the
|
|
810
|
+
* reliability and observability layers use (composing, never clobbering).
|
|
811
|
+
*/
|
|
812
|
+
const registerQueueControl = (config, options, reliability) => {
|
|
813
|
+
const deps = {
|
|
814
|
+
access: options.access,
|
|
815
|
+
queues: options.queues,
|
|
816
|
+
reliability
|
|
817
|
+
};
|
|
818
|
+
const controlEndpoints = [
|
|
819
|
+
statusEndpoint(deps),
|
|
820
|
+
runControlEndpoint(deps),
|
|
821
|
+
sweepEndpoint(deps)
|
|
822
|
+
];
|
|
823
|
+
const existingOverride = config.jobs?.jobsCollectionOverrides;
|
|
824
|
+
config.jobs = {
|
|
825
|
+
...config.jobs,
|
|
826
|
+
access: {
|
|
827
|
+
...config.jobs?.access,
|
|
828
|
+
run: options.access
|
|
829
|
+
},
|
|
830
|
+
jobsCollectionOverrides: ({ defaultJobsCollection }) => {
|
|
831
|
+
const base = existingOverride ? existingOverride({ defaultJobsCollection }) : defaultJobsCollection;
|
|
832
|
+
const baseEndpoints = Array.isArray(base.endpoints) ? base.endpoints : [];
|
|
833
|
+
return {
|
|
834
|
+
...base,
|
|
835
|
+
endpoints: [...baseEndpoints, ...controlEndpoints]
|
|
836
|
+
};
|
|
837
|
+
}
|
|
838
|
+
};
|
|
839
|
+
};
|
|
840
|
+
//#endregion
|
|
841
|
+
//#region src/reliability/options.ts
|
|
842
|
+
/** Resolve user reliability options to a fully-defaulted object, or `null` when disabled. */
|
|
843
|
+
const resolveReliabilityOptions = (options) => {
|
|
844
|
+
if (options === void 0 || options === false) return null;
|
|
845
|
+
const jobLeaseTtlMs = options.jobLeaseTtlMs ?? 3e5;
|
|
846
|
+
return {
|
|
847
|
+
heartbeatIntervalMs: options.heartbeatIntervalMs ?? Math.floor(jobLeaseTtlMs / 3),
|
|
848
|
+
jobLeaseTtlMs,
|
|
849
|
+
leaderId: options.leaderId ?? null,
|
|
850
|
+
leaderLeaseTtlMs: options.leaderLeaseTtlMs ?? 3e4,
|
|
851
|
+
maxRecoveries: options.maxRecoveries ?? 3,
|
|
852
|
+
requireConcurrencyControl: options.requireConcurrencyControl ?? false,
|
|
853
|
+
serverlessMaxDurationMs: options.serverless?.maxDurationMs ?? null,
|
|
854
|
+
sweepIntervalMs: options.sweepIntervalMs ?? 6e4
|
|
855
|
+
};
|
|
856
|
+
};
|
|
857
|
+
//#endregion
|
|
858
|
+
//#region src/reliability/concurrencyContract.ts
|
|
859
|
+
/**
|
|
860
|
+
* When `reliability.requireConcurrencyControl` is set, refuse to start unless
|
|
861
|
+
* Payload's own `jobs.enableConcurrencyControl` is on. The at-least-once contract
|
|
862
|
+
* relies on it for app-level mutual exclusion under multi-node, and enabling it
|
|
863
|
+
* changes the jobs schema (adds a `concurrencyKey` field), so we fail loudly rather
|
|
864
|
+
* than silently mutate the adopter's schema. The config parameter is a structural
|
|
865
|
+
* subset (Payload's `Config` satisfies it), which keeps the function trivial to unit
|
|
866
|
+
* test with a plain object literal.
|
|
867
|
+
*/
|
|
868
|
+
const enforceConcurrencyControl = (config, options) => {
|
|
869
|
+
if (!options.requireConcurrencyControl) return;
|
|
870
|
+
if (config.jobs?.enableConcurrencyControl === true) return;
|
|
871
|
+
throw new Error("@10x-media/jobs: reliability.requireConcurrencyControl is set but jobs.enableConcurrencyControl is not true. Enable it in your Payload config (it adds a concurrencyKey field, so run a migration) or unset requireConcurrencyControl.");
|
|
872
|
+
};
|
|
873
|
+
/**
|
|
874
|
+
* Wrap a job handler so its side effect runs at most once per idempotency key, even
|
|
875
|
+
* when at-least-once delivery re-runs the job. The race this guards is not only
|
|
876
|
+
* redelivery: when the sweeper requeues a job whose lease expired while it was still
|
|
877
|
+
* running, the original execution cannot be aborted (Payload exposes no AbortSignal),
|
|
878
|
+
* so a second worker can run the handler concurrently with the first. Key on the
|
|
879
|
+
* stable unit of work, never the job id, so both executions resolve to the same key.
|
|
880
|
+
* The check and the mark are the caller's to make transactional (Payload's queue and
|
|
881
|
+
* the app data share one database, which makes that natural); this helper only
|
|
882
|
+
* sequences them around the handler. A skipped run returns `{ output: {} }`.
|
|
883
|
+
*/
|
|
884
|
+
const withIdempotencyKey = (handler, idem) => {
|
|
885
|
+
return async (args) => {
|
|
886
|
+
const key = idem.keyFor(args);
|
|
887
|
+
if (await idem.store.has(key)) return { output: {} };
|
|
888
|
+
const out = await handler(args);
|
|
889
|
+
await idem.store.mark(key);
|
|
890
|
+
return out;
|
|
891
|
+
};
|
|
892
|
+
};
|
|
893
|
+
//#endregion
|
|
894
|
+
//#region src/reliability/fields.ts
|
|
895
|
+
/**
|
|
896
|
+
* Fields added to the built-in `payload-jobs` collection (via jobsCollectionOverrides)
|
|
897
|
+
* when reliability is enabled. `leaseExpiresAt` is the liveness signal renewed by the
|
|
898
|
+
* worker heartbeat; the sweeper reclaims a job when `leaseExpiresAt < now`. All are
|
|
899
|
+
* sidebar, read-only-in-admin diagnostics (the runtime, not a human, writes them).
|
|
900
|
+
*/
|
|
901
|
+
const reliabilityJobFields = () => [
|
|
902
|
+
{
|
|
903
|
+
name: "startedAt",
|
|
904
|
+
type: "date",
|
|
905
|
+
admin: {
|
|
906
|
+
position: "sidebar",
|
|
907
|
+
readOnly: true
|
|
908
|
+
},
|
|
909
|
+
index: true
|
|
910
|
+
},
|
|
911
|
+
{
|
|
912
|
+
name: "leaseExpiresAt",
|
|
913
|
+
type: "date",
|
|
914
|
+
admin: {
|
|
915
|
+
position: "sidebar",
|
|
916
|
+
readOnly: true
|
|
917
|
+
},
|
|
918
|
+
index: true
|
|
919
|
+
},
|
|
920
|
+
{
|
|
921
|
+
name: "claimedBy",
|
|
922
|
+
type: "text",
|
|
923
|
+
admin: {
|
|
924
|
+
position: "sidebar",
|
|
925
|
+
readOnly: true
|
|
926
|
+
},
|
|
927
|
+
index: true
|
|
928
|
+
},
|
|
929
|
+
{
|
|
930
|
+
name: "fenceToken",
|
|
931
|
+
type: "number",
|
|
932
|
+
admin: {
|
|
933
|
+
position: "sidebar",
|
|
934
|
+
readOnly: true
|
|
935
|
+
}
|
|
936
|
+
},
|
|
937
|
+
{
|
|
938
|
+
name: "recoveryAttempts",
|
|
939
|
+
type: "number",
|
|
940
|
+
admin: {
|
|
941
|
+
position: "sidebar",
|
|
942
|
+
readOnly: true
|
|
943
|
+
},
|
|
944
|
+
defaultValue: 0
|
|
945
|
+
}
|
|
946
|
+
];
|
|
947
|
+
//#endregion
|
|
948
|
+
//#region src/reliability/heartbeat.ts
|
|
949
|
+
/**
|
|
950
|
+
* Wrap one job handler so the job keeps its lease fresh while it runs. On entry the
|
|
951
|
+
* wrapper stamps the just-claimed row (fenced on `processing = true`) and, in
|
|
952
|
+
* heartbeat mode, renews the lease on a self-scheduling timer at
|
|
953
|
+
* `heartbeatIntervalMs`. A renew that matches nothing means the sweeper reclaimed the
|
|
954
|
+
* job (its fence token moved): the wrapper records the loss, warns, and stops
|
|
955
|
+
* renewing, but cannot abort an opaque handler (Payload exposes no AbortSignal), so
|
|
956
|
+
* correctness under that race is the idempotency contract's job. The timer is always
|
|
957
|
+
* cleared in `finally`. If the initial stamp fails (already completed, cancelled, or
|
|
958
|
+
* reclaimed) the handler still runs, just without a heartbeat.
|
|
959
|
+
*/
|
|
960
|
+
const withHeartbeat = (args) => {
|
|
961
|
+
const { getStore, handler, onLeaseLost, options, ownerId } = args;
|
|
962
|
+
const ttlMs = initialLeaseTtlMs(options);
|
|
963
|
+
const beats = isHeartbeatMode(options);
|
|
964
|
+
const intervalMs = options.heartbeatIntervalMs;
|
|
965
|
+
return async (handlerArgs) => {
|
|
966
|
+
const payload = handlerArgs.req.payload;
|
|
967
|
+
const jobId = handlerArgs.job.id;
|
|
968
|
+
const store = getStore(payload);
|
|
969
|
+
const stamp = await store.stampClaim(jobId, ownerId, ttlMs, getCurrentDate());
|
|
970
|
+
if (!stamp.ok) return handler(handlerArgs);
|
|
971
|
+
const fence = stamp.fenceToken;
|
|
972
|
+
let done = false;
|
|
973
|
+
let timer;
|
|
974
|
+
const tick = async () => {
|
|
975
|
+
if (done) return;
|
|
976
|
+
let ok = true;
|
|
977
|
+
try {
|
|
978
|
+
ok = (await store.renew(jobId, fence, ttlMs, getCurrentDate())).ok;
|
|
979
|
+
} catch (err) {
|
|
980
|
+
payload.logger?.warn(`@10x-media/jobs: heartbeat renew error for job ${jobId}: ${String(err)}`);
|
|
981
|
+
schedule();
|
|
982
|
+
return;
|
|
983
|
+
}
|
|
984
|
+
if (!ok) {
|
|
985
|
+
payload.logger?.warn(`@10x-media/jobs: lost lease for job ${jobId} (reclaimed)`);
|
|
986
|
+
onLeaseLost?.(jobId);
|
|
987
|
+
return;
|
|
988
|
+
}
|
|
989
|
+
schedule();
|
|
990
|
+
};
|
|
991
|
+
function schedule() {
|
|
992
|
+
if (done) return;
|
|
993
|
+
timer = setTimeout(() => {
|
|
994
|
+
tick();
|
|
995
|
+
}, intervalMs);
|
|
996
|
+
}
|
|
997
|
+
if (beats) schedule();
|
|
998
|
+
try {
|
|
999
|
+
return await handler(handlerArgs);
|
|
1000
|
+
} finally {
|
|
1001
|
+
done = true;
|
|
1002
|
+
if (timer) clearTimeout(timer);
|
|
1003
|
+
}
|
|
1004
|
+
};
|
|
1005
|
+
};
|
|
1006
|
+
/**
|
|
1007
|
+
* Wrap every task and workflow handler on the config with the heartbeat. Only
|
|
1008
|
+
* function handlers are wrapped (Payload also allows a string path for controlled
|
|
1009
|
+
* handlers, which we leave alone). One job-lease store is reused per Payload instance
|
|
1010
|
+
* via a WeakMap. All other task and workflow properties are preserved.
|
|
1011
|
+
*/
|
|
1012
|
+
const registerHeartbeat = (config, options, ownerId) => {
|
|
1013
|
+
const jobs = config.jobs;
|
|
1014
|
+
if (!jobs) return;
|
|
1015
|
+
const storeCache = /* @__PURE__ */ new WeakMap();
|
|
1016
|
+
const getStore = (payload) => {
|
|
1017
|
+
const hit = storeCache.get(payload);
|
|
1018
|
+
if (hit) return hit;
|
|
1019
|
+
const store = createJobLeaseStore(payload);
|
|
1020
|
+
storeCache.set(payload, store);
|
|
1021
|
+
return store;
|
|
1022
|
+
};
|
|
1023
|
+
const wrapEntry = (entry) => {
|
|
1024
|
+
if (typeof entry.handler !== "function") return entry;
|
|
1025
|
+
const wrapped = withHeartbeat({
|
|
1026
|
+
getStore,
|
|
1027
|
+
handler: entry.handler,
|
|
1028
|
+
options,
|
|
1029
|
+
ownerId
|
|
1030
|
+
});
|
|
1031
|
+
return {
|
|
1032
|
+
...entry,
|
|
1033
|
+
handler: wrapped
|
|
1034
|
+
};
|
|
1035
|
+
};
|
|
1036
|
+
if (Array.isArray(jobs.tasks)) jobs.tasks = jobs.tasks.map(wrapEntry);
|
|
1037
|
+
if (Array.isArray(jobs.workflows)) jobs.workflows = jobs.workflows.map(wrapEntry);
|
|
1038
|
+
};
|
|
1039
|
+
//#endregion
|
|
1040
|
+
//#region src/reliability/locksCollection.ts
|
|
1041
|
+
/** Slug of the plugin-owned leases collection. Table name: payload_jobs_locks. */
|
|
1042
|
+
const JOBS_LOCKS_SLUG = "payload-jobs-locks";
|
|
1043
|
+
/** The two singleton leadership roles. */
|
|
1044
|
+
const LEADER_ROLES = ["scheduler", "sweeper"];
|
|
1045
|
+
/**
|
|
1046
|
+
* A hidden collection holding one row per leadership role. Acquire/renew/steal are
|
|
1047
|
+
* conditional updates against these rows (see the lease store). Not edit-locked,
|
|
1048
|
+
* not shown in admin, and access is closed by default (the plugin mutates it
|
|
1049
|
+
* directly through the db adapter, never the REST/GraphQL API).
|
|
1050
|
+
*/
|
|
1051
|
+
const buildJobsLocksCollection = () => ({
|
|
1052
|
+
slug: JOBS_LOCKS_SLUG,
|
|
1053
|
+
access: {
|
|
1054
|
+
create: () => false,
|
|
1055
|
+
delete: () => false,
|
|
1056
|
+
read: () => false,
|
|
1057
|
+
update: () => false
|
|
1058
|
+
},
|
|
1059
|
+
admin: { hidden: true },
|
|
1060
|
+
fields: [
|
|
1061
|
+
{
|
|
1062
|
+
name: "role",
|
|
1063
|
+
type: "text",
|
|
1064
|
+
index: true,
|
|
1065
|
+
required: true,
|
|
1066
|
+
unique: true
|
|
1067
|
+
},
|
|
1068
|
+
{
|
|
1069
|
+
name: "owner",
|
|
1070
|
+
type: "text"
|
|
1071
|
+
},
|
|
1072
|
+
{
|
|
1073
|
+
name: "leaseExpiresAt",
|
|
1074
|
+
type: "date"
|
|
1075
|
+
},
|
|
1076
|
+
{
|
|
1077
|
+
name: "fenceToken",
|
|
1078
|
+
type: "number",
|
|
1079
|
+
defaultValue: 0,
|
|
1080
|
+
required: true
|
|
1081
|
+
}
|
|
1082
|
+
],
|
|
1083
|
+
lockDocuments: false
|
|
1084
|
+
});
|
|
1085
|
+
//#endregion
|
|
1086
|
+
//#region src/reliability/nodeId.ts
|
|
1087
|
+
/**
|
|
1088
|
+
* A stable-per-process identity for job claims and leadership. An explicit
|
|
1089
|
+
* `leaderId` wins; otherwise derive `hostname:pid`, which is stable within a process
|
|
1090
|
+
* and distinguishes nodes in a cluster.
|
|
1091
|
+
*/
|
|
1092
|
+
const resolveNodeId = (leaderId) => leaderId ?? `${hostname()}:${process.pid}`;
|
|
1093
|
+
//#endregion
|
|
1094
|
+
//#region src/reliability/registerReliability.ts
|
|
1095
|
+
/** Idempotently ensure one lock row per leadership role. Race-safe via the unique `role`. */
|
|
1096
|
+
const ensureLockRows = async (payload) => {
|
|
1097
|
+
const create = payload.create;
|
|
1098
|
+
for (const role of LEADER_ROLES) try {
|
|
1099
|
+
await create({
|
|
1100
|
+
collection: JOBS_LOCKS_SLUG,
|
|
1101
|
+
data: {
|
|
1102
|
+
fenceToken: 0,
|
|
1103
|
+
leaseExpiresAt: null,
|
|
1104
|
+
owner: null,
|
|
1105
|
+
role
|
|
1106
|
+
},
|
|
1107
|
+
overrideAccess: true
|
|
1108
|
+
});
|
|
1109
|
+
} catch (err) {
|
|
1110
|
+
if (!(err instanceof ValidationError)) throw err;
|
|
1111
|
+
}
|
|
1112
|
+
};
|
|
1113
|
+
/**
|
|
1114
|
+
* Register the reliability layer on the incoming config: add diagnostic fields to
|
|
1115
|
+
* `payload-jobs` through the same `jobsCollectionOverrides` seam the observability
|
|
1116
|
+
* layer uses (composing, never clobbering), add the locks collection, and ensure the
|
|
1117
|
+
* lock rows at init (preserving any host onInit).
|
|
1118
|
+
*/
|
|
1119
|
+
const registerReliability = (config, options) => {
|
|
1120
|
+
enforceConcurrencyControl(config, options);
|
|
1121
|
+
const existingOverride = config.jobs?.jobsCollectionOverrides;
|
|
1122
|
+
config.jobs = {
|
|
1123
|
+
...config.jobs,
|
|
1124
|
+
jobsCollectionOverrides: ({ defaultJobsCollection }) => {
|
|
1125
|
+
const base = existingOverride ? existingOverride({ defaultJobsCollection }) : defaultJobsCollection;
|
|
1126
|
+
return {
|
|
1127
|
+
...base,
|
|
1128
|
+
fields: [...base.fields, ...reliabilityJobFields()]
|
|
1129
|
+
};
|
|
1130
|
+
}
|
|
1131
|
+
};
|
|
1132
|
+
config.collections = [...config.collections ?? [], buildJobsLocksCollection()];
|
|
1133
|
+
const previousOnInit = config.onInit;
|
|
1134
|
+
config.onInit = async (payload) => {
|
|
1135
|
+
await previousOnInit?.(payload);
|
|
1136
|
+
await ensureLockRows(payload);
|
|
1137
|
+
};
|
|
1138
|
+
registerHeartbeat(config, options, resolveNodeId(options.leaderId));
|
|
1139
|
+
};
|
|
1140
|
+
//#endregion
|
|
1141
|
+
//#region src/execution/autoRunConfig.ts
|
|
1142
|
+
const DEFAULT_CRON = "* * * * *";
|
|
1143
|
+
const DEFAULT_LIMIT = 10;
|
|
1144
|
+
/**
|
|
1145
|
+
* Build a production `jobs.autoRun` array: one Croner config per queue, silent by
|
|
1146
|
+
* default, with Payload's own `protect: true` preventing overlap. This is the simple
|
|
1147
|
+
* single-node and serverless-adjacent path where native autoRun safely handles both
|
|
1148
|
+
* scheduling and running in one process. Multi-node deployments use `createWorker`
|
|
1149
|
+
* instead, because native autoRun cannot gate scheduling to one elected leader (its
|
|
1150
|
+
* only dynamic lever, `shouldAutoRun`, permanently stops the cron rather than pausing).
|
|
1151
|
+
*/
|
|
1152
|
+
const autoRunConfig = (options = {}) => {
|
|
1153
|
+
const { disableScheduling = false, queues = [{ queue: "default" }], silent = true } = options;
|
|
1154
|
+
return queues.map((q) => ({
|
|
1155
|
+
cron: q.cron ?? DEFAULT_CRON,
|
|
1156
|
+
disableScheduling,
|
|
1157
|
+
limit: q.limit ?? DEFAULT_LIMIT,
|
|
1158
|
+
queue: q.queue,
|
|
1159
|
+
silent
|
|
1160
|
+
}));
|
|
1161
|
+
};
|
|
1162
|
+
//#endregion
|
|
1163
|
+
//#region src/execution/drain.ts
|
|
1164
|
+
/**
|
|
1165
|
+
* Run the graceful-drain sequence: stop claiming, await this node's in-flight jobs up
|
|
1166
|
+
* to a wall-clock budget, requeue any stragglers, release leadership, and destroy. The
|
|
1167
|
+
* clock (`now`/`sleep`) is injected so tests drive it deterministically without real
|
|
1168
|
+
* waiting. Always releases leadership and destroys, even when nothing was in flight.
|
|
1169
|
+
*/
|
|
1170
|
+
const drainWorker = async (deps, options) => {
|
|
1171
|
+
deps.stopLoops();
|
|
1172
|
+
const start = deps.now();
|
|
1173
|
+
const inFlightAtStart = await deps.countInFlight();
|
|
1174
|
+
let remaining = inFlightAtStart;
|
|
1175
|
+
while (remaining > 0 && deps.now() - start < options.drainTimeoutMs) {
|
|
1176
|
+
await deps.sleep(options.pollIntervalMs);
|
|
1177
|
+
remaining = await deps.countInFlight();
|
|
1178
|
+
}
|
|
1179
|
+
const timedOut = remaining > 0;
|
|
1180
|
+
const requeued = timedOut ? await deps.requeueStragglers() : 0;
|
|
1181
|
+
await deps.releaseLeadership();
|
|
1182
|
+
await deps.destroy();
|
|
1183
|
+
deps.logger?.info?.(`@10x-media/jobs: drain complete (started ${inFlightAtStart}, requeued ${requeued}, timedOut ${timedOut})`);
|
|
1184
|
+
return {
|
|
1185
|
+
inFlightAtStart,
|
|
1186
|
+
remaining,
|
|
1187
|
+
requeued,
|
|
1188
|
+
timedOut
|
|
1189
|
+
};
|
|
1190
|
+
};
|
|
1191
|
+
//#endregion
|
|
1192
|
+
//#region src/reliability/leaderController.ts
|
|
1193
|
+
/**
|
|
1194
|
+
* Turns a lease into leadership. On each `tick`: if not leading, try to acquire/steal;
|
|
1195
|
+
* if leading, renew. A failed renew (the lease was stolen while this node was paused
|
|
1196
|
+
* past expiry) drops leadership immediately, so a zombie never keeps acting. No timer
|
|
1197
|
+
* lives here; a caller schedules `tick` at `ttlMs / 3`.
|
|
1198
|
+
*/
|
|
1199
|
+
const createLeaderController = (args) => {
|
|
1200
|
+
const { ownerId, role, store, ttlMs } = args;
|
|
1201
|
+
let leading = false;
|
|
1202
|
+
let token = 0;
|
|
1203
|
+
return {
|
|
1204
|
+
fenceToken: () => leading ? token : 0,
|
|
1205
|
+
isLeader: () => leading,
|
|
1206
|
+
release: async () => {
|
|
1207
|
+
await store.release(role, ownerId);
|
|
1208
|
+
leading = false;
|
|
1209
|
+
token = 0;
|
|
1210
|
+
},
|
|
1211
|
+
tick: async (now) => {
|
|
1212
|
+
if (leading) {
|
|
1213
|
+
if (!(await store.renew(role, ownerId, ttlMs, now)).ok) {
|
|
1214
|
+
leading = false;
|
|
1215
|
+
token = 0;
|
|
1216
|
+
}
|
|
1217
|
+
return;
|
|
1218
|
+
}
|
|
1219
|
+
const acquired = await store.acquireOrSteal(role, ownerId, ttlMs, now);
|
|
1220
|
+
if (acquired.ok) {
|
|
1221
|
+
leading = true;
|
|
1222
|
+
token = acquired.fenceToken;
|
|
1223
|
+
}
|
|
1224
|
+
}
|
|
1225
|
+
};
|
|
1226
|
+
};
|
|
1227
|
+
//#endregion
|
|
1228
|
+
//#region src/reliability/leaseStore.mongo.ts
|
|
1229
|
+
/** Reach the raw Mongoose model the adapter exposes on `db.collections[slug]`. */
|
|
1230
|
+
const model = (payload) => {
|
|
1231
|
+
const found = payload.db.collections[JOBS_LOCKS_SLUG];
|
|
1232
|
+
if (!found) throw new Error(`@10x-media/jobs: missing Mongoose model for "${JOBS_LOCKS_SLUG}"`);
|
|
1233
|
+
return found;
|
|
1234
|
+
};
|
|
1235
|
+
/** Single-document `findOneAndUpdate` is an atomic compare-and-set on MongoDB. */
|
|
1236
|
+
const createMongoLeaseStore = (payload) => {
|
|
1237
|
+
const m = model(payload);
|
|
1238
|
+
const toRecord = (doc) => doc === null ? null : {
|
|
1239
|
+
fenceToken: doc.fenceToken,
|
|
1240
|
+
leaseExpiresAt: doc.leaseExpiresAt ? new Date(doc.leaseExpiresAt) : null,
|
|
1241
|
+
owner: doc.owner ?? null,
|
|
1242
|
+
role: doc.role
|
|
1243
|
+
};
|
|
1244
|
+
return {
|
|
1245
|
+
acquireOrSteal: async (role, owner, ttlMs, now) => {
|
|
1246
|
+
const doc = await m.findOneAndUpdate({
|
|
1247
|
+
$or: [{ owner: null }, { leaseExpiresAt: { $lt: now } }],
|
|
1248
|
+
role
|
|
1249
|
+
}, {
|
|
1250
|
+
$inc: { fenceToken: 1 },
|
|
1251
|
+
$set: {
|
|
1252
|
+
leaseExpiresAt: leaseExpiry(now, ttlMs),
|
|
1253
|
+
owner
|
|
1254
|
+
}
|
|
1255
|
+
}, { new: true });
|
|
1256
|
+
return {
|
|
1257
|
+
fenceToken: doc?.fenceToken ?? 0,
|
|
1258
|
+
ok: doc !== null
|
|
1259
|
+
};
|
|
1260
|
+
},
|
|
1261
|
+
read: async (role) => toRecord(await m.findOne({ role }).lean()),
|
|
1262
|
+
release: async (role, owner) => {
|
|
1263
|
+
await m.findOneAndUpdate({
|
|
1264
|
+
owner,
|
|
1265
|
+
role
|
|
1266
|
+
}, { $set: {
|
|
1267
|
+
leaseExpiresAt: null,
|
|
1268
|
+
owner: null
|
|
1269
|
+
} });
|
|
1270
|
+
},
|
|
1271
|
+
renew: async (role, owner, ttlMs, now) => {
|
|
1272
|
+
const doc = await m.findOneAndUpdate({
|
|
1273
|
+
owner,
|
|
1274
|
+
role
|
|
1275
|
+
}, { $set: { leaseExpiresAt: leaseExpiry(now, ttlMs) } }, { new: true });
|
|
1276
|
+
return {
|
|
1277
|
+
fenceToken: doc?.fenceToken ?? 0,
|
|
1278
|
+
ok: doc !== null
|
|
1279
|
+
};
|
|
1280
|
+
}
|
|
1281
|
+
};
|
|
1282
|
+
};
|
|
1283
|
+
//#endregion
|
|
1284
|
+
//#region src/reliability/leaseStore.postgres.ts
|
|
1285
|
+
/** Payload tables for kebab slugs are the slug with hyphens replaced by underscores. */
|
|
1286
|
+
const tableName = (payload) => {
|
|
1287
|
+
const defaultTable = JOBS_LOCKS_SLUG.replace(/-/g, "_");
|
|
1288
|
+
const map = payload.db.tableNameMap;
|
|
1289
|
+
return (map && typeof map.get === "function" ? map.get(defaultTable) : void 0) ?? defaultTable;
|
|
1290
|
+
};
|
|
1291
|
+
const pool = (payload) => {
|
|
1292
|
+
const found = payload.db.pool;
|
|
1293
|
+
if (!found) throw new Error(`@10x-media/jobs: missing pg pool on the "${payload.db.name}" adapter`);
|
|
1294
|
+
return found;
|
|
1295
|
+
};
|
|
1296
|
+
/** Single-statement `UPDATE ... WHERE <guard> RETURNING` is atomic under READ COMMITTED. */
|
|
1297
|
+
const createPostgresLeaseStore = (payload) => {
|
|
1298
|
+
const table = tableName(payload);
|
|
1299
|
+
const db = pool(payload);
|
|
1300
|
+
const toRecord = (row) => row === void 0 ? null : {
|
|
1301
|
+
fenceToken: Number(row.fence_token),
|
|
1302
|
+
leaseExpiresAt: row.lease_expires_at ? new Date(row.lease_expires_at) : null,
|
|
1303
|
+
owner: row.owner ?? null,
|
|
1304
|
+
role: row.role
|
|
1305
|
+
};
|
|
1306
|
+
return {
|
|
1307
|
+
acquireOrSteal: async (role, owner, ttlMs, now) => {
|
|
1308
|
+
const res = await db.query(`UPDATE ${table}
|
|
1309
|
+
SET owner = $1, lease_expires_at = $2, fence_token = fence_token + 1
|
|
1310
|
+
WHERE role = $3 AND (owner IS NULL OR lease_expires_at < $4)
|
|
1311
|
+
RETURNING fence_token`, [
|
|
1312
|
+
owner,
|
|
1313
|
+
leaseExpiry(now, ttlMs),
|
|
1314
|
+
role,
|
|
1315
|
+
now
|
|
1316
|
+
]);
|
|
1317
|
+
return {
|
|
1318
|
+
fenceToken: res.rowCount === 1 ? Number(res.rows[0]?.fence_token) : 0,
|
|
1319
|
+
ok: res.rowCount === 1
|
|
1320
|
+
};
|
|
1321
|
+
},
|
|
1322
|
+
read: async (role) => {
|
|
1323
|
+
return toRecord((await db.query(`SELECT role, owner, lease_expires_at, fence_token FROM ${table} WHERE role = $1`, [role])).rows[0]);
|
|
1324
|
+
},
|
|
1325
|
+
release: async (role, owner) => {
|
|
1326
|
+
await db.query(`UPDATE ${table} SET owner = NULL, lease_expires_at = NULL WHERE role = $1 AND owner = $2`, [role, owner]);
|
|
1327
|
+
},
|
|
1328
|
+
renew: async (role, owner, ttlMs, now) => {
|
|
1329
|
+
const res = await db.query(`UPDATE ${table}
|
|
1330
|
+
SET lease_expires_at = $1
|
|
1331
|
+
WHERE role = $2 AND owner = $3
|
|
1332
|
+
RETURNING fence_token`, [
|
|
1333
|
+
leaseExpiry(now, ttlMs),
|
|
1334
|
+
role,
|
|
1335
|
+
owner
|
|
1336
|
+
]);
|
|
1337
|
+
return {
|
|
1338
|
+
fenceToken: res.rowCount === 1 ? Number(res.rows[0]?.fence_token) : 0,
|
|
1339
|
+
ok: res.rowCount === 1
|
|
1340
|
+
};
|
|
1341
|
+
}
|
|
1342
|
+
};
|
|
1343
|
+
};
|
|
1344
|
+
//#endregion
|
|
1345
|
+
//#region src/reliability/leaseStore.ts
|
|
1346
|
+
/** Build the lease store for the running adapter. Throws for an unsupported adapter. */
|
|
1347
|
+
const createLeaseStore = (payload) => {
|
|
1348
|
+
if (payload.db.name === "mongoose") return createMongoLeaseStore(payload);
|
|
1349
|
+
if (payload.db.name === "postgres") return createPostgresLeaseStore(payload);
|
|
1350
|
+
throw new Error(`@10x-media/jobs reliability does not support db adapter "${payload.db.name}"`);
|
|
1351
|
+
};
|
|
1352
|
+
//#endregion
|
|
1353
|
+
//#region src/execution/inFlight.ts
|
|
1354
|
+
const JOBS_SLUG = "payload-jobs";
|
|
1355
|
+
/** Count this node's in-flight jobs: `processing: true` AND claimed by `claimedBy`. */
|
|
1356
|
+
const countInFlight = async (payload, claimedBy) => {
|
|
1357
|
+
const count = payload.count;
|
|
1358
|
+
return (await count({
|
|
1359
|
+
collection: JOBS_SLUG,
|
|
1360
|
+
where: { and: [{ processing: { equals: true } }, { claimedBy: { equals: claimedBy } }] }
|
|
1361
|
+
})).totalDocs;
|
|
1362
|
+
};
|
|
1363
|
+
//#endregion
|
|
1364
|
+
//#region src/execution/signals.ts
|
|
1365
|
+
/**
|
|
1366
|
+
* Register `handler` for each signal and return a cleanup that removes them. The
|
|
1367
|
+
* handler fires at most once across all signals (a second signal during drain is
|
|
1368
|
+
* ignored). Payload installs no signal handlers of its own, so these never conflict.
|
|
1369
|
+
* `target` defaults to `process`; tests pass an EventEmitter.
|
|
1370
|
+
*/
|
|
1371
|
+
const installSignalHandlers = (signals, handler, target = process) => {
|
|
1372
|
+
let fired = false;
|
|
1373
|
+
const registered = signals.map((signal) => {
|
|
1374
|
+
const listener = () => {
|
|
1375
|
+
if (fired) return;
|
|
1376
|
+
fired = true;
|
|
1377
|
+
handler(signal);
|
|
1378
|
+
};
|
|
1379
|
+
target.on(signal, listener);
|
|
1380
|
+
return [signal, listener];
|
|
1381
|
+
});
|
|
1382
|
+
return () => {
|
|
1383
|
+
for (const [signal, listener] of registered) target.removeListener(signal, listener);
|
|
1384
|
+
};
|
|
1385
|
+
};
|
|
1386
|
+
//#endregion
|
|
1387
|
+
//#region src/execution/workerCycles.ts
|
|
1388
|
+
const guard = async (logger, label, fn) => {
|
|
1389
|
+
try {
|
|
1390
|
+
await fn();
|
|
1391
|
+
} catch (err) {
|
|
1392
|
+
logger?.error?.(`@10x-media/jobs: worker ${label} cycle error: ${String(err)}`);
|
|
1393
|
+
}
|
|
1394
|
+
};
|
|
1395
|
+
/** One run cycle: claim and execute jobs on this node. Errors are logged, not thrown. */
|
|
1396
|
+
const runCycle = async (deps) => {
|
|
1397
|
+
await guard(deps.logger, "run", deps.runJobs);
|
|
1398
|
+
};
|
|
1399
|
+
/** One maintenance cycle: advance leadership, then schedule only if scheduler-leader. */
|
|
1400
|
+
const maintenanceCycle = async (deps) => {
|
|
1401
|
+
await guard(deps.logger, "leadership", () => deps.tickLeaders(deps.now()));
|
|
1402
|
+
if (deps.isSchedulerLeader()) await guard(deps.logger, "schedule", deps.handleSchedules);
|
|
1403
|
+
};
|
|
1404
|
+
/** One sweep cycle: reclaim stuck jobs only if sweeper-leader. */
|
|
1405
|
+
const sweepCycle = async (deps) => {
|
|
1406
|
+
if (deps.isSweeperLeader()) await guard(deps.logger, "sweep", deps.sweep);
|
|
1407
|
+
};
|
|
1408
|
+
//#endregion
|
|
1409
|
+
//#region src/execution/worker.ts
|
|
1410
|
+
const realSleep = (ms) => new Promise((resolve) => {
|
|
1411
|
+
setTimeout(resolve, ms);
|
|
1412
|
+
});
|
|
1413
|
+
/** A self-skipping interval: a slow tick never overlaps the next (like Croner's `protect`). */
|
|
1414
|
+
const guardedInterval = (fn, ms) => {
|
|
1415
|
+
let busy = false;
|
|
1416
|
+
return setInterval(() => {
|
|
1417
|
+
if (busy) return;
|
|
1418
|
+
busy = true;
|
|
1419
|
+
fn().finally(() => {
|
|
1420
|
+
busy = false;
|
|
1421
|
+
});
|
|
1422
|
+
}, ms);
|
|
1423
|
+
};
|
|
1424
|
+
/**
|
|
1425
|
+
* A plugin-driven worker: runs jobs on every node, schedules and sweeps only while
|
|
1426
|
+
* holding the corresponding leadership lease, and drains gracefully on SIGTERM/SIGINT.
|
|
1427
|
+
* It owns its own timers (never Payload's autoRun cron), because Payload's
|
|
1428
|
+
* `shouldAutoRun` gate permanently stops a cron and cannot follow dynamic leadership.
|
|
1429
|
+
* Leadership timestamps use Payload's swappable `getCurrentDate()`; the drain budget
|
|
1430
|
+
* uses the injected wall clock.
|
|
1431
|
+
*/
|
|
1432
|
+
const createWorker = (args) => {
|
|
1433
|
+
const { payload, reliability } = args;
|
|
1434
|
+
if (!payload.collections["payload-jobs"]) throw new Error("@10x-media/jobs: createWorker requires at least one configured job task. Payload only registers the payload-jobs collection when jobs.tasks or jobs.workflows is non-empty.");
|
|
1435
|
+
const nodeId = resolveNodeId(reliability.leaderId);
|
|
1436
|
+
const leaseStore = createLeaseStore(payload);
|
|
1437
|
+
const jobLeaseStore = createJobLeaseStore(payload);
|
|
1438
|
+
const scheduler = createLeaderController({
|
|
1439
|
+
ownerId: nodeId,
|
|
1440
|
+
role: "scheduler",
|
|
1441
|
+
store: leaseStore,
|
|
1442
|
+
ttlMs: reliability.leaderLeaseTtlMs
|
|
1443
|
+
});
|
|
1444
|
+
const sweeper = createLeaderController({
|
|
1445
|
+
ownerId: nodeId,
|
|
1446
|
+
role: "sweeper",
|
|
1447
|
+
store: leaseStore,
|
|
1448
|
+
ttlMs: reliability.leaderLeaseTtlMs
|
|
1449
|
+
});
|
|
1450
|
+
const runIntervalMs = args.runIntervalMs ?? 2e3;
|
|
1451
|
+
const maintenanceIntervalMs = args.maintenanceIntervalMs ?? Math.max(1e3, Math.floor(reliability.leaderLeaseTtlMs / 3));
|
|
1452
|
+
const sweepIntervalMs = reliability.sweepIntervalMs;
|
|
1453
|
+
const runLimit = args.runLimit ?? 10;
|
|
1454
|
+
const drainTimeoutMs = args.drainTimeoutMs ?? 3e4;
|
|
1455
|
+
const pollIntervalMs = args.pollIntervalMs ?? 500;
|
|
1456
|
+
const destroy = args.destroy ?? (() => payload.destroy());
|
|
1457
|
+
const now = args.now ?? (() => Date.now());
|
|
1458
|
+
const sleep = args.sleep ?? realSleep;
|
|
1459
|
+
const logger = payload.logger;
|
|
1460
|
+
let timers = [];
|
|
1461
|
+
let signalCleanup;
|
|
1462
|
+
let draining;
|
|
1463
|
+
const runJobs = async () => {
|
|
1464
|
+
const state = args.pauseStore ? await args.pauseStore.getState() : emptyPauseState();
|
|
1465
|
+
for (const target of runTargetsForPause(args.queues, state)) await payload.jobs.run({
|
|
1466
|
+
...target,
|
|
1467
|
+
limit: runLimit,
|
|
1468
|
+
silent: true
|
|
1469
|
+
});
|
|
1470
|
+
};
|
|
1471
|
+
const handleSchedules = async () => {
|
|
1472
|
+
await payload.jobs.handleSchedules({ allQueues: true });
|
|
1473
|
+
};
|
|
1474
|
+
const sweep = async () => {
|
|
1475
|
+
await runSweep({
|
|
1476
|
+
isLeader: sweeper.isLeader(),
|
|
1477
|
+
now: getCurrentDate(),
|
|
1478
|
+
options: reliability,
|
|
1479
|
+
payload,
|
|
1480
|
+
store: jobLeaseStore
|
|
1481
|
+
});
|
|
1482
|
+
};
|
|
1483
|
+
const tickLeaders = async (at) => {
|
|
1484
|
+
await scheduler.tick(at);
|
|
1485
|
+
await sweeper.tick(at);
|
|
1486
|
+
};
|
|
1487
|
+
const stopLoops = () => {
|
|
1488
|
+
for (const timer of timers) clearInterval(timer);
|
|
1489
|
+
timers = [];
|
|
1490
|
+
};
|
|
1491
|
+
const removeSignals = () => {
|
|
1492
|
+
signalCleanup?.();
|
|
1493
|
+
signalCleanup = void 0;
|
|
1494
|
+
};
|
|
1495
|
+
const drain = () => {
|
|
1496
|
+
if (draining) return draining;
|
|
1497
|
+
removeSignals();
|
|
1498
|
+
draining = drainWorker({
|
|
1499
|
+
countInFlight: () => countInFlight(payload, nodeId),
|
|
1500
|
+
destroy,
|
|
1501
|
+
logger,
|
|
1502
|
+
now,
|
|
1503
|
+
releaseLeadership: async () => {
|
|
1504
|
+
await scheduler.release();
|
|
1505
|
+
await sweeper.release();
|
|
1506
|
+
},
|
|
1507
|
+
requeueStragglers: async () => (await jobLeaseStore.releaseAllClaims(nodeId)).released,
|
|
1508
|
+
sleep,
|
|
1509
|
+
stopLoops
|
|
1510
|
+
}, {
|
|
1511
|
+
drainTimeoutMs,
|
|
1512
|
+
pollIntervalMs
|
|
1513
|
+
});
|
|
1514
|
+
return draining;
|
|
1515
|
+
};
|
|
1516
|
+
const worker = {
|
|
1517
|
+
drain,
|
|
1518
|
+
isLeader: (role) => role === "scheduler" ? scheduler.isLeader() : sweeper.isLeader(),
|
|
1519
|
+
start: () => {
|
|
1520
|
+
stopLoops();
|
|
1521
|
+
timers = [
|
|
1522
|
+
guardedInterval(() => runCycle({
|
|
1523
|
+
logger,
|
|
1524
|
+
runJobs
|
|
1525
|
+
}), runIntervalMs),
|
|
1526
|
+
guardedInterval(() => maintenanceCycle({
|
|
1527
|
+
handleSchedules,
|
|
1528
|
+
isSchedulerLeader: scheduler.isLeader,
|
|
1529
|
+
logger,
|
|
1530
|
+
now: getCurrentDate,
|
|
1531
|
+
tickLeaders
|
|
1532
|
+
}), maintenanceIntervalMs),
|
|
1533
|
+
guardedInterval(() => sweepCycle({
|
|
1534
|
+
isSweeperLeader: sweeper.isLeader,
|
|
1535
|
+
logger,
|
|
1536
|
+
sweep
|
|
1537
|
+
}), sweepIntervalMs)
|
|
1538
|
+
];
|
|
1539
|
+
},
|
|
1540
|
+
stop: () => {
|
|
1541
|
+
stopLoops();
|
|
1542
|
+
removeSignals();
|
|
1543
|
+
}
|
|
1544
|
+
};
|
|
1545
|
+
if (args.installSignals !== false) {
|
|
1546
|
+
const exit = args.exit ?? ((code) => process.exit(code));
|
|
1547
|
+
signalCleanup = installSignalHandlers(args.signals ?? ["SIGTERM", "SIGINT"], () => {
|
|
1548
|
+
drain().then(() => exit(0)).catch(() => exit(1));
|
|
1549
|
+
});
|
|
1550
|
+
}
|
|
1551
|
+
return worker;
|
|
1552
|
+
};
|
|
1553
|
+
//#endregion
|
|
1554
|
+
//#region src/presets/presets.ts
|
|
1555
|
+
/**
|
|
1556
|
+
* Single-node Docker: one serial claimer, so claim races are moot. Reliability and
|
|
1557
|
+
* queue control on with defaults; the sweeper runs from the in-process worker.
|
|
1558
|
+
*/
|
|
1559
|
+
const singleNodePreset = () => ({
|
|
1560
|
+
queueControl: {},
|
|
1561
|
+
reliability: {}
|
|
1562
|
+
});
|
|
1563
|
+
/**
|
|
1564
|
+
* Multi-node Docker: leader-elected scheduling and sweeping (the default). Pass a
|
|
1565
|
+
* stable `leaderId` per node if you do not want the generated hostname:pid identity.
|
|
1566
|
+
*/
|
|
1567
|
+
const multiNodePreset = (options = {}) => ({
|
|
1568
|
+
queueControl: {},
|
|
1569
|
+
reliability: options.leaderId === void 0 ? {} : { leaderId: options.leaderId }
|
|
1570
|
+
});
|
|
1571
|
+
/**
|
|
1572
|
+
* Serverless (Vercel): no long-running worker, so staleness derives from the platform
|
|
1573
|
+
* hard-kill duration (the function maxDuration) rather than a heartbeat, and the run
|
|
1574
|
+
* and sweep endpoints are guarded by the cron secret. Drive them with Vercel Cron
|
|
1575
|
+
* (see `vercelCrons`).
|
|
1576
|
+
*/
|
|
1577
|
+
const serverlessPreset = (options) => ({
|
|
1578
|
+
queueControl: { access: cronSecretAccess({ envVar: options.cronSecretEnvVar }) },
|
|
1579
|
+
reliability: {
|
|
1580
|
+
jobLeaseTtlMs: options.maxDurationMs,
|
|
1581
|
+
serverless: { maxDurationMs: options.maxDurationMs }
|
|
1582
|
+
}
|
|
1583
|
+
});
|
|
1584
|
+
/**
|
|
1585
|
+
* Build the `vercel.json` `crons` array: one entry hitting the hardened run endpoint
|
|
1586
|
+
* (all queues) and one hitting the sweep endpoint. Defaults to every minute (Vercel
|
|
1587
|
+
* Pro). Vercel sends the `CRON_SECRET` as a Bearer token, which `cronSecretAccess`
|
|
1588
|
+
* checks.
|
|
1589
|
+
*/
|
|
1590
|
+
const vercelCrons = (options = {}) => [{
|
|
1591
|
+
path: options.runPath ?? "/api/payload-jobs/queue-run?allQueues=true",
|
|
1592
|
+
schedule: options.runSchedule ?? "* * * * *"
|
|
1593
|
+
}, {
|
|
1594
|
+
path: options.sweepPath ?? "/api/payload-jobs/queue-sweep",
|
|
1595
|
+
schedule: options.sweepSchedule ?? "* * * * *"
|
|
1596
|
+
}];
|
|
1597
|
+
//#endregion
|
|
1598
|
+
//#region src/index.ts
|
|
1599
|
+
/**
|
|
1600
|
+
* Jobs plugin for Payload v3. Enhances the built-in `payload-jobs` collection
|
|
1601
|
+
* with an ops dashboard (status, queue health, error and log panels) and the
|
|
1602
|
+
* supporting i18n. Authored with `definePlugin` so the automations and webhooks
|
|
1603
|
+
* plugins can detect it by slug. Runs first (`order: 0`).
|
|
1604
|
+
*/
|
|
1605
|
+
const jobs = definePlugin({
|
|
1606
|
+
slug: "@10x-media/jobs",
|
|
1607
|
+
order: 0,
|
|
1608
|
+
plugin: ({ config, plugins: _plugins, ...options }) => {
|
|
1609
|
+
if (options.disabled === true) return config;
|
|
1610
|
+
registerTranslations(config);
|
|
1611
|
+
registerJobsEnhancements(config, options);
|
|
1612
|
+
const reliability = resolveReliabilityOptions(options.reliability);
|
|
1613
|
+
if (reliability) registerReliability(config, reliability);
|
|
1614
|
+
const queueControl = resolveQueueControlOptions(options.queueControl);
|
|
1615
|
+
if (queueControl) registerQueueControl(config, queueControl, reliability);
|
|
1616
|
+
return config;
|
|
1617
|
+
}
|
|
1618
|
+
});
|
|
1619
|
+
//#endregion
|
|
1620
|
+
export { JOBS_LOCKS_SLUG, LEADER_ROLES, autoRunConfig, createJobLeaseStore, createLeaderController, createLeaseStore, createPauseStore, createWorker, cronSecretAccess, decideRecovery, deriveJobStatus, drainWorker, getQueueHealth, jobs, loggedInAccess, multiNodePreset, resolveReliabilityOptions, runSweep, serverlessPreset, singleNodePreset, vercelCrons, withIdempotencyKey };
|
|
1621
|
+
|
|
1622
|
+
//# sourceMappingURL=index.js.map
|