@10x-media/jobs 0.1.0-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js ADDED
@@ -0,0 +1,1622 @@
1
+ import { t as keys } from "./keys-BYs8GTJy.js";
2
+ import { n as labelForKey } from "./server-JsvYf5vy.js";
3
+ import { t as translations } from "./translations-KyQ3962W.js";
4
+ import { t as deriveJobStatus } from "./deriveJobStatus-CM_sCsgm.js";
5
+ import { ValidationError, definePlugin, getCurrentDate } from "payload";
6
+ import { deepMergeSimple } from "payload/shared";
7
+ import { hostname } from "node:os";
8
+ //#region src/plugin/resolve.ts
9
+ /** Apply an `Override`, falling back to `defaults` when none was provided. */
10
+ const resolve = (override, defaults) => {
11
+ if (override === void 0) return defaults;
12
+ return typeof override === "function" ? override(defaults) : override;
13
+ };
14
+ //#endregion
15
+ //#region src/plugin/registerJobsEnhancements.ts
16
+ const STATUS_CELL = "@10x-media/jobs/client#JobStatusCell";
17
+ const STATUS_HEADER = "@10x-media/jobs/client#JobStatusHeader";
18
+ const RELATIVE_CELL = "@10x-media/jobs/client#RelativeTimeCell";
19
+ const ERROR_PANEL = "@10x-media/jobs/client#JobErrorPanel";
20
+ const LOG_TIMELINE = "@10x-media/jobs/client#JobLogTimeline";
21
+ const DOC_DESCRIPTION = "@10x-media/jobs/client#JobDocDescription";
22
+ const HEALTH_BAR = "@10x-media/jobs/rsc#JobsHealthBar";
23
+ /** Stored field that titles the document: the workflow or task the job runs. */
24
+ const TITLE_FIELD = {
25
+ name: "jobTitle",
26
+ type: "text",
27
+ admin: {
28
+ disableListColumn: true,
29
+ hidden: true
30
+ }
31
+ };
32
+ /**
33
+ * Keep `jobTitle` in sync with the job's workflow or task on every write, falling
34
+ * back to the existing doc so partial state updates never clobber it.
35
+ */
36
+ const setJobTitle = ({ data, originalDoc }) => {
37
+ const source = {
38
+ ...originalDoc ?? {},
39
+ ...data
40
+ };
41
+ const workflow = typeof source.workflowSlug === "string" ? source.workflowSlug : void 0;
42
+ const task = typeof source.taskSlug === "string" ? source.taskSlug : void 0;
43
+ const title = workflow || task;
44
+ return title ? {
45
+ ...data,
46
+ jobTitle: title
47
+ } : data;
48
+ };
49
+ /** Our default document Field component for specific job fields. */
50
+ const FIELD_COMPONENTS = {
51
+ error: ERROR_PANEL,
52
+ log: LOG_TIMELINE
53
+ };
54
+ const DEFAULT_JOBS_COLUMNS = [
55
+ "workflowSlug",
56
+ "status",
57
+ "queue",
58
+ "totalTried",
59
+ "updatedAt"
60
+ ];
61
+ /** Friendlier labels for a few of Payload's default job fields. */
62
+ const FIELD_LABELS = {
63
+ totalTried: labelForKey(keys.fieldAttempts),
64
+ workflowSlug: labelForKey(keys.fieldWorkflow)
65
+ };
66
+ /**
67
+ * Payload appends `createdAt`/`updatedAt` after overrides run, but only when they
68
+ * are absent, so we inject our own to attach the relative-time cell to them.
69
+ */
70
+ const TIMESTAMP_FIELDS = [{
71
+ name: "createdAt",
72
+ type: "date",
73
+ label: labelForKey(keys.fieldCreated),
74
+ index: true,
75
+ admin: {
76
+ disableBulkEdit: true,
77
+ hidden: true
78
+ }
79
+ }, {
80
+ name: "updatedAt",
81
+ type: "date",
82
+ label: labelForKey(keys.fieldUpdated),
83
+ index: true,
84
+ admin: {
85
+ disableBulkEdit: true,
86
+ hidden: true
87
+ }
88
+ }];
89
+ const fieldName = (field) => "name" in field && typeof field.name === "string" ? field.name : void 0;
90
+ const setCell = (field, cell) => ({
91
+ ...field,
92
+ admin: {
93
+ ...field.admin,
94
+ components: {
95
+ ...field.admin?.components,
96
+ Cell: cell
97
+ }
98
+ }
99
+ });
100
+ /**
101
+ * Resolve a field's list cell: an explicit `cells` override wins (`false` keeps
102
+ * Payload's default), otherwise date fields get the relative-time cell.
103
+ */
104
+ const resolveCell = (field, cells) => {
105
+ const name = fieldName(field);
106
+ const override = name ? cells?.[name] : void 0;
107
+ if (override === false) return field;
108
+ if (override) return setCell(field, override);
109
+ if (field.type === "date") return setCell(field, RELATIVE_CELL);
110
+ return field;
111
+ };
112
+ /** Execution-state fields surfaced in the header; hidden from the form, kept as data/columns. */
113
+ const HIDE_ALWAYS = new Set([
114
+ "completedAt",
115
+ "totalTried",
116
+ "hasError",
117
+ "processing",
118
+ "taskStatus"
119
+ ]);
120
+ /** Inputs shown only when creating a job; hidden once it exists. */
121
+ const HIDE_ON_EDIT = new Set(["waitUntil"]);
122
+ /** Execution output: read-only in the admin (admin-only, so the runner can still write it). */
123
+ const READONLY_OUTPUTS = new Set(["error", "log"]);
124
+ /** Config inputs: editable on Create, read-only on edit; kept in the sidebar. */
125
+ const READONLY_INPUTS = new Set([
126
+ "input",
127
+ "workflowSlug",
128
+ "taskSlug",
129
+ "queue"
130
+ ]);
131
+ /**
132
+ * Lock a field for the read-only-record model. Runner-written fields use
133
+ * admin-only props (`hidden`/`readOnly`) so the jobs runner is never blocked;
134
+ * inputs use field `access.update` so they stay editable on Create but lock once
135
+ * the job exists.
136
+ */
137
+ const lockField = (field, name) => {
138
+ if (HIDE_ALWAYS.has(name)) return {
139
+ ...field,
140
+ admin: {
141
+ ...field.admin,
142
+ hidden: true
143
+ }
144
+ };
145
+ if (HIDE_ON_EDIT.has(name)) return {
146
+ ...field,
147
+ admin: {
148
+ ...field.admin,
149
+ condition: (_data, _siblingData, { operation }) => operation === "create"
150
+ }
151
+ };
152
+ if (READONLY_OUTPUTS.has(name)) return {
153
+ ...field,
154
+ admin: {
155
+ ...field.admin,
156
+ readOnly: true
157
+ }
158
+ };
159
+ if (READONLY_INPUTS.has(name)) {
160
+ const access = "access" in field && field.access ? field.access : {};
161
+ return {
162
+ ...field,
163
+ access: {
164
+ ...access,
165
+ update: () => false
166
+ }
167
+ };
168
+ }
169
+ return field;
170
+ };
171
+ /** Recursively relabel fields, apply cell overrides, and lock the record, descending into tabs. */
172
+ const enhanceFields = (fields, cells, lockRecord) => fields.map((field) => {
173
+ let next = field;
174
+ const name = fieldName(field);
175
+ if (name && FIELD_LABELS[name]) next = {
176
+ ...field,
177
+ label: FIELD_LABELS[name]
178
+ };
179
+ next = resolveCell(next, cells);
180
+ if (lockRecord && name) next = lockField(next, name);
181
+ if (name && FIELD_COMPONENTS[name]) next = {
182
+ ...next,
183
+ admin: {
184
+ ...next.admin,
185
+ components: {
186
+ ...next.admin?.components,
187
+ Field: FIELD_COMPONENTS[name]
188
+ }
189
+ }
190
+ };
191
+ if ("tabs" in next) next = {
192
+ ...next,
193
+ tabs: next.tabs.map((tab) => ({
194
+ ...tab,
195
+ fields: enhanceFields(tab.fields, cells, lockRecord)
196
+ }))
197
+ };
198
+ else if ("fields" in next) next = {
199
+ ...next,
200
+ fields: enhanceFields(next.fields, cells, lockRecord)
201
+ };
202
+ return next;
203
+ });
204
+ /** Build the derived Status column, honoring the `status` override (`false` removes it). */
205
+ const buildStatusField = (status) => {
206
+ if (status === false) return null;
207
+ return {
208
+ name: "status",
209
+ type: "ui",
210
+ label: labelForKey(keys.fieldStatus),
211
+ admin: { components: {
212
+ Cell: status ?? STATUS_CELL,
213
+ Field: STATUS_HEADER
214
+ } }
215
+ };
216
+ };
217
+ /**
218
+ * Enhance Payload's built-in `payload-jobs` collection through its sanctioned
219
+ * `jobs.jobsCollectionOverrides` seam. Composes with a host-provided override
220
+ * (ours layer on top of theirs) so neither clobbers the other, and every piece
221
+ * is replaceable through `options` (`status`, `cells`, `beforeListTable`).
222
+ */
223
+ const registerJobsEnhancements = (config, options) => {
224
+ const existing = config.jobs?.jobsCollectionOverrides;
225
+ const enhance = ({ defaultJobsCollection }) => {
226
+ const base = existing ? existing({ defaultJobsCollection }) : defaultJobsCollection;
227
+ const existingNames = new Set(base.fields.flatMap((field) => {
228
+ const name = fieldName(field);
229
+ return name ? [name] : [];
230
+ }));
231
+ const timestamps = TIMESTAMP_FIELDS.filter((field) => {
232
+ const name = fieldName(field);
233
+ return name ? !existingNames.has(name) : false;
234
+ }).map((field) => resolveCell(field, options.cells));
235
+ const statusField = buildStatusField(options.status);
236
+ const lockRecord = options.readOnlyRecord !== false;
237
+ const hostBeforeTable = base.admin?.components?.beforeListTable ?? [];
238
+ const ourBeforeTable = options.beforeListTable === false ? [] : options.beforeListTable ?? [{
239
+ path: HEALTH_BAR,
240
+ serverProps: { cap: options.healthBarCap }
241
+ }];
242
+ return {
243
+ ...base,
244
+ admin: {
245
+ ...base.admin,
246
+ hidden: options.hidden ?? false,
247
+ useAsTitle: "jobTitle",
248
+ defaultColumns: resolve(options.defaultColumns, DEFAULT_JOBS_COLUMNS),
249
+ components: {
250
+ ...base.admin?.components,
251
+ Description: DOC_DESCRIPTION,
252
+ beforeListTable: [...hostBeforeTable, ...ourBeforeTable]
253
+ }
254
+ },
255
+ hooks: {
256
+ ...base.hooks,
257
+ beforeChange: [...base.hooks?.beforeChange ?? [], setJobTitle]
258
+ },
259
+ fields: [
260
+ ...statusField ? [statusField] : [],
261
+ TITLE_FIELD,
262
+ ...enhanceFields(base.fields, options.cells, lockRecord),
263
+ ...timestamps
264
+ ]
265
+ };
266
+ };
267
+ config.jobs = {
268
+ ...config.jobs,
269
+ jobsCollectionOverrides: enhance
270
+ };
271
+ };
272
+ //#endregion
273
+ //#region src/plugin/registerTranslations.ts
274
+ /**
275
+ * Merge this plugin's translations into the host config. A host value wins on
276
+ * conflict (`deepMergeSimple` lets the second argument override), so projects can
277
+ * override any string.
278
+ */
279
+ const registerTranslations = (config) => {
280
+ config.i18n ??= {};
281
+ config.i18n.translations = deepMergeSimple(translations, config.i18n.translations ?? {});
282
+ };
283
+ //#endregion
284
+ //#region src/queueControl/access.ts
285
+ /** The default endpoint guard: logged-in users only (safer than Payload's open default). */
286
+ const loggedInAccess = ({ req }) => Boolean(req.user);
287
+ /**
288
+ * An access checker for serverless cron triggers: a logged-in user passes, otherwise
289
+ * the request must carry `Authorization: Bearer ${process.env[envVar]}`. Payload ships
290
+ * no built-in secret check, so this implements the canonical Vercel-cron pattern.
291
+ */
292
+ const cronSecretAccess = (options = {}) => {
293
+ const envVar = options.envVar ?? "CRON_SECRET";
294
+ return ({ req }) => {
295
+ if (req.user) return true;
296
+ const secret = process.env[envVar];
297
+ if (!secret) return false;
298
+ return req.headers.get("authorization") === `Bearer ${secret}`;
299
+ };
300
+ };
301
+ //#endregion
302
+ //#region src/queueControl/options.ts
303
+ /** Resolve queue-control options to a fully-defaulted object, or `null` when disabled. */
304
+ const resolveQueueControlOptions = (options) => {
305
+ if (options === void 0 || options === false) return null;
306
+ return {
307
+ access: options.access ?? loggedInAccess,
308
+ queues: options.queues ?? ["default"]
309
+ };
310
+ };
311
+ //#endregion
312
+ //#region src/reliability/leaseLogic.ts
313
+ /** The expiry instant for a lease acquired/renewed at `now` for `ttlMs`. */
314
+ const leaseExpiry = (now, ttlMs) => new Date(now.getTime() + ttlMs);
315
+ //#endregion
316
+ //#region src/reliability/jobLeaseStore.mongo.ts
317
+ const JOBS_SLUG$4 = "payload-jobs";
318
+ /** Reach the raw Mongoose model the adapter exposes on `db.collections[slug]`. */
319
+ const model$1 = (payload) => {
320
+ const found = payload.db.collections[JOBS_SLUG$4];
321
+ if (!found) throw new Error(`@10x-media/jobs: missing Mongoose model for "${JOBS_SLUG$4}"`);
322
+ return found;
323
+ };
324
+ /** The stale-orphan predicate shared by requeue and dead-letter. */
325
+ const stale = (now, fallbackMs) => ({
326
+ hasError: { $ne: true },
327
+ processing: true,
328
+ $or: [{ leaseExpiresAt: { $lt: now } }, {
329
+ leaseExpiresAt: null,
330
+ updatedAt: { $lt: new Date(now.getTime() - fallbackMs) }
331
+ }]
332
+ });
333
+ /** Single-document `findOneAndUpdate` is an atomic compare-and-set on MongoDB. */
334
+ const createMongoJobLeaseStore = (payload) => {
335
+ const m = model$1(payload);
336
+ return {
337
+ deadLetter: async ({ error, fallbackMs, jobId, now }) => {
338
+ return { ok: await m.findOneAndUpdate({
339
+ _id: jobId,
340
+ ...stale(now, fallbackMs)
341
+ }, {
342
+ $inc: { fenceToken: 1 },
343
+ $set: {
344
+ claimedBy: null,
345
+ error,
346
+ hasError: true,
347
+ leaseExpiresAt: null,
348
+ processing: false
349
+ }
350
+ }, { new: true }) !== null };
351
+ },
352
+ read: async (jobId) => {
353
+ const doc = await m.findOne({ _id: jobId }).lean();
354
+ if (doc === null) return null;
355
+ return {
356
+ claimedBy: doc.claimedBy ?? null,
357
+ fenceToken: doc.fenceToken ?? 0,
358
+ leaseExpiresAt: doc.leaseExpiresAt ? new Date(doc.leaseExpiresAt) : null,
359
+ processing: doc.processing === true,
360
+ recoveryAttempts: doc.recoveryAttempts ?? 0,
361
+ updatedAt: doc.updatedAt ? new Date(doc.updatedAt) : /* @__PURE__ */ new Date(0)
362
+ };
363
+ },
364
+ renew: async (jobId, fenceToken, ttlMs, now) => {
365
+ return { ok: await m.findOneAndUpdate({
366
+ _id: jobId,
367
+ fenceToken
368
+ }, { $set: { leaseExpiresAt: leaseExpiry(now, ttlMs) } }, { new: true }) !== null };
369
+ },
370
+ releaseAllClaims: async (owner) => {
371
+ return { released: (await m.updateMany({
372
+ claimedBy: owner,
373
+ processing: true
374
+ }, {
375
+ $inc: {
376
+ fenceToken: 1,
377
+ recoveryAttempts: 1
378
+ },
379
+ $set: {
380
+ claimedBy: null,
381
+ leaseExpiresAt: null,
382
+ processing: false,
383
+ waitUntil: null
384
+ }
385
+ })).modifiedCount ?? 0 };
386
+ },
387
+ requeue: async (jobId, now, fallbackMs) => {
388
+ return { ok: await m.findOneAndUpdate({
389
+ _id: jobId,
390
+ ...stale(now, fallbackMs)
391
+ }, {
392
+ $inc: {
393
+ fenceToken: 1,
394
+ recoveryAttempts: 1
395
+ },
396
+ $set: {
397
+ claimedBy: null,
398
+ leaseExpiresAt: null,
399
+ processing: false,
400
+ waitUntil: null
401
+ }
402
+ }, { new: true }) !== null };
403
+ },
404
+ stampClaim: async (jobId, owner, ttlMs, now) => {
405
+ const doc = await m.findOneAndUpdate({
406
+ _id: jobId,
407
+ processing: true
408
+ }, {
409
+ $inc: { fenceToken: 1 },
410
+ $set: {
411
+ claimedBy: owner,
412
+ leaseExpiresAt: leaseExpiry(now, ttlMs),
413
+ startedAt: now
414
+ }
415
+ }, { new: true });
416
+ return {
417
+ fenceToken: doc?.fenceToken ?? 0,
418
+ ok: doc !== null
419
+ };
420
+ }
421
+ };
422
+ };
423
+ //#endregion
424
+ //#region src/reliability/jobLeaseStore.postgres.ts
425
+ const JOBS_SLUG$3 = "payload-jobs";
426
+ /** Payload tables for kebab slugs are the slug with hyphens replaced by underscores. */
427
+ const tableName$1 = (payload) => {
428
+ const defaultTable = JOBS_SLUG$3.replace(/-/g, "_");
429
+ const map = payload.db.tableNameMap;
430
+ return (map && typeof map.get === "function" ? map.get(defaultTable) : void 0) ?? defaultTable;
431
+ };
432
+ const pool$1 = (payload) => {
433
+ const found = payload.db.pool;
434
+ if (!found) throw new Error(`@10x-media/jobs: missing pg pool on the "${payload.db.name}" adapter`);
435
+ return found;
436
+ };
437
+ /** Single-statement `UPDATE ... WHERE <guard> RETURNING` is atomic under READ COMMITTED. */
438
+ const createPostgresJobLeaseStore = (payload) => {
439
+ const table = tableName$1(payload);
440
+ const db = pool$1(payload);
441
+ const staleGuard = "processing = true AND has_error IS NOT TRUE AND (lease_expires_at < $2 OR (lease_expires_at IS NULL AND updated_at < $3))";
442
+ return {
443
+ deadLetter: async ({ error, fallbackMs, jobId, now }) => {
444
+ return { ok: (await db.query(`UPDATE ${table}
445
+ SET processing = false, has_error = true, error = $4::jsonb,
446
+ lease_expires_at = NULL, claimed_by = NULL, fence_token = COALESCE(fence_token, 0) + 1
447
+ WHERE id = $1 AND ${staleGuard}
448
+ RETURNING id`, [
449
+ jobId,
450
+ now,
451
+ new Date(now.getTime() - fallbackMs),
452
+ JSON.stringify(error)
453
+ ])).rowCount === 1 };
454
+ },
455
+ read: async (jobId) => {
456
+ const row = (await db.query(`SELECT processing, lease_expires_at, claimed_by, fence_token, recovery_attempts, updated_at
457
+ FROM ${table} WHERE id = $1`, [jobId])).rows[0];
458
+ if (row === void 0) return null;
459
+ return {
460
+ claimedBy: row.claimed_by ?? null,
461
+ fenceToken: Number(row.fence_token ?? 0),
462
+ leaseExpiresAt: row.lease_expires_at ? new Date(row.lease_expires_at) : null,
463
+ processing: row.processing === true,
464
+ recoveryAttempts: Number(row.recovery_attempts ?? 0),
465
+ updatedAt: row.updated_at ? new Date(row.updated_at) : /* @__PURE__ */ new Date(0)
466
+ };
467
+ },
468
+ renew: async (jobId, fenceToken, ttlMs, now) => {
469
+ return { ok: (await db.query(`UPDATE ${table} SET lease_expires_at = $1
470
+ WHERE id = $2 AND fence_token = $3
471
+ RETURNING id`, [
472
+ leaseExpiry(now, ttlMs),
473
+ jobId,
474
+ fenceToken
475
+ ])).rowCount === 1 };
476
+ },
477
+ releaseAllClaims: async (owner) => {
478
+ return { released: (await db.query(`UPDATE ${table}
479
+ SET processing = false, claimed_by = NULL, lease_expires_at = NULL, wait_until = NULL,
480
+ recovery_attempts = COALESCE(recovery_attempts, 0) + 1,
481
+ fence_token = COALESCE(fence_token, 0) + 1
482
+ WHERE claimed_by = $1 AND processing = true
483
+ RETURNING id`, [owner])).rowCount ?? 0 };
484
+ },
485
+ requeue: async (jobId, now, fallbackMs) => {
486
+ return { ok: (await db.query(`UPDATE ${table}
487
+ SET processing = false, lease_expires_at = NULL, claimed_by = NULL, wait_until = NULL,
488
+ recovery_attempts = COALESCE(recovery_attempts, 0) + 1,
489
+ fence_token = COALESCE(fence_token, 0) + 1
490
+ WHERE id = $1 AND ${staleGuard}
491
+ RETURNING id`, [
492
+ jobId,
493
+ now,
494
+ new Date(now.getTime() - fallbackMs)
495
+ ])).rowCount === 1 };
496
+ },
497
+ stampClaim: async (jobId, owner, ttlMs, now) => {
498
+ const res = await db.query(`UPDATE ${table}
499
+ SET started_at = $1, claimed_by = $2, lease_expires_at = $3,
500
+ fence_token = COALESCE(fence_token, 0) + 1
501
+ WHERE id = $4 AND processing = true
502
+ RETURNING fence_token`, [
503
+ now,
504
+ owner,
505
+ leaseExpiry(now, ttlMs),
506
+ jobId
507
+ ]);
508
+ const ok = res.rowCount === 1;
509
+ return {
510
+ fenceToken: ok ? Number(res.rows[0]?.fence_token) : 0,
511
+ ok
512
+ };
513
+ }
514
+ };
515
+ };
516
+ //#endregion
517
+ //#region src/reliability/jobLeaseStore.ts
518
+ /** Build the job-lease store for the running adapter. Throws for an unsupported adapter. */
519
+ const createJobLeaseStore = (payload) => {
520
+ if (payload.db.name === "mongoose") return createMongoJobLeaseStore(payload);
521
+ if (payload.db.name === "postgres") return createPostgresJobLeaseStore(payload);
522
+ throw new Error(`@10x-media/jobs reliability does not support db adapter "${payload.db.name}"`);
523
+ };
524
+ //#endregion
525
+ //#region src/reliability/leaseMode.ts
526
+ /**
527
+ * Heartbeat mode renews a running job's lease from a side timer. Serverless mode
528
+ * never renews (the function is hard-killed at maxDuration with no SIGTERM), so the
529
+ * stamp encodes the platform limit and the sweeper recovers from it.
530
+ */
531
+ const isHeartbeatMode = (options) => options.serverlessMaxDurationMs === null;
532
+ /**
533
+ * The lease TTL stamped when a worker claims a job, and the grace used to age out a
534
+ * claimed-but-never-stamped (null lease) job by its updatedAt. In serverless mode
535
+ * this is the platform hard-kill duration; otherwise the configured job lease TTL.
536
+ */
537
+ const initialLeaseTtlMs = (options) => options.serverlessMaxDurationMs ?? options.jobLeaseTtlMs;
538
+ //#endregion
539
+ //#region src/reliability/recoveryDecision.ts
540
+ /**
541
+ * Requeue an orphaned job while it is below the recovery cap; dead-letter once it
542
+ * reaches the cap, to stop a poison job from thrashing the queue. `recoveryAttempts`
543
+ * is the count before this pass (a requeue increments it). `maxRecoveries` of 0
544
+ * dead-letters on the first orphan.
545
+ */
546
+ const decideRecovery = (recoveryAttempts, maxRecoveries) => recoveryAttempts < maxRecoveries ? "requeue" : "deadLetter";
547
+ //#endregion
548
+ //#region src/reliability/sweeper.ts
549
+ const JOBS_SLUG$2 = "payload-jobs";
550
+ /** The dead-letter error payload written when a job reaches the recovery cap. */
551
+ const deadLetterError = (attempts) => ({
552
+ cancelled: false,
553
+ message: `@10x-media/jobs sweeper: dead-lettered after reaching the recovery cap (${attempts} attempts)`,
554
+ recovered: false
555
+ });
556
+ /**
557
+ * One sweep pass: find stale `processing: true` orphans and either requeue them
558
+ * (below the recovery cap) or dead-letter them (at the cap), each through a fenced
559
+ * conditional write that re-checks staleness, so a job that renewed between the find
560
+ * and the write is skipped and two overlapping sweepers never both reclaim the same
561
+ * orphan. Gated by leadership: callers pass `isLeader` from the Plan 1 sweeper lease.
562
+ */
563
+ const runSweep = async (args) => {
564
+ const { isLeader = true, limit = 100, options, payload } = args;
565
+ const result = {
566
+ deadLettered: 0,
567
+ requeued: 0,
568
+ scanned: 0
569
+ };
570
+ if (!isLeader) return result;
571
+ const now = args.now ?? getCurrentDate();
572
+ const store = args.store ?? createJobLeaseStore(payload);
573
+ const fallbackMs = initialLeaseTtlMs(options);
574
+ const cutoff = new Date(now.getTime() - fallbackMs).toISOString();
575
+ const where = { and: [
576
+ { processing: { equals: true } },
577
+ { completedAt: { exists: false } },
578
+ { hasError: { not_equals: true } },
579
+ { or: [{ leaseExpiresAt: { less_than: now.toISOString() } }, { and: [{ leaseExpiresAt: { exists: false } }, { updatedAt: { less_than: cutoff } }] }] }
580
+ ] };
581
+ const find = payload.find;
582
+ const { docs } = await find({
583
+ collection: JOBS_SLUG$2,
584
+ depth: 0,
585
+ limit,
586
+ where
587
+ });
588
+ result.scanned = docs.length;
589
+ for (const doc of docs) {
590
+ const attempts = typeof doc.recoveryAttempts === "number" ? doc.recoveryAttempts : 0;
591
+ if (decideRecovery(attempts, options.maxRecoveries) === "requeue") {
592
+ if ((await store.requeue(doc.id, now, fallbackMs)).ok) result.requeued += 1;
593
+ } else if ((await store.deadLetter({
594
+ error: deadLetterError(attempts),
595
+ fallbackMs,
596
+ jobId: doc.id,
597
+ now
598
+ })).ok) result.deadLettered += 1;
599
+ }
600
+ return result;
601
+ };
602
+ //#endregion
603
+ //#region src/queueControl/pauseState.ts
604
+ /** The default (nothing paused) state. */
605
+ const emptyPauseState = () => ({
606
+ global: false,
607
+ queues: []
608
+ });
609
+ /** Pause everything (no queue) or a single queue (idempotent). */
610
+ const applyPause = (state, queue) => {
611
+ if (queue === void 0) return {
612
+ ...state,
613
+ global: true
614
+ };
615
+ return state.queues.includes(queue) ? state : {
616
+ ...state,
617
+ queues: [...state.queues, queue]
618
+ };
619
+ };
620
+ /** Resume everything (no queue, clears the global flag) or a single queue. */
621
+ const applyResume = (state, queue) => {
622
+ if (queue === void 0) return {
623
+ ...state,
624
+ global: false
625
+ };
626
+ return {
627
+ ...state,
628
+ queues: state.queues.filter((q) => q !== queue)
629
+ };
630
+ };
631
+ /** Whether `queue` is currently paused (a global pause covers every queue). */
632
+ const isPaused = (state, queue) => state.global || state.queues.includes(queue);
633
+ //#endregion
634
+ //#region src/queueControl/pauseStore.ts
635
+ const PAUSE_KEY = "@10x-media/jobs:pause-state";
636
+ /**
637
+ * Build the pause store over `payload.kv` (always available, durable, cluster-wide).
638
+ * Pause and resume read-modify-write the single state value; this is last-writer-wins
639
+ * (kv has no atomic compare-and-set), which is acceptable for rare admin actions.
640
+ */
641
+ const createPauseStore = (payload) => {
642
+ const getState = async () => await payload.kv.get(PAUSE_KEY) ?? emptyPauseState();
643
+ return {
644
+ getState,
645
+ isPaused: async (queue) => isPaused(await getState(), queue),
646
+ pause: async (queue) => {
647
+ await payload.kv.set(PAUSE_KEY, applyPause(await getState(), queue));
648
+ },
649
+ resume: async (queue) => {
650
+ await payload.kv.set(PAUSE_KEY, applyResume(await getState(), queue));
651
+ }
652
+ };
653
+ };
654
+ //#endregion
655
+ //#region src/queueControl/queueHealth.ts
656
+ const JOBS_SLUG$1 = "payload-jobs";
657
+ const STATS_SLUG = "payload-jobs-stats";
658
+ const queueClause = (queue) => queue ? [{ queue: { equals: queue } }] : [];
659
+ const pendingWhere = (queue) => ({ and: [
660
+ { completedAt: { exists: false } },
661
+ { hasError: { not_equals: true } },
662
+ { processing: { equals: false } },
663
+ ...queueClause(queue)
664
+ ] });
665
+ const processingWhere = (queue) => ({ and: [{ processing: { equals: true } }, ...queueClause(queue)] });
666
+ const failedWhere = (queue) => ({ and: [{ hasError: { equals: true } }, ...queueClause(queue)] });
667
+ const recoveredWhere = (queue) => ({ and: [{ recoveryAttempts: { greater_than: 0 } }, ...queueClause(queue)] });
668
+ const readStats = async (payload) => {
669
+ const db = payload.db;
670
+ try {
671
+ return await db.findGlobal({ slug: STATS_SLUG });
672
+ } catch {
673
+ return null;
674
+ }
675
+ };
676
+ const lastScheduledRunFor = (stats, queue) => {
677
+ const entry = stats?.stats?.scheduledRuns?.queues?.[queue];
678
+ if (!entry) return null;
679
+ const runs = [...Object.values(entry.tasks ?? {}), ...Object.values(entry.workflows ?? {})].map((r) => r.lastScheduledRun).filter((r) => typeof r === "string");
680
+ return runs.length > 0 ? runs.sort().at(-1) ?? null : null;
681
+ };
682
+ /**
683
+ * Aggregate queue health via `payload.count` per state (pending, processing, failed,
684
+ * and, when reliability is on, recovered), plus the oldest-pending age and the last
685
+ * scheduled run per queue. Reuses Payload's own run-selector predicates. The stats
686
+ * global is read through `payload.db.findGlobal` inside a try/catch because that call
687
+ * throws when the global does not exist (no schedule has run).
688
+ */
689
+ const getQueueHealth = async (payload, options = {}) => {
690
+ const queues = options.queues ?? ["default"];
691
+ const includeRecovered = options.includeRecovered ?? false;
692
+ const count = payload.count;
693
+ const find = payload.find;
694
+ const countWhere = async (where) => (await count({
695
+ collection: JOBS_SLUG$1,
696
+ where
697
+ })).totalDocs;
698
+ const totals = {
699
+ failed: await countWhere(failedWhere()),
700
+ pending: await countWhere(pendingWhere()),
701
+ processing: await countWhere(processingWhere()),
702
+ recovered: includeRecovered ? await countWhere(recoveredWhere()) : 0
703
+ };
704
+ const oldestCreated = (await find({
705
+ collection: JOBS_SLUG$1,
706
+ depth: 0,
707
+ limit: 1,
708
+ sort: "createdAt",
709
+ where: pendingWhere()
710
+ })).docs[0]?.createdAt;
711
+ const nowMs = (options.now ?? /* @__PURE__ */ new Date()).getTime();
712
+ const oldestPendingAgeMs = oldestCreated ? nowMs - new Date(oldestCreated).getTime() : null;
713
+ const stats = await readStats(payload);
714
+ const perQueue = [];
715
+ for (const queue of queues) perQueue.push({
716
+ failed: await countWhere(failedWhere(queue)),
717
+ lastScheduledRun: lastScheduledRunFor(stats, queue),
718
+ pending: await countWhere(pendingWhere(queue)),
719
+ processing: await countWhere(processingWhere(queue)),
720
+ queue,
721
+ recovered: includeRecovered ? await countWhere(recoveredWhere(queue)) : 0
722
+ });
723
+ return {
724
+ oldestPendingAgeMs,
725
+ queues: perQueue,
726
+ totals
727
+ };
728
+ };
729
+ //#endregion
730
+ //#region src/queueControl/runTargets.ts
731
+ /**
732
+ * Compute the run targets for one cycle, given the worker's configured queues (or
733
+ * undefined for all queues) and the current pause state. A global pause yields no
734
+ * targets. A specific queue list drops the paused queues. All-queues running excludes
735
+ * paused queues via `not_in` paired with `allQueues: true` (the exclusion requires
736
+ * allQueues, because Payload's built-in single-queue filter only applies otherwise).
737
+ */
738
+ const runTargetsForPause = (queues, state) => {
739
+ if (state.global) return [];
740
+ if (queues && queues.length > 0) return queues.filter((queue) => !state.queues.includes(queue)).map((queue) => ({ queue }));
741
+ if (state.queues.length > 0) return [{
742
+ allQueues: true,
743
+ where: { queue: { not_in: state.queues } }
744
+ }];
745
+ return [{ allQueues: true }];
746
+ };
747
+ //#endregion
748
+ //#region src/queueControl/endpoints.ts
749
+ const unauthorized = () => Response.json({ message: "Unauthorized" }, { status: 401 });
750
+ /** GET queue health, grouped by queue. */
751
+ const statusEndpoint = (deps) => ({
752
+ handler: async (req) => {
753
+ if (!await deps.access({ req })) return unauthorized();
754
+ const report = await getQueueHealth(req.payload, {
755
+ includeRecovered: deps.reliability !== null,
756
+ queues: deps.queues
757
+ });
758
+ return Response.json(report, { status: 200 });
759
+ },
760
+ method: "get",
761
+ path: "/queue-status"
762
+ });
763
+ /** GET a hardened, pause-aware run. Mirrors the native run params. */
764
+ const runControlEndpoint = (deps) => ({
765
+ handler: async (req) => {
766
+ if (!await deps.access({ req })) return unauthorized();
767
+ const query = req.query;
768
+ const limit = query.limit ? Number(query.limit) : void 0;
769
+ const silent = query.silent === "true";
770
+ const state = await createPauseStore(req.payload).getState();
771
+ const targets = runTargetsForPause(query.allQueues === "true" ? void 0 : [query.queue ?? "default"], state);
772
+ if (query.disableScheduling !== "true") await req.payload.jobs.handleSchedules({ allQueues: true });
773
+ const results = [];
774
+ for (const target of targets) results.push(await req.payload.jobs.run({
775
+ ...target,
776
+ ...limit !== void 0 ? { limit } : {},
777
+ silent
778
+ }));
779
+ return Response.json({
780
+ paused: state,
781
+ ran: targets.length,
782
+ results
783
+ }, { status: 200 });
784
+ },
785
+ method: "get",
786
+ path: "/queue-run"
787
+ });
788
+ /** GET a one-shot sweep for serverless (one cron invocation, so no leader election). */
789
+ const sweepEndpoint = (deps) => ({
790
+ handler: async (req) => {
791
+ if (!await deps.access({ req })) return unauthorized();
792
+ if (!deps.reliability) return Response.json({ message: "reliability is not enabled; the sweeper is unavailable" }, { status: 400 });
793
+ const result = await runSweep({
794
+ isLeader: true,
795
+ options: deps.reliability,
796
+ payload: req.payload,
797
+ store: createJobLeaseStore(req.payload)
798
+ });
799
+ return Response.json(result, { status: 200 });
800
+ },
801
+ method: "get",
802
+ path: "/queue-sweep"
803
+ });
804
+ //#endregion
805
+ //#region src/queueControl/registerQueueControl.ts
806
+ /**
807
+ * Register the queue-control layer: harden the native run endpoint by setting
808
+ * `jobs.access.run` to the configured checker, and add the status, hardened-run, and
809
+ * sweep endpoints to `payload-jobs` through the same `jobsCollectionOverrides` seam the
810
+ * reliability and observability layers use (composing, never clobbering).
811
+ */
812
+ const registerQueueControl = (config, options, reliability) => {
813
+ const deps = {
814
+ access: options.access,
815
+ queues: options.queues,
816
+ reliability
817
+ };
818
+ const controlEndpoints = [
819
+ statusEndpoint(deps),
820
+ runControlEndpoint(deps),
821
+ sweepEndpoint(deps)
822
+ ];
823
+ const existingOverride = config.jobs?.jobsCollectionOverrides;
824
+ config.jobs = {
825
+ ...config.jobs,
826
+ access: {
827
+ ...config.jobs?.access,
828
+ run: options.access
829
+ },
830
+ jobsCollectionOverrides: ({ defaultJobsCollection }) => {
831
+ const base = existingOverride ? existingOverride({ defaultJobsCollection }) : defaultJobsCollection;
832
+ const baseEndpoints = Array.isArray(base.endpoints) ? base.endpoints : [];
833
+ return {
834
+ ...base,
835
+ endpoints: [...baseEndpoints, ...controlEndpoints]
836
+ };
837
+ }
838
+ };
839
+ };
840
+ //#endregion
841
+ //#region src/reliability/options.ts
842
+ /** Resolve user reliability options to a fully-defaulted object, or `null` when disabled. */
843
+ const resolveReliabilityOptions = (options) => {
844
+ if (options === void 0 || options === false) return null;
845
+ const jobLeaseTtlMs = options.jobLeaseTtlMs ?? 3e5;
846
+ return {
847
+ heartbeatIntervalMs: options.heartbeatIntervalMs ?? Math.floor(jobLeaseTtlMs / 3),
848
+ jobLeaseTtlMs,
849
+ leaderId: options.leaderId ?? null,
850
+ leaderLeaseTtlMs: options.leaderLeaseTtlMs ?? 3e4,
851
+ maxRecoveries: options.maxRecoveries ?? 3,
852
+ requireConcurrencyControl: options.requireConcurrencyControl ?? false,
853
+ serverlessMaxDurationMs: options.serverless?.maxDurationMs ?? null,
854
+ sweepIntervalMs: options.sweepIntervalMs ?? 6e4
855
+ };
856
+ };
857
+ //#endregion
858
+ //#region src/reliability/concurrencyContract.ts
859
+ /**
860
+ * When `reliability.requireConcurrencyControl` is set, refuse to start unless
861
+ * Payload's own `jobs.enableConcurrencyControl` is on. The at-least-once contract
862
+ * relies on it for app-level mutual exclusion under multi-node, and enabling it
863
+ * changes the jobs schema (adds a `concurrencyKey` field), so we fail loudly rather
864
+ * than silently mutate the adopter's schema. The config parameter is a structural
865
+ * subset (Payload's `Config` satisfies it), which keeps the function trivial to unit
866
+ * test with a plain object literal.
867
+ */
868
+ const enforceConcurrencyControl = (config, options) => {
869
+ if (!options.requireConcurrencyControl) return;
870
+ if (config.jobs?.enableConcurrencyControl === true) return;
871
+ throw new Error("@10x-media/jobs: reliability.requireConcurrencyControl is set but jobs.enableConcurrencyControl is not true. Enable it in your Payload config (it adds a concurrencyKey field, so run a migration) or unset requireConcurrencyControl.");
872
+ };
873
+ /**
874
+ * Wrap a job handler so its side effect runs at most once per idempotency key, even
875
+ * when at-least-once delivery re-runs the job. The race this guards is not only
876
+ * redelivery: when the sweeper requeues a job whose lease expired while it was still
877
+ * running, the original execution cannot be aborted (Payload exposes no AbortSignal),
878
+ * so a second worker can run the handler concurrently with the first. Key on the
879
+ * stable unit of work, never the job id, so both executions resolve to the same key.
880
+ * The check and the mark are the caller's to make transactional (Payload's queue and
881
+ * the app data share one database, which makes that natural); this helper only
882
+ * sequences them around the handler. A skipped run returns `{ output: {} }`.
883
+ */
884
+ const withIdempotencyKey = (handler, idem) => {
885
+ return async (args) => {
886
+ const key = idem.keyFor(args);
887
+ if (await idem.store.has(key)) return { output: {} };
888
+ const out = await handler(args);
889
+ await idem.store.mark(key);
890
+ return out;
891
+ };
892
+ };
893
+ //#endregion
894
+ //#region src/reliability/fields.ts
895
+ /**
896
+ * Fields added to the built-in `payload-jobs` collection (via jobsCollectionOverrides)
897
+ * when reliability is enabled. `leaseExpiresAt` is the liveness signal renewed by the
898
+ * worker heartbeat; the sweeper reclaims a job when `leaseExpiresAt < now`. All are
899
+ * sidebar, read-only-in-admin diagnostics (the runtime, not a human, writes them).
900
+ */
901
+ const reliabilityJobFields = () => [
902
+ {
903
+ name: "startedAt",
904
+ type: "date",
905
+ admin: {
906
+ position: "sidebar",
907
+ readOnly: true
908
+ },
909
+ index: true
910
+ },
911
+ {
912
+ name: "leaseExpiresAt",
913
+ type: "date",
914
+ admin: {
915
+ position: "sidebar",
916
+ readOnly: true
917
+ },
918
+ index: true
919
+ },
920
+ {
921
+ name: "claimedBy",
922
+ type: "text",
923
+ admin: {
924
+ position: "sidebar",
925
+ readOnly: true
926
+ },
927
+ index: true
928
+ },
929
+ {
930
+ name: "fenceToken",
931
+ type: "number",
932
+ admin: {
933
+ position: "sidebar",
934
+ readOnly: true
935
+ }
936
+ },
937
+ {
938
+ name: "recoveryAttempts",
939
+ type: "number",
940
+ admin: {
941
+ position: "sidebar",
942
+ readOnly: true
943
+ },
944
+ defaultValue: 0
945
+ }
946
+ ];
947
+ //#endregion
948
+ //#region src/reliability/heartbeat.ts
949
+ /**
950
+ * Wrap one job handler so the job keeps its lease fresh while it runs. On entry the
951
+ * wrapper stamps the just-claimed row (fenced on `processing = true`) and, in
952
+ * heartbeat mode, renews the lease on a self-scheduling timer at
953
+ * `heartbeatIntervalMs`. A renew that matches nothing means the sweeper reclaimed the
954
+ * job (its fence token moved): the wrapper records the loss, warns, and stops
955
+ * renewing, but cannot abort an opaque handler (Payload exposes no AbortSignal), so
956
+ * correctness under that race is the idempotency contract's job. The timer is always
957
+ * cleared in `finally`. If the initial stamp fails (already completed, cancelled, or
958
+ * reclaimed) the handler still runs, just without a heartbeat.
959
+ */
960
+ const withHeartbeat = (args) => {
961
+ const { getStore, handler, onLeaseLost, options, ownerId } = args;
962
+ const ttlMs = initialLeaseTtlMs(options);
963
+ const beats = isHeartbeatMode(options);
964
+ const intervalMs = options.heartbeatIntervalMs;
965
+ return async (handlerArgs) => {
966
+ const payload = handlerArgs.req.payload;
967
+ const jobId = handlerArgs.job.id;
968
+ const store = getStore(payload);
969
+ const stamp = await store.stampClaim(jobId, ownerId, ttlMs, getCurrentDate());
970
+ if (!stamp.ok) return handler(handlerArgs);
971
+ const fence = stamp.fenceToken;
972
+ let done = false;
973
+ let timer;
974
+ const tick = async () => {
975
+ if (done) return;
976
+ let ok = true;
977
+ try {
978
+ ok = (await store.renew(jobId, fence, ttlMs, getCurrentDate())).ok;
979
+ } catch (err) {
980
+ payload.logger?.warn(`@10x-media/jobs: heartbeat renew error for job ${jobId}: ${String(err)}`);
981
+ schedule();
982
+ return;
983
+ }
984
+ if (!ok) {
985
+ payload.logger?.warn(`@10x-media/jobs: lost lease for job ${jobId} (reclaimed)`);
986
+ onLeaseLost?.(jobId);
987
+ return;
988
+ }
989
+ schedule();
990
+ };
991
+ function schedule() {
992
+ if (done) return;
993
+ timer = setTimeout(() => {
994
+ tick();
995
+ }, intervalMs);
996
+ }
997
+ if (beats) schedule();
998
+ try {
999
+ return await handler(handlerArgs);
1000
+ } finally {
1001
+ done = true;
1002
+ if (timer) clearTimeout(timer);
1003
+ }
1004
+ };
1005
+ };
1006
+ /**
1007
+ * Wrap every task and workflow handler on the config with the heartbeat. Only
1008
+ * function handlers are wrapped (Payload also allows a string path for controlled
1009
+ * handlers, which we leave alone). One job-lease store is reused per Payload instance
1010
+ * via a WeakMap. All other task and workflow properties are preserved.
1011
+ */
1012
+ const registerHeartbeat = (config, options, ownerId) => {
1013
+ const jobs = config.jobs;
1014
+ if (!jobs) return;
1015
+ const storeCache = /* @__PURE__ */ new WeakMap();
1016
+ const getStore = (payload) => {
1017
+ const hit = storeCache.get(payload);
1018
+ if (hit) return hit;
1019
+ const store = createJobLeaseStore(payload);
1020
+ storeCache.set(payload, store);
1021
+ return store;
1022
+ };
1023
+ const wrapEntry = (entry) => {
1024
+ if (typeof entry.handler !== "function") return entry;
1025
+ const wrapped = withHeartbeat({
1026
+ getStore,
1027
+ handler: entry.handler,
1028
+ options,
1029
+ ownerId
1030
+ });
1031
+ return {
1032
+ ...entry,
1033
+ handler: wrapped
1034
+ };
1035
+ };
1036
+ if (Array.isArray(jobs.tasks)) jobs.tasks = jobs.tasks.map(wrapEntry);
1037
+ if (Array.isArray(jobs.workflows)) jobs.workflows = jobs.workflows.map(wrapEntry);
1038
+ };
1039
+ //#endregion
1040
+ //#region src/reliability/locksCollection.ts
1041
+ /** Slug of the plugin-owned leases collection. Table name: payload_jobs_locks. */
1042
+ const JOBS_LOCKS_SLUG = "payload-jobs-locks";
1043
+ /** The two singleton leadership roles. */
1044
+ const LEADER_ROLES = ["scheduler", "sweeper"];
1045
+ /**
1046
+ * A hidden collection holding one row per leadership role. Acquire/renew/steal are
1047
+ * conditional updates against these rows (see the lease store). Not edit-locked,
1048
+ * not shown in admin, and access is closed by default (the plugin mutates it
1049
+ * directly through the db adapter, never the REST/GraphQL API).
1050
+ */
1051
+ const buildJobsLocksCollection = () => ({
1052
+ slug: JOBS_LOCKS_SLUG,
1053
+ access: {
1054
+ create: () => false,
1055
+ delete: () => false,
1056
+ read: () => false,
1057
+ update: () => false
1058
+ },
1059
+ admin: { hidden: true },
1060
+ fields: [
1061
+ {
1062
+ name: "role",
1063
+ type: "text",
1064
+ index: true,
1065
+ required: true,
1066
+ unique: true
1067
+ },
1068
+ {
1069
+ name: "owner",
1070
+ type: "text"
1071
+ },
1072
+ {
1073
+ name: "leaseExpiresAt",
1074
+ type: "date"
1075
+ },
1076
+ {
1077
+ name: "fenceToken",
1078
+ type: "number",
1079
+ defaultValue: 0,
1080
+ required: true
1081
+ }
1082
+ ],
1083
+ lockDocuments: false
1084
+ });
1085
+ //#endregion
1086
+ //#region src/reliability/nodeId.ts
1087
+ /**
1088
+ * A stable-per-process identity for job claims and leadership. An explicit
1089
+ * `leaderId` wins; otherwise derive `hostname:pid`, which is stable within a process
1090
+ * and distinguishes nodes in a cluster.
1091
+ */
1092
+ const resolveNodeId = (leaderId) => leaderId ?? `${hostname()}:${process.pid}`;
1093
+ //#endregion
1094
+ //#region src/reliability/registerReliability.ts
1095
+ /** Idempotently ensure one lock row per leadership role. Race-safe via the unique `role`. */
1096
+ const ensureLockRows = async (payload) => {
1097
+ const create = payload.create;
1098
+ for (const role of LEADER_ROLES) try {
1099
+ await create({
1100
+ collection: JOBS_LOCKS_SLUG,
1101
+ data: {
1102
+ fenceToken: 0,
1103
+ leaseExpiresAt: null,
1104
+ owner: null,
1105
+ role
1106
+ },
1107
+ overrideAccess: true
1108
+ });
1109
+ } catch (err) {
1110
+ if (!(err instanceof ValidationError)) throw err;
1111
+ }
1112
+ };
1113
+ /**
1114
+ * Register the reliability layer on the incoming config: add diagnostic fields to
1115
+ * `payload-jobs` through the same `jobsCollectionOverrides` seam the observability
1116
+ * layer uses (composing, never clobbering), add the locks collection, and ensure the
1117
+ * lock rows at init (preserving any host onInit).
1118
+ */
1119
+ const registerReliability = (config, options) => {
1120
+ enforceConcurrencyControl(config, options);
1121
+ const existingOverride = config.jobs?.jobsCollectionOverrides;
1122
+ config.jobs = {
1123
+ ...config.jobs,
1124
+ jobsCollectionOverrides: ({ defaultJobsCollection }) => {
1125
+ const base = existingOverride ? existingOverride({ defaultJobsCollection }) : defaultJobsCollection;
1126
+ return {
1127
+ ...base,
1128
+ fields: [...base.fields, ...reliabilityJobFields()]
1129
+ };
1130
+ }
1131
+ };
1132
+ config.collections = [...config.collections ?? [], buildJobsLocksCollection()];
1133
+ const previousOnInit = config.onInit;
1134
+ config.onInit = async (payload) => {
1135
+ await previousOnInit?.(payload);
1136
+ await ensureLockRows(payload);
1137
+ };
1138
+ registerHeartbeat(config, options, resolveNodeId(options.leaderId));
1139
+ };
1140
+ //#endregion
1141
+ //#region src/execution/autoRunConfig.ts
1142
+ const DEFAULT_CRON = "* * * * *";
1143
+ const DEFAULT_LIMIT = 10;
1144
+ /**
1145
+ * Build a production `jobs.autoRun` array: one Croner config per queue, silent by
1146
+ * default, with Payload's own `protect: true` preventing overlap. This is the simple
1147
+ * single-node and serverless-adjacent path where native autoRun safely handles both
1148
+ * scheduling and running in one process. Multi-node deployments use `createWorker`
1149
+ * instead, because native autoRun cannot gate scheduling to one elected leader (its
1150
+ * only dynamic lever, `shouldAutoRun`, permanently stops the cron rather than pausing).
1151
+ */
1152
+ const autoRunConfig = (options = {}) => {
1153
+ const { disableScheduling = false, queues = [{ queue: "default" }], silent = true } = options;
1154
+ return queues.map((q) => ({
1155
+ cron: q.cron ?? DEFAULT_CRON,
1156
+ disableScheduling,
1157
+ limit: q.limit ?? DEFAULT_LIMIT,
1158
+ queue: q.queue,
1159
+ silent
1160
+ }));
1161
+ };
1162
+ //#endregion
1163
+ //#region src/execution/drain.ts
1164
+ /**
1165
+ * Run the graceful-drain sequence: stop claiming, await this node's in-flight jobs up
1166
+ * to a wall-clock budget, requeue any stragglers, release leadership, and destroy. The
1167
+ * clock (`now`/`sleep`) is injected so tests drive it deterministically without real
1168
+ * waiting. Always releases leadership and destroys, even when nothing was in flight.
1169
+ */
1170
+ const drainWorker = async (deps, options) => {
1171
+ deps.stopLoops();
1172
+ const start = deps.now();
1173
+ const inFlightAtStart = await deps.countInFlight();
1174
+ let remaining = inFlightAtStart;
1175
+ while (remaining > 0 && deps.now() - start < options.drainTimeoutMs) {
1176
+ await deps.sleep(options.pollIntervalMs);
1177
+ remaining = await deps.countInFlight();
1178
+ }
1179
+ const timedOut = remaining > 0;
1180
+ const requeued = timedOut ? await deps.requeueStragglers() : 0;
1181
+ await deps.releaseLeadership();
1182
+ await deps.destroy();
1183
+ deps.logger?.info?.(`@10x-media/jobs: drain complete (started ${inFlightAtStart}, requeued ${requeued}, timedOut ${timedOut})`);
1184
+ return {
1185
+ inFlightAtStart,
1186
+ remaining,
1187
+ requeued,
1188
+ timedOut
1189
+ };
1190
+ };
1191
+ //#endregion
1192
+ //#region src/reliability/leaderController.ts
1193
+ /**
1194
+ * Turns a lease into leadership. On each `tick`: if not leading, try to acquire/steal;
1195
+ * if leading, renew. A failed renew (the lease was stolen while this node was paused
1196
+ * past expiry) drops leadership immediately, so a zombie never keeps acting. No timer
1197
+ * lives here; a caller schedules `tick` at `ttlMs / 3`.
1198
+ */
1199
+ const createLeaderController = (args) => {
1200
+ const { ownerId, role, store, ttlMs } = args;
1201
+ let leading = false;
1202
+ let token = 0;
1203
+ return {
1204
+ fenceToken: () => leading ? token : 0,
1205
+ isLeader: () => leading,
1206
+ release: async () => {
1207
+ await store.release(role, ownerId);
1208
+ leading = false;
1209
+ token = 0;
1210
+ },
1211
+ tick: async (now) => {
1212
+ if (leading) {
1213
+ if (!(await store.renew(role, ownerId, ttlMs, now)).ok) {
1214
+ leading = false;
1215
+ token = 0;
1216
+ }
1217
+ return;
1218
+ }
1219
+ const acquired = await store.acquireOrSteal(role, ownerId, ttlMs, now);
1220
+ if (acquired.ok) {
1221
+ leading = true;
1222
+ token = acquired.fenceToken;
1223
+ }
1224
+ }
1225
+ };
1226
+ };
1227
+ //#endregion
1228
+ //#region src/reliability/leaseStore.mongo.ts
1229
+ /** Reach the raw Mongoose model the adapter exposes on `db.collections[slug]`. */
1230
+ const model = (payload) => {
1231
+ const found = payload.db.collections[JOBS_LOCKS_SLUG];
1232
+ if (!found) throw new Error(`@10x-media/jobs: missing Mongoose model for "${JOBS_LOCKS_SLUG}"`);
1233
+ return found;
1234
+ };
1235
+ /** Single-document `findOneAndUpdate` is an atomic compare-and-set on MongoDB. */
1236
+ const createMongoLeaseStore = (payload) => {
1237
+ const m = model(payload);
1238
+ const toRecord = (doc) => doc === null ? null : {
1239
+ fenceToken: doc.fenceToken,
1240
+ leaseExpiresAt: doc.leaseExpiresAt ? new Date(doc.leaseExpiresAt) : null,
1241
+ owner: doc.owner ?? null,
1242
+ role: doc.role
1243
+ };
1244
+ return {
1245
+ acquireOrSteal: async (role, owner, ttlMs, now) => {
1246
+ const doc = await m.findOneAndUpdate({
1247
+ $or: [{ owner: null }, { leaseExpiresAt: { $lt: now } }],
1248
+ role
1249
+ }, {
1250
+ $inc: { fenceToken: 1 },
1251
+ $set: {
1252
+ leaseExpiresAt: leaseExpiry(now, ttlMs),
1253
+ owner
1254
+ }
1255
+ }, { new: true });
1256
+ return {
1257
+ fenceToken: doc?.fenceToken ?? 0,
1258
+ ok: doc !== null
1259
+ };
1260
+ },
1261
+ read: async (role) => toRecord(await m.findOne({ role }).lean()),
1262
+ release: async (role, owner) => {
1263
+ await m.findOneAndUpdate({
1264
+ owner,
1265
+ role
1266
+ }, { $set: {
1267
+ leaseExpiresAt: null,
1268
+ owner: null
1269
+ } });
1270
+ },
1271
+ renew: async (role, owner, ttlMs, now) => {
1272
+ const doc = await m.findOneAndUpdate({
1273
+ owner,
1274
+ role
1275
+ }, { $set: { leaseExpiresAt: leaseExpiry(now, ttlMs) } }, { new: true });
1276
+ return {
1277
+ fenceToken: doc?.fenceToken ?? 0,
1278
+ ok: doc !== null
1279
+ };
1280
+ }
1281
+ };
1282
+ };
1283
+ //#endregion
1284
+ //#region src/reliability/leaseStore.postgres.ts
1285
+ /** Payload tables for kebab slugs are the slug with hyphens replaced by underscores. */
1286
+ const tableName = (payload) => {
1287
+ const defaultTable = JOBS_LOCKS_SLUG.replace(/-/g, "_");
1288
+ const map = payload.db.tableNameMap;
1289
+ return (map && typeof map.get === "function" ? map.get(defaultTable) : void 0) ?? defaultTable;
1290
+ };
1291
+ const pool = (payload) => {
1292
+ const found = payload.db.pool;
1293
+ if (!found) throw new Error(`@10x-media/jobs: missing pg pool on the "${payload.db.name}" adapter`);
1294
+ return found;
1295
+ };
1296
+ /** Single-statement `UPDATE ... WHERE <guard> RETURNING` is atomic under READ COMMITTED. */
1297
+ const createPostgresLeaseStore = (payload) => {
1298
+ const table = tableName(payload);
1299
+ const db = pool(payload);
1300
+ const toRecord = (row) => row === void 0 ? null : {
1301
+ fenceToken: Number(row.fence_token),
1302
+ leaseExpiresAt: row.lease_expires_at ? new Date(row.lease_expires_at) : null,
1303
+ owner: row.owner ?? null,
1304
+ role: row.role
1305
+ };
1306
+ return {
1307
+ acquireOrSteal: async (role, owner, ttlMs, now) => {
1308
+ const res = await db.query(`UPDATE ${table}
1309
+ SET owner = $1, lease_expires_at = $2, fence_token = fence_token + 1
1310
+ WHERE role = $3 AND (owner IS NULL OR lease_expires_at < $4)
1311
+ RETURNING fence_token`, [
1312
+ owner,
1313
+ leaseExpiry(now, ttlMs),
1314
+ role,
1315
+ now
1316
+ ]);
1317
+ return {
1318
+ fenceToken: res.rowCount === 1 ? Number(res.rows[0]?.fence_token) : 0,
1319
+ ok: res.rowCount === 1
1320
+ };
1321
+ },
1322
+ read: async (role) => {
1323
+ return toRecord((await db.query(`SELECT role, owner, lease_expires_at, fence_token FROM ${table} WHERE role = $1`, [role])).rows[0]);
1324
+ },
1325
+ release: async (role, owner) => {
1326
+ await db.query(`UPDATE ${table} SET owner = NULL, lease_expires_at = NULL WHERE role = $1 AND owner = $2`, [role, owner]);
1327
+ },
1328
+ renew: async (role, owner, ttlMs, now) => {
1329
+ const res = await db.query(`UPDATE ${table}
1330
+ SET lease_expires_at = $1
1331
+ WHERE role = $2 AND owner = $3
1332
+ RETURNING fence_token`, [
1333
+ leaseExpiry(now, ttlMs),
1334
+ role,
1335
+ owner
1336
+ ]);
1337
+ return {
1338
+ fenceToken: res.rowCount === 1 ? Number(res.rows[0]?.fence_token) : 0,
1339
+ ok: res.rowCount === 1
1340
+ };
1341
+ }
1342
+ };
1343
+ };
1344
+ //#endregion
1345
+ //#region src/reliability/leaseStore.ts
1346
+ /** Build the lease store for the running adapter. Throws for an unsupported adapter. */
1347
+ const createLeaseStore = (payload) => {
1348
+ if (payload.db.name === "mongoose") return createMongoLeaseStore(payload);
1349
+ if (payload.db.name === "postgres") return createPostgresLeaseStore(payload);
1350
+ throw new Error(`@10x-media/jobs reliability does not support db adapter "${payload.db.name}"`);
1351
+ };
1352
+ //#endregion
1353
+ //#region src/execution/inFlight.ts
1354
+ const JOBS_SLUG = "payload-jobs";
1355
+ /** Count this node's in-flight jobs: `processing: true` AND claimed by `claimedBy`. */
1356
+ const countInFlight = async (payload, claimedBy) => {
1357
+ const count = payload.count;
1358
+ return (await count({
1359
+ collection: JOBS_SLUG,
1360
+ where: { and: [{ processing: { equals: true } }, { claimedBy: { equals: claimedBy } }] }
1361
+ })).totalDocs;
1362
+ };
1363
+ //#endregion
1364
+ //#region src/execution/signals.ts
1365
+ /**
1366
+ * Register `handler` for each signal and return a cleanup that removes them. The
1367
+ * handler fires at most once across all signals (a second signal during drain is
1368
+ * ignored). Payload installs no signal handlers of its own, so these never conflict.
1369
+ * `target` defaults to `process`; tests pass an EventEmitter.
1370
+ */
1371
+ const installSignalHandlers = (signals, handler, target = process) => {
1372
+ let fired = false;
1373
+ const registered = signals.map((signal) => {
1374
+ const listener = () => {
1375
+ if (fired) return;
1376
+ fired = true;
1377
+ handler(signal);
1378
+ };
1379
+ target.on(signal, listener);
1380
+ return [signal, listener];
1381
+ });
1382
+ return () => {
1383
+ for (const [signal, listener] of registered) target.removeListener(signal, listener);
1384
+ };
1385
+ };
1386
+ //#endregion
1387
+ //#region src/execution/workerCycles.ts
1388
+ const guard = async (logger, label, fn) => {
1389
+ try {
1390
+ await fn();
1391
+ } catch (err) {
1392
+ logger?.error?.(`@10x-media/jobs: worker ${label} cycle error: ${String(err)}`);
1393
+ }
1394
+ };
1395
+ /** One run cycle: claim and execute jobs on this node. Errors are logged, not thrown. */
1396
+ const runCycle = async (deps) => {
1397
+ await guard(deps.logger, "run", deps.runJobs);
1398
+ };
1399
+ /** One maintenance cycle: advance leadership, then schedule only if scheduler-leader. */
1400
+ const maintenanceCycle = async (deps) => {
1401
+ await guard(deps.logger, "leadership", () => deps.tickLeaders(deps.now()));
1402
+ if (deps.isSchedulerLeader()) await guard(deps.logger, "schedule", deps.handleSchedules);
1403
+ };
1404
+ /** One sweep cycle: reclaim stuck jobs only if sweeper-leader. */
1405
+ const sweepCycle = async (deps) => {
1406
+ if (deps.isSweeperLeader()) await guard(deps.logger, "sweep", deps.sweep);
1407
+ };
1408
+ //#endregion
1409
+ //#region src/execution/worker.ts
1410
+ const realSleep = (ms) => new Promise((resolve) => {
1411
+ setTimeout(resolve, ms);
1412
+ });
1413
+ /** A self-skipping interval: a slow tick never overlaps the next (like Croner's `protect`). */
1414
+ const guardedInterval = (fn, ms) => {
1415
+ let busy = false;
1416
+ return setInterval(() => {
1417
+ if (busy) return;
1418
+ busy = true;
1419
+ fn().finally(() => {
1420
+ busy = false;
1421
+ });
1422
+ }, ms);
1423
+ };
1424
+ /**
1425
+ * A plugin-driven worker: runs jobs on every node, schedules and sweeps only while
1426
+ * holding the corresponding leadership lease, and drains gracefully on SIGTERM/SIGINT.
1427
+ * It owns its own timers (never Payload's autoRun cron), because Payload's
1428
+ * `shouldAutoRun` gate permanently stops a cron and cannot follow dynamic leadership.
1429
+ * Leadership timestamps use Payload's swappable `getCurrentDate()`; the drain budget
1430
+ * uses the injected wall clock.
1431
+ */
1432
+ const createWorker = (args) => {
1433
+ const { payload, reliability } = args;
1434
+ if (!payload.collections["payload-jobs"]) throw new Error("@10x-media/jobs: createWorker requires at least one configured job task. Payload only registers the payload-jobs collection when jobs.tasks or jobs.workflows is non-empty.");
1435
+ const nodeId = resolveNodeId(reliability.leaderId);
1436
+ const leaseStore = createLeaseStore(payload);
1437
+ const jobLeaseStore = createJobLeaseStore(payload);
1438
+ const scheduler = createLeaderController({
1439
+ ownerId: nodeId,
1440
+ role: "scheduler",
1441
+ store: leaseStore,
1442
+ ttlMs: reliability.leaderLeaseTtlMs
1443
+ });
1444
+ const sweeper = createLeaderController({
1445
+ ownerId: nodeId,
1446
+ role: "sweeper",
1447
+ store: leaseStore,
1448
+ ttlMs: reliability.leaderLeaseTtlMs
1449
+ });
1450
+ const runIntervalMs = args.runIntervalMs ?? 2e3;
1451
+ const maintenanceIntervalMs = args.maintenanceIntervalMs ?? Math.max(1e3, Math.floor(reliability.leaderLeaseTtlMs / 3));
1452
+ const sweepIntervalMs = reliability.sweepIntervalMs;
1453
+ const runLimit = args.runLimit ?? 10;
1454
+ const drainTimeoutMs = args.drainTimeoutMs ?? 3e4;
1455
+ const pollIntervalMs = args.pollIntervalMs ?? 500;
1456
+ const destroy = args.destroy ?? (() => payload.destroy());
1457
+ const now = args.now ?? (() => Date.now());
1458
+ const sleep = args.sleep ?? realSleep;
1459
+ const logger = payload.logger;
1460
+ let timers = [];
1461
+ let signalCleanup;
1462
+ let draining;
1463
+ const runJobs = async () => {
1464
+ const state = args.pauseStore ? await args.pauseStore.getState() : emptyPauseState();
1465
+ for (const target of runTargetsForPause(args.queues, state)) await payload.jobs.run({
1466
+ ...target,
1467
+ limit: runLimit,
1468
+ silent: true
1469
+ });
1470
+ };
1471
+ const handleSchedules = async () => {
1472
+ await payload.jobs.handleSchedules({ allQueues: true });
1473
+ };
1474
+ const sweep = async () => {
1475
+ await runSweep({
1476
+ isLeader: sweeper.isLeader(),
1477
+ now: getCurrentDate(),
1478
+ options: reliability,
1479
+ payload,
1480
+ store: jobLeaseStore
1481
+ });
1482
+ };
1483
+ const tickLeaders = async (at) => {
1484
+ await scheduler.tick(at);
1485
+ await sweeper.tick(at);
1486
+ };
1487
+ const stopLoops = () => {
1488
+ for (const timer of timers) clearInterval(timer);
1489
+ timers = [];
1490
+ };
1491
+ const removeSignals = () => {
1492
+ signalCleanup?.();
1493
+ signalCleanup = void 0;
1494
+ };
1495
+ const drain = () => {
1496
+ if (draining) return draining;
1497
+ removeSignals();
1498
+ draining = drainWorker({
1499
+ countInFlight: () => countInFlight(payload, nodeId),
1500
+ destroy,
1501
+ logger,
1502
+ now,
1503
+ releaseLeadership: async () => {
1504
+ await scheduler.release();
1505
+ await sweeper.release();
1506
+ },
1507
+ requeueStragglers: async () => (await jobLeaseStore.releaseAllClaims(nodeId)).released,
1508
+ sleep,
1509
+ stopLoops
1510
+ }, {
1511
+ drainTimeoutMs,
1512
+ pollIntervalMs
1513
+ });
1514
+ return draining;
1515
+ };
1516
+ const worker = {
1517
+ drain,
1518
+ isLeader: (role) => role === "scheduler" ? scheduler.isLeader() : sweeper.isLeader(),
1519
+ start: () => {
1520
+ stopLoops();
1521
+ timers = [
1522
+ guardedInterval(() => runCycle({
1523
+ logger,
1524
+ runJobs
1525
+ }), runIntervalMs),
1526
+ guardedInterval(() => maintenanceCycle({
1527
+ handleSchedules,
1528
+ isSchedulerLeader: scheduler.isLeader,
1529
+ logger,
1530
+ now: getCurrentDate,
1531
+ tickLeaders
1532
+ }), maintenanceIntervalMs),
1533
+ guardedInterval(() => sweepCycle({
1534
+ isSweeperLeader: sweeper.isLeader,
1535
+ logger,
1536
+ sweep
1537
+ }), sweepIntervalMs)
1538
+ ];
1539
+ },
1540
+ stop: () => {
1541
+ stopLoops();
1542
+ removeSignals();
1543
+ }
1544
+ };
1545
+ if (args.installSignals !== false) {
1546
+ const exit = args.exit ?? ((code) => process.exit(code));
1547
+ signalCleanup = installSignalHandlers(args.signals ?? ["SIGTERM", "SIGINT"], () => {
1548
+ drain().then(() => exit(0)).catch(() => exit(1));
1549
+ });
1550
+ }
1551
+ return worker;
1552
+ };
1553
+ //#endregion
1554
+ //#region src/presets/presets.ts
1555
+ /**
1556
+ * Single-node Docker: one serial claimer, so claim races are moot. Reliability and
1557
+ * queue control on with defaults; the sweeper runs from the in-process worker.
1558
+ */
1559
+ const singleNodePreset = () => ({
1560
+ queueControl: {},
1561
+ reliability: {}
1562
+ });
1563
+ /**
1564
+ * Multi-node Docker: leader-elected scheduling and sweeping (the default). Pass a
1565
+ * stable `leaderId` per node if you do not want the generated hostname:pid identity.
1566
+ */
1567
+ const multiNodePreset = (options = {}) => ({
1568
+ queueControl: {},
1569
+ reliability: options.leaderId === void 0 ? {} : { leaderId: options.leaderId }
1570
+ });
1571
+ /**
1572
+ * Serverless (Vercel): no long-running worker, so staleness derives from the platform
1573
+ * hard-kill duration (the function maxDuration) rather than a heartbeat, and the run
1574
+ * and sweep endpoints are guarded by the cron secret. Drive them with Vercel Cron
1575
+ * (see `vercelCrons`).
1576
+ */
1577
+ const serverlessPreset = (options) => ({
1578
+ queueControl: { access: cronSecretAccess({ envVar: options.cronSecretEnvVar }) },
1579
+ reliability: {
1580
+ jobLeaseTtlMs: options.maxDurationMs,
1581
+ serverless: { maxDurationMs: options.maxDurationMs }
1582
+ }
1583
+ });
1584
+ /**
1585
+ * Build the `vercel.json` `crons` array: one entry hitting the hardened run endpoint
1586
+ * (all queues) and one hitting the sweep endpoint. Defaults to every minute (Vercel
1587
+ * Pro). Vercel sends the `CRON_SECRET` as a Bearer token, which `cronSecretAccess`
1588
+ * checks.
1589
+ */
1590
+ const vercelCrons = (options = {}) => [{
1591
+ path: options.runPath ?? "/api/payload-jobs/queue-run?allQueues=true",
1592
+ schedule: options.runSchedule ?? "* * * * *"
1593
+ }, {
1594
+ path: options.sweepPath ?? "/api/payload-jobs/queue-sweep",
1595
+ schedule: options.sweepSchedule ?? "* * * * *"
1596
+ }];
1597
+ //#endregion
1598
+ //#region src/index.ts
1599
+ /**
1600
+ * Jobs plugin for Payload v3. Enhances the built-in `payload-jobs` collection
1601
+ * with an ops dashboard (status, queue health, error and log panels) and the
1602
+ * supporting i18n. Authored with `definePlugin` so the automations and webhooks
1603
+ * plugins can detect it by slug. Runs first (`order: 0`).
1604
+ */
1605
+ const jobs = definePlugin({
1606
+ slug: "@10x-media/jobs",
1607
+ order: 0,
1608
+ plugin: ({ config, plugins: _plugins, ...options }) => {
1609
+ if (options.disabled === true) return config;
1610
+ registerTranslations(config);
1611
+ registerJobsEnhancements(config, options);
1612
+ const reliability = resolveReliabilityOptions(options.reliability);
1613
+ if (reliability) registerReliability(config, reliability);
1614
+ const queueControl = resolveQueueControlOptions(options.queueControl);
1615
+ if (queueControl) registerQueueControl(config, queueControl, reliability);
1616
+ return config;
1617
+ }
1618
+ });
1619
+ //#endregion
1620
+ export { JOBS_LOCKS_SLUG, LEADER_ROLES, autoRunConfig, createJobLeaseStore, createLeaderController, createLeaseStore, createPauseStore, createWorker, cronSecretAccess, decideRecovery, deriveJobStatus, drainWorker, getQueueHealth, jobs, loggedInAccess, multiNodePreset, resolveReliabilityOptions, runSweep, serverlessPreset, singleNodePreset, vercelCrons, withIdempotencyKey };
1621
+
1622
+ //# sourceMappingURL=index.js.map