@alexify/migronaut 1.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +409 -1
  2. package/README.md +248 -24
  3. package/bin/migronaut.js +11 -3
  4. package/bullmq.d.ts +845 -0
  5. package/bullmq.js +1 -0
  6. package/index.d.ts +757 -29
  7. package/migronaut.schema.json +191 -1
  8. package/package.json +27 -6
  9. package/src/bullmq/index.js +55 -0
  10. package/src/bullmq/jobs.js +454 -0
  11. package/src/bullmq/processor.js +608 -0
  12. package/src/bullmq/producer.js +424 -0
  13. package/src/bullmq/service.js +653 -0
  14. package/src/bullmq/wait.js +124 -0
  15. package/src/cli/args.js +12 -2
  16. package/src/cli/commands/baseline.js +45 -0
  17. package/src/cli/commands/converge.js +160 -0
  18. package/src/cli/commands/down.js +2 -0
  19. package/src/cli/commands/lock.js +2 -1
  20. package/src/cli/commands/redo.js +8 -1
  21. package/src/cli/commands/unlock.js +12 -2
  22. package/src/cli/commands/up.js +14 -1
  23. package/src/cli/exit-codes.js +10 -2
  24. package/src/cli/index.js +4 -0
  25. package/src/cli/shared.js +29 -7
  26. package/src/cli/table.js +105 -0
  27. package/src/core/audit.js +17 -3
  28. package/src/core/baseline.js +80 -0
  29. package/src/core/changelog.js +140 -24
  30. package/src/core/collections.js +372 -0
  31. package/src/core/config.js +125 -27
  32. package/src/core/converge-log.js +47 -0
  33. package/src/core/converge-plan.js +483 -0
  34. package/src/core/converge.js +867 -0
  35. package/src/core/import-runner.js +34 -6
  36. package/src/core/import.js +14 -7
  37. package/src/core/index-spec.js +496 -0
  38. package/src/core/lock-wait.js +260 -0
  39. package/src/core/lock.js +71 -20
  40. package/src/core/migrator.js +805 -304
  41. package/src/core/options.js +251 -0
  42. package/src/core/run-recorder.js +157 -0
  43. package/src/core/run.js +71 -71
  44. package/src/core/runner.js +70 -20
  45. package/src/core/sequence.js +134 -0
  46. package/src/errors/index.js +71 -1
  47. package/src/index.js +16 -0
  48. package/src/utils/actor.js +48 -0
  49. package/src/utils/canonical.js +179 -0
  50. package/src/utils/collection-name.js +21 -0
  51. package/src/utils/error.js +18 -1
  52. package/src/utils/id.js +77 -0
  53. package/src/utils/loader.js +39 -21
  54. package/src/utils/logger.js +30 -12
  55. package/src/utils/migration-name.js +32 -0
  56. package/src/utils/redact.js +57 -4
  57. package/src/utils/sanitize.js +8 -3
  58. package/src/utils/telemetry.js +393 -0
  59. package/src/utils/template.js +60 -12
@@ -0,0 +1,608 @@
1
+ const { MigratorKit, RECORD_LOCK_WAIT } = require('../core/migrator.js');
2
+ const { assertLockWaitOptions, withLockWait } = require('../core/lock-wait.js');
3
+ const {
4
+ ConfigInvalidError,
5
+ LockAlreadyHeldError,
6
+ MigrationBlockedError,
7
+ MigronautError,
8
+ NotAppliedError,
9
+ RunAbortedError,
10
+ } = require('../errors/index.js');
11
+ const { pickActor } = require('../utils/actor.js');
12
+ const { errorText } = require('../utils/error.js');
13
+ const { redactDeep, redactOutbound } = require('../utils/redact.js');
14
+ const {
15
+ JOB_NAMES,
16
+ assertAllowed,
17
+ isPlainObject,
18
+ parseJobData,
19
+ resolveAllow,
20
+ } = require('./jobs.js');
21
+ const { assertJobOptions, enqueueConverge, enqueueUp } = require('./producer.js');
22
+
23
+ /**
24
+ * Failures a later attempt can get past without anything being fixed: the lock
25
+ * was busy or lost, the database was unreachable, the run was stopped. Every
26
+ * other MigronautError needs a change first — to the migration, the files or
27
+ * the changelog — so retrying it only repeats the failure (and, in a FIFO
28
+ * queue, reorders the line). Errors of unknown origin stay retryable: not
29
+ * recognising a failure is no reason to rule a retry out.
30
+ */
31
+ const RETRYABLE_CODES = Object.freeze([
32
+ 'LOCK_ALREADY_HELD',
33
+ 'LOCK_LOST',
34
+ 'LOCK_RELEASE_FAILED',
35
+ 'RUN_ABORTED',
36
+ 'CONNECTION_FAILED',
37
+ ]);
38
+
39
+ /**
40
+ * What a job waits out instead of failing: a held lock — and a block by
41
+ * migrations that have not failed. With more than one worker taking jobs
42
+ * (global concurrency off or unsupported, several processors), the job for an
43
+ * earlier migration can still be in flight on another worker when a later one
44
+ * takes the lock; failing the later job would fail its whole group although
45
+ * nothing went wrong. A blocker with a `'failed'` trace stopped the line for
46
+ * real, and that job fails at once, as before.
47
+ */
48
+ function isTransientForJob(error) {
49
+ if (error instanceof LockAlreadyHeldError) return true;
50
+ return (
51
+ error instanceof MigrationBlockedError &&
52
+ Array.isArray(error.context?.failed) &&
53
+ error.context.failed.length === 0
54
+ );
55
+ }
56
+
57
+ function isRetryableError(error) {
58
+ return !(error instanceof MigronautError) || RETRYABLE_CODES.includes(error.code);
59
+ }
60
+
61
+ /** The least time between two `lock-wait` progress updates of one job */
62
+ const WAIT_PROGRESS_INTERVAL_MS = 2000;
63
+
64
+ /** What BullMQ checks (by name, not only by class) to fail a job without retrying it */
65
+ const UNRECOVERABLE_ERROR_NAME = 'UnrecoverableError';
66
+
67
+ /**
68
+ * What BullMQ checks (by name) for a job the processor has moved back to the
69
+ * wait list itself: neither failed nor completed, nothing more to record.
70
+ */
71
+ const WAITING_ERROR_NAME = 'WaitingError';
72
+
73
+ /** The correlation ids of a job, for its error context and log lines */
74
+ function jobIds(ctx) {
75
+ return {
76
+ ...(ctx.job?.id !== undefined ? { jobId: String(ctx.job.id) } : {}),
77
+ ...(ctx.data?.groupId !== undefined ? { groupId: ctx.data.groupId } : {}),
78
+ ...(ctx.runId ? { runId: ctx.runId } : {}),
79
+ };
80
+ }
81
+
82
+ /** A job in a few words, for log lines: `up 20260101-x.js (1/3)`, `converge`, `sync` */
83
+ function describeJob(data) {
84
+ if (data.kind !== 'migration') return data.kind;
85
+ return `${data.direction} ${data.migration} (${data.index + 1}/${data.total})`;
86
+ }
87
+
88
+ /** The structured fields of a job's log lines */
89
+ function jobFields(data) {
90
+ if (data.kind !== 'migration') return { kind: data.kind };
91
+ return {
92
+ migration: data.migration,
93
+ direction: data.direction,
94
+ index: data.index,
95
+ total: data.total,
96
+ };
97
+ }
98
+
99
+ /**
100
+ * Stamp the job's ids onto a typed error before it leaves for the queue — the
101
+ * one place a failed job's run id survives, since a failed job has no return
102
+ * value. Copy-on-write: the error may be a shared abort reason.
103
+ */
104
+ function attachJobIds(error, ctx) {
105
+ if (!(error instanceof MigronautError)) return;
106
+ const ids = jobIds(ctx);
107
+ if (Object.keys(ids).length > 0) error.context = { ...error.context, ...ids };
108
+ }
109
+
110
+ /**
111
+ * Prepare an error for the queue. BullMQ stores its message (as the job's
112
+ * `failedReason`) and stack, so both leave the process and must be redacted.
113
+ * The message is also all a dashboard shows: a wrapper says WHICH migration
114
+ * failed and keeps the WHY in `context.cause`, which the queue never sees —
115
+ * so the cause is folded into the message, as the changelog's failure trace
116
+ * already does.
117
+ */
118
+ function prepareErrorForQueue(error) {
119
+ if (!(error instanceof Error)) return;
120
+ const cause = error instanceof MigronautError ? error.context?.cause : undefined;
121
+ try {
122
+ const message = redactOutbound(error.message);
123
+ error.message =
124
+ typeof cause === 'string' && cause.length > 0 && !message.includes(cause)
125
+ ? `${message} — ${redactOutbound(cause)}`
126
+ : message;
127
+ if (typeof error.stack === 'string') error.stack = redactOutbound(error.stack);
128
+ } catch {
129
+ // A frozen error cannot be rewritten; its message was written by us anyway.
130
+ }
131
+ }
132
+
133
+ /**
134
+ * Validate the processor's options and return the resolved lock-wait budget.
135
+ * Pure — nothing is constructed — so a caller can reject bad input before it
136
+ * opens any connection of its own.
137
+ */
138
+ function resolveProcessorOptions(options) {
139
+ if (!isPlainObject(options)) {
140
+ throw new ConfigInvalidError('createMigrationProcessor options must be an object');
141
+ }
142
+ const { kit, config, lockWait = {}, jobOptions, ordered = true, allow } = options;
143
+ if (kit !== undefined && config !== undefined) {
144
+ throw new ConfigInvalidError('Pass either `kit` or `config`, not both');
145
+ }
146
+ if (kit !== undefined && typeof kit?.up !== 'function') {
147
+ throw new ConfigInvalidError('kit must be a MigratorKit instance');
148
+ }
149
+ if (typeof ordered !== 'boolean') {
150
+ throw new ConfigInvalidError('ordered must be a boolean', { ordered });
151
+ }
152
+ if (!isPlainObject(lockWait)) {
153
+ throw new ConfigInvalidError('lockWait must be an object', { lockWait: typeof lockWait });
154
+ }
155
+ // Unlike runMigrations, waiting is the default: nothing is blocked on this
156
+ // job, and the budget only burns while the lock's holder is stalled.
157
+ const waitOptions = { onLockHeld: 'wait', ...lockWait };
158
+ assertLockWaitOptions(waitOptions);
159
+ assertJobOptions(jobOptions);
160
+ return { waitOptions, defaultOrdered: ordered, allow: resolveAllow(allow) };
161
+ }
162
+
163
+ /**
164
+ * Build the function a BullMQ Worker runs for each migration job.
165
+ *
166
+ * One long-lived MigratorKit serves every job, and jobs are run one at a time
167
+ * even when the Worker is configured for more: a kit rejects overlapping runs,
168
+ * and migrations are sequential by nature. Across processes the MongoDB lock
169
+ * does that job, and the `ordered` guard keeps the sequence.
170
+ *
171
+ * Declared with exactly three parameters: BullMQ passes the cancellation
172
+ * signal only to processors whose `length` is at least 3.
173
+ */
174
+ function createMigrationProcessor(options = {}) {
175
+ const { waitOptions, defaultOrdered, allow } = resolveProcessorOptions(options);
176
+ const { kit: injectedKit, config, kitOptions, queue, jobOptions } = options;
177
+
178
+ const ownsKit = injectedKit === undefined;
179
+ const kit = injectedKit ?? new MigratorKit(config ?? {}, kitOptions);
180
+ const shutdownController = new AbortController();
181
+ /** Serializes jobs in this process — see the factory's doc comment */
182
+ let chain = Promise.resolve();
183
+ /** The job being processed, for the kit's event listeners */
184
+ let current;
185
+
186
+ /**
187
+ * Fire-and-forget write to the job (a log row, a progress update). Never
188
+ * awaited on the run path — Redis must not pace a migration — and never
189
+ * allowed to fail the job; `flush` drains what is still in flight before the
190
+ * job settles, so nothing lands on a job BullMQ already finished.
191
+ */
192
+ function write(ctx, action) {
193
+ let pending;
194
+ try {
195
+ pending = Promise.resolve(action(ctx.job)).catch(() => undefined);
196
+ } catch {
197
+ return;
198
+ }
199
+ ctx.writes.add(pending);
200
+ pending.finally(() => ctx.writes.delete(pending));
201
+ }
202
+ const log = (ctx, row) => write(ctx, (job) => job.log?.(redactOutbound(row)));
203
+ const progress = (ctx, phase, extra) =>
204
+ write(ctx, (job) =>
205
+ job.updateProgress?.({
206
+ phase,
207
+ ...(ctx.data.kind === 'migration'
208
+ ? {
209
+ migration: ctx.data.migration,
210
+ direction: ctx.data.direction,
211
+ groupId: ctx.data.groupId,
212
+ index: ctx.data.index,
213
+ total: ctx.data.total,
214
+ }
215
+ : { kind: ctx.data.kind }),
216
+ ...extra,
217
+ }),
218
+ );
219
+ const flush = (ctx) => Promise.allSettled([...ctx.writes]);
220
+
221
+ // Subscribed once, for the processor's lifetime: the kit emits per run, and
222
+ // `current` says which job that run belongs to.
223
+ const listeners = {
224
+ 'run:start': (event) => {
225
+ if (current) current.runId = event.runId;
226
+ },
227
+ 'lock:acquired': (event) => {
228
+ if (current && !event.skipped) log(current, `🔒 Lock acquired (${event.acquireMs ?? 0}ms)`);
229
+ },
230
+ 'lock:lost': (event) => {
231
+ if (current) log(current, `⚠ Lock lost: ${event.reason ?? 'unknown reason'}`);
232
+ },
233
+ 'migration:start': (event) => {
234
+ if (!current) return;
235
+ // From here on the job is doing its work: a shutdown lets it finish
236
+ // rather than putting it back in the queue.
237
+ current.started = true;
238
+ progress(current, 'running');
239
+ log(current, `▶ ${event.direction} ${event.migration}`);
240
+ },
241
+ 'migration:success': (event) => {
242
+ if (!current) return;
243
+ const label = event.direction === 'up' ? '✔ Applied' : '↩ Reverted';
244
+ log(current, `${label} ${event.migration} [${event.durationMs ?? 0}ms]`);
245
+ },
246
+ 'migration:skipped': (event) => {
247
+ if (current) log(current, `⏭ Skipped ${event.migration} (${event.reason ?? 'skipped'})`);
248
+ },
249
+ 'converge:start': () => {
250
+ if (!current) return;
251
+ current.started = true;
252
+ progress(current, 'running');
253
+ },
254
+ 'converge:action': (event) => {
255
+ if (!current) return;
256
+ const target = event.target === 'index' ? `index ${event.name}` : event.target;
257
+ const what = `${event.action} ${target} on ${event.collection}`;
258
+ if (event.status === 'started') log(current, `… ${what}`);
259
+ else if (event.status === 'applied') log(current, `✔ ${what} [${event.durationMs ?? 0}ms]`);
260
+ else log(current, `✖ ${what}: ${event.error ?? 'failed'}`);
261
+ },
262
+ 'converge:end': (event) => {
263
+ if (current && event.success) log(current, `✔ Converged ${event.changed} change(s)`);
264
+ },
265
+ };
266
+ for (const [event, listener] of Object.entries(listeners)) kit.on(event, listener);
267
+
268
+ function resultOf(ctx, rows, waitedMs) {
269
+ const { data } = ctx;
270
+ const row = rows[0];
271
+ return {
272
+ migration: data.migration,
273
+ direction: data.direction,
274
+ status: row?.status ?? 'skipped',
275
+ ...(row?.duration !== undefined ? { duration: row.duration } : {}),
276
+ ...(row?.batch !== undefined ? { batch: row.batch } : {}),
277
+ ...(ctx.runId ? { runId: ctx.runId } : {}),
278
+ ...(row?.reason ? { reason: row.reason } : {}),
279
+ lockWaitMs: waitedMs,
280
+ };
281
+ }
282
+
283
+ async function runMigrationJob(ctx, signal) {
284
+ const { data } = ctx;
285
+ const ordered = data.ordered ?? defaultOrdered;
286
+ const attempt = () =>
287
+ data.direction === JOB_NAMES.UP
288
+ ? kit.up(data.migration, {
289
+ batch: data.batch,
290
+ ...(ordered ? { ordered: true } : {}),
291
+ ...(data.force ? { force: true } : {}),
292
+ ...(data.checksum ? { checksum: data.checksum } : {}),
293
+ ...pickActor(data),
294
+ })
295
+ : kit.down(data.migration, { ...(ordered ? { ordered: true } : {}), ...pickActor(data) });
296
+
297
+ try {
298
+ const { result, waitedMs } = await waitForLock(ctx, attempt, signal);
299
+ return resultOf(ctx, result, waitedMs);
300
+ } catch (error) {
301
+ // A duplicate rollback job: the first one already reverted it. Same
302
+ // outcome as a duplicate `up` job, which the kit reports as skipped.
303
+ if (data.direction === JOB_NAMES.DOWN && error instanceof NotAppliedError) {
304
+ return {
305
+ migration: data.migration,
306
+ direction: data.direction,
307
+ status: 'skipped',
308
+ reason: 'Not applied',
309
+ lockWaitMs: 0,
310
+ };
311
+ }
312
+ throw error;
313
+ }
314
+ }
315
+
316
+ /** The lock-wait loop every run-kind job shares */
317
+ function waitForLock(ctx, attempt, signal) {
318
+ return withLockWait(attempt, {
319
+ ...waitOptions,
320
+ logger: kit.logger,
321
+ signal,
322
+ isTransient: isTransientForJob,
323
+ // A duck-typed kit has no telemetry to report to.
324
+ onSettle: (wait) => kit[RECORD_LOCK_WAIT]?.(wait),
325
+ onWait: ({ attempts, waitedMs: soFar, code }) => {
326
+ // Each progress update is a Redis write and an event-stream entry:
327
+ // a long wait reports every few seconds, not on every poll.
328
+ const now = Date.now();
329
+ if (attempts > 1 && now - (ctx.lastWaitProgressAt ?? 0) < WAIT_PROGRESS_INTERVAL_MS) {
330
+ return;
331
+ }
332
+ ctx.lastWaitProgressAt = now;
333
+ if (attempts === 1) {
334
+ log(
335
+ ctx,
336
+ code === 'MIGRATION_BLOCKED'
337
+ ? 'Earlier migration(s) not applied yet — waiting for them…'
338
+ : 'Migration lock held by another process — waiting…',
339
+ );
340
+ }
341
+ progress(ctx, 'lock-wait', { attempts, waitedMs: soFar });
342
+ },
343
+ });
344
+ }
345
+
346
+ async function runConvergeJob(ctx, signal) {
347
+ const { data } = ctx;
348
+ // Ordered by default: the tail of a deploy must not converge to a schema
349
+ // its own migrations have not reached yet — and the check is the kit's,
350
+ // under the lock, so a peer still applying the last migration is waited
351
+ // out rather than raced.
352
+ const ordered = data.ordered ?? defaultOrdered;
353
+ const { result, waitedMs } = await waitForLock(
354
+ ctx,
355
+ () => kit.converge({ ...(ordered ? { ordered: true } : {}), ...pickActor(data) }),
356
+ signal,
357
+ );
358
+ return redactDeep({
359
+ kind: 'converge',
360
+ ...(data.groupId !== undefined ? { groupId: data.groupId } : {}),
361
+ changed: result.changed,
362
+ inSync: result.inSync,
363
+ collections: result.collections,
364
+ ...(result.unstable ? { unstable: result.unstable } : {}),
365
+ ...(ctx.runId ? { runId: ctx.runId } : {}),
366
+ lockWaitMs: waitedMs,
367
+ });
368
+ }
369
+
370
+ /**
371
+ * The circuit breaker of a schedule: the next migration in line failed, and
372
+ * its file is still the version that failed. Re-enqueueing it every tick
373
+ * would re-run a migration that may have half-applied its changes, again and
374
+ * again, adding a failed job each time — until a fix is deployed (the
375
+ * checksum changes) or someone asks for it explicitly (`enqueueUp(name)`,
376
+ * which never consults this). Returns `{ migration, reason, failedAt? }`.
377
+ */
378
+ async function heldFailure(first, to) {
379
+ if (!first || (to !== undefined && first.file > to)) return undefined;
380
+ if (typeof kit.list !== 'function' || typeof kit.dryRun !== 'function') return undefined;
381
+ const row = (await kit.list('all', { checksums: false })).find(
382
+ (candidate) => candidate.file === first.file,
383
+ );
384
+ if (row?.status !== 'failed' || typeof row.failedChecksum !== 'string') return undefined;
385
+ const [planned] = await kit.dryRun('up', first.file);
386
+ if (planned?.checksum !== row.failedChecksum) return undefined;
387
+ return {
388
+ migration: first.file,
389
+ reason: 'failed, and unchanged since',
390
+ ...(row.failedAt ? { failedAt: row.failedAt } : {}),
391
+ };
392
+ }
393
+
394
+ async function runSyncJob(ctx) {
395
+ if (!queue) {
396
+ throw new ConfigInvalidError(
397
+ 'A sync job needs the queue to enqueue into — pass `queue` to createMigrationProcessor',
398
+ );
399
+ }
400
+ const { to } = ctx.data;
401
+ // The cheap probe first: a scheduler ticks far more often than there is
402
+ // anything to do, and planning proper re-reads the whole directory.
403
+ const pending = await kit.list('pending');
404
+ if (pending.length === 0) {
405
+ const result = {
406
+ kind: 'sync',
407
+ groupId: null,
408
+ batch: null,
409
+ enqueued: 0,
410
+ upToDate: true,
411
+ migrations: [],
412
+ };
413
+ // With `convergeAfterUp`, a tick that finds no migration still checks
414
+ // the declared collections — a deploy that only changed a definition
415
+ // converges on the next tick. A dry run first, so an idle tick takes no
416
+ // lock; a job only when something differs.
417
+ if (
418
+ to === undefined &&
419
+ typeof kit.convergesAfterUp === 'function' &&
420
+ (await kit.convergesAfterUp()) &&
421
+ !(await kit.converge({ dryRun: true })).inSync
422
+ ) {
423
+ const handle = await enqueueConverge(queue, kit, jobOptions ? { jobOptions } : {});
424
+ result.converge = { jobId: handle.jobId, deduplicated: handle.deduplicated };
425
+ }
426
+ return result;
427
+ }
428
+ const held = await heldFailure(pending[0], to);
429
+ if (held) {
430
+ kit.logger.warn(
431
+ `⚠ sync: ${held.migration} failed and has not changed since — not enqueued again ` +
432
+ `until the file changes (or enqueueUp('${held.migration}') asks for it)`,
433
+ { migration: held.migration },
434
+ );
435
+ return {
436
+ kind: 'sync',
437
+ groupId: null,
438
+ batch: null,
439
+ enqueued: 0,
440
+ upToDate: false,
441
+ migrations: [],
442
+ held,
443
+ };
444
+ }
445
+ const group = await enqueueUp(queue, kit, {
446
+ ...(to !== undefined ? { to } : {}),
447
+ ...(jobOptions ? { jobOptions } : {}),
448
+ });
449
+ const migrations = [];
450
+ for (const job of group.jobs) migrations.push(job.migration);
451
+ return {
452
+ kind: 'sync',
453
+ groupId: group.upToDate ? null : group.groupId,
454
+ batch: group.batch,
455
+ enqueued: group.jobs.length,
456
+ upToDate: group.upToDate,
457
+ migrations,
458
+ ...(group.converge
459
+ ? { converge: { jobId: group.converge.id, deduplicated: group.converge.deduplicated } }
460
+ : {}),
461
+ };
462
+ }
463
+
464
+ /**
465
+ * Put a job that a shutdown stopped before it started its work back at the
466
+ * head of the queue, and return the error that tells BullMQ so — or
467
+ * undefined when the job is to fail as usual.
468
+ *
469
+ * Failing it would be for good (one attempt), and every job of its group
470
+ * behind it would then fail as blocked: a rolling deploy would end each
471
+ * enqueue it interrupts. Moved back, the job keeps its place in line and the
472
+ * next worker runs it. A job whose migration (or converge) had begun is
473
+ * never put back — it ran, and its outcome is what it is.
474
+ */
475
+ async function requeueOnShutdown(ctx, error, token) {
476
+ if (!(error instanceof RunAbortedError) || !shutdownController.signal.aborted) return undefined;
477
+ if (ctx.started || typeof ctx.job?.moveToWait !== 'function' || typeof token !== 'string') {
478
+ return undefined;
479
+ }
480
+ const reason = errorText(shutdownController.signal.reason ?? 'shutting down');
481
+ log(ctx, `↩ Returned to the queue: ${reason}`);
482
+ await flush(ctx);
483
+ try {
484
+ await ctx.job.moveToWait(token);
485
+ } catch {
486
+ // The lock is gone (a stall already moved it) or Redis is: fail as usual.
487
+ return undefined;
488
+ }
489
+ kit.logger.debug(`↩ Returned job ${ctx.job.id} to the queue: ${reason}`, {
490
+ ...jobIds(ctx),
491
+ reason,
492
+ });
493
+ const requeued = new RunAbortedError(`Returned to the queue: ${reason}`, {
494
+ reason,
495
+ requeued: true,
496
+ ...jobIds(ctx),
497
+ });
498
+ requeued.name = WAITING_ERROR_NAME;
499
+ return requeued;
500
+ }
501
+
502
+ async function handle(job, token, signal) {
503
+ const ctx = { job, data: undefined, runId: undefined, started: false, writes: new Set() };
504
+ const startedAt = Date.now();
505
+ const signals = [shutdownController.signal];
506
+ if (signal) signals.push(signal);
507
+ const abort = AbortSignal.any(signals);
508
+ // Reaches a run that is setting up or between migrations; one already
509
+ // executing a migration body finishes it — interrupting a body mid-write
510
+ // is what leaves a database half-migrated.
511
+ const onAbort = () => kit.stop(errorText(abort.reason ?? 'Queue job cancelled'));
512
+ try {
513
+ // Validated before anything connects: a payload that fails the contract
514
+ // must not reach the kit, let alone the filesystem.
515
+ ctx.data = parseJobData(job);
516
+ // Well-formed is not the same as permitted: what a payload may ask for
517
+ // beyond the ordinary is this worker's decision, not Redis's.
518
+ assertAllowed(job, ctx.data, allow);
519
+ current = ctx;
520
+ // A job fetched while this process shuts down goes straight back.
521
+ if (shutdownController.signal.aborted) throw shutdownController.signal.reason;
522
+ abort.addEventListener('abort', onAbort, { once: true });
523
+ kit.logger.debug(`▶ Job ${job?.id} (${describeJob(ctx.data)})`, {
524
+ ...jobIds(ctx),
525
+ ...jobFields(ctx.data),
526
+ });
527
+ await kit.connect();
528
+ let result;
529
+ if (ctx.data.kind === 'sync') result = await runSyncJob(ctx);
530
+ else if (ctx.data.kind === 'converge') result = await runConvergeJob(ctx, abort);
531
+ else result = await runMigrationJob(ctx, abort);
532
+ progress(ctx, 'completed', ctx.runId ? { runId: ctx.runId } : {});
533
+ await flush(ctx);
534
+ kit.logger.debug(`✔ Job ${job?.id} done`, {
535
+ ...jobIds(ctx),
536
+ ...jobFields(ctx.data),
537
+ durationMs: Date.now() - startedAt,
538
+ });
539
+ return result;
540
+ } catch (error) {
541
+ const requeued = await requeueOnShutdown(ctx, error, token);
542
+ if (requeued) throw requeued;
543
+ // BullMQ only retries when the job was given more than one attempt — the
544
+ // adapter's own jobs never are, so this matters for jobs enqueued some
545
+ // other way. Renaming is how a library that never imports bullmq says
546
+ // "do not retry"; the typed `code` and class are untouched.
547
+ if (!isRetryableError(error) && (job?.opts?.attempts ?? 1) > 1) {
548
+ error.name = UNRECOVERABLE_ERROR_NAME;
549
+ }
550
+ attachJobIds(error, ctx);
551
+ prepareErrorForQueue(error);
552
+ if (ctx.data) {
553
+ progress(ctx, 'failed', {
554
+ code: error instanceof MigronautError ? error.code : 'UNKNOWN',
555
+ ...(ctx.runId ? { runId: ctx.runId } : {}),
556
+ });
557
+ }
558
+ write(ctx, (target) =>
559
+ target.log?.(
560
+ redactOutbound(`✖ ${errorText(error)}${ctx.runId ? ` [run ${ctx.runId}]` : ''}`),
561
+ ),
562
+ );
563
+ await flush(ctx);
564
+ throw error;
565
+ } finally {
566
+ abort.removeEventListener('abort', onAbort);
567
+ current = undefined;
568
+ }
569
+ }
570
+
571
+ // Three declared parameters, on purpose — see the factory's doc comment.
572
+ async function processor(job, token, signal) {
573
+ const run = chain.then(() => handle(job, token, signal));
574
+ chain = run.catch(() => undefined);
575
+ return run;
576
+ }
577
+
578
+ /**
579
+ * Stop taking the lock: a job waiting for it fails with RunAbortedError (a
580
+ * retryable failure), and a run that has not started its migration is
581
+ * stopped. Irreversible — meant for process shutdown.
582
+ */
583
+ processor.shutdown = (reason = 'Migration worker shutting down') => {
584
+ if (!shutdownController.signal.aborted) {
585
+ shutdownController.abort(new RunAbortedError(reason, { reason }));
586
+ }
587
+ };
588
+
589
+ /** Shut down, let the job in flight settle, and disconnect a kit this processor created */
590
+ processor.close = async () => {
591
+ processor.shutdown();
592
+ await chain;
593
+ for (const [event, listener] of Object.entries(listeners)) kit.off(event, listener);
594
+ if (ownsKit) await kit.disconnect();
595
+ };
596
+
597
+ Object.defineProperty(processor, 'kit', { value: kit, enumerable: true });
598
+ return processor;
599
+ }
600
+
601
+ module.exports = {
602
+ RETRYABLE_CODES,
603
+ WAITING_ERROR_NAME,
604
+ createMigrationProcessor,
605
+ isTransientForJob,
606
+ isRetryableError,
607
+ resolveProcessorOptions,
608
+ };