@alexify/migronaut 1.0.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +409 -1
- package/README.md +248 -24
- package/bin/migronaut.js +11 -3
- package/bullmq.d.ts +845 -0
- package/bullmq.js +1 -0
- package/index.d.ts +757 -29
- package/migronaut.schema.json +191 -1
- package/package.json +27 -6
- package/src/bullmq/index.js +55 -0
- package/src/bullmq/jobs.js +454 -0
- package/src/bullmq/processor.js +608 -0
- package/src/bullmq/producer.js +424 -0
- package/src/bullmq/service.js +653 -0
- package/src/bullmq/wait.js +124 -0
- package/src/cli/args.js +12 -2
- package/src/cli/commands/baseline.js +45 -0
- package/src/cli/commands/converge.js +160 -0
- package/src/cli/commands/down.js +2 -0
- package/src/cli/commands/lock.js +2 -1
- package/src/cli/commands/redo.js +8 -1
- package/src/cli/commands/unlock.js +12 -2
- package/src/cli/commands/up.js +14 -1
- package/src/cli/exit-codes.js +10 -2
- package/src/cli/index.js +4 -0
- package/src/cli/shared.js +29 -7
- package/src/cli/table.js +105 -0
- package/src/core/audit.js +17 -3
- package/src/core/baseline.js +80 -0
- package/src/core/changelog.js +140 -24
- package/src/core/collections.js +372 -0
- package/src/core/config.js +125 -27
- package/src/core/converge-log.js +47 -0
- package/src/core/converge-plan.js +483 -0
- package/src/core/converge.js +867 -0
- package/src/core/import-runner.js +34 -6
- package/src/core/import.js +14 -7
- package/src/core/index-spec.js +496 -0
- package/src/core/lock-wait.js +260 -0
- package/src/core/lock.js +71 -20
- package/src/core/migrator.js +805 -304
- package/src/core/options.js +251 -0
- package/src/core/run-recorder.js +157 -0
- package/src/core/run.js +71 -71
- package/src/core/runner.js +70 -20
- package/src/core/sequence.js +134 -0
- package/src/errors/index.js +71 -1
- package/src/index.js +16 -0
- package/src/utils/actor.js +48 -0
- package/src/utils/canonical.js +179 -0
- package/src/utils/collection-name.js +21 -0
- package/src/utils/error.js +18 -1
- package/src/utils/id.js +77 -0
- package/src/utils/loader.js +39 -21
- package/src/utils/logger.js +30 -12
- package/src/utils/migration-name.js +32 -0
- package/src/utils/redact.js +57 -4
- package/src/utils/sanitize.js +8 -3
- package/src/utils/telemetry.js +393 -0
- package/src/utils/template.js +60 -12
|
@@ -0,0 +1,608 @@
|
|
|
1
|
+
const { MigratorKit, RECORD_LOCK_WAIT } = require('../core/migrator.js');
|
|
2
|
+
const { assertLockWaitOptions, withLockWait } = require('../core/lock-wait.js');
|
|
3
|
+
const {
|
|
4
|
+
ConfigInvalidError,
|
|
5
|
+
LockAlreadyHeldError,
|
|
6
|
+
MigrationBlockedError,
|
|
7
|
+
MigronautError,
|
|
8
|
+
NotAppliedError,
|
|
9
|
+
RunAbortedError,
|
|
10
|
+
} = require('../errors/index.js');
|
|
11
|
+
const { pickActor } = require('../utils/actor.js');
|
|
12
|
+
const { errorText } = require('../utils/error.js');
|
|
13
|
+
const { redactDeep, redactOutbound } = require('../utils/redact.js');
|
|
14
|
+
const {
|
|
15
|
+
JOB_NAMES,
|
|
16
|
+
assertAllowed,
|
|
17
|
+
isPlainObject,
|
|
18
|
+
parseJobData,
|
|
19
|
+
resolveAllow,
|
|
20
|
+
} = require('./jobs.js');
|
|
21
|
+
const { assertJobOptions, enqueueConverge, enqueueUp } = require('./producer.js');
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Failures a later attempt can get past without anything being fixed: the lock
|
|
25
|
+
* was busy or lost, the database was unreachable, the run was stopped. Every
|
|
26
|
+
* other MigronautError needs a change first — to the migration, the files or
|
|
27
|
+
* the changelog — so retrying it only repeats the failure (and, in a FIFO
|
|
28
|
+
* queue, reorders the line). Errors of unknown origin stay retryable: not
|
|
29
|
+
* recognising a failure is no reason to rule a retry out.
|
|
30
|
+
*/
|
|
31
|
+
const RETRYABLE_CODES = Object.freeze([
|
|
32
|
+
'LOCK_ALREADY_HELD',
|
|
33
|
+
'LOCK_LOST',
|
|
34
|
+
'LOCK_RELEASE_FAILED',
|
|
35
|
+
'RUN_ABORTED',
|
|
36
|
+
'CONNECTION_FAILED',
|
|
37
|
+
]);
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* What a job waits out instead of failing: a held lock — and a block by
|
|
41
|
+
* migrations that have not failed. With more than one worker taking jobs
|
|
42
|
+
* (global concurrency off or unsupported, several processors), the job for an
|
|
43
|
+
* earlier migration can still be in flight on another worker when a later one
|
|
44
|
+
* takes the lock; failing the later job would fail its whole group although
|
|
45
|
+
* nothing went wrong. A blocker with a `'failed'` trace stopped the line for
|
|
46
|
+
* real, and that job fails at once, as before.
|
|
47
|
+
*/
|
|
48
|
+
function isTransientForJob(error) {
|
|
49
|
+
if (error instanceof LockAlreadyHeldError) return true;
|
|
50
|
+
return (
|
|
51
|
+
error instanceof MigrationBlockedError &&
|
|
52
|
+
Array.isArray(error.context?.failed) &&
|
|
53
|
+
error.context.failed.length === 0
|
|
54
|
+
);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function isRetryableError(error) {
|
|
58
|
+
return !(error instanceof MigronautError) || RETRYABLE_CODES.includes(error.code);
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** The least time between two `lock-wait` progress updates of one job */
|
|
62
|
+
const WAIT_PROGRESS_INTERVAL_MS = 2000;
|
|
63
|
+
|
|
64
|
+
/** What BullMQ checks (by name, not only by class) to fail a job without retrying it */
|
|
65
|
+
const UNRECOVERABLE_ERROR_NAME = 'UnrecoverableError';
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* What BullMQ checks (by name) for a job the processor has moved back to the
|
|
69
|
+
* wait list itself: neither failed nor completed, nothing more to record.
|
|
70
|
+
*/
|
|
71
|
+
const WAITING_ERROR_NAME = 'WaitingError';
|
|
72
|
+
|
|
73
|
+
/** The correlation ids of a job, for its error context and log lines */
|
|
74
|
+
function jobIds(ctx) {
|
|
75
|
+
return {
|
|
76
|
+
...(ctx.job?.id !== undefined ? { jobId: String(ctx.job.id) } : {}),
|
|
77
|
+
...(ctx.data?.groupId !== undefined ? { groupId: ctx.data.groupId } : {}),
|
|
78
|
+
...(ctx.runId ? { runId: ctx.runId } : {}),
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/** A job in a few words, for log lines: `up 20260101-x.js (1/3)`, `converge`, `sync` */
|
|
83
|
+
function describeJob(data) {
|
|
84
|
+
if (data.kind !== 'migration') return data.kind;
|
|
85
|
+
return `${data.direction} ${data.migration} (${data.index + 1}/${data.total})`;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** The structured fields of a job's log lines */
|
|
89
|
+
function jobFields(data) {
|
|
90
|
+
if (data.kind !== 'migration') return { kind: data.kind };
|
|
91
|
+
return {
|
|
92
|
+
migration: data.migration,
|
|
93
|
+
direction: data.direction,
|
|
94
|
+
index: data.index,
|
|
95
|
+
total: data.total,
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Stamp the job's ids onto a typed error before it leaves for the queue — the
|
|
101
|
+
* one place a failed job's run id survives, since a failed job has no return
|
|
102
|
+
* value. Copy-on-write: the error may be a shared abort reason.
|
|
103
|
+
*/
|
|
104
|
+
function attachJobIds(error, ctx) {
|
|
105
|
+
if (!(error instanceof MigronautError)) return;
|
|
106
|
+
const ids = jobIds(ctx);
|
|
107
|
+
if (Object.keys(ids).length > 0) error.context = { ...error.context, ...ids };
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Prepare an error for the queue. BullMQ stores its message (as the job's
|
|
112
|
+
* `failedReason`) and stack, so both leave the process and must be redacted.
|
|
113
|
+
* The message is also all a dashboard shows: a wrapper says WHICH migration
|
|
114
|
+
* failed and keeps the WHY in `context.cause`, which the queue never sees —
|
|
115
|
+
* so the cause is folded into the message, as the changelog's failure trace
|
|
116
|
+
* already does.
|
|
117
|
+
*/
|
|
118
|
+
function prepareErrorForQueue(error) {
|
|
119
|
+
if (!(error instanceof Error)) return;
|
|
120
|
+
const cause = error instanceof MigronautError ? error.context?.cause : undefined;
|
|
121
|
+
try {
|
|
122
|
+
const message = redactOutbound(error.message);
|
|
123
|
+
error.message =
|
|
124
|
+
typeof cause === 'string' && cause.length > 0 && !message.includes(cause)
|
|
125
|
+
? `${message} — ${redactOutbound(cause)}`
|
|
126
|
+
: message;
|
|
127
|
+
if (typeof error.stack === 'string') error.stack = redactOutbound(error.stack);
|
|
128
|
+
} catch {
|
|
129
|
+
// A frozen error cannot be rewritten; its message was written by us anyway.
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Validate the processor's options and return the resolved lock-wait budget.
|
|
135
|
+
* Pure — nothing is constructed — so a caller can reject bad input before it
|
|
136
|
+
* opens any connection of its own.
|
|
137
|
+
*/
|
|
138
|
+
function resolveProcessorOptions(options) {
|
|
139
|
+
if (!isPlainObject(options)) {
|
|
140
|
+
throw new ConfigInvalidError('createMigrationProcessor options must be an object');
|
|
141
|
+
}
|
|
142
|
+
const { kit, config, lockWait = {}, jobOptions, ordered = true, allow } = options;
|
|
143
|
+
if (kit !== undefined && config !== undefined) {
|
|
144
|
+
throw new ConfigInvalidError('Pass either `kit` or `config`, not both');
|
|
145
|
+
}
|
|
146
|
+
if (kit !== undefined && typeof kit?.up !== 'function') {
|
|
147
|
+
throw new ConfigInvalidError('kit must be a MigratorKit instance');
|
|
148
|
+
}
|
|
149
|
+
if (typeof ordered !== 'boolean') {
|
|
150
|
+
throw new ConfigInvalidError('ordered must be a boolean', { ordered });
|
|
151
|
+
}
|
|
152
|
+
if (!isPlainObject(lockWait)) {
|
|
153
|
+
throw new ConfigInvalidError('lockWait must be an object', { lockWait: typeof lockWait });
|
|
154
|
+
}
|
|
155
|
+
// Unlike runMigrations, waiting is the default: nothing is blocked on this
|
|
156
|
+
// job, and the budget only burns while the lock's holder is stalled.
|
|
157
|
+
const waitOptions = { onLockHeld: 'wait', ...lockWait };
|
|
158
|
+
assertLockWaitOptions(waitOptions);
|
|
159
|
+
assertJobOptions(jobOptions);
|
|
160
|
+
return { waitOptions, defaultOrdered: ordered, allow: resolveAllow(allow) };
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Build the function a BullMQ Worker runs for each migration job.
|
|
165
|
+
*
|
|
166
|
+
* One long-lived MigratorKit serves every job, and jobs are run one at a time
|
|
167
|
+
* even when the Worker is configured for more: a kit rejects overlapping runs,
|
|
168
|
+
* and migrations are sequential by nature. Across processes the MongoDB lock
|
|
169
|
+
* does that job, and the `ordered` guard keeps the sequence.
|
|
170
|
+
*
|
|
171
|
+
* Declared with exactly three parameters: BullMQ passes the cancellation
|
|
172
|
+
* signal only to processors whose `length` is at least 3.
|
|
173
|
+
*/
|
|
174
|
+
function createMigrationProcessor(options = {}) {
|
|
175
|
+
const { waitOptions, defaultOrdered, allow } = resolveProcessorOptions(options);
|
|
176
|
+
const { kit: injectedKit, config, kitOptions, queue, jobOptions } = options;
|
|
177
|
+
|
|
178
|
+
const ownsKit = injectedKit === undefined;
|
|
179
|
+
const kit = injectedKit ?? new MigratorKit(config ?? {}, kitOptions);
|
|
180
|
+
const shutdownController = new AbortController();
|
|
181
|
+
/** Serializes jobs in this process — see the factory's doc comment */
|
|
182
|
+
let chain = Promise.resolve();
|
|
183
|
+
/** The job being processed, for the kit's event listeners */
|
|
184
|
+
let current;
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* Fire-and-forget write to the job (a log row, a progress update). Never
|
|
188
|
+
* awaited on the run path — Redis must not pace a migration — and never
|
|
189
|
+
* allowed to fail the job; `flush` drains what is still in flight before the
|
|
190
|
+
* job settles, so nothing lands on a job BullMQ already finished.
|
|
191
|
+
*/
|
|
192
|
+
function write(ctx, action) {
|
|
193
|
+
let pending;
|
|
194
|
+
try {
|
|
195
|
+
pending = Promise.resolve(action(ctx.job)).catch(() => undefined);
|
|
196
|
+
} catch {
|
|
197
|
+
return;
|
|
198
|
+
}
|
|
199
|
+
ctx.writes.add(pending);
|
|
200
|
+
pending.finally(() => ctx.writes.delete(pending));
|
|
201
|
+
}
|
|
202
|
+
const log = (ctx, row) => write(ctx, (job) => job.log?.(redactOutbound(row)));
|
|
203
|
+
const progress = (ctx, phase, extra) =>
|
|
204
|
+
write(ctx, (job) =>
|
|
205
|
+
job.updateProgress?.({
|
|
206
|
+
phase,
|
|
207
|
+
...(ctx.data.kind === 'migration'
|
|
208
|
+
? {
|
|
209
|
+
migration: ctx.data.migration,
|
|
210
|
+
direction: ctx.data.direction,
|
|
211
|
+
groupId: ctx.data.groupId,
|
|
212
|
+
index: ctx.data.index,
|
|
213
|
+
total: ctx.data.total,
|
|
214
|
+
}
|
|
215
|
+
: { kind: ctx.data.kind }),
|
|
216
|
+
...extra,
|
|
217
|
+
}),
|
|
218
|
+
);
|
|
219
|
+
const flush = (ctx) => Promise.allSettled([...ctx.writes]);
|
|
220
|
+
|
|
221
|
+
// Subscribed once, for the processor's lifetime: the kit emits per run, and
|
|
222
|
+
// `current` says which job that run belongs to.
|
|
223
|
+
const listeners = {
|
|
224
|
+
'run:start': (event) => {
|
|
225
|
+
if (current) current.runId = event.runId;
|
|
226
|
+
},
|
|
227
|
+
'lock:acquired': (event) => {
|
|
228
|
+
if (current && !event.skipped) log(current, `🔒 Lock acquired (${event.acquireMs ?? 0}ms)`);
|
|
229
|
+
},
|
|
230
|
+
'lock:lost': (event) => {
|
|
231
|
+
if (current) log(current, `⚠ Lock lost: ${event.reason ?? 'unknown reason'}`);
|
|
232
|
+
},
|
|
233
|
+
'migration:start': (event) => {
|
|
234
|
+
if (!current) return;
|
|
235
|
+
// From here on the job is doing its work: a shutdown lets it finish
|
|
236
|
+
// rather than putting it back in the queue.
|
|
237
|
+
current.started = true;
|
|
238
|
+
progress(current, 'running');
|
|
239
|
+
log(current, `▶ ${event.direction} ${event.migration}`);
|
|
240
|
+
},
|
|
241
|
+
'migration:success': (event) => {
|
|
242
|
+
if (!current) return;
|
|
243
|
+
const label = event.direction === 'up' ? '✔ Applied' : '↩ Reverted';
|
|
244
|
+
log(current, `${label} ${event.migration} [${event.durationMs ?? 0}ms]`);
|
|
245
|
+
},
|
|
246
|
+
'migration:skipped': (event) => {
|
|
247
|
+
if (current) log(current, `⏭ Skipped ${event.migration} (${event.reason ?? 'skipped'})`);
|
|
248
|
+
},
|
|
249
|
+
'converge:start': () => {
|
|
250
|
+
if (!current) return;
|
|
251
|
+
current.started = true;
|
|
252
|
+
progress(current, 'running');
|
|
253
|
+
},
|
|
254
|
+
'converge:action': (event) => {
|
|
255
|
+
if (!current) return;
|
|
256
|
+
const target = event.target === 'index' ? `index ${event.name}` : event.target;
|
|
257
|
+
const what = `${event.action} ${target} on ${event.collection}`;
|
|
258
|
+
if (event.status === 'started') log(current, `… ${what}`);
|
|
259
|
+
else if (event.status === 'applied') log(current, `✔ ${what} [${event.durationMs ?? 0}ms]`);
|
|
260
|
+
else log(current, `✖ ${what}: ${event.error ?? 'failed'}`);
|
|
261
|
+
},
|
|
262
|
+
'converge:end': (event) => {
|
|
263
|
+
if (current && event.success) log(current, `✔ Converged ${event.changed} change(s)`);
|
|
264
|
+
},
|
|
265
|
+
};
|
|
266
|
+
for (const [event, listener] of Object.entries(listeners)) kit.on(event, listener);
|
|
267
|
+
|
|
268
|
+
function resultOf(ctx, rows, waitedMs) {
|
|
269
|
+
const { data } = ctx;
|
|
270
|
+
const row = rows[0];
|
|
271
|
+
return {
|
|
272
|
+
migration: data.migration,
|
|
273
|
+
direction: data.direction,
|
|
274
|
+
status: row?.status ?? 'skipped',
|
|
275
|
+
...(row?.duration !== undefined ? { duration: row.duration } : {}),
|
|
276
|
+
...(row?.batch !== undefined ? { batch: row.batch } : {}),
|
|
277
|
+
...(ctx.runId ? { runId: ctx.runId } : {}),
|
|
278
|
+
...(row?.reason ? { reason: row.reason } : {}),
|
|
279
|
+
lockWaitMs: waitedMs,
|
|
280
|
+
};
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
async function runMigrationJob(ctx, signal) {
|
|
284
|
+
const { data } = ctx;
|
|
285
|
+
const ordered = data.ordered ?? defaultOrdered;
|
|
286
|
+
const attempt = () =>
|
|
287
|
+
data.direction === JOB_NAMES.UP
|
|
288
|
+
? kit.up(data.migration, {
|
|
289
|
+
batch: data.batch,
|
|
290
|
+
...(ordered ? { ordered: true } : {}),
|
|
291
|
+
...(data.force ? { force: true } : {}),
|
|
292
|
+
...(data.checksum ? { checksum: data.checksum } : {}),
|
|
293
|
+
...pickActor(data),
|
|
294
|
+
})
|
|
295
|
+
: kit.down(data.migration, { ...(ordered ? { ordered: true } : {}), ...pickActor(data) });
|
|
296
|
+
|
|
297
|
+
try {
|
|
298
|
+
const { result, waitedMs } = await waitForLock(ctx, attempt, signal);
|
|
299
|
+
return resultOf(ctx, result, waitedMs);
|
|
300
|
+
} catch (error) {
|
|
301
|
+
// A duplicate rollback job: the first one already reverted it. Same
|
|
302
|
+
// outcome as a duplicate `up` job, which the kit reports as skipped.
|
|
303
|
+
if (data.direction === JOB_NAMES.DOWN && error instanceof NotAppliedError) {
|
|
304
|
+
return {
|
|
305
|
+
migration: data.migration,
|
|
306
|
+
direction: data.direction,
|
|
307
|
+
status: 'skipped',
|
|
308
|
+
reason: 'Not applied',
|
|
309
|
+
lockWaitMs: 0,
|
|
310
|
+
};
|
|
311
|
+
}
|
|
312
|
+
throw error;
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
/** The lock-wait loop every run-kind job shares */
|
|
317
|
+
function waitForLock(ctx, attempt, signal) {
|
|
318
|
+
return withLockWait(attempt, {
|
|
319
|
+
...waitOptions,
|
|
320
|
+
logger: kit.logger,
|
|
321
|
+
signal,
|
|
322
|
+
isTransient: isTransientForJob,
|
|
323
|
+
// A duck-typed kit has no telemetry to report to.
|
|
324
|
+
onSettle: (wait) => kit[RECORD_LOCK_WAIT]?.(wait),
|
|
325
|
+
onWait: ({ attempts, waitedMs: soFar, code }) => {
|
|
326
|
+
// Each progress update is a Redis write and an event-stream entry:
|
|
327
|
+
// a long wait reports every few seconds, not on every poll.
|
|
328
|
+
const now = Date.now();
|
|
329
|
+
if (attempts > 1 && now - (ctx.lastWaitProgressAt ?? 0) < WAIT_PROGRESS_INTERVAL_MS) {
|
|
330
|
+
return;
|
|
331
|
+
}
|
|
332
|
+
ctx.lastWaitProgressAt = now;
|
|
333
|
+
if (attempts === 1) {
|
|
334
|
+
log(
|
|
335
|
+
ctx,
|
|
336
|
+
code === 'MIGRATION_BLOCKED'
|
|
337
|
+
? 'Earlier migration(s) not applied yet — waiting for them…'
|
|
338
|
+
: 'Migration lock held by another process — waiting…',
|
|
339
|
+
);
|
|
340
|
+
}
|
|
341
|
+
progress(ctx, 'lock-wait', { attempts, waitedMs: soFar });
|
|
342
|
+
},
|
|
343
|
+
});
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
async function runConvergeJob(ctx, signal) {
|
|
347
|
+
const { data } = ctx;
|
|
348
|
+
// Ordered by default: the tail of a deploy must not converge to a schema
|
|
349
|
+
// its own migrations have not reached yet — and the check is the kit's,
|
|
350
|
+
// under the lock, so a peer still applying the last migration is waited
|
|
351
|
+
// out rather than raced.
|
|
352
|
+
const ordered = data.ordered ?? defaultOrdered;
|
|
353
|
+
const { result, waitedMs } = await waitForLock(
|
|
354
|
+
ctx,
|
|
355
|
+
() => kit.converge({ ...(ordered ? { ordered: true } : {}), ...pickActor(data) }),
|
|
356
|
+
signal,
|
|
357
|
+
);
|
|
358
|
+
return redactDeep({
|
|
359
|
+
kind: 'converge',
|
|
360
|
+
...(data.groupId !== undefined ? { groupId: data.groupId } : {}),
|
|
361
|
+
changed: result.changed,
|
|
362
|
+
inSync: result.inSync,
|
|
363
|
+
collections: result.collections,
|
|
364
|
+
...(result.unstable ? { unstable: result.unstable } : {}),
|
|
365
|
+
...(ctx.runId ? { runId: ctx.runId } : {}),
|
|
366
|
+
lockWaitMs: waitedMs,
|
|
367
|
+
});
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
/**
|
|
371
|
+
* The circuit breaker of a schedule: the next migration in line failed, and
|
|
372
|
+
* its file is still the version that failed. Re-enqueueing it every tick
|
|
373
|
+
* would re-run a migration that may have half-applied its changes, again and
|
|
374
|
+
* again, adding a failed job each time — until a fix is deployed (the
|
|
375
|
+
* checksum changes) or someone asks for it explicitly (`enqueueUp(name)`,
|
|
376
|
+
* which never consults this). Returns `{ migration, reason, failedAt? }`.
|
|
377
|
+
*/
|
|
378
|
+
async function heldFailure(first, to) {
|
|
379
|
+
if (!first || (to !== undefined && first.file > to)) return undefined;
|
|
380
|
+
if (typeof kit.list !== 'function' || typeof kit.dryRun !== 'function') return undefined;
|
|
381
|
+
const row = (await kit.list('all', { checksums: false })).find(
|
|
382
|
+
(candidate) => candidate.file === first.file,
|
|
383
|
+
);
|
|
384
|
+
if (row?.status !== 'failed' || typeof row.failedChecksum !== 'string') return undefined;
|
|
385
|
+
const [planned] = await kit.dryRun('up', first.file);
|
|
386
|
+
if (planned?.checksum !== row.failedChecksum) return undefined;
|
|
387
|
+
return {
|
|
388
|
+
migration: first.file,
|
|
389
|
+
reason: 'failed, and unchanged since',
|
|
390
|
+
...(row.failedAt ? { failedAt: row.failedAt } : {}),
|
|
391
|
+
};
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
async function runSyncJob(ctx) {
|
|
395
|
+
if (!queue) {
|
|
396
|
+
throw new ConfigInvalidError(
|
|
397
|
+
'A sync job needs the queue to enqueue into — pass `queue` to createMigrationProcessor',
|
|
398
|
+
);
|
|
399
|
+
}
|
|
400
|
+
const { to } = ctx.data;
|
|
401
|
+
// The cheap probe first: a scheduler ticks far more often than there is
|
|
402
|
+
// anything to do, and planning proper re-reads the whole directory.
|
|
403
|
+
const pending = await kit.list('pending');
|
|
404
|
+
if (pending.length === 0) {
|
|
405
|
+
const result = {
|
|
406
|
+
kind: 'sync',
|
|
407
|
+
groupId: null,
|
|
408
|
+
batch: null,
|
|
409
|
+
enqueued: 0,
|
|
410
|
+
upToDate: true,
|
|
411
|
+
migrations: [],
|
|
412
|
+
};
|
|
413
|
+
// With `convergeAfterUp`, a tick that finds no migration still checks
|
|
414
|
+
// the declared collections — a deploy that only changed a definition
|
|
415
|
+
// converges on the next tick. A dry run first, so an idle tick takes no
|
|
416
|
+
// lock; a job only when something differs.
|
|
417
|
+
if (
|
|
418
|
+
to === undefined &&
|
|
419
|
+
typeof kit.convergesAfterUp === 'function' &&
|
|
420
|
+
(await kit.convergesAfterUp()) &&
|
|
421
|
+
!(await kit.converge({ dryRun: true })).inSync
|
|
422
|
+
) {
|
|
423
|
+
const handle = await enqueueConverge(queue, kit, jobOptions ? { jobOptions } : {});
|
|
424
|
+
result.converge = { jobId: handle.jobId, deduplicated: handle.deduplicated };
|
|
425
|
+
}
|
|
426
|
+
return result;
|
|
427
|
+
}
|
|
428
|
+
const held = await heldFailure(pending[0], to);
|
|
429
|
+
if (held) {
|
|
430
|
+
kit.logger.warn(
|
|
431
|
+
`⚠ sync: ${held.migration} failed and has not changed since — not enqueued again ` +
|
|
432
|
+
`until the file changes (or enqueueUp('${held.migration}') asks for it)`,
|
|
433
|
+
{ migration: held.migration },
|
|
434
|
+
);
|
|
435
|
+
return {
|
|
436
|
+
kind: 'sync',
|
|
437
|
+
groupId: null,
|
|
438
|
+
batch: null,
|
|
439
|
+
enqueued: 0,
|
|
440
|
+
upToDate: false,
|
|
441
|
+
migrations: [],
|
|
442
|
+
held,
|
|
443
|
+
};
|
|
444
|
+
}
|
|
445
|
+
const group = await enqueueUp(queue, kit, {
|
|
446
|
+
...(to !== undefined ? { to } : {}),
|
|
447
|
+
...(jobOptions ? { jobOptions } : {}),
|
|
448
|
+
});
|
|
449
|
+
const migrations = [];
|
|
450
|
+
for (const job of group.jobs) migrations.push(job.migration);
|
|
451
|
+
return {
|
|
452
|
+
kind: 'sync',
|
|
453
|
+
groupId: group.upToDate ? null : group.groupId,
|
|
454
|
+
batch: group.batch,
|
|
455
|
+
enqueued: group.jobs.length,
|
|
456
|
+
upToDate: group.upToDate,
|
|
457
|
+
migrations,
|
|
458
|
+
...(group.converge
|
|
459
|
+
? { converge: { jobId: group.converge.id, deduplicated: group.converge.deduplicated } }
|
|
460
|
+
: {}),
|
|
461
|
+
};
|
|
462
|
+
}
|
|
463
|
+
|
|
464
|
+
/**
|
|
465
|
+
* Put a job that a shutdown stopped before it started its work back at the
|
|
466
|
+
* head of the queue, and return the error that tells BullMQ so — or
|
|
467
|
+
* undefined when the job is to fail as usual.
|
|
468
|
+
*
|
|
469
|
+
* Failing it would be for good (one attempt), and every job of its group
|
|
470
|
+
* behind it would then fail as blocked: a rolling deploy would end each
|
|
471
|
+
* enqueue it interrupts. Moved back, the job keeps its place in line and the
|
|
472
|
+
* next worker runs it. A job whose migration (or converge) had begun is
|
|
473
|
+
* never put back — it ran, and its outcome is what it is.
|
|
474
|
+
*/
|
|
475
|
+
async function requeueOnShutdown(ctx, error, token) {
|
|
476
|
+
if (!(error instanceof RunAbortedError) || !shutdownController.signal.aborted) return undefined;
|
|
477
|
+
if (ctx.started || typeof ctx.job?.moveToWait !== 'function' || typeof token !== 'string') {
|
|
478
|
+
return undefined;
|
|
479
|
+
}
|
|
480
|
+
const reason = errorText(shutdownController.signal.reason ?? 'shutting down');
|
|
481
|
+
log(ctx, `↩ Returned to the queue: ${reason}`);
|
|
482
|
+
await flush(ctx);
|
|
483
|
+
try {
|
|
484
|
+
await ctx.job.moveToWait(token);
|
|
485
|
+
} catch {
|
|
486
|
+
// The lock is gone (a stall already moved it) or Redis is: fail as usual.
|
|
487
|
+
return undefined;
|
|
488
|
+
}
|
|
489
|
+
kit.logger.debug(`↩ Returned job ${ctx.job.id} to the queue: ${reason}`, {
|
|
490
|
+
...jobIds(ctx),
|
|
491
|
+
reason,
|
|
492
|
+
});
|
|
493
|
+
const requeued = new RunAbortedError(`Returned to the queue: ${reason}`, {
|
|
494
|
+
reason,
|
|
495
|
+
requeued: true,
|
|
496
|
+
...jobIds(ctx),
|
|
497
|
+
});
|
|
498
|
+
requeued.name = WAITING_ERROR_NAME;
|
|
499
|
+
return requeued;
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
async function handle(job, token, signal) {
|
|
503
|
+
const ctx = { job, data: undefined, runId: undefined, started: false, writes: new Set() };
|
|
504
|
+
const startedAt = Date.now();
|
|
505
|
+
const signals = [shutdownController.signal];
|
|
506
|
+
if (signal) signals.push(signal);
|
|
507
|
+
const abort = AbortSignal.any(signals);
|
|
508
|
+
// Reaches a run that is setting up or between migrations; one already
|
|
509
|
+
// executing a migration body finishes it — interrupting a body mid-write
|
|
510
|
+
// is what leaves a database half-migrated.
|
|
511
|
+
const onAbort = () => kit.stop(errorText(abort.reason ?? 'Queue job cancelled'));
|
|
512
|
+
try {
|
|
513
|
+
// Validated before anything connects: a payload that fails the contract
|
|
514
|
+
// must not reach the kit, let alone the filesystem.
|
|
515
|
+
ctx.data = parseJobData(job);
|
|
516
|
+
// Well-formed is not the same as permitted: what a payload may ask for
|
|
517
|
+
// beyond the ordinary is this worker's decision, not Redis's.
|
|
518
|
+
assertAllowed(job, ctx.data, allow);
|
|
519
|
+
current = ctx;
|
|
520
|
+
// A job fetched while this process shuts down goes straight back.
|
|
521
|
+
if (shutdownController.signal.aborted) throw shutdownController.signal.reason;
|
|
522
|
+
abort.addEventListener('abort', onAbort, { once: true });
|
|
523
|
+
kit.logger.debug(`▶ Job ${job?.id} (${describeJob(ctx.data)})`, {
|
|
524
|
+
...jobIds(ctx),
|
|
525
|
+
...jobFields(ctx.data),
|
|
526
|
+
});
|
|
527
|
+
await kit.connect();
|
|
528
|
+
let result;
|
|
529
|
+
if (ctx.data.kind === 'sync') result = await runSyncJob(ctx);
|
|
530
|
+
else if (ctx.data.kind === 'converge') result = await runConvergeJob(ctx, abort);
|
|
531
|
+
else result = await runMigrationJob(ctx, abort);
|
|
532
|
+
progress(ctx, 'completed', ctx.runId ? { runId: ctx.runId } : {});
|
|
533
|
+
await flush(ctx);
|
|
534
|
+
kit.logger.debug(`✔ Job ${job?.id} done`, {
|
|
535
|
+
...jobIds(ctx),
|
|
536
|
+
...jobFields(ctx.data),
|
|
537
|
+
durationMs: Date.now() - startedAt,
|
|
538
|
+
});
|
|
539
|
+
return result;
|
|
540
|
+
} catch (error) {
|
|
541
|
+
const requeued = await requeueOnShutdown(ctx, error, token);
|
|
542
|
+
if (requeued) throw requeued;
|
|
543
|
+
// BullMQ only retries when the job was given more than one attempt — the
|
|
544
|
+
// adapter's own jobs never are, so this matters for jobs enqueued some
|
|
545
|
+
// other way. Renaming is how a library that never imports bullmq says
|
|
546
|
+
// "do not retry"; the typed `code` and class are untouched.
|
|
547
|
+
if (!isRetryableError(error) && (job?.opts?.attempts ?? 1) > 1) {
|
|
548
|
+
error.name = UNRECOVERABLE_ERROR_NAME;
|
|
549
|
+
}
|
|
550
|
+
attachJobIds(error, ctx);
|
|
551
|
+
prepareErrorForQueue(error);
|
|
552
|
+
if (ctx.data) {
|
|
553
|
+
progress(ctx, 'failed', {
|
|
554
|
+
code: error instanceof MigronautError ? error.code : 'UNKNOWN',
|
|
555
|
+
...(ctx.runId ? { runId: ctx.runId } : {}),
|
|
556
|
+
});
|
|
557
|
+
}
|
|
558
|
+
write(ctx, (target) =>
|
|
559
|
+
target.log?.(
|
|
560
|
+
redactOutbound(`✖ ${errorText(error)}${ctx.runId ? ` [run ${ctx.runId}]` : ''}`),
|
|
561
|
+
),
|
|
562
|
+
);
|
|
563
|
+
await flush(ctx);
|
|
564
|
+
throw error;
|
|
565
|
+
} finally {
|
|
566
|
+
abort.removeEventListener('abort', onAbort);
|
|
567
|
+
current = undefined;
|
|
568
|
+
}
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
// Three declared parameters, on purpose — see the factory's doc comment.
|
|
572
|
+
async function processor(job, token, signal) {
|
|
573
|
+
const run = chain.then(() => handle(job, token, signal));
|
|
574
|
+
chain = run.catch(() => undefined);
|
|
575
|
+
return run;
|
|
576
|
+
}
|
|
577
|
+
|
|
578
|
+
/**
|
|
579
|
+
* Stop taking the lock: a job waiting for it fails with RunAbortedError (a
|
|
580
|
+
* retryable failure), and a run that has not started its migration is
|
|
581
|
+
* stopped. Irreversible — meant for process shutdown.
|
|
582
|
+
*/
|
|
583
|
+
processor.shutdown = (reason = 'Migration worker shutting down') => {
|
|
584
|
+
if (!shutdownController.signal.aborted) {
|
|
585
|
+
shutdownController.abort(new RunAbortedError(reason, { reason }));
|
|
586
|
+
}
|
|
587
|
+
};
|
|
588
|
+
|
|
589
|
+
/** Shut down, let the job in flight settle, and disconnect a kit this processor created */
|
|
590
|
+
processor.close = async () => {
|
|
591
|
+
processor.shutdown();
|
|
592
|
+
await chain;
|
|
593
|
+
for (const [event, listener] of Object.entries(listeners)) kit.off(event, listener);
|
|
594
|
+
if (ownsKit) await kit.disconnect();
|
|
595
|
+
};
|
|
596
|
+
|
|
597
|
+
Object.defineProperty(processor, 'kit', { value: kit, enumerable: true });
|
|
598
|
+
return processor;
|
|
599
|
+
}
|
|
600
|
+
|
|
601
|
+
module.exports = {
|
|
602
|
+
RETRYABLE_CODES,
|
|
603
|
+
WAITING_ERROR_NAME,
|
|
604
|
+
createMigrationProcessor,
|
|
605
|
+
isTransientForJob,
|
|
606
|
+
isRetryableError,
|
|
607
|
+
resolveProcessorOptions,
|
|
608
|
+
};
|