@alexify/migronaut 2.2.0 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +190 -0
  2. package/README.md +41 -3
  3. package/bullmq.d.ts +484 -8
  4. package/index.d.ts +1264 -9
  5. package/migronaut.schema.json +93 -1
  6. package/package.json +9 -2
  7. package/src/bullmq/background-processor.js +541 -0
  8. package/src/bullmq/index.js +12 -0
  9. package/src/bullmq/jobs.js +254 -7
  10. package/src/bullmq/processor.js +348 -21
  11. package/src/bullmq/producer.js +185 -13
  12. package/src/bullmq/service.js +484 -45
  13. package/src/cli/commands/background.js +500 -0
  14. package/src/cli/commands/create.js +6 -0
  15. package/src/cli/exit-codes.js +6 -0
  16. package/src/cli/index.js +2 -0
  17. package/src/core/audit.js +11 -1
  18. package/src/core/background-audit.js +139 -0
  19. package/src/core/background-drift.js +126 -0
  20. package/src/core/background-dry-run.js +375 -0
  21. package/src/core/background-engine.js +849 -0
  22. package/src/core/background-kit.js +432 -0
  23. package/src/core/background-partition.js +298 -0
  24. package/src/core/background-runner.js +305 -0
  25. package/src/core/background-sandbox.js +701 -0
  26. package/src/core/background-shard.js +542 -0
  27. package/src/core/background-spec.js +597 -0
  28. package/src/core/background-store.js +951 -0
  29. package/src/core/background-throttle.js +269 -0
  30. package/src/core/background-watch-plan.js +164 -0
  31. package/src/core/background-watch-store.js +78 -0
  32. package/src/core/background-watch.js +610 -0
  33. package/src/core/background.js +1127 -0
  34. package/src/core/bson-peer.js +23 -0
  35. package/src/core/changelog.js +32 -0
  36. package/src/core/collections.js +78 -8
  37. package/src/core/config.js +102 -12
  38. package/src/core/converge-plan.js +86 -7
  39. package/src/core/converge.js +88 -0
  40. package/src/core/lock.js +48 -21
  41. package/src/core/migration-logger.js +279 -0
  42. package/src/core/migrator.js +1027 -22
  43. package/src/core/options.js +36 -0
  44. package/src/core/run-recorder.js +6 -1
  45. package/src/core/run.js +26 -12
  46. package/src/core/runner.js +34 -8
  47. package/src/core/server-info.js +9 -2
  48. package/src/core/shard-info.js +76 -0
  49. package/src/core/versioning-spec.js +181 -0
  50. package/src/errors/index.js +88 -0
  51. package/src/index.js +16 -0
  52. package/src/utils/error.js +11 -2
  53. package/src/utils/job-ref.js +44 -0
  54. package/src/utils/loader.js +77 -9
  55. package/src/utils/migration-name.js +33 -1
  56. package/src/utils/redact.js +140 -3
  57. package/src/utils/telemetry.js +110 -0
  58. package/src/utils/template.js +62 -1
  59. package/src/versioning/config.js +155 -0
  60. package/src/versioning/document.js +326 -0
  61. package/src/versioning/index.js +50 -0
  62. package/src/versioning/internal.js +279 -0
  63. package/src/versioning/mongoose.js +151 -0
  64. package/src/versioning/occ.js +318 -0
  65. package/src/versioning/registry.js +187 -0
  66. package/src/versioning/upcaster.js +213 -0
  67. package/versioning.d.ts +666 -0
  68. package/versioning.js +1 -0
@@ -0,0 +1,1127 @@
1
+ const {
2
+ BackgroundConflictError,
3
+ BackgroundFailedError,
4
+ ChecksumMismatchError,
5
+ LockAlreadyHeldError,
6
+ LockLostError,
7
+ MigronautError,
8
+ TransactionsUnsupportedError,
9
+ } = require('../errors/index.js');
10
+ const { documentErrorText, errorText } = require('../utils/error.js');
11
+ const { versionIndexKey } = require('../versioning/document.js');
12
+ const {
13
+ addCounters,
14
+ excludeBadIds,
15
+ idKey,
16
+ matchOf,
17
+ processPartition,
18
+ } = require('./background-engine.js');
19
+ const { idRangePartitioner } = require('./background-partition.js');
20
+ const { createShardPartitioner, withShardKey } = require('./background-shard.js');
21
+ const { TERMINAL, matchHash, transition } = require('./background-spec.js');
22
+ const { MAX_BAD_IDS, STATE_SCHEMA } = require('./background-store.js');
23
+ const { createAdaptive, createThrottle, sleep } = require('./background-throttle.js');
24
+ const { runWithLock } = require('./lock.js');
25
+ const { READ_OPTIONS } = require('./server-info.js');
26
+ const { shardedVersionIndexKey } = require('./versioning-spec.js');
27
+
28
+ /**
29
+ * Background migrations, orchestrated: the coordinator's steps (plan the
30
+ * partitions, wait for the lanes, finalize a pass), a lane's slice over the
31
+ * partitions, and the controls (pause, resume, cancel, retry, repin).
32
+ *
33
+ * Pure orchestration over what the kit injects (`deps`):
34
+ * `{ db, client, store, logger, fields, emit, lockFor(name), load(name),
35
+ * ttlMs, now?, adaptiveCache?, warned? }` — `load(name)` resolves the
36
+ * migration file to `{ spec, fns, checksum }`; `lockFor(name)` is the
37
+ * coordinator's `MigrationLock` (`background:<name>`), held only for one step.
38
+ *
39
+ * Every step is idempotent and starts from what MongoDB says, so any number
40
+ * of coordinators and lanes, in any number of processes, may run at once:
41
+ * the coordinator lock serializes the steps, a compare-and-set on the plan
42
+ * token keeps a stale plan from committing, and the leases cap the lanes.
43
+ */
44
+
45
+ /**
46
+ * How long a collection's index keys are reused (`deps.indexCache`): every
47
+ * slice plans its reads by them, and indexes change on a deploy, not a batch.
48
+ */
49
+ const INDEX_CACHE_MS = 30_000;
50
+
51
+ /** What a lane's control read needs of the state */
52
+ const CONTROL_FIELDS = Object.freeze({ status: 1, generation: 1, 'plan.token': 1 });
53
+
54
+ /** What frequent readers of every state leave out — the history and the document errors */
55
+ const STATE_SUMMARY = Object.freeze({ history: 0, docErrors: 0 });
56
+
57
+ /** What a degraded plan means, for its warning */
58
+ const DEGRADED = {
59
+ 'sample-timeout': 'the sample ran out of time — some ranges stay whole this pass (fewer lanes)',
60
+ ungrouped:
61
+ 'more runs of chunks than maxPartitions — merged blocks span shards and are not capped by ' +
62
+ 'shardConcurrency',
63
+ };
64
+
65
+ /** How long a coordinator waits before it looks at its lanes again, without a driver that wakes it */
66
+ const POLL_MS = 5_000;
67
+
68
+ /** What a coordinator says when a plan, a lock or a deploy is in the way */
69
+ const retrySoon = (deps, reason) => ({
70
+ next: 'wait',
71
+ reason,
72
+ retryAfterMs: Math.max(1, Math.floor(deps.ttlMs / 2)),
73
+ });
74
+
75
+ /** The keys of a collection's indexes — none for one that does not exist yet */
76
+ async function indexKeys(deps, collection) {
77
+ const cache = deps.indexCache;
78
+ const hit = cache?.get(collection);
79
+ const now = Date.now();
80
+ if (hit !== undefined && now - hit.at < INDEX_CACHE_MS) return hit.keys;
81
+ try {
82
+ const indexes = await deps.db.collection(collection).listIndexes(READ_OPTIONS).toArray();
83
+ const keys = [];
84
+ for (const index of indexes) keys.push(index.key);
85
+ cache?.set(collection, { at: now, keys });
86
+ return keys;
87
+ } catch {
88
+ return [];
89
+ }
90
+ }
91
+
92
+ /** Whether a live index key is `wanted` — same fields, same order, same directions or `hashed` */
93
+ function sameIndexKey(live, wanted) {
94
+ const a = Object.entries(live);
95
+ const b = Object.entries(wanted);
96
+ if (a.length !== b.length) return false;
97
+ for (let i = 0; i < a.length; i++) {
98
+ if (a[i][0] !== b[i][0]) return false;
99
+ const [x, y] = [a[i][1], b[i][1]];
100
+ if (x === 'hashed' || y === 'hashed' ? x !== y : Number(x) !== Number(y)) return false;
101
+ }
102
+ return true;
103
+ }
104
+
105
+ const hasIndex = (keys, wanted) => keys.some((key) => sameIndexKey(key, wanted));
106
+
107
+ /**
108
+ * What a drift probe hints: the version index — or, on a sharded collection,
109
+ * the version index that carries the shard key (any index led by the version
110
+ * field ascending serves an equality probe on it).
111
+ */
112
+ async function probeHint(deps, spec) {
113
+ const keys = await indexKeys(deps, spec.collection);
114
+ const exact = versionIndexKey(spec);
115
+ if (hasIndex(keys, exact)) return exact;
116
+ for (const key of keys) {
117
+ const [first] = Object.entries(key);
118
+ if (first !== undefined && first[0] === spec.field && Number(first[1]) === 1) return key;
119
+ }
120
+ return versionHint(deps, spec, keys);
121
+ }
122
+
123
+ /** Say something once per process — `deps.warned` remembers */
124
+ function warnOnce(deps, id, message, fields) {
125
+ if (deps.warned?.has(id)) return;
126
+ deps.warned?.add(id);
127
+ deps.logger.warn(message, deps.fields(fields));
128
+ }
129
+
130
+ /** The version index of a collection, when it has one — every scan hints it */
131
+ async function versionHint(deps, spec, keys) {
132
+ if (spec.mode === 'step') return undefined;
133
+ const key = versionIndexKey(spec);
134
+ if (hasIndex(keys ?? (await indexKeys(deps, spec.collection)), key)) return key;
135
+ warnOnce(
136
+ deps,
137
+ `index:${spec.collection}`,
138
+ `⚠ ${spec.collection} has no { ${spec.field}: 1, _id: 1 } index — background migrations ` +
139
+ 'over it scan the collection (declare versioning in its definition and converge)',
140
+ { collection: spec.collection },
141
+ );
142
+ return undefined;
143
+ }
144
+
145
+ /**
146
+ * A transactional background migration needs a replica set or a mongos —
147
+ * checked once per process (`deps.topology()` caches the answer).
148
+ *
149
+ * @throws {TransactionsUnsupportedError} on a standalone server
150
+ */
151
+ async function assertTransactions(deps, name, spec) {
152
+ if (!spec.transaction || typeof deps.topology !== 'function') return;
153
+ if ((await deps.topology()) === 'standalone') {
154
+ throw new TransactionsUnsupportedError(
155
+ `Background migration ${name} asks for transactions, which need a replica set or a ` +
156
+ 'mongos — this server is standalone (run it with transaction: false)',
157
+ { migration: name, background: true },
158
+ );
159
+ }
160
+ }
161
+
162
+ /** The job a lane or a coordinator works with: the spec, the functions, and what they scan */
163
+ async function jobFor(deps, name, state, { direction } = {}) {
164
+ // The state's spec wins below: a definition file broken since the
165
+ // registration must not stop its lanes (`tolerant`).
166
+ const loaded = await deps.load(name, { tolerant: true });
167
+ if (state.checksum !== undefined && loaded.checksum !== state.checksum) {
168
+ throw new ChecksumMismatchError(
169
+ `Background migration ${name} changed on disk since it was registered — finish the ` +
170
+ 'deploy, or pin the new version (migronaut background repin)',
171
+ { migration: name, background: true, expected: state.checksum, actual: loaded.checksum },
172
+ );
173
+ }
174
+ const spec = { ...loaded.spec, ...(state.spec ?? {}) };
175
+ const dir = direction ?? state.direction ?? 'forward';
176
+ const match = spec.mode === 'step' ? undefined : excludeBadIds(matchOf(spec, dir), state.badIds);
177
+ return {
178
+ name,
179
+ spec,
180
+ fns: loaded.fns,
181
+ direction: dir === 'revert' ? 'revert' : 'forward',
182
+ match,
183
+ ...(await partitionerFor(deps, spec, dir)),
184
+ logger: deps.logger,
185
+ // For ctx.background and ctx.logger: the lane working it and its queue job
186
+ // (with its group when a run drives it inline).
187
+ ...(deps.logs ? { logs: deps.logs } : {}),
188
+ ...(deps.runId !== undefined ? { runId: deps.runId } : {}),
189
+ ...(deps.job?.id !== undefined ? { jobId: deps.job.id } : {}),
190
+ ...(deps.job?.groupId !== undefined ? { groupId: deps.job.groupId } : {}),
191
+ };
192
+ }
193
+
194
+ /**
195
+ * How a background migration's collection is split: along the shard key on
196
+ * a sharded collection whose key can be read and whose version index carries
197
+ * it (`sharding.mode` `chunks` or `sampled`, decided at plan time), by `_id`
198
+ * everywhere else — `untargeted` when the collection is sharded but that is
199
+ * all that is known, said once per process. `backgroundShardAware: 'off'`
200
+ * keeps every collection on `_id`.
201
+ */
202
+ async function partitionerFor(deps, spec, direction) {
203
+ const keys = spec.mode === 'step' ? [] : await indexKeys(deps, spec.collection);
204
+ const byId = async (mode) => ({
205
+ partitioner: idRangePartitioner,
206
+ hint: await versionHint(deps, spec, keys),
207
+ sharding: { mode },
208
+ });
209
+ if (spec.mode === 'step' || deps.shardAware === 'off' || deps.shardKeyOf === undefined) {
210
+ return byId('off');
211
+ }
212
+ if ((await deps.topology?.()) !== 'sharded') return byId('off');
213
+ const sharding = await deps.shardKeyOf(spec.collection);
214
+ if (sharding === null) return byId('off');
215
+ const where = { background: spec.collection, collection: spec.collection };
216
+ if (sharding === undefined) {
217
+ warnOnce(
218
+ deps,
219
+ `shard:${spec.collection}:privileges`,
220
+ `⚠ ${spec.collection}: its shard key cannot be read (config needs clusterMonitor) — ` +
221
+ 'background migrations over it are not shard-aware',
222
+ where,
223
+ );
224
+ return byId('untargeted');
225
+ }
226
+ const indexKey = shardedVersionIndexKey(spec, sharding.key);
227
+ if (!hasIndex(keys, indexKey)) {
228
+ // The key is known: writes are still targeted, and the guard holds.
229
+ const untargeted = await byId('untargeted');
230
+ warnOnce(
231
+ deps,
232
+ `shard:${spec.collection}:index`,
233
+ `⚠ ${spec.collection} is sharded but has no ${JSON.stringify(indexKey)} index — ` +
234
+ 'background migrations over it are not shard-aware (declare versioning in its ' +
235
+ 'definition and converge)',
236
+ where,
237
+ );
238
+ return {
239
+ ...untargeted,
240
+ partitioner: withShardKey(idRangePartitioner, sharding.key),
241
+ sharding: { mode: 'untargeted', key: sharding.key },
242
+ };
243
+ }
244
+ return {
245
+ partitioner: createShardPartitioner({
246
+ key: sharding.key,
247
+ field: spec.field,
248
+ source: direction === 'revert' ? spec.to : spec.from,
249
+ readChunks: () => deps.chunksOf(spec.collection, sharding),
250
+ epoch: sharding,
251
+ }),
252
+ hint: indexKey,
253
+ sharding: { mode: 'shard', key: sharding.key },
254
+ };
255
+ }
256
+
257
+ /** The shard-aware facts a plan records, for status: how it split, by which key, into how many groups */
258
+ function shardingOf(job, plan) {
259
+ if (job.sharding === undefined || job.sharding.mode === 'off') return undefined;
260
+ if (job.sharding.mode === 'untargeted') {
261
+ return { mode: 'untargeted', ...(job.sharding.key ? { key: job.sharding.key } : {}) };
262
+ }
263
+ const groups = new Set();
264
+ for (const partition of plan.partitions) {
265
+ if (partition.group !== undefined) groups.add(partition.group);
266
+ }
267
+ return { mode: plan.method, key: job.sharding.key, groups: groups.size };
268
+ }
269
+
270
+ // ─── The coordinator ──────────────────────────────────────────────────────────
271
+
272
+ /**
273
+ * One coordinator step, under the coordinator lock: whatever the state
274
+ * needs next. Resolves to `{ next, … }`:
275
+ * - `process` — lanes have work (`lanes`: how many could start now);
276
+ * - `wait` — nothing to do yet (`retryAfterMs`): leases draining, a deploy
277
+ * in progress, a lost commit race;
278
+ * - `done` — terminal, paused, or blocked (`status`);
279
+ * - `busy` — another coordinator holds the lock;
280
+ * - `superseded` — a newer BullMQ coordinator round took over.
281
+ *
282
+ * `driver`: `{ kind, ref?, round? }` — who runs this coordinator.
283
+ */
284
+ async function coordinate(deps, name, { signal, driver = { kind: 'local' } } = {}) {
285
+ const lock = deps.lockFor(name);
286
+ try {
287
+ return await runWithLock(lock, { logger: deps.logger, owner: deps.owner?.() }, (lockSignal) =>
288
+ // The span opens once the lock is held: a busy coordinator makes none.
289
+ withSpan(deps, 'coordinate', name, () =>
290
+ coordinateStep(deps, name, { signal: anySignal(signal, lockSignal), driver }),
291
+ ),
292
+ );
293
+ } catch (error) {
294
+ if (error instanceof LockAlreadyHeldError) {
295
+ return { next: 'busy', retryAfterMs: Math.max(1, Math.floor(lock.ttlMs / 2)) };
296
+ }
297
+ throw error;
298
+ }
299
+ }
300
+
301
+ /** `fn` inside the kit's span for `kind` (`slice`, `coordinate`), when it gives one */
302
+ function withSpan(deps, kind, name, fn) {
303
+ return typeof deps.span === 'function' ? deps.span(kind, name, fn) : fn();
304
+ }
305
+
306
+ /** A signal aborted when either is */
307
+ function anySignal(a, b) {
308
+ if (!a) return b;
309
+ if (!b) return a;
310
+ return AbortSignal.any([a, b]);
311
+ }
312
+
313
+ async function coordinateStep(deps, name, { signal, driver }) {
314
+ const { store } = deps;
315
+ let state = await store.get(name);
316
+ if (state === null) return { next: 'done', status: 'unregistered' };
317
+ if (state.schema > STATE_SCHEMA) {
318
+ warnOnce(
319
+ deps,
320
+ `schema:${name}`,
321
+ `⚠ ${name} was registered by a newer release (state schema ${state.schema}, this one ` +
322
+ `knows ${STATE_SCHEMA}) — waiting for a process of that release to drive it`,
323
+ { background: name },
324
+ );
325
+ return retrySoon(deps, 'newer-schema');
326
+ }
327
+ if (TERMINAL.has(state.status) || state.status === 'paused') {
328
+ return { next: 'done', status: state.status };
329
+ }
330
+ if (state.status === 'blocked') {
331
+ const unblocked = await tryUnblock(deps, state);
332
+ if (!unblocked) return { next: 'done', status: 'blocked', waitsFor: state.waitsFor };
333
+ state = unblocked;
334
+ }
335
+
336
+ // Who drives it — recorded for status. A BullMQ chain of coordinator jobs
337
+ // also carries a round, which only this step hands out (under the
338
+ // coordinator lock): a new chain gets the next round; a chain whose round
339
+ // is not the latest bows out — a newer one took over (two chains started
340
+ // at once each get their own), or it was registered again since. A round
341
+ // the kit never handed out (forged in Redis) is never written.
342
+ let round;
343
+ if (driver.kind === 'bullmq') {
344
+ const current = state.round;
345
+ if (driver.round === undefined) round = (current ?? 0) + 1;
346
+ else if (driver.round !== current) return { next: 'superseded' };
347
+ else round = current;
348
+ }
349
+ await store.set(name, {
350
+ coordinator: {
351
+ kind: driver.kind,
352
+ ...(driver.ref !== undefined ? { ref: driver.ref } : {}),
353
+ at: new Date(),
354
+ },
355
+ ...(round !== undefined ? { round } : {}),
356
+ });
357
+ const answer = await coordinatePass(deps, name, state, { signal });
358
+ return round !== undefined ? { ...answer, round } : answer;
359
+ }
360
+
361
+ /** The rest of a coordinator step, once its driver is recorded */
362
+ async function coordinatePass(deps, name, initial, { signal }) {
363
+ const { store } = deps;
364
+ let state = initial;
365
+ let job;
366
+ try {
367
+ job = await jobFor(deps, name, state);
368
+ } catch (error) {
369
+ if (error instanceof ChecksumMismatchError || error?.code === 'MIGRATION_FILE_NOT_FOUND') {
370
+ // Mid-deploy: an older or newer pod holds the other version of the file.
371
+ deps.logger.warn(`⚠ ${errorText(error)}`, deps.fields({ background: name }));
372
+ return retrySoon(deps, 'checksum');
373
+ }
374
+ throw error;
375
+ }
376
+ try {
377
+ await assertTransactions(deps, name, job.spec);
378
+ } catch (error) {
379
+ // Not something a retry fixes: the deployment cannot do it. Anything
380
+ // else (the server could not be asked) is the step's to retry.
381
+ if (!(error instanceof TransactionsUnsupportedError)) throw error;
382
+ return failState(deps, state, error.message);
383
+ }
384
+ const hash = matchHash(job.spec, job.direction === 'revert' ? 'revert' : 'forward');
385
+
386
+ for (let guard = 0; guard < job.spec.maxPasses + 2; guard++) {
387
+ if (signal?.aborted) return retrySoon(deps, 'stopping');
388
+ // Resharded (or its key refined) since this plan: its ranges mean nothing
389
+ // now — re-split the same pass (safe: the version filter skips what is done).
390
+ if (state.phase === 'process' && job.partitioner.stale?.(state.plan?.epoch)) {
391
+ deps.logger.warn(
392
+ `⚠ ${name}: ${job.spec.collection} was resharded since its plan — replanning`,
393
+ deps.fields({ background: name }),
394
+ );
395
+ state =
396
+ (await store.cas(
397
+ name,
398
+ { phase: 'process', 'plan.token': state.plan.token },
399
+ { $set: { phase: 'replan' } },
400
+ )) ?? (await store.get(name));
401
+ }
402
+ // A replan re-splits the same pass; it is not a new one.
403
+ const replanning = state.phase === 'replan';
404
+ if (replanning && state.plan !== undefined) {
405
+ await store.supersede(name, { generation: state.generation, plan: state.plan.token });
406
+ const leases = await store.leases(name);
407
+ if (leases.live > 0) return retrySoon(deps, 'replan-draining');
408
+ state =
409
+ (await store.cas(name, { phase: 'replan' }, { $set: { phase: 'partition' } })) ?? state;
410
+ }
411
+ if (state.phase !== 'process' || state.plan === undefined || state.plan.match !== hash) {
412
+ const planned = await planPass(deps, job, state, hash, { newPass: !replanning });
413
+ if (planned.next) return planned;
414
+ state = planned.state;
415
+ if (planned.empty) {
416
+ const finalized = await finalize(deps, job, state);
417
+ if (finalized.next) return finalized;
418
+ state = finalized.state;
419
+ continue;
420
+ }
421
+ }
422
+ if (state.status === 'pending') {
423
+ state =
424
+ (await store.move(name, { from: ['pending'], to: 'running', action: 'plan' })) ?? state;
425
+ }
426
+ const reaped = await store.reap(name);
427
+ if (reaped > 0) deps.telemetry?.backgroundLeasesReclaimed({ name, count: reaped });
428
+ const counts = await store.partitionCounts(name, {
429
+ generation: state.generation,
430
+ plan: state.plan.token,
431
+ });
432
+ const open = counts.pending + counts.running;
433
+ if (open > 0) {
434
+ const claimable = counts.pending + (counts.running - counts.leased);
435
+ return {
436
+ next: 'process',
437
+ lanes: Math.max(0, Math.min(claimable, job.spec.maxParallel - counts.leased)),
438
+ generation: state.generation,
439
+ registration: state.registration,
440
+ counts,
441
+ retryAfterMs: POLL_MS,
442
+ };
443
+ }
444
+ const finalized = await finalize(deps, job, state);
445
+ if (finalized.next) return finalized;
446
+ state = finalized.state;
447
+ }
448
+ return retrySoon(deps, 'passes');
449
+ }
450
+
451
+ /**
452
+ * Plan the next pass: partitions over what still matches, committed under a
453
+ * fresh plan token. Resolves to `{ state }` (with `empty` when nothing is
454
+ * left to plan), or a coordinator answer (`{ next }`).
455
+ */
456
+ async function planPass(deps, job, state, hash, { newPass = true } = {}) {
457
+ const { store } = deps;
458
+ const spec = job.spec;
459
+ let plan;
460
+ if (spec.mode === 'step') {
461
+ plan = { method: 'step', estimate: 0, partitions: [{ scope: { kind: 'step' } }] };
462
+ } else {
463
+ plan = await job.partitioner.plan({
464
+ collection: deps.db.collection(spec.collection),
465
+ match: job.match,
466
+ hint: job.hint,
467
+ maxParallel: spec.maxParallel,
468
+ settings: spec.partitions,
469
+ shardConcurrency: spec.shardConcurrency,
470
+ });
471
+ }
472
+ const pass = (state.pass ?? 0) + (newPass ? 1 : 0);
473
+ const sharding = spec.mode === 'step' ? undefined : shardingOf(job, plan);
474
+ const committed = await store.commitPlan(state._id, {
475
+ generation: state.generation,
476
+ previousToken: state.plan?.token,
477
+ // A repin while this pass was planned changed the spec under it: the
478
+ // commit loses, and the next step plans with the new one.
479
+ filter: {
480
+ status: { $in: ['pending', 'running'] },
481
+ ...(state.checksum !== undefined ? { checksum: state.checksum } : {}),
482
+ },
483
+ plan: {
484
+ partitioner: spec.mode === 'step' ? 'step' : job.partitioner.id,
485
+ method: plan.method,
486
+ estimate: plan.estimate,
487
+ ...(plan.atLeast ? { atLeast: true } : {}),
488
+ match: hash,
489
+ ...(plan.degraded ? { degraded: plan.degraded } : {}),
490
+ ...(plan.epoch ? { epoch: plan.epoch } : {}),
491
+ ...(sharding !== undefined ? { sharding } : {}),
492
+ },
493
+ partitions: plan.partitions,
494
+ fields: {
495
+ pass,
496
+ ...(state.startedAt === undefined ? { startedAt: new Date() } : {}),
497
+ lastProgressAt: new Date(),
498
+ },
499
+ });
500
+ if (committed === null) return retrySoon(deps, 'plan-race');
501
+ // A coordinator that lost a commit race earlier left partitions nobody can claim.
502
+ await store.dropForeignPlans(state._id, committed.plan.token);
503
+ deps.emit('background:partitioned', {
504
+ migration: state._id,
505
+ generation: committed.generation,
506
+ pass,
507
+ partitions: plan.partitions.length,
508
+ estimate: plan.estimate,
509
+ ...(plan.atLeast ? { atLeast: true } : {}),
510
+ method: plan.method,
511
+ ...(plan.degraded ? { degraded: plan.degraded } : {}),
512
+ });
513
+ if (plan.degraded) {
514
+ deps.logger.warn(
515
+ `⚠ ${state._id}: ${DEGRADED[plan.degraded] ?? plan.degraded}`,
516
+ deps.fields({ background: state._id, degraded: plan.degraded }),
517
+ );
518
+ }
519
+ return { state: committed, empty: plan.partitions.length === 0 };
520
+ }
521
+
522
+ /**
523
+ * Close a pass: roll the partitions' counters into the state (once per
524
+ * generation — `rolledGeneration` makes it idempotent after a crash), then
525
+ * decide. A failed partition fails the background migration; nothing left
526
+ * to rewrite completes it; anything left starts another pass — unless
527
+ * `maxPasses` passes have not drained it, which means something keeps
528
+ * writing the old shape.
529
+ */
530
+ async function finalize(deps, job, state) {
531
+ const { store } = deps;
532
+ const name = state._id;
533
+ const generation = state.generation;
534
+ if ((state.rolledGeneration ?? 0) < generation) {
535
+ const rolled = await store.generationTotals(name, generation);
536
+ const inc = {};
537
+ for (const [key, value] of Object.entries(rolled?.totals ?? {})) {
538
+ if (typeof value === 'number' && key !== 'failedPartitions' && value !== 0) {
539
+ inc[`totals.${key}`] = value;
540
+ }
541
+ }
542
+ // One id, one entry — by its canonical form, so a string _id never
543
+ // stands in for an ObjectId with the same hex.
544
+ const badIds = [];
545
+ const seen = new Set();
546
+ for (const list of [state.badIds ?? [], rolled?.badIds ?? []]) {
547
+ for (const id of list) {
548
+ const key = idKey(id);
549
+ if (!seen.has(key)) {
550
+ seen.add(key);
551
+ badIds.push(id);
552
+ }
553
+ }
554
+ }
555
+ const updated = await store.cas(
556
+ name,
557
+ { rolledGeneration: { $lt: generation } },
558
+ {
559
+ ...(Object.keys(inc).length > 0 ? { $inc: inc } : {}),
560
+ $set: {
561
+ rolledGeneration: generation,
562
+ badIds: badIds.slice(0, MAX_BAD_IDS + 1),
563
+ ...(rolled?.docErrors?.length ? { docErrors: rolled.docErrors } : {}),
564
+ ...(rolled?.totals?.failedPartitions > 0
565
+ ? {
566
+ failedPartitions: rolled.totals.failedPartitions,
567
+ lastError: rolled.totals.lastError,
568
+ }
569
+ : {}),
570
+ updatedAt: new Date(),
571
+ },
572
+ },
573
+ );
574
+ state = updated ?? (await store.get(name));
575
+ }
576
+
577
+ if ((state.failedPartitions ?? 0) > 0) {
578
+ return failState(
579
+ deps,
580
+ state,
581
+ `${state.failedPartitions} partition(s) failed: ${state.lastError}`,
582
+ );
583
+ }
584
+
585
+ let remaining = 0;
586
+ if (job.spec.mode !== 'step') {
587
+ remaining = await deps.db
588
+ .collection(job.spec.collection)
589
+ .countDocuments(excludeBadIds(matchOf(job.spec, job.direction), state.badIds), {
590
+ limit: 1,
591
+ ...(job.hint ? { hint: job.hint } : {}),
592
+ ...READ_OPTIONS,
593
+ });
594
+ }
595
+ if (remaining === 0) {
596
+ const completed = await store.move(name, {
597
+ from: ['running', 'pending'],
598
+ to: 'completed',
599
+ action: 'complete',
600
+ fields: { completedAt: new Date(), phase: 'process' },
601
+ });
602
+ await store.dropGenerations(name, generation - 1);
603
+ if (completed !== null) {
604
+ if ((state.badIds ?? []).length > 0) {
605
+ deps.logger.warn(
606
+ `⚠ ${name} completed with ${state.badIds.length} document(s) it could not migrate ` +
607
+ '(within maxDocumentErrors) — they are still in the old shape',
608
+ deps.fields({ background: name, failedDocuments: state.badIds.length }),
609
+ );
610
+ }
611
+ deps.emit('background:completed', {
612
+ migration: name,
613
+ totals: completed.totals,
614
+ passes: completed.pass,
615
+ });
616
+ deps.logger.info(
617
+ `✔ Background migration ${name} completed (${completed.totals?.migrated ?? 0} migrated, ` +
618
+ `${completed.pass} pass(es))`,
619
+ deps.fields({ background: name }),
620
+ );
621
+ // A completed revert unblocks nothing: what required it needs the forward shape.
622
+ if (job.direction !== 'revert') await deps.onCompleted?.(name);
623
+ }
624
+ return { next: 'done', status: completed?.status ?? (await store.get(name))?.status };
625
+ }
626
+ if ((state.pass ?? 0) >= job.spec.maxPasses) {
627
+ return failState(
628
+ deps,
629
+ state,
630
+ `old-shape documents keep appearing after ${state.pass} pass(es) — is an old version ` +
631
+ 'of the application still writing them?',
632
+ );
633
+ }
634
+ deps.emit('background:pass', {
635
+ migration: name,
636
+ pass: state.pass,
637
+ generation,
638
+ totals: state.totals,
639
+ });
640
+ const next = await store.cas(
641
+ name,
642
+ { generation, phase: 'process' },
643
+ { $set: { phase: 'partition', updatedAt: new Date() } },
644
+ );
645
+ await store.dropGenerations(name, generation - 1);
646
+ return { state: next ?? (await store.get(name)) };
647
+ }
648
+
649
+ /** Move a state to `failed` with `message`, and say so */
650
+ async function failState(deps, state, reason) {
651
+ // Kept, emitted and logged: the application's data stays out of it.
652
+ const message = documentErrorText(reason);
653
+ const failed = await deps.store.move(state._id, {
654
+ from: ['running', 'pending', 'blocked'],
655
+ to: 'failed',
656
+ action: 'fail',
657
+ fields: { lastError: message, failedAt: new Date() },
658
+ });
659
+ if (failed !== null) {
660
+ deps.emit('background:failed', { migration: state._id, error: message });
661
+ deps.logger.error(
662
+ `✖ Background migration ${state._id} failed: ${message}`,
663
+ deps.fields({ background: state._id }),
664
+ );
665
+ }
666
+ return { next: 'done', status: 'failed', error: message };
667
+ }
668
+
669
+ /**
670
+ * Where each of `requires` stands, in order — the one rule every guard
671
+ * shares: `[{ migration, status, done, state? }]`. A background migration is
672
+ * done when its state completed forward, or — with no state at all — when
673
+ * the changelog has it from a history that predates migronaut (a baseline,
674
+ * an import: `deps.adoptedOf(names)`). A completed revert is not done: what
675
+ * required it needs the forward shape.
676
+ */
677
+ async function requiresStatus(deps, requires) {
678
+ if (requires.length === 0) return [];
679
+ const states = await deps.store.getMany(requires);
680
+ const missing = [];
681
+ for (const name of requires) if (!states.has(name)) missing.push(name);
682
+ const adopted = new Set(missing.length > 0 ? ((await deps.adoptedOf?.(missing)) ?? []) : []);
683
+ const rows = [];
684
+ for (const migration of requires) {
685
+ const state = states.get(migration);
686
+ if (state === undefined) {
687
+ const done = adopted.has(migration);
688
+ rows.push({ migration, status: done ? 'adopted' : 'unregistered', done });
689
+ } else if (state.direction === 'revert') {
690
+ rows.push({ migration, status: 'reverted', done: false, state });
691
+ } else {
692
+ rows.push({ migration, status: state.status, done: state.status === 'completed', state });
693
+ }
694
+ }
695
+ return rows;
696
+ }
697
+
698
+ /** The names among `requires` not done yet, in order */
699
+ async function waitingFor(deps, requires) {
700
+ const waitsFor = [];
701
+ for (const row of await requiresStatus(deps, requires)) {
702
+ if (!row.done) waitsFor.push(row.migration);
703
+ }
704
+ return waitsFor;
705
+ }
706
+
707
+ /**
708
+ * A blocked one whose `requires` are all done moves to `pending`. Returns
709
+ * the state after, or `null`.
710
+ */
711
+ async function tryUnblock(deps, state) {
712
+ const waitsFor = await waitingFor(deps, state.requires ?? []);
713
+ if (waitsFor.length > 0) {
714
+ if (JSON.stringify(waitsFor) !== JSON.stringify(state.waitsFor ?? [])) {
715
+ await deps.store.set(state._id, { waitsFor });
716
+ }
717
+ return null;
718
+ }
719
+ const unblocked = await deps.store.move(state._id, {
720
+ from: ['blocked'],
721
+ to: 'pending',
722
+ action: 'unblock',
723
+ fields: { waitsFor: [] },
724
+ });
725
+ if (unblocked !== null) deps.emit('background:unblocked', { migration: state._id });
726
+ return unblocked;
727
+ }
728
+
729
+ // ─── A lane's slice ───────────────────────────────────────────────────────────
730
+
731
+ /**
732
+ * One slice of a lane: claim a partition (and its slot), work it until the
733
+ * slice ends, release — then claim another while time is left. Resolves to
734
+ * `{ outcome, counters, … }`: `yielded` (time is up, work is left),
735
+ * `exhausted` (nothing left to claim), `busy` (every slot is held —
736
+ * `retryAfterMs`), `stale` (no current plan: the coordinator must run),
737
+ * `paused`, `cancelled`, `failed`, `stopped` (the signal) or `lost` (the
738
+ * lease was taken). A slice that fails is counted on its partition and
739
+ * thrown.
740
+ */
741
+ async function runSlice(deps, name, { signal, sliceMs, owner } = {}) {
742
+ const { store } = deps;
743
+ const state = await store.get(name);
744
+ if (state === null) return { outcome: 'cancelled', counters: {} };
745
+ if (state.status === 'paused' || state.status === 'cancelled' || state.status === 'failed') {
746
+ return { outcome: state.status, counters: {} };
747
+ }
748
+ if (state.status !== 'running' || state.phase !== 'process' || state.plan === undefined) {
749
+ return { outcome: state.status === 'completed' ? 'exhausted' : 'stale', counters: {} };
750
+ }
751
+ // A newer release's state (an old pod mid-deploy): its plan may hold scopes
752
+ // this one cannot read. The coordinator waits; so does the lane.
753
+ if (state.schema > STATE_SCHEMA) return { outcome: 'stale', counters: {} };
754
+ let job;
755
+ try {
756
+ job = await jobFor(deps, name, state);
757
+ // A plan made before a reshard: the coordinator must re-split first.
758
+ if (job.partitioner.stale?.(state.plan.epoch)) return { outcome: 'stale', counters: {} };
759
+ await assertTransactions(deps, name, job.spec);
760
+ } catch (error) {
761
+ // Before any claim (the file changed or is gone, the deployment cannot
762
+ // run it): no partition to count it on and no slice:start to pair an
763
+ // event with — but measured like any failed slice; the drivers log it.
764
+ deps.telemetry?.backgroundSliceEnded({ name, durationMs: 0, outcome: 'error', error });
765
+ throw error;
766
+ }
767
+ job.generation = state.generation;
768
+ const now = deps.now ?? Date.now;
769
+ const deadline = now() + (sliceMs ?? job.spec.sliceMs);
770
+ const counters = {};
771
+ const throttle = createThrottle({
772
+ spec: { ...job.spec, throttle: job.fns.throttle },
773
+ db: deps.db,
774
+ logger: deps.logger,
775
+ name,
776
+ warned: deps.warned,
777
+ });
778
+ let worked = false;
779
+ for (;;) {
780
+ if (signal?.aborted) return { outcome: 'stopped', counters };
781
+ // The first claim is always made: a slice shorter than its setup still works a batch.
782
+ if (worked && now() >= deadline) return { outcome: 'yielded', counters };
783
+ const claimed = await store.claim(name, {
784
+ generation: state.generation,
785
+ plan: state.plan.token,
786
+ maxParallel: job.spec.maxParallel,
787
+ ttlMs: deps.ttlMs,
788
+ owner,
789
+ // Lanes per shard, on partitions that belong to one.
790
+ ...(job.partitioner.id === 'shard' ? { shardConcurrency: job.spec.shardConcurrency } : {}),
791
+ onReaped: (count) => deps.telemetry?.backgroundLeasesReclaimed({ name, count }),
792
+ });
793
+ if (claimed.exhausted) return { outcome: 'exhausted', counters };
794
+ if (claimed.busy) {
795
+ return worked
796
+ ? { outcome: 'yielded', counters }
797
+ : { outcome: 'busy', counters, retryAfterMs: claimed.retryAfterMs };
798
+ }
799
+ worked = true;
800
+ const { partition, lease } = claimed;
801
+ job.partitionId = partition._id;
802
+ const adaptive =
803
+ job.spec.adaptive === false
804
+ ? undefined
805
+ : adaptiveFor(deps, name, partition, job.spec.adaptive);
806
+ deps.emit('background:slice:start', {
807
+ migration: name,
808
+ generation: state.generation,
809
+ partition: String(partition._id),
810
+ slot: partition.lease.slot,
811
+ });
812
+ const runLease = () =>
813
+ runWithLock(lease, { logger: deps.logger, owner }, (leaseSignal) => {
814
+ job.signal = anySignal(signal, leaseSignal);
815
+ return processPartition(job, {
816
+ store,
817
+ lease,
818
+ partition,
819
+ db: deps.db,
820
+ client: deps.client,
821
+ signal: job.signal,
822
+ deadline,
823
+ throttle,
824
+ adaptive,
825
+ now,
826
+ // Every batch of every lane asks: only what it decides on is read.
827
+ readControl: () =>
828
+ store
829
+ .get(name, { projection: CONTROL_FIELDS })
830
+ .then((current) =>
831
+ current === null
832
+ ? null
833
+ : { status: current.status, generation: current.generation, plan: current.plan },
834
+ ),
835
+ onBatch: (info) => {
836
+ deps.telemetry?.backgroundBatchWritten({
837
+ name,
838
+ durationMs: info.latencyMs,
839
+ shard: partition.group,
840
+ });
841
+ deps.emit('background:batch', {
842
+ migration: name,
843
+ partition: String(partition._id),
844
+ ...(partition.group !== undefined ? { group: partition.group } : {}),
845
+ ...info,
846
+ });
847
+ },
848
+ onTransactionRetries: (count) =>
849
+ deps.telemetry?.backgroundTransactionRetried({ name, reason: 'transient', count }),
850
+ onThrottle: (change) => {
851
+ deps.telemetry?.backgroundThrottled({ name, reason: change.reason });
852
+ deps.emit('background:throttle', {
853
+ migration: name,
854
+ ...(partition.group !== undefined ? { group: partition.group } : {}),
855
+ ...change,
856
+ });
857
+ },
858
+ });
859
+ });
860
+ let result;
861
+ const sliceStarted = now();
862
+ try {
863
+ // One span per lease held — it exists only once the claim succeeded.
864
+ result = await withSpan(deps, 'slice', name, runLease);
865
+ } catch (error) {
866
+ const lost = error instanceof LockLostError;
867
+ // The caller stopped the lane (a shutdown, a deploy): whatever a wait
868
+ // rejected with is the stop, not a fault of the partition — counting it
869
+ // would fail a partition after a few rolling deploys.
870
+ const stopped = !lost && signal?.aborted === true;
871
+ deps.telemetry?.backgroundSliceEnded({
872
+ name,
873
+ durationMs: now() - sliceStarted,
874
+ outcome: lost ? 'lost' : stopped ? 'stopped' : 'error',
875
+ ...(stopped ? {} : { error }),
876
+ });
877
+ if (lost) {
878
+ deps.emit('background:lease:lost', { migration: name, partition: String(partition._id) });
879
+ return { outcome: 'lost', counters };
880
+ }
881
+ if (stopped) {
882
+ deps.emit('background:slice:end', {
883
+ migration: name,
884
+ partition: String(partition._id),
885
+ outcome: 'stopped',
886
+ });
887
+ return { outcome: 'stopped', counters };
888
+ }
889
+ const partitionFailed = await store
890
+ .failSlice(lease, {
891
+ error: documentErrorText(error),
892
+ maxSliceFailures: job.spec.maxSliceFailures,
893
+ })
894
+ .catch((countError) => {
895
+ // The slice's own error is what is thrown; this one is only said.
896
+ deps.logger.debug(
897
+ `Could not count a failed slice of ${name}: ${errorText(countError)}`,
898
+ deps.fields({ background: name }),
899
+ );
900
+ return false;
901
+ });
902
+ deps.emit('background:slice:end', {
903
+ migration: name,
904
+ partition: String(partition._id),
905
+ outcome: 'error',
906
+ error: documentErrorText(error),
907
+ });
908
+ if (error instanceof MigronautError) {
909
+ error.context = { ...error.context, background: name, partitionFailed };
910
+ }
911
+ throw error;
912
+ }
913
+ deps.telemetry?.backgroundSliceEnded({
914
+ name,
915
+ durationMs: now() - sliceStarted,
916
+ outcome: result.outcome,
917
+ counters: result.counters,
918
+ });
919
+ addCounters(counters, result.counters ?? {});
920
+ await store.set(name, { lastProgressAt: new Date() });
921
+ deps.emit('background:slice:end', {
922
+ migration: name,
923
+ partition: String(partition._id),
924
+ outcome: result.outcome,
925
+ counters: result.counters,
926
+ });
927
+ if (result.outcome === 'failed' && result.error) {
928
+ await store
929
+ .failPartition(lease, { error: documentErrorText(result.error) })
930
+ .catch(() => undefined);
931
+ await failState(deps, state, result.error.message);
932
+ return { outcome: 'failed', counters, error: result.error };
933
+ }
934
+ if (result.outcome !== 'exhausted') return { outcome: result.outcome, counters };
935
+ // That partition is done: claim another while the slice lasts.
936
+ }
937
+ }
938
+
939
+ /** The process's AIMD controller for one background migration and group — shared by its lanes */
940
+ function adaptiveFor(deps, name, partition, settings) {
941
+ const cache = deps.adaptiveCache;
942
+ const key = `${name}:${partition.group ?? ''}`;
943
+ if (cache?.has(key)) return cache.get(key);
944
+ const controller = createAdaptive(settings, {
945
+ ...(partition.throttle ? { initial: partition.throttle } : {}),
946
+ });
947
+ cache?.set(key, controller);
948
+ return controller;
949
+ }
950
+
951
+ // ─── Controls ─────────────────────────────────────────────────────────────────
952
+
953
+ /**
954
+ * Apply a control action — `pause`, `resume`, `cancel`, `retry` (with
955
+ * `fromStart`) — to a background migration. Resolves to
956
+ * `{ applied: 'changed' | 'unchanged', status }`.
957
+ *
958
+ * @throws {BackgroundConflictError} when the action does not fit its state
959
+ * @throws {NotAppliedError} via the kit when it is not registered
960
+ */
961
+ async function control(deps, name, action, { requestedBy, reason, fromStart = false } = {}) {
962
+ const { store } = deps;
963
+ const state = await store.get(name);
964
+ if (state === null) {
965
+ throw new BackgroundConflictError(`Background migration ${name} is not registered`, {
966
+ action,
967
+ migration: name,
968
+ status: 'unregistered',
969
+ });
970
+ }
971
+ const effective = action === 'retry' && state.status === 'completed' ? 'reopen' : action;
972
+ const { to, applied } = transition(state.status, effective, { migration: name });
973
+ if (applied === 'unchanged') return { applied, status: state.status };
974
+
975
+ const stamp = {
976
+ action,
977
+ ...(requestedBy !== undefined ? { requestedBy } : {}),
978
+ ...(reason !== undefined ? { reason } : {}),
979
+ at: new Date(),
980
+ };
981
+ let target = to;
982
+ const fields = { control: stamp };
983
+ if (action === 'resume') {
984
+ // Back to where it was: blocked if what it requires is still not done.
985
+ const waitsFor = await waitingFor(deps, state.requires ?? []);
986
+ if (waitsFor.length > 0) {
987
+ target = 'blocked';
988
+ fields.waitsFor = waitsFor;
989
+ } else if (state.plan !== undefined && state.phase === 'process') {
990
+ target = 'running';
991
+ }
992
+ }
993
+ if (effective === 'reopen') {
994
+ fields.phase = 'partition';
995
+ fields.reopened = (state.reopened ?? 0) + 1;
996
+ }
997
+ // A pass that already closed (its counters rolled up) is not resumed: its
998
+ // partitions' work since would never be counted. A new pass takes what is left.
999
+ const closed = state.plan !== undefined && (state.rolledGeneration ?? 0) >= state.generation;
1000
+ if (action === 'retry' && fromStart) {
1001
+ fields.phase = 'partition';
1002
+ fields.pass = 0;
1003
+ fields.badIds = [];
1004
+ fields.failedPartitions = 0;
1005
+ // Diagnostics of the runs before; `totals` stays — it is what was rewritten.
1006
+ fields.docErrors = [];
1007
+ fields.lastError = null;
1008
+ } else if (action === 'retry') {
1009
+ fields.failedPartitions = 0;
1010
+ if (closed) fields.phase = 'partition';
1011
+ }
1012
+ const moved = await store.move(name, {
1013
+ from: [state.status],
1014
+ to: target,
1015
+ action,
1016
+ fields,
1017
+ by: requestedBy,
1018
+ reason,
1019
+ });
1020
+ if (moved === null) {
1021
+ // Someone moved it first: judge again from what it is now.
1022
+ return control(deps, name, action, { requestedBy, reason, fromStart });
1023
+ }
1024
+
1025
+ const plan = state.plan;
1026
+ if (plan !== undefined) {
1027
+ if (action === 'cancel') {
1028
+ await store.setOpenPartitions(name, {
1029
+ generation: state.generation,
1030
+ plan: plan.token,
1031
+ status: 'cancelled',
1032
+ });
1033
+ } else if (action === 'retry' && fromStart) {
1034
+ await store.dropGenerations(name, state.generation);
1035
+ } else if (action === 'retry' && effective !== 'reopen' && !closed) {
1036
+ await store.setOpenPartitions(name, {
1037
+ generation: state.generation,
1038
+ plan: plan.token,
1039
+ from: ['failed', 'cancelled'],
1040
+ status: 'pending',
1041
+ });
1042
+ }
1043
+ }
1044
+ deps.emit('background:control', { migration: name, action, from: state.status, to: target });
1045
+ deps.logger.info(
1046
+ `${name}: ${action} (${state.status} → ${target})`,
1047
+ deps.fields({ background: name, action }),
1048
+ );
1049
+ return { applied: 'changed', status: target };
1050
+ }
1051
+
1052
+ /**
1053
+ * Pin what is on disk now: its checksum and spec become the state's. A
1054
+ * change to what is matched, or how it is split, means a new plan.
1055
+ */
1056
+ async function repin(deps, name, { requestedBy, reason } = {}) {
1057
+ const { store } = deps;
1058
+ const state = await store.get(name);
1059
+ if (state === null) {
1060
+ throw new BackgroundConflictError(`Background migration ${name} is not registered`, {
1061
+ action: 'repin',
1062
+ migration: name,
1063
+ status: 'unregistered',
1064
+ });
1065
+ }
1066
+ const loaded = await deps.load(name);
1067
+ const before = state.spec ?? {};
1068
+ const after = loaded.spec;
1069
+ const dir = state.direction === 'revert' ? 'revert' : 'forward';
1070
+ const replan =
1071
+ matchHash(before, dir) !== matchHash(after, dir) ||
1072
+ before.maxParallel !== after.maxParallel ||
1073
+ JSON.stringify(before.partitions) !== JSON.stringify(after.partitions);
1074
+ const fields = {
1075
+ checksum: loaded.checksum,
1076
+ spec: after,
1077
+ ...(replan && !TERMINAL.has(state.status) && state.plan !== undefined
1078
+ ? { phase: 'replan' }
1079
+ : {}),
1080
+ control: { action: 'repin', requestedBy, reason, at: new Date() },
1081
+ };
1082
+ await store.set(name, fields);
1083
+ deps.emit('background:control', { migration: name, action: 'repin', replan });
1084
+ return { applied: 'changed', status: state.status, replan, checksum: loaded.checksum };
1085
+ }
1086
+
1087
+ /**
1088
+ * Wait until no lane holds a lease any more — after a pause or a cancel,
1089
+ * for a caller that wants "stopped" to mean stopped.
1090
+ */
1091
+ async function waitForLanes(deps, name, { signal, timeoutMs = 120_000, pollMs = 250 } = {}) {
1092
+ const now = deps.now ?? Date.now;
1093
+ const until = now() + timeoutMs;
1094
+ for (;;) {
1095
+ const { live } = await deps.store.leases(name);
1096
+ if (live === 0) return true;
1097
+ if (now() >= until) return false;
1098
+ // One listener per wait, removed with it — and an aborted signal ends it at once.
1099
+ await sleep(pollMs, signal);
1100
+ }
1101
+ }
1102
+
1103
+ /** Why a failed state failed, for a thrown error */
1104
+ function failedError(state) {
1105
+ return new BackgroundFailedError(
1106
+ `Background migration ${state._id} failed${state.lastError ? `: ${state.lastError}` : ''}`,
1107
+ { migration: state._id, lastError: state.lastError },
1108
+ );
1109
+ }
1110
+
1111
+ module.exports = {
1112
+ STATE_SUMMARY,
1113
+ control,
1114
+ coordinate,
1115
+ failedError,
1116
+ finalize,
1117
+ jobFor,
1118
+ partitionerFor,
1119
+ probeHint,
1120
+ repin,
1121
+ requiresStatus,
1122
+ runSlice,
1123
+ tryUnblock,
1124
+ versionHint,
1125
+ waitForLanes,
1126
+ waitingFor,
1127
+ };