@alexify/migronaut 2.2.0 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/CHANGELOG.md +107 -0
  2. package/README.md +33 -2
  3. package/bullmq.d.ts +449 -6
  4. package/index.d.ts +1010 -9
  5. package/migronaut.schema.json +93 -1
  6. package/package.json +8 -2
  7. package/src/bullmq/background-processor.js +469 -0
  8. package/src/bullmq/index.js +12 -0
  9. package/src/bullmq/jobs.js +254 -7
  10. package/src/bullmq/processor.js +128 -14
  11. package/src/bullmq/producer.js +185 -13
  12. package/src/bullmq/service.js +480 -45
  13. package/src/cli/commands/background.js +500 -0
  14. package/src/cli/commands/create.js +6 -0
  15. package/src/cli/exit-codes.js +6 -0
  16. package/src/cli/index.js +2 -0
  17. package/src/core/audit.js +11 -1
  18. package/src/core/background-audit.js +139 -0
  19. package/src/core/background-drift.js +126 -0
  20. package/src/core/background-dry-run.js +366 -0
  21. package/src/core/background-engine.js +818 -0
  22. package/src/core/background-kit.js +425 -0
  23. package/src/core/background-partition.js +298 -0
  24. package/src/core/background-runner.js +305 -0
  25. package/src/core/background-sandbox.js +701 -0
  26. package/src/core/background-shard.js +542 -0
  27. package/src/core/background-spec.js +597 -0
  28. package/src/core/background-store.js +951 -0
  29. package/src/core/background-throttle.js +269 -0
  30. package/src/core/background-watch-plan.js +164 -0
  31. package/src/core/background-watch-store.js +78 -0
  32. package/src/core/background-watch.js +605 -0
  33. package/src/core/background.js +1121 -0
  34. package/src/core/bson-peer.js +23 -0
  35. package/src/core/changelog.js +32 -0
  36. package/src/core/collections.js +78 -8
  37. package/src/core/config.js +102 -12
  38. package/src/core/converge-plan.js +86 -7
  39. package/src/core/converge.js +88 -0
  40. package/src/core/lock.js +48 -21
  41. package/src/core/migrator.js +904 -12
  42. package/src/core/options.js +16 -0
  43. package/src/core/run.js +26 -12
  44. package/src/core/runner.js +1 -1
  45. package/src/core/server-info.js +9 -2
  46. package/src/core/shard-info.js +76 -0
  47. package/src/core/versioning-spec.js +181 -0
  48. package/src/errors/index.js +88 -0
  49. package/src/index.js +16 -0
  50. package/src/utils/error.js +11 -2
  51. package/src/utils/loader.js +77 -9
  52. package/src/utils/migration-name.js +33 -1
  53. package/src/utils/telemetry.js +107 -0
  54. package/src/utils/template.js +62 -1
  55. package/src/versioning/config.js +155 -0
  56. package/src/versioning/document.js +326 -0
  57. package/src/versioning/index.js +50 -0
  58. package/src/versioning/internal.js +279 -0
  59. package/src/versioning/mongoose.js +151 -0
  60. package/src/versioning/occ.js +318 -0
  61. package/src/versioning/registry.js +187 -0
  62. package/src/versioning/upcaster.js +213 -0
  63. package/versioning.d.ts +666 -0
  64. package/versioning.js +1 -0
@@ -0,0 +1,1121 @@
1
+ const {
2
+ BackgroundConflictError,
3
+ BackgroundFailedError,
4
+ ChecksumMismatchError,
5
+ LockAlreadyHeldError,
6
+ LockLostError,
7
+ MigronautError,
8
+ TransactionsUnsupportedError,
9
+ } = require('../errors/index.js');
10
+ const { documentErrorText, errorText } = require('../utils/error.js');
11
+ const { versionIndexKey } = require('../versioning/document.js');
12
+ const {
13
+ addCounters,
14
+ excludeBadIds,
15
+ idKey,
16
+ matchOf,
17
+ processPartition,
18
+ } = require('./background-engine.js');
19
+ const { idRangePartitioner } = require('./background-partition.js');
20
+ const { createShardPartitioner, withShardKey } = require('./background-shard.js');
21
+ const { TERMINAL, matchHash, transition } = require('./background-spec.js');
22
+ const { MAX_BAD_IDS, STATE_SCHEMA } = require('./background-store.js');
23
+ const { createAdaptive, createThrottle, sleep } = require('./background-throttle.js');
24
+ const { runWithLock } = require('./lock.js');
25
+ const { READ_OPTIONS } = require('./server-info.js');
26
+ const { shardedVersionIndexKey } = require('./versioning-spec.js');
27
+
28
+ /**
29
+ * Background migrations, orchestrated: the coordinator's steps (plan the
30
+ * partitions, wait for the lanes, finalize a pass), a lane's slice over the
31
+ * partitions, and the controls (pause, resume, cancel, retry, repin).
32
+ *
33
+ * Pure orchestration over what the kit injects (`deps`):
34
+ * `{ db, client, store, logger, fields, emit, lockFor(name), load(name),
35
+ * ttlMs, now?, adaptiveCache?, warned? }` — `load(name)` resolves the
36
+ * migration file to `{ spec, fns, checksum }`; `lockFor(name)` is the
37
+ * coordinator's `MigrationLock` (`background:<name>`), held only for one step.
38
+ *
39
+ * Every step is idempotent and starts from what MongoDB says, so any number
40
+ * of coordinators and lanes, in any number of processes, may run at once:
41
+ * the coordinator lock serializes the steps, a compare-and-set on the plan
42
+ * token keeps a stale plan from committing, and the leases cap the lanes.
43
+ */
44
+
45
+ /**
46
+ * How long a collection's index keys are reused (`deps.indexCache`): every
47
+ * slice plans its reads by them, and indexes change on a deploy, not a batch.
48
+ */
49
+ const INDEX_CACHE_MS = 30_000;
50
+
51
+ /** What a lane's control read needs of the state */
52
+ const CONTROL_FIELDS = Object.freeze({ status: 1, generation: 1, 'plan.token': 1 });
53
+
54
+ /** What frequent readers of every state leave out — the history and the document errors */
55
+ const STATE_SUMMARY = Object.freeze({ history: 0, docErrors: 0 });
56
+
57
+ /** What a degraded plan means, for its warning */
58
+ const DEGRADED = {
59
+ 'sample-timeout': 'the sample ran out of time — some ranges stay whole this pass (fewer lanes)',
60
+ ungrouped:
61
+ 'more runs of chunks than maxPartitions — merged blocks span shards and are not capped by ' +
62
+ 'shardConcurrency',
63
+ };
64
+
65
+ /** How long a coordinator waits before it looks at its lanes again, without a driver that wakes it */
66
+ const POLL_MS = 5_000;
67
+
68
+ /** What a coordinator says when a plan, a lock or a deploy is in the way */
69
+ const retrySoon = (deps, reason) => ({
70
+ next: 'wait',
71
+ reason,
72
+ retryAfterMs: Math.max(1, Math.floor(deps.ttlMs / 2)),
73
+ });
74
+
75
+ /** The keys of a collection's indexes — none for one that does not exist yet */
76
+ async function indexKeys(deps, collection) {
77
+ const cache = deps.indexCache;
78
+ const hit = cache?.get(collection);
79
+ const now = Date.now();
80
+ if (hit !== undefined && now - hit.at < INDEX_CACHE_MS) return hit.keys;
81
+ try {
82
+ const indexes = await deps.db.collection(collection).listIndexes(READ_OPTIONS).toArray();
83
+ const keys = [];
84
+ for (const index of indexes) keys.push(index.key);
85
+ cache?.set(collection, { at: now, keys });
86
+ return keys;
87
+ } catch {
88
+ return [];
89
+ }
90
+ }
91
+
92
+ /** Whether a live index key is `wanted` — same fields, same order, same directions or `hashed` */
93
+ function sameIndexKey(live, wanted) {
94
+ const a = Object.entries(live);
95
+ const b = Object.entries(wanted);
96
+ if (a.length !== b.length) return false;
97
+ for (let i = 0; i < a.length; i++) {
98
+ if (a[i][0] !== b[i][0]) return false;
99
+ const [x, y] = [a[i][1], b[i][1]];
100
+ if (x === 'hashed' || y === 'hashed' ? x !== y : Number(x) !== Number(y)) return false;
101
+ }
102
+ return true;
103
+ }
104
+
105
+ const hasIndex = (keys, wanted) => keys.some((key) => sameIndexKey(key, wanted));
106
+
107
+ /**
108
+ * What a drift probe hints: the version index — or, on a sharded collection,
109
+ * the version index that carries the shard key (any index led by the version
110
+ * field ascending serves an equality probe on it).
111
+ */
112
+ async function probeHint(deps, spec) {
113
+ const keys = await indexKeys(deps, spec.collection);
114
+ const exact = versionIndexKey(spec);
115
+ if (hasIndex(keys, exact)) return exact;
116
+ for (const key of keys) {
117
+ const [first] = Object.entries(key);
118
+ if (first !== undefined && first[0] === spec.field && Number(first[1]) === 1) return key;
119
+ }
120
+ return versionHint(deps, spec, keys);
121
+ }
122
+
123
+ /** Say something once per process — `deps.warned` remembers */
124
+ function warnOnce(deps, id, message, fields) {
125
+ if (deps.warned?.has(id)) return;
126
+ deps.warned?.add(id);
127
+ deps.logger.warn(message, deps.fields(fields));
128
+ }
129
+
130
+ /** The version index of a collection, when it has one — every scan hints it */
131
+ async function versionHint(deps, spec, keys) {
132
+ if (spec.mode === 'step') return undefined;
133
+ const key = versionIndexKey(spec);
134
+ if (hasIndex(keys ?? (await indexKeys(deps, spec.collection)), key)) return key;
135
+ warnOnce(
136
+ deps,
137
+ `index:${spec.collection}`,
138
+ `⚠ ${spec.collection} has no { ${spec.field}: 1, _id: 1 } index — background migrations ` +
139
+ 'over it scan the collection (declare versioning in its definition and converge)',
140
+ { collection: spec.collection },
141
+ );
142
+ return undefined;
143
+ }
144
+
145
+ /**
146
+ * A transactional background migration needs a replica set or a mongos —
147
+ * checked once per process (`deps.topology()` caches the answer).
148
+ *
149
+ * @throws {TransactionsUnsupportedError} on a standalone server
150
+ */
151
+ async function assertTransactions(deps, name, spec) {
152
+ if (!spec.transaction || typeof deps.topology !== 'function') return;
153
+ if ((await deps.topology()) === 'standalone') {
154
+ throw new TransactionsUnsupportedError(
155
+ `Background migration ${name} asks for transactions, which need a replica set or a ` +
156
+ 'mongos — this server is standalone (run it with transaction: false)',
157
+ { migration: name, background: true },
158
+ );
159
+ }
160
+ }
161
+
162
+ /** The job a lane or a coordinator works with: the spec, the functions, and what they scan */
163
+ async function jobFor(deps, name, state, { direction } = {}) {
164
+ // The state's spec wins below: a definition file broken since the
165
+ // registration must not stop its lanes (`tolerant`).
166
+ const loaded = await deps.load(name, { tolerant: true });
167
+ if (state.checksum !== undefined && loaded.checksum !== state.checksum) {
168
+ throw new ChecksumMismatchError(
169
+ `Background migration ${name} changed on disk since it was registered — finish the ` +
170
+ 'deploy, or pin the new version (migronaut background repin)',
171
+ { migration: name, background: true, expected: state.checksum, actual: loaded.checksum },
172
+ );
173
+ }
174
+ const spec = { ...loaded.spec, ...(state.spec ?? {}) };
175
+ const dir = direction ?? state.direction ?? 'forward';
176
+ const match = spec.mode === 'step' ? undefined : excludeBadIds(matchOf(spec, dir), state.badIds);
177
+ return {
178
+ name,
179
+ spec,
180
+ fns: loaded.fns,
181
+ direction: dir === 'revert' ? 'revert' : 'forward',
182
+ match,
183
+ ...(await partitionerFor(deps, spec, dir)),
184
+ logger: deps.logger,
185
+ };
186
+ }
187
+
188
+ /**
189
+ * How a background migration's collection is split: along the shard key on
190
+ * a sharded collection whose key can be read and whose version index carries
191
+ * it (`sharding.mode` `chunks` or `sampled`, decided at plan time), by `_id`
192
+ * everywhere else — `untargeted` when the collection is sharded but that is
193
+ * all that is known, said once per process. `backgroundShardAware: 'off'`
194
+ * keeps every collection on `_id`.
195
+ */
196
+ async function partitionerFor(deps, spec, direction) {
197
+ const keys = spec.mode === 'step' ? [] : await indexKeys(deps, spec.collection);
198
+ const byId = async (mode) => ({
199
+ partitioner: idRangePartitioner,
200
+ hint: await versionHint(deps, spec, keys),
201
+ sharding: { mode },
202
+ });
203
+ if (spec.mode === 'step' || deps.shardAware === 'off' || deps.shardKeyOf === undefined) {
204
+ return byId('off');
205
+ }
206
+ if ((await deps.topology?.()) !== 'sharded') return byId('off');
207
+ const sharding = await deps.shardKeyOf(spec.collection);
208
+ if (sharding === null) return byId('off');
209
+ const where = { background: spec.collection, collection: spec.collection };
210
+ if (sharding === undefined) {
211
+ warnOnce(
212
+ deps,
213
+ `shard:${spec.collection}:privileges`,
214
+ `⚠ ${spec.collection}: its shard key cannot be read (config needs clusterMonitor) — ` +
215
+ 'background migrations over it are not shard-aware',
216
+ where,
217
+ );
218
+ return byId('untargeted');
219
+ }
220
+ const indexKey = shardedVersionIndexKey(spec, sharding.key);
221
+ if (!hasIndex(keys, indexKey)) {
222
+ // The key is known: writes are still targeted, and the guard holds.
223
+ const untargeted = await byId('untargeted');
224
+ warnOnce(
225
+ deps,
226
+ `shard:${spec.collection}:index`,
227
+ `⚠ ${spec.collection} is sharded but has no ${JSON.stringify(indexKey)} index — ` +
228
+ 'background migrations over it are not shard-aware (declare versioning in its ' +
229
+ 'definition and converge)',
230
+ where,
231
+ );
232
+ return {
233
+ ...untargeted,
234
+ partitioner: withShardKey(idRangePartitioner, sharding.key),
235
+ sharding: { mode: 'untargeted', key: sharding.key },
236
+ };
237
+ }
238
+ return {
239
+ partitioner: createShardPartitioner({
240
+ key: sharding.key,
241
+ field: spec.field,
242
+ source: direction === 'revert' ? spec.to : spec.from,
243
+ readChunks: () => deps.chunksOf(spec.collection, sharding),
244
+ epoch: sharding,
245
+ }),
246
+ hint: indexKey,
247
+ sharding: { mode: 'shard', key: sharding.key },
248
+ };
249
+ }
250
+
251
+ /** The shard-aware facts a plan records, for status: how it split, by which key, into how many groups */
252
+ function shardingOf(job, plan) {
253
+ if (job.sharding === undefined || job.sharding.mode === 'off') return undefined;
254
+ if (job.sharding.mode === 'untargeted') {
255
+ return { mode: 'untargeted', ...(job.sharding.key ? { key: job.sharding.key } : {}) };
256
+ }
257
+ const groups = new Set();
258
+ for (const partition of plan.partitions) {
259
+ if (partition.group !== undefined) groups.add(partition.group);
260
+ }
261
+ return { mode: plan.method, key: job.sharding.key, groups: groups.size };
262
+ }
263
+
264
+ // ─── The coordinator ──────────────────────────────────────────────────────────
265
+
266
+ /**
267
+ * One coordinator step, under the coordinator lock: whatever the state
268
+ * needs next. Resolves to `{ next, … }`:
269
+ * - `process` — lanes have work (`lanes`: how many could start now);
270
+ * - `wait` — nothing to do yet (`retryAfterMs`): leases draining, a deploy
271
+ * in progress, a lost commit race;
272
+ * - `done` — terminal, paused, or blocked (`status`);
273
+ * - `busy` — another coordinator holds the lock;
274
+ * - `superseded` — a newer BullMQ coordinator round took over.
275
+ *
276
+ * `driver`: `{ kind, ref?, round? }` — who runs this coordinator.
277
+ */
278
+ async function coordinate(deps, name, { signal, driver = { kind: 'local' } } = {}) {
279
+ const lock = deps.lockFor(name);
280
+ try {
281
+ return await runWithLock(lock, { logger: deps.logger, owner: deps.owner?.() }, (lockSignal) =>
282
+ // The span opens once the lock is held: a busy coordinator makes none.
283
+ withSpan(deps, 'coordinate', name, () =>
284
+ coordinateStep(deps, name, { signal: anySignal(signal, lockSignal), driver }),
285
+ ),
286
+ );
287
+ } catch (error) {
288
+ if (error instanceof LockAlreadyHeldError) {
289
+ return { next: 'busy', retryAfterMs: Math.max(1, Math.floor(lock.ttlMs / 2)) };
290
+ }
291
+ throw error;
292
+ }
293
+ }
294
+
295
+ /** `fn` inside the kit's span for `kind` (`slice`, `coordinate`), when it gives one */
296
+ function withSpan(deps, kind, name, fn) {
297
+ return typeof deps.span === 'function' ? deps.span(kind, name, fn) : fn();
298
+ }
299
+
300
+ /** A signal aborted when either is */
301
+ function anySignal(a, b) {
302
+ if (!a) return b;
303
+ if (!b) return a;
304
+ return AbortSignal.any([a, b]);
305
+ }
306
+
307
+ async function coordinateStep(deps, name, { signal, driver }) {
308
+ const { store } = deps;
309
+ let state = await store.get(name);
310
+ if (state === null) return { next: 'done', status: 'unregistered' };
311
+ if (state.schema > STATE_SCHEMA) {
312
+ warnOnce(
313
+ deps,
314
+ `schema:${name}`,
315
+ `⚠ ${name} was registered by a newer release (state schema ${state.schema}, this one ` +
316
+ `knows ${STATE_SCHEMA}) — waiting for a process of that release to drive it`,
317
+ { background: name },
318
+ );
319
+ return retrySoon(deps, 'newer-schema');
320
+ }
321
+ if (TERMINAL.has(state.status) || state.status === 'paused') {
322
+ return { next: 'done', status: state.status };
323
+ }
324
+ if (state.status === 'blocked') {
325
+ const unblocked = await tryUnblock(deps, state);
326
+ if (!unblocked) return { next: 'done', status: 'blocked', waitsFor: state.waitsFor };
327
+ state = unblocked;
328
+ }
329
+
330
+ // Who drives it — recorded for status. A BullMQ chain of coordinator jobs
331
+ // also carries a round, which only this step hands out (under the
332
+ // coordinator lock): a new chain gets the next round; a chain whose round
333
+ // is not the latest bows out — a newer one took over (two chains started
334
+ // at once each get their own), or it was registered again since. A round
335
+ // the kit never handed out (forged in Redis) is never written.
336
+ let round;
337
+ if (driver.kind === 'bullmq') {
338
+ const current = state.round;
339
+ if (driver.round === undefined) round = (current ?? 0) + 1;
340
+ else if (driver.round !== current) return { next: 'superseded' };
341
+ else round = current;
342
+ }
343
+ await store.set(name, {
344
+ coordinator: {
345
+ kind: driver.kind,
346
+ ...(driver.ref !== undefined ? { ref: driver.ref } : {}),
347
+ at: new Date(),
348
+ },
349
+ ...(round !== undefined ? { round } : {}),
350
+ });
351
+ const answer = await coordinatePass(deps, name, state, { signal });
352
+ return round !== undefined ? { ...answer, round } : answer;
353
+ }
354
+
355
+ /** The rest of a coordinator step, once its driver is recorded */
356
+ async function coordinatePass(deps, name, initial, { signal }) {
357
+ const { store } = deps;
358
+ let state = initial;
359
+ let job;
360
+ try {
361
+ job = await jobFor(deps, name, state);
362
+ } catch (error) {
363
+ if (error instanceof ChecksumMismatchError || error?.code === 'MIGRATION_FILE_NOT_FOUND') {
364
+ // Mid-deploy: an older or newer pod holds the other version of the file.
365
+ deps.logger.warn(`⚠ ${errorText(error)}`, deps.fields({ background: name }));
366
+ return retrySoon(deps, 'checksum');
367
+ }
368
+ throw error;
369
+ }
370
+ try {
371
+ await assertTransactions(deps, name, job.spec);
372
+ } catch (error) {
373
+ // Not something a retry fixes: the deployment cannot do it. Anything
374
+ // else (the server could not be asked) is the step's to retry.
375
+ if (!(error instanceof TransactionsUnsupportedError)) throw error;
376
+ return failState(deps, state, error.message);
377
+ }
378
+ const hash = matchHash(job.spec, job.direction === 'revert' ? 'revert' : 'forward');
379
+
380
+ for (let guard = 0; guard < job.spec.maxPasses + 2; guard++) {
381
+ if (signal?.aborted) return retrySoon(deps, 'stopping');
382
+ // Resharded (or its key refined) since this plan: its ranges mean nothing
383
+ // now — re-split the same pass (safe: the version filter skips what is done).
384
+ if (state.phase === 'process' && job.partitioner.stale?.(state.plan?.epoch)) {
385
+ deps.logger.warn(
386
+ `⚠ ${name}: ${job.spec.collection} was resharded since its plan — replanning`,
387
+ deps.fields({ background: name }),
388
+ );
389
+ state =
390
+ (await store.cas(
391
+ name,
392
+ { phase: 'process', 'plan.token': state.plan.token },
393
+ { $set: { phase: 'replan' } },
394
+ )) ?? (await store.get(name));
395
+ }
396
+ // A replan re-splits the same pass; it is not a new one.
397
+ const replanning = state.phase === 'replan';
398
+ if (replanning && state.plan !== undefined) {
399
+ await store.supersede(name, { generation: state.generation, plan: state.plan.token });
400
+ const leases = await store.leases(name);
401
+ if (leases.live > 0) return retrySoon(deps, 'replan-draining');
402
+ state =
403
+ (await store.cas(name, { phase: 'replan' }, { $set: { phase: 'partition' } })) ?? state;
404
+ }
405
+ if (state.phase !== 'process' || state.plan === undefined || state.plan.match !== hash) {
406
+ const planned = await planPass(deps, job, state, hash, { newPass: !replanning });
407
+ if (planned.next) return planned;
408
+ state = planned.state;
409
+ if (planned.empty) {
410
+ const finalized = await finalize(deps, job, state);
411
+ if (finalized.next) return finalized;
412
+ state = finalized.state;
413
+ continue;
414
+ }
415
+ }
416
+ if (state.status === 'pending') {
417
+ state =
418
+ (await store.move(name, { from: ['pending'], to: 'running', action: 'plan' })) ?? state;
419
+ }
420
+ const reaped = await store.reap(name);
421
+ if (reaped > 0) deps.telemetry?.backgroundLeasesReclaimed({ name, count: reaped });
422
+ const counts = await store.partitionCounts(name, {
423
+ generation: state.generation,
424
+ plan: state.plan.token,
425
+ });
426
+ const open = counts.pending + counts.running;
427
+ if (open > 0) {
428
+ const claimable = counts.pending + (counts.running - counts.leased);
429
+ return {
430
+ next: 'process',
431
+ lanes: Math.max(0, Math.min(claimable, job.spec.maxParallel - counts.leased)),
432
+ generation: state.generation,
433
+ registration: state.registration,
434
+ counts,
435
+ retryAfterMs: POLL_MS,
436
+ };
437
+ }
438
+ const finalized = await finalize(deps, job, state);
439
+ if (finalized.next) return finalized;
440
+ state = finalized.state;
441
+ }
442
+ return retrySoon(deps, 'passes');
443
+ }
444
+
445
+ /**
446
+ * Plan the next pass: partitions over what still matches, committed under a
447
+ * fresh plan token. Resolves to `{ state }` (with `empty` when nothing is
448
+ * left to plan), or a coordinator answer (`{ next }`).
449
+ */
450
+ async function planPass(deps, job, state, hash, { newPass = true } = {}) {
451
+ const { store } = deps;
452
+ const spec = job.spec;
453
+ let plan;
454
+ if (spec.mode === 'step') {
455
+ plan = { method: 'step', estimate: 0, partitions: [{ scope: { kind: 'step' } }] };
456
+ } else {
457
+ plan = await job.partitioner.plan({
458
+ collection: deps.db.collection(spec.collection),
459
+ match: job.match,
460
+ hint: job.hint,
461
+ maxParallel: spec.maxParallel,
462
+ settings: spec.partitions,
463
+ shardConcurrency: spec.shardConcurrency,
464
+ });
465
+ }
466
+ const pass = (state.pass ?? 0) + (newPass ? 1 : 0);
467
+ const sharding = spec.mode === 'step' ? undefined : shardingOf(job, plan);
468
+ const committed = await store.commitPlan(state._id, {
469
+ generation: state.generation,
470
+ previousToken: state.plan?.token,
471
+ // A repin while this pass was planned changed the spec under it: the
472
+ // commit loses, and the next step plans with the new one.
473
+ filter: {
474
+ status: { $in: ['pending', 'running'] },
475
+ ...(state.checksum !== undefined ? { checksum: state.checksum } : {}),
476
+ },
477
+ plan: {
478
+ partitioner: spec.mode === 'step' ? 'step' : job.partitioner.id,
479
+ method: plan.method,
480
+ estimate: plan.estimate,
481
+ ...(plan.atLeast ? { atLeast: true } : {}),
482
+ match: hash,
483
+ ...(plan.degraded ? { degraded: plan.degraded } : {}),
484
+ ...(plan.epoch ? { epoch: plan.epoch } : {}),
485
+ ...(sharding !== undefined ? { sharding } : {}),
486
+ },
487
+ partitions: plan.partitions,
488
+ fields: {
489
+ pass,
490
+ ...(state.startedAt === undefined ? { startedAt: new Date() } : {}),
491
+ lastProgressAt: new Date(),
492
+ },
493
+ });
494
+ if (committed === null) return retrySoon(deps, 'plan-race');
495
+ // A coordinator that lost a commit race earlier left partitions nobody can claim.
496
+ await store.dropForeignPlans(state._id, committed.plan.token);
497
+ deps.emit('background:partitioned', {
498
+ migration: state._id,
499
+ generation: committed.generation,
500
+ pass,
501
+ partitions: plan.partitions.length,
502
+ estimate: plan.estimate,
503
+ ...(plan.atLeast ? { atLeast: true } : {}),
504
+ method: plan.method,
505
+ ...(plan.degraded ? { degraded: plan.degraded } : {}),
506
+ });
507
+ if (plan.degraded) {
508
+ deps.logger.warn(
509
+ `⚠ ${state._id}: ${DEGRADED[plan.degraded] ?? plan.degraded}`,
510
+ deps.fields({ background: state._id, degraded: plan.degraded }),
511
+ );
512
+ }
513
+ return { state: committed, empty: plan.partitions.length === 0 };
514
+ }
515
+
516
+ /**
517
+ * Close a pass: roll the partitions' counters into the state (once per
518
+ * generation — `rolledGeneration` makes it idempotent after a crash), then
519
+ * decide. A failed partition fails the background migration; nothing left
520
+ * to rewrite completes it; anything left starts another pass — unless
521
+ * `maxPasses` passes have not drained it, which means something keeps
522
+ * writing the old shape.
523
+ */
524
+ async function finalize(deps, job, state) {
525
+ const { store } = deps;
526
+ const name = state._id;
527
+ const generation = state.generation;
528
+ if ((state.rolledGeneration ?? 0) < generation) {
529
+ const rolled = await store.generationTotals(name, generation);
530
+ const inc = {};
531
+ for (const [key, value] of Object.entries(rolled?.totals ?? {})) {
532
+ if (typeof value === 'number' && key !== 'failedPartitions' && value !== 0) {
533
+ inc[`totals.${key}`] = value;
534
+ }
535
+ }
536
+ // One id, one entry — by its canonical form, so a string _id never
537
+ // stands in for an ObjectId with the same hex.
538
+ const badIds = [];
539
+ const seen = new Set();
540
+ for (const list of [state.badIds ?? [], rolled?.badIds ?? []]) {
541
+ for (const id of list) {
542
+ const key = idKey(id);
543
+ if (!seen.has(key)) {
544
+ seen.add(key);
545
+ badIds.push(id);
546
+ }
547
+ }
548
+ }
549
+ const updated = await store.cas(
550
+ name,
551
+ { rolledGeneration: { $lt: generation } },
552
+ {
553
+ ...(Object.keys(inc).length > 0 ? { $inc: inc } : {}),
554
+ $set: {
555
+ rolledGeneration: generation,
556
+ badIds: badIds.slice(0, MAX_BAD_IDS + 1),
557
+ ...(rolled?.docErrors?.length ? { docErrors: rolled.docErrors } : {}),
558
+ ...(rolled?.totals?.failedPartitions > 0
559
+ ? {
560
+ failedPartitions: rolled.totals.failedPartitions,
561
+ lastError: rolled.totals.lastError,
562
+ }
563
+ : {}),
564
+ updatedAt: new Date(),
565
+ },
566
+ },
567
+ );
568
+ state = updated ?? (await store.get(name));
569
+ }
570
+
571
+ if ((state.failedPartitions ?? 0) > 0) {
572
+ return failState(
573
+ deps,
574
+ state,
575
+ `${state.failedPartitions} partition(s) failed: ${state.lastError}`,
576
+ );
577
+ }
578
+
579
+ let remaining = 0;
580
+ if (job.spec.mode !== 'step') {
581
+ remaining = await deps.db
582
+ .collection(job.spec.collection)
583
+ .countDocuments(excludeBadIds(matchOf(job.spec, job.direction), state.badIds), {
584
+ limit: 1,
585
+ ...(job.hint ? { hint: job.hint } : {}),
586
+ ...READ_OPTIONS,
587
+ });
588
+ }
589
+ if (remaining === 0) {
590
+ const completed = await store.move(name, {
591
+ from: ['running', 'pending'],
592
+ to: 'completed',
593
+ action: 'complete',
594
+ fields: { completedAt: new Date(), phase: 'process' },
595
+ });
596
+ await store.dropGenerations(name, generation - 1);
597
+ if (completed !== null) {
598
+ if ((state.badIds ?? []).length > 0) {
599
+ deps.logger.warn(
600
+ `⚠ ${name} completed with ${state.badIds.length} document(s) it could not migrate ` +
601
+ '(within maxDocumentErrors) — they are still in the old shape',
602
+ deps.fields({ background: name, failedDocuments: state.badIds.length }),
603
+ );
604
+ }
605
+ deps.emit('background:completed', {
606
+ migration: name,
607
+ totals: completed.totals,
608
+ passes: completed.pass,
609
+ });
610
+ deps.logger.info(
611
+ `✔ Background migration ${name} completed (${completed.totals?.migrated ?? 0} migrated, ` +
612
+ `${completed.pass} pass(es))`,
613
+ deps.fields({ background: name }),
614
+ );
615
+ // A completed revert unblocks nothing: what required it needs the forward shape.
616
+ if (job.direction !== 'revert') await deps.onCompleted?.(name);
617
+ }
618
+ return { next: 'done', status: completed?.status ?? (await store.get(name))?.status };
619
+ }
620
+ if ((state.pass ?? 0) >= job.spec.maxPasses) {
621
+ return failState(
622
+ deps,
623
+ state,
624
+ `old-shape documents keep appearing after ${state.pass} pass(es) — is an old version ` +
625
+ 'of the application still writing them?',
626
+ );
627
+ }
628
+ deps.emit('background:pass', {
629
+ migration: name,
630
+ pass: state.pass,
631
+ generation,
632
+ totals: state.totals,
633
+ });
634
+ const next = await store.cas(
635
+ name,
636
+ { generation, phase: 'process' },
637
+ { $set: { phase: 'partition', updatedAt: new Date() } },
638
+ );
639
+ await store.dropGenerations(name, generation - 1);
640
+ return { state: next ?? (await store.get(name)) };
641
+ }
642
+
643
+ /** Move a state to `failed` with `message`, and say so */
644
+ async function failState(deps, state, reason) {
645
+ // Kept, emitted and logged: the application's data stays out of it.
646
+ const message = documentErrorText(reason);
647
+ const failed = await deps.store.move(state._id, {
648
+ from: ['running', 'pending', 'blocked'],
649
+ to: 'failed',
650
+ action: 'fail',
651
+ fields: { lastError: message, failedAt: new Date() },
652
+ });
653
+ if (failed !== null) {
654
+ deps.emit('background:failed', { migration: state._id, error: message });
655
+ deps.logger.error(
656
+ `✖ Background migration ${state._id} failed: ${message}`,
657
+ deps.fields({ background: state._id }),
658
+ );
659
+ }
660
+ return { next: 'done', status: 'failed', error: message };
661
+ }
662
+
663
+ /**
664
+ * Where each of `requires` stands, in order — the one rule every guard
665
+ * shares: `[{ migration, status, done, state? }]`. A background migration is
666
+ * done when its state completed forward, or — with no state at all — when
667
+ * the changelog has it from a history that predates migronaut (a baseline,
668
+ * an import: `deps.adoptedOf(names)`). A completed revert is not done: what
669
+ * required it needs the forward shape.
670
+ */
671
+ async function requiresStatus(deps, requires) {
672
+ if (requires.length === 0) return [];
673
+ const states = await deps.store.getMany(requires);
674
+ const missing = [];
675
+ for (const name of requires) if (!states.has(name)) missing.push(name);
676
+ const adopted = new Set(missing.length > 0 ? ((await deps.adoptedOf?.(missing)) ?? []) : []);
677
+ const rows = [];
678
+ for (const migration of requires) {
679
+ const state = states.get(migration);
680
+ if (state === undefined) {
681
+ const done = adopted.has(migration);
682
+ rows.push({ migration, status: done ? 'adopted' : 'unregistered', done });
683
+ } else if (state.direction === 'revert') {
684
+ rows.push({ migration, status: 'reverted', done: false, state });
685
+ } else {
686
+ rows.push({ migration, status: state.status, done: state.status === 'completed', state });
687
+ }
688
+ }
689
+ return rows;
690
+ }
691
+
692
+ /** The names among `requires` not done yet, in order */
693
+ async function waitingFor(deps, requires) {
694
+ const waitsFor = [];
695
+ for (const row of await requiresStatus(deps, requires)) {
696
+ if (!row.done) waitsFor.push(row.migration);
697
+ }
698
+ return waitsFor;
699
+ }
700
+
701
+ /**
702
+ * A blocked one whose `requires` are all done moves to `pending`. Returns
703
+ * the state after, or `null`.
704
+ */
705
+ async function tryUnblock(deps, state) {
706
+ const waitsFor = await waitingFor(deps, state.requires ?? []);
707
+ if (waitsFor.length > 0) {
708
+ if (JSON.stringify(waitsFor) !== JSON.stringify(state.waitsFor ?? [])) {
709
+ await deps.store.set(state._id, { waitsFor });
710
+ }
711
+ return null;
712
+ }
713
+ const unblocked = await deps.store.move(state._id, {
714
+ from: ['blocked'],
715
+ to: 'pending',
716
+ action: 'unblock',
717
+ fields: { waitsFor: [] },
718
+ });
719
+ if (unblocked !== null) deps.emit('background:unblocked', { migration: state._id });
720
+ return unblocked;
721
+ }
722
+
723
+ // ─── A lane's slice ───────────────────────────────────────────────────────────
724
+
725
+ /**
726
+ * One slice of a lane: claim a partition (and its slot), work it until the
727
+ * slice ends, release — then claim another while time is left. Resolves to
728
+ * `{ outcome, counters, … }`: `yielded` (time is up, work is left),
729
+ * `exhausted` (nothing left to claim), `busy` (every slot is held —
730
+ * `retryAfterMs`), `stale` (no current plan: the coordinator must run),
731
+ * `paused`, `cancelled`, `failed`, `stopped` (the signal) or `lost` (the
732
+ * lease was taken). A slice that fails is counted on its partition and
733
+ * thrown.
734
+ */
735
+ async function runSlice(deps, name, { signal, sliceMs, owner } = {}) {
736
+ const { store } = deps;
737
+ const state = await store.get(name);
738
+ if (state === null) return { outcome: 'cancelled', counters: {} };
739
+ if (state.status === 'paused' || state.status === 'cancelled' || state.status === 'failed') {
740
+ return { outcome: state.status, counters: {} };
741
+ }
742
+ if (state.status !== 'running' || state.phase !== 'process' || state.plan === undefined) {
743
+ return { outcome: state.status === 'completed' ? 'exhausted' : 'stale', counters: {} };
744
+ }
745
+ // A newer release's state (an old pod mid-deploy): its plan may hold scopes
746
+ // this one cannot read. The coordinator waits; so does the lane.
747
+ if (state.schema > STATE_SCHEMA) return { outcome: 'stale', counters: {} };
748
+ let job;
749
+ try {
750
+ job = await jobFor(deps, name, state);
751
+ // A plan made before a reshard: the coordinator must re-split first.
752
+ if (job.partitioner.stale?.(state.plan.epoch)) return { outcome: 'stale', counters: {} };
753
+ await assertTransactions(deps, name, job.spec);
754
+ } catch (error) {
755
+ // Before any claim (the file changed or is gone, the deployment cannot
756
+ // run it): no partition to count it on and no slice:start to pair an
757
+ // event with — but measured like any failed slice; the drivers log it.
758
+ deps.telemetry?.backgroundSliceEnded({ name, durationMs: 0, outcome: 'error', error });
759
+ throw error;
760
+ }
761
+ job.generation = state.generation;
762
+ const now = deps.now ?? Date.now;
763
+ const deadline = now() + (sliceMs ?? job.spec.sliceMs);
764
+ const counters = {};
765
+ const throttle = createThrottle({
766
+ spec: { ...job.spec, throttle: job.fns.throttle },
767
+ db: deps.db,
768
+ logger: deps.logger,
769
+ name,
770
+ warned: deps.warned,
771
+ });
772
+ let worked = false;
773
+ for (;;) {
774
+ if (signal?.aborted) return { outcome: 'stopped', counters };
775
+ // The first claim is always made: a slice shorter than its setup still works a batch.
776
+ if (worked && now() >= deadline) return { outcome: 'yielded', counters };
777
+ const claimed = await store.claim(name, {
778
+ generation: state.generation,
779
+ plan: state.plan.token,
780
+ maxParallel: job.spec.maxParallel,
781
+ ttlMs: deps.ttlMs,
782
+ owner,
783
+ // Lanes per shard, on partitions that belong to one.
784
+ ...(job.partitioner.id === 'shard' ? { shardConcurrency: job.spec.shardConcurrency } : {}),
785
+ onReaped: (count) => deps.telemetry?.backgroundLeasesReclaimed({ name, count }),
786
+ });
787
+ if (claimed.exhausted) return { outcome: 'exhausted', counters };
788
+ if (claimed.busy) {
789
+ return worked
790
+ ? { outcome: 'yielded', counters }
791
+ : { outcome: 'busy', counters, retryAfterMs: claimed.retryAfterMs };
792
+ }
793
+ worked = true;
794
+ const { partition, lease } = claimed;
795
+ job.partitionId = partition._id;
796
+ const adaptive =
797
+ job.spec.adaptive === false
798
+ ? undefined
799
+ : adaptiveFor(deps, name, partition, job.spec.adaptive);
800
+ deps.emit('background:slice:start', {
801
+ migration: name,
802
+ generation: state.generation,
803
+ partition: String(partition._id),
804
+ slot: partition.lease.slot,
805
+ });
806
+ const runLease = () =>
807
+ runWithLock(lease, { logger: deps.logger, owner }, (leaseSignal) => {
808
+ job.signal = anySignal(signal, leaseSignal);
809
+ return processPartition(job, {
810
+ store,
811
+ lease,
812
+ partition,
813
+ db: deps.db,
814
+ client: deps.client,
815
+ signal: job.signal,
816
+ deadline,
817
+ throttle,
818
+ adaptive,
819
+ now,
820
+ // Every batch of every lane asks: only what it decides on is read.
821
+ readControl: () =>
822
+ store
823
+ .get(name, { projection: CONTROL_FIELDS })
824
+ .then((current) =>
825
+ current === null
826
+ ? null
827
+ : { status: current.status, generation: current.generation, plan: current.plan },
828
+ ),
829
+ onBatch: (info) => {
830
+ deps.telemetry?.backgroundBatchWritten({
831
+ name,
832
+ durationMs: info.latencyMs,
833
+ shard: partition.group,
834
+ });
835
+ deps.emit('background:batch', {
836
+ migration: name,
837
+ partition: String(partition._id),
838
+ ...(partition.group !== undefined ? { group: partition.group } : {}),
839
+ ...info,
840
+ });
841
+ },
842
+ onTransactionRetries: (count) =>
843
+ deps.telemetry?.backgroundTransactionRetried({ name, reason: 'transient', count }),
844
+ onThrottle: (change) => {
845
+ deps.telemetry?.backgroundThrottled({ name, reason: change.reason });
846
+ deps.emit('background:throttle', {
847
+ migration: name,
848
+ ...(partition.group !== undefined ? { group: partition.group } : {}),
849
+ ...change,
850
+ });
851
+ },
852
+ });
853
+ });
854
+ let result;
855
+ const sliceStarted = now();
856
+ try {
857
+ // One span per lease held — it exists only once the claim succeeded.
858
+ result = await withSpan(deps, 'slice', name, runLease);
859
+ } catch (error) {
860
+ const lost = error instanceof LockLostError;
861
+ // The caller stopped the lane (a shutdown, a deploy): whatever a wait
862
+ // rejected with is the stop, not a fault of the partition — counting it
863
+ // would fail a partition after a few rolling deploys.
864
+ const stopped = !lost && signal?.aborted === true;
865
+ deps.telemetry?.backgroundSliceEnded({
866
+ name,
867
+ durationMs: now() - sliceStarted,
868
+ outcome: lost ? 'lost' : stopped ? 'stopped' : 'error',
869
+ ...(stopped ? {} : { error }),
870
+ });
871
+ if (lost) {
872
+ deps.emit('background:lease:lost', { migration: name, partition: String(partition._id) });
873
+ return { outcome: 'lost', counters };
874
+ }
875
+ if (stopped) {
876
+ deps.emit('background:slice:end', {
877
+ migration: name,
878
+ partition: String(partition._id),
879
+ outcome: 'stopped',
880
+ });
881
+ return { outcome: 'stopped', counters };
882
+ }
883
+ const partitionFailed = await store
884
+ .failSlice(lease, {
885
+ error: documentErrorText(error),
886
+ maxSliceFailures: job.spec.maxSliceFailures,
887
+ })
888
+ .catch((countError) => {
889
+ // The slice's own error is what is thrown; this one is only said.
890
+ deps.logger.debug(
891
+ `Could not count a failed slice of ${name}: ${errorText(countError)}`,
892
+ deps.fields({ background: name }),
893
+ );
894
+ return false;
895
+ });
896
+ deps.emit('background:slice:end', {
897
+ migration: name,
898
+ partition: String(partition._id),
899
+ outcome: 'error',
900
+ error: documentErrorText(error),
901
+ });
902
+ if (error instanceof MigronautError) {
903
+ error.context = { ...error.context, background: name, partitionFailed };
904
+ }
905
+ throw error;
906
+ }
907
+ deps.telemetry?.backgroundSliceEnded({
908
+ name,
909
+ durationMs: now() - sliceStarted,
910
+ outcome: result.outcome,
911
+ counters: result.counters,
912
+ });
913
+ addCounters(counters, result.counters ?? {});
914
+ await store.set(name, { lastProgressAt: new Date() });
915
+ deps.emit('background:slice:end', {
916
+ migration: name,
917
+ partition: String(partition._id),
918
+ outcome: result.outcome,
919
+ counters: result.counters,
920
+ });
921
+ if (result.outcome === 'failed' && result.error) {
922
+ await store
923
+ .failPartition(lease, { error: documentErrorText(result.error) })
924
+ .catch(() => undefined);
925
+ await failState(deps, state, result.error.message);
926
+ return { outcome: 'failed', counters, error: result.error };
927
+ }
928
+ if (result.outcome !== 'exhausted') return { outcome: result.outcome, counters };
929
+ // That partition is done: claim another while the slice lasts.
930
+ }
931
+ }
932
+
933
+ /** The process's AIMD controller for one background migration and group — shared by its lanes */
934
+ function adaptiveFor(deps, name, partition, settings) {
935
+ const cache = deps.adaptiveCache;
936
+ const key = `${name}:${partition.group ?? ''}`;
937
+ if (cache?.has(key)) return cache.get(key);
938
+ const controller = createAdaptive(settings, {
939
+ ...(partition.throttle ? { initial: partition.throttle } : {}),
940
+ });
941
+ cache?.set(key, controller);
942
+ return controller;
943
+ }
944
+
945
+ // ─── Controls ─────────────────────────────────────────────────────────────────
946
+
947
+ /**
948
+ * Apply a control action — `pause`, `resume`, `cancel`, `retry` (with
949
+ * `fromStart`) — to a background migration. Resolves to
950
+ * `{ applied: 'changed' | 'unchanged', status }`.
951
+ *
952
+ * @throws {BackgroundConflictError} when the action does not fit its state
953
+ * @throws {NotAppliedError} via the kit when it is not registered
954
+ */
955
+ async function control(deps, name, action, { requestedBy, reason, fromStart = false } = {}) {
956
+ const { store } = deps;
957
+ const state = await store.get(name);
958
+ if (state === null) {
959
+ throw new BackgroundConflictError(`Background migration ${name} is not registered`, {
960
+ action,
961
+ migration: name,
962
+ status: 'unregistered',
963
+ });
964
+ }
965
+ const effective = action === 'retry' && state.status === 'completed' ? 'reopen' : action;
966
+ const { to, applied } = transition(state.status, effective, { migration: name });
967
+ if (applied === 'unchanged') return { applied, status: state.status };
968
+
969
+ const stamp = {
970
+ action,
971
+ ...(requestedBy !== undefined ? { requestedBy } : {}),
972
+ ...(reason !== undefined ? { reason } : {}),
973
+ at: new Date(),
974
+ };
975
+ let target = to;
976
+ const fields = { control: stamp };
977
+ if (action === 'resume') {
978
+ // Back to where it was: blocked if what it requires is still not done.
979
+ const waitsFor = await waitingFor(deps, state.requires ?? []);
980
+ if (waitsFor.length > 0) {
981
+ target = 'blocked';
982
+ fields.waitsFor = waitsFor;
983
+ } else if (state.plan !== undefined && state.phase === 'process') {
984
+ target = 'running';
985
+ }
986
+ }
987
+ if (effective === 'reopen') {
988
+ fields.phase = 'partition';
989
+ fields.reopened = (state.reopened ?? 0) + 1;
990
+ }
991
+ // A pass that already closed (its counters rolled up) is not resumed: its
992
+ // partitions' work since would never be counted. A new pass takes what is left.
993
+ const closed = state.plan !== undefined && (state.rolledGeneration ?? 0) >= state.generation;
994
+ if (action === 'retry' && fromStart) {
995
+ fields.phase = 'partition';
996
+ fields.pass = 0;
997
+ fields.badIds = [];
998
+ fields.failedPartitions = 0;
999
+ // Diagnostics of the runs before; `totals` stays — it is what was rewritten.
1000
+ fields.docErrors = [];
1001
+ fields.lastError = null;
1002
+ } else if (action === 'retry') {
1003
+ fields.failedPartitions = 0;
1004
+ if (closed) fields.phase = 'partition';
1005
+ }
1006
+ const moved = await store.move(name, {
1007
+ from: [state.status],
1008
+ to: target,
1009
+ action,
1010
+ fields,
1011
+ by: requestedBy,
1012
+ reason,
1013
+ });
1014
+ if (moved === null) {
1015
+ // Someone moved it first: judge again from what it is now.
1016
+ return control(deps, name, action, { requestedBy, reason, fromStart });
1017
+ }
1018
+
1019
+ const plan = state.plan;
1020
+ if (plan !== undefined) {
1021
+ if (action === 'cancel') {
1022
+ await store.setOpenPartitions(name, {
1023
+ generation: state.generation,
1024
+ plan: plan.token,
1025
+ status: 'cancelled',
1026
+ });
1027
+ } else if (action === 'retry' && fromStart) {
1028
+ await store.dropGenerations(name, state.generation);
1029
+ } else if (action === 'retry' && effective !== 'reopen' && !closed) {
1030
+ await store.setOpenPartitions(name, {
1031
+ generation: state.generation,
1032
+ plan: plan.token,
1033
+ from: ['failed', 'cancelled'],
1034
+ status: 'pending',
1035
+ });
1036
+ }
1037
+ }
1038
+ deps.emit('background:control', { migration: name, action, from: state.status, to: target });
1039
+ deps.logger.info(
1040
+ `${name}: ${action} (${state.status} → ${target})`,
1041
+ deps.fields({ background: name, action }),
1042
+ );
1043
+ return { applied: 'changed', status: target };
1044
+ }
1045
+
1046
+ /**
1047
+ * Pin what is on disk now: its checksum and spec become the state's. A
1048
+ * change to what is matched, or how it is split, means a new plan.
1049
+ */
1050
+ async function repin(deps, name, { requestedBy, reason } = {}) {
1051
+ const { store } = deps;
1052
+ const state = await store.get(name);
1053
+ if (state === null) {
1054
+ throw new BackgroundConflictError(`Background migration ${name} is not registered`, {
1055
+ action: 'repin',
1056
+ migration: name,
1057
+ status: 'unregistered',
1058
+ });
1059
+ }
1060
+ const loaded = await deps.load(name);
1061
+ const before = state.spec ?? {};
1062
+ const after = loaded.spec;
1063
+ const dir = state.direction === 'revert' ? 'revert' : 'forward';
1064
+ const replan =
1065
+ matchHash(before, dir) !== matchHash(after, dir) ||
1066
+ before.maxParallel !== after.maxParallel ||
1067
+ JSON.stringify(before.partitions) !== JSON.stringify(after.partitions);
1068
+ const fields = {
1069
+ checksum: loaded.checksum,
1070
+ spec: after,
1071
+ ...(replan && !TERMINAL.has(state.status) && state.plan !== undefined
1072
+ ? { phase: 'replan' }
1073
+ : {}),
1074
+ control: { action: 'repin', requestedBy, reason, at: new Date() },
1075
+ };
1076
+ await store.set(name, fields);
1077
+ deps.emit('background:control', { migration: name, action: 'repin', replan });
1078
+ return { applied: 'changed', status: state.status, replan, checksum: loaded.checksum };
1079
+ }
1080
+
1081
+ /**
1082
+ * Wait until no lane holds a lease any more — after a pause or a cancel,
1083
+ * for a caller that wants "stopped" to mean stopped.
1084
+ */
1085
+ async function waitForLanes(deps, name, { signal, timeoutMs = 120_000, pollMs = 250 } = {}) {
1086
+ const now = deps.now ?? Date.now;
1087
+ const until = now() + timeoutMs;
1088
+ for (;;) {
1089
+ const { live } = await deps.store.leases(name);
1090
+ if (live === 0) return true;
1091
+ if (now() >= until) return false;
1092
+ // One listener per wait, removed with it — and an aborted signal ends it at once.
1093
+ await sleep(pollMs, signal);
1094
+ }
1095
+ }
1096
+
1097
+ /** Why a failed state failed, for a thrown error */
1098
+ function failedError(state) {
1099
+ return new BackgroundFailedError(
1100
+ `Background migration ${state._id} failed${state.lastError ? `: ${state.lastError}` : ''}`,
1101
+ { migration: state._id, lastError: state.lastError },
1102
+ );
1103
+ }
1104
+
1105
+ module.exports = {
1106
+ STATE_SUMMARY,
1107
+ control,
1108
+ coordinate,
1109
+ failedError,
1110
+ finalize,
1111
+ jobFor,
1112
+ partitionerFor,
1113
+ probeHint,
1114
+ repin,
1115
+ requiresStatus,
1116
+ runSlice,
1117
+ tryUnblock,
1118
+ versionHint,
1119
+ waitForLanes,
1120
+ waitingFor,
1121
+ };