@alexify/migronaut 2.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CHANGELOG.md +320 -0
  2. package/README.md +208 -6
  3. package/bullmq.d.ts +845 -0
  4. package/bullmq.js +1 -0
  5. package/index.d.ts +634 -18
  6. package/migronaut.schema.json +182 -1
  7. package/package.json +21 -5
  8. package/src/bullmq/index.js +55 -0
  9. package/src/bullmq/jobs.js +454 -0
  10. package/src/bullmq/processor.js +608 -0
  11. package/src/bullmq/producer.js +424 -0
  12. package/src/bullmq/service.js +653 -0
  13. package/src/bullmq/wait.js +124 -0
  14. package/src/cli/args.js +12 -2
  15. package/src/cli/commands/converge.js +160 -0
  16. package/src/cli/commands/down.js +2 -0
  17. package/src/cli/commands/lock.js +2 -1
  18. package/src/cli/commands/redo.js +8 -1
  19. package/src/cli/commands/up.js +14 -1
  20. package/src/cli/exit-codes.js +9 -2
  21. package/src/cli/index.js +2 -0
  22. package/src/cli/shared.js +14 -4
  23. package/src/cli/table.js +105 -0
  24. package/src/core/changelog.js +71 -6
  25. package/src/core/collections.js +372 -0
  26. package/src/core/config.js +100 -25
  27. package/src/core/converge-log.js +47 -0
  28. package/src/core/converge-plan.js +483 -0
  29. package/src/core/converge.js +867 -0
  30. package/src/core/index-spec.js +496 -0
  31. package/src/core/lock-wait.js +260 -0
  32. package/src/core/lock.js +45 -16
  33. package/src/core/migrator.js +563 -283
  34. package/src/core/options.js +251 -0
  35. package/src/core/run-recorder.js +157 -0
  36. package/src/core/run.js +58 -90
  37. package/src/core/sequence.js +134 -0
  38. package/src/errors/index.js +56 -0
  39. package/src/index.js +8 -0
  40. package/src/utils/actor.js +48 -0
  41. package/src/utils/canonical.js +179 -0
  42. package/src/utils/collection-name.js +21 -0
  43. package/src/utils/error.js +18 -1
  44. package/src/utils/id.js +77 -0
  45. package/src/utils/loader.js +39 -21
  46. package/src/utils/migration-name.js +32 -0
  47. package/src/utils/redact.js +21 -1
  48. package/src/utils/telemetry.js +393 -0
  49. package/src/utils/template.js +36 -2
@@ -0,0 +1,260 @@
1
+ const { setTimeout: delay } = require('node:timers/promises');
2
+ const {
3
+ ConfigInvalidError,
4
+ LockAlreadyHeldError,
5
+ MigronautError,
6
+ RunAbortedError,
7
+ } = require('../errors/index.js');
8
+ const { errorText } = require('../utils/error.js');
9
+
10
+ /**
11
+ * Longer than the default 60s lock TTL on purpose: with a shorter budget, a peer
12
+ * migration that outlives it makes every waiting instance fail to boot, even
13
+ * though the peer is healthy and still holding a valid lock.
14
+ */
15
+ const DEFAULT_LOCK_WAIT_TIMEOUT_MS = 90_000;
16
+ const DEFAULT_LOCK_POLL_INTERVAL_MS = 500;
17
+ /**
18
+ * The longest a waiter sleeps between two polls, however long it has waited.
19
+ * Polls back off from `lockPollIntervalMs` up to this, so a hundred instances
20
+ * waiting out a long deploy do not hammer the one lock document — while a
21
+ * released lock is still noticed within a few seconds.
22
+ */
23
+ const MAX_LOCK_POLL_INTERVAL_MS = 5000;
24
+ /** The longest delay a Node timer honours — anything above it fires after 1ms */
25
+ const MAX_TIMER_MS = 2 ** 31 - 1;
26
+ /** ±25% jitter so N instances booting together stop polling in lockstep */
27
+ const POLL_JITTER_RATIO = 0.25;
28
+ /**
29
+ * How much longer than the holder's TTL a waiter is patient by default: the
30
+ * holder's heartbeat moves `lockedAt` every TTL/2, and a crashed holder's lock
31
+ * is only reclaimable after a full TTL — a default below that would give up
32
+ * on healthy holders, or before a dead one could be replaced.
33
+ */
34
+ const TTL_PATIENCE_FACTOR = 1.5;
35
+
36
+ function jitteredDelay(baseMs) {
37
+ const spread = baseMs * POLL_JITTER_RATIO;
38
+ return Math.max(1, Math.round(baseMs - spread + Math.random() * spread * 2));
39
+ }
40
+
41
+ /** Polls that must fit in one wait budget: enough to see a live holder's heartbeat move */
42
+ const POLLS_PER_BUDGET = 4;
43
+
44
+ /**
45
+ * The sleep before poll number `attempt` (1-based): doubling from `baseMs`,
46
+ * jittered, and capped — at `MAX_LOCK_POLL_INTERVAL_MS`, and at a quarter of
47
+ * the wait budget, so a long sleep can never outlast the budget and make a
48
+ * live holder look stalled. Never below `baseMs`.
49
+ */
50
+ function backoffDelay(baseMs, attempt, budgetMs = Number.POSITIVE_INFINITY) {
51
+ const cap = Math.max(baseMs, Math.min(MAX_LOCK_POLL_INTERVAL_MS, budgetMs / POLLS_PER_BUDGET));
52
+ const grown = Math.min(cap, baseMs * 2 ** Math.min(Math.max(attempt - 1, 0), 30));
53
+ return Math.min(jitteredDelay(grown), MAX_TIMER_MS);
54
+ }
55
+
56
+ /**
57
+ * Validate the wait options. A NaN here (or any non-positive value) disables
58
+ * every deadline comparison in the loop — `NaN > deadline` is always false —
59
+ * turning the wait into an unbounded retry storm against the lock collection
60
+ * that never returns and never surfaces the real error. An interval above the
61
+ * largest timer would fire after 1ms: the same storm. `lockWaitTimeoutMs` may
62
+ * be left out — the default then follows the holder's TTL.
63
+ */
64
+ function assertLockWaitOptions({
65
+ onLockHeld,
66
+ lockWaitTimeoutMs,
67
+ lockPollIntervalMs = DEFAULT_LOCK_POLL_INTERVAL_MS,
68
+ } = {}) {
69
+ if (onLockHeld !== undefined && onLockHeld !== 'wait' && onLockHeld !== 'throw') {
70
+ throw new ConfigInvalidError("onLockHeld must be 'wait' or 'throw'", { onLockHeld });
71
+ }
72
+ if (
73
+ lockWaitTimeoutMs !== undefined &&
74
+ (!Number.isFinite(lockWaitTimeoutMs) || lockWaitTimeoutMs <= 0)
75
+ ) {
76
+ throw new ConfigInvalidError('lockWaitTimeoutMs must be a positive finite number', {
77
+ lockWaitTimeoutMs,
78
+ });
79
+ }
80
+ if (
81
+ !Number.isFinite(lockPollIntervalMs) ||
82
+ lockPollIntervalMs <= 0 ||
83
+ lockPollIntervalMs > MAX_TIMER_MS
84
+ ) {
85
+ throw new ConfigInvalidError(
86
+ `lockPollIntervalMs must be a positive number of milliseconds, at most ${MAX_TIMER_MS}`,
87
+ { lockPollIntervalMs },
88
+ );
89
+ }
90
+ }
91
+
92
+ /** The error an aborted wait rejects with — the signal's own typed reason when it has one */
93
+ function abortedError(signal, extra) {
94
+ const reason = signal.reason;
95
+ if (reason instanceof MigronautError) return reason;
96
+ return new RunAbortedError('Lock wait aborted', {
97
+ reason: errorText(reason ?? 'aborted'),
98
+ ...extra,
99
+ });
100
+ }
101
+
102
+ /**
103
+ * The wait budget: the caller's, or — left out — the default, stretched to
104
+ * outlast the holder's heartbeat and its stale-reclaim window. The holder's
105
+ * TTL comes from its lock document (written since 2.1), else this process's.
106
+ */
107
+ function waitBudget(lockWaitTimeoutMs, error) {
108
+ if (lockWaitTimeoutMs !== undefined) return lockWaitTimeoutMs;
109
+ const ttlMs = error.context?.holder?.ttlMs ?? error.context?.ttlMs;
110
+ return Number.isFinite(ttlMs)
111
+ ? Math.max(DEFAULT_LOCK_WAIT_TIMEOUT_MS, Math.round(ttlMs * TTL_PATIENCE_FACTOR))
112
+ : DEFAULT_LOCK_WAIT_TIMEOUT_MS;
113
+ }
114
+
115
+ /**
116
+ * Run `attempt()`; while it rejects with LockAlreadyHeldError and `onLockHeld`
117
+ * is `'wait'`, poll again. Resolves `{ result, waited, waitedMs, attempts }`.
118
+ *
119
+ * The one wait-for-the-lock loop, shared by `runMigrations` (instances booting
120
+ * together) and the queue processor (a job that finds a CLI run or a peer
121
+ * worker holding the lock). `signal` aborts the wait *between* attempts — it
122
+ * never interrupts an attempt in progress — and `onWait` fires before each
123
+ * sleep, for callers that report progress. `isTransient(error)` widens what is
124
+ * waited out beyond a held lock (the queue's "an earlier job is still in
125
+ * flight"); `onSettle({ waitedMs, attempts, outcome })` fires once when a wait
126
+ * that happened ends — `'acquired'`, `'timeout'` or `'aborted'`.
127
+ */
128
+ async function withLockWait(attempt, options = {}) {
129
+ const {
130
+ onLockHeld = 'throw',
131
+ lockWaitTimeoutMs,
132
+ lockPollIntervalMs = DEFAULT_LOCK_POLL_INTERVAL_MS,
133
+ logger,
134
+ signal,
135
+ onWait,
136
+ onSettle,
137
+ isTransient = (error) => error instanceof LockAlreadyHeldError,
138
+ } = options;
139
+ let waited = false;
140
+ let attempts = 0;
141
+ // The clock starts at the first contention, not before the first attempt —
142
+ // otherwise a slow initial attempt eats the whole waiting budget.
143
+ let firstRefusalAt;
144
+ let deadline;
145
+ // The holder's lockedAt from the last refusal: its heartbeat advances it
146
+ // every TTL/2, so a change between polls is proof of a live, progressing
147
+ // peer.
148
+ let lastHolderLockedAt;
149
+ let warnedShortBudget = false;
150
+ const waitedMs = () => (firstRefusalAt === undefined ? 0 : Date.now() - firstRefusalAt);
151
+ const settle = (outcome) => {
152
+ if (!waited) return;
153
+ try {
154
+ onSettle?.({ waitedMs: waitedMs(), attempts, outcome });
155
+ } catch {
156
+ // A reporting callback must never change how the wait ends.
157
+ }
158
+ };
159
+
160
+ for (;;) {
161
+ if (signal?.aborted) {
162
+ settle('aborted');
163
+ throw abortedError(signal, { attempts, waitedMs: waitedMs() });
164
+ }
165
+ try {
166
+ attempts += 1;
167
+ const result = await attempt(attempts);
168
+ const total = waitedMs();
169
+ settle('acquired');
170
+ return { result, waited, waitedMs: total, attempts };
171
+ } catch (error) {
172
+ if (onLockHeld !== 'wait' || !isTransient(error)) {
173
+ throw error;
174
+ }
175
+ firstRefusalAt ??= Date.now();
176
+ const budget = waitBudget(lockWaitTimeoutMs, error);
177
+ const holderTtlMs = error.context?.holder?.ttlMs ?? error.context?.ttlMs;
178
+ if (
179
+ !warnedShortBudget &&
180
+ lockWaitTimeoutMs !== undefined &&
181
+ Number.isFinite(holderTtlMs) &&
182
+ lockWaitTimeoutMs <= holderTtlMs / 2
183
+ ) {
184
+ warnedShortBudget = true;
185
+ logger?.warn(
186
+ `⚠ lockWaitTimeoutMs (${lockWaitTimeoutMs}ms) is no longer than the holder's heartbeat ` +
187
+ `(${holderTtlMs / 2}ms) — a healthy holder may look stalled and the wait give up`,
188
+ { lockWaitTimeoutMs, holderTtlMs },
189
+ );
190
+ }
191
+ // The timeout bounds *stall* time, not total wait: while the holder's
192
+ // heartbeat visibly advances, it is healthy and working through its
193
+ // backlog — timing out then would crash-loop every waiting instance
194
+ // on exactly the deploys (a large first backlog) that take longest.
195
+ // Only a holder that stops renewing runs the deadline down.
196
+ const holderLockedAt = error.context?.holder?.lockedAt?.getTime?.();
197
+ const holderAdvanced =
198
+ holderLockedAt !== undefined &&
199
+ lastHolderLockedAt !== undefined &&
200
+ holderLockedAt > lastHolderLockedAt;
201
+ if (deadline === undefined || holderAdvanced) {
202
+ deadline = Date.now() + budget;
203
+ }
204
+ if (holderLockedAt !== undefined) lastHolderLockedAt = holderLockedAt;
205
+ const nextDelay = backoffDelay(lockPollIntervalMs, attempts, budget);
206
+ if (Date.now() + nextDelay > deadline) {
207
+ settle('timeout');
208
+ if (error instanceof MigronautError) {
209
+ // Copy-on-write: the context may be shared with whoever threw it.
210
+ error.context = { ...error.context, attempts, waitedMs: waitedMs(), timedOut: true };
211
+ }
212
+ throw error;
213
+ }
214
+ if (!waited) {
215
+ logger?.info(
216
+ error instanceof LockAlreadyHeldError
217
+ ? 'Migration lock held by another process — waiting for it to release…'
218
+ : `Waiting: ${errorText(error)}`,
219
+ );
220
+ }
221
+ waited = true;
222
+ logger?.debug('Migration lock still held — retrying', {
223
+ attempts,
224
+ waitedMs: waitedMs(),
225
+ nextDelayMs: nextDelay,
226
+ });
227
+ try {
228
+ onWait?.({
229
+ attempts,
230
+ waitedMs: waitedMs(),
231
+ nextDelayMs: nextDelay,
232
+ holder: error.context?.holder,
233
+ code: error.code,
234
+ });
235
+ } catch {
236
+ // Progress reporting must never end the wait.
237
+ }
238
+ try {
239
+ await delay(nextDelay, undefined, signal ? { signal } : undefined);
240
+ } catch (delayError) {
241
+ if (signal?.aborted) {
242
+ settle('aborted');
243
+ throw abortedError(signal, { attempts, waitedMs: waitedMs() });
244
+ }
245
+ throw delayError;
246
+ }
247
+ }
248
+ }
249
+ }
250
+
251
+ module.exports = {
252
+ DEFAULT_LOCK_POLL_INTERVAL_MS,
253
+ DEFAULT_LOCK_WAIT_TIMEOUT_MS,
254
+ MAX_LOCK_POLL_INTERVAL_MS,
255
+ assertLockWaitOptions,
256
+ backoffDelay,
257
+ jitteredDelay,
258
+ waitBudget,
259
+ withLockWait,
260
+ };
package/src/core/lock.js CHANGED
@@ -1,4 +1,3 @@
1
- const { randomUUID } = require('node:crypto');
2
1
  const os = require('node:os');
3
2
  const {
4
3
  LockAlreadyHeldError,
@@ -6,6 +5,7 @@ const {
6
5
  LockReleaseFailedError,
7
6
  } = require('../errors/index.js');
8
7
  const { errorText } = require('../utils/error.js');
8
+ const { randomId } = require('../utils/id.js');
9
9
  const { safeUsername } = require('../utils/user.js');
10
10
 
11
11
  /** Fixed `_id` of the singleton lock document */
@@ -17,13 +17,23 @@ function isDuplicateKeyError(error) {
17
17
  }
18
18
 
19
19
  /**
20
- * Map a raw lock document to the public LockInfo shape. Strips internal fields —
21
- * most importantly the `owner` token, which proves lock ownership and must never
22
- * leak into error context or CLI output.
20
+ * Map a raw lock document to the public LockInfo shape. The `nonce` — what
21
+ * proves ownership — never leaves this module. The `owner` is the holder's run
22
+ * id, the same value its events, log lines and changelog records carry, so it
23
+ * is reported as `runId`: "which run holds the lock?" is the first question
24
+ * when one is stuck. `ttlMs` is the holder's own TTL, which paces its
25
+ * heartbeat (written since 2.1; absent from an older holder's document).
23
26
  */
24
27
  function toLockInfo(doc) {
25
28
  if (!doc) return null;
26
- return { lockedAt: doc.lockedAt, pid: doc.pid, host: doc.host, executedBy: doc.executedBy };
29
+ return {
30
+ lockedAt: doc.lockedAt,
31
+ pid: doc.pid,
32
+ host: doc.host,
33
+ executedBy: doc.executedBy,
34
+ ...(typeof doc.owner === 'string' ? { runId: doc.owner } : {}),
35
+ ...(typeof doc.ttlMs === 'number' ? { ttlMs: doc.ttlMs } : {}),
36
+ };
27
37
  }
28
38
 
29
39
  /**
@@ -35,8 +45,15 @@ class MigrationLock {
35
45
  #db;
36
46
  #collectionName;
37
47
  #ttlSeconds;
38
- /** Token proving this instance is the current holder; set on acquire, cleared on release */
48
+ /** Token naming this instance as the current holder; set on acquire, cleared on release */
39
49
  #owner;
50
+ /**
51
+ * Minted here on every acquire and matched alongside `owner`. The owner token
52
+ * is the run id, whose format — and therefore whose uniqueness — a user's
53
+ * `generateId` decides; the nonce is what keeps two holders apart even when
54
+ * that generator hands both the same id.
55
+ */
56
+ #nonce;
40
57
 
41
58
  constructor(db, collectionName, ttlSeconds) {
42
59
  this.#db = db;
@@ -68,7 +85,8 @@ class MigrationLock {
68
85
  const collection = this.#db.collection(this.#collectionName);
69
86
  // The caller may supply the run id so the lock document, the changelog
70
87
  // records and the log lines of one run all carry the same token.
71
- const owner = token ?? randomUUID();
88
+ const owner = token ?? randomId();
89
+ const nonce = randomId();
72
90
  // The replacement document for the taken branch. $literal guards the
73
91
  // strings: a pipeline expression would otherwise interpret a leading `$`
74
92
  // in a value as a field path.
@@ -79,6 +97,8 @@ class MigrationLock {
79
97
  host: { $literal: os.hostname() },
80
98
  executedBy: { $literal: safeUsername() },
81
99
  owner: { $literal: owner },
100
+ nonce: { $literal: nonce },
101
+ ttlMs: { $literal: this.ttlMs },
82
102
  };
83
103
 
84
104
  let result;
@@ -114,6 +134,7 @@ class MigrationLock {
114
134
  const holder = await collection.findOne({ _id: LOCK_ID });
115
135
  throw new LockAlreadyHeldError('Migration lock is already held', {
116
136
  holder: toLockInfo(holder) ?? undefined,
137
+ ttlMs: this.ttlMs,
117
138
  });
118
139
  }
119
140
  throw error;
@@ -125,27 +146,31 @@ class MigrationLock {
125
146
  // read-back halves that path's round trips.
126
147
  if (result.upsertedCount === 1) {
127
148
  this.#owner = owner;
149
+ this.#nonce = nonce;
128
150
  return;
129
151
  }
130
152
 
131
153
  // Confirm we are the holder. A fresh lock left the document untouched, and
132
154
  // if two processes raced to reclaim the same stale lock only the last
133
- // writer's `owner` wins; either way the loser reads a different token here
134
- // and backs off instead of running concurrently.
155
+ // writer's document wins; either way the loser reads a different one here
156
+ // and backs off instead of running concurrently. The nonce is what makes
157
+ // that hold when both carry the same owner token.
135
158
  const current = await collection.findOne({ _id: LOCK_ID });
136
- if (!current || current.owner !== owner) {
159
+ if (!current || current.owner !== owner || current.nonce !== nonce) {
137
160
  throw new LockAlreadyHeldError('Migration lock is already held', {
138
161
  holder: toLockInfo(current) ?? undefined,
162
+ ttlMs: this.ttlMs,
139
163
  });
140
164
  }
141
165
  this.#owner = owner;
166
+ this.#nonce = nonce;
142
167
  }
143
168
 
144
169
  /**
145
170
  * Refresh `lockedAt` so a long-running migration's lock never goes stale and
146
- * gets reclaimed mid-run. Scoped to our `owner` token, so it is a no-op if the
147
- * lock was already lost. Server time, for the same reason as acquire().
148
- * Returns true while we still hold the lock.
171
+ * gets reclaimed mid-run. Scoped to our `owner` token and nonce, so it is a
172
+ * no-op if the lock was already lost. Server time, for the same reason as
173
+ * acquire(). Returns true while we still hold the lock.
149
174
  */
150
175
  async renew() {
151
176
  if (!this.#owner) {
@@ -153,7 +178,9 @@ class MigrationLock {
153
178
  }
154
179
  const result = await this.#db
155
180
  .collection(this.#collectionName)
156
- .updateOne({ _id: LOCK_ID, owner: this.#owner }, [{ $set: { lockedAt: '$$NOW' } }]);
181
+ .updateOne({ _id: LOCK_ID, owner: this.#owner, nonce: this.#nonce }, [
182
+ { $set: { lockedAt: '$$NOW' } },
183
+ ]);
157
184
  return result.matchedCount === 1;
158
185
  }
159
186
 
@@ -176,7 +203,8 @@ class MigrationLock {
176
203
 
177
204
  /**
178
205
  * Release the lock by deleting the lock document. Scoped to our `owner` token
179
- * so we never delete a lock that has since been reclaimed by another process.
206
+ * and nonce so we never delete a lock that has since been reclaimed by another
207
+ * process — not even one that was handed the same owner token.
180
208
  * With no token held this is a no-op — an unscoped delete here would be
181
209
  * `forceRelease()` without its deliberate opt-in, and a future caller
182
210
  * releasing twice (or before acquiring) must not silently steal a peer's
@@ -185,10 +213,11 @@ class MigrationLock {
185
213
  */
186
214
  async release() {
187
215
  if (!this.#owner) return;
188
- const filter = { _id: LOCK_ID, owner: this.#owner };
216
+ const filter = { _id: LOCK_ID, owner: this.#owner, nonce: this.#nonce };
189
217
  try {
190
218
  await this.#db.collection(this.#collectionName).deleteOne(filter);
191
219
  this.#owner = undefined;
220
+ this.#nonce = undefined;
192
221
  } catch (error) {
193
222
  throw new LockReleaseFailedError(
194
223
  'Failed to release migration lock',