@alexify/migronaut 1.0.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +409 -1
- package/README.md +248 -24
- package/bin/migronaut.js +11 -3
- package/bullmq.d.ts +845 -0
- package/bullmq.js +1 -0
- package/index.d.ts +757 -29
- package/migronaut.schema.json +191 -1
- package/package.json +27 -6
- package/src/bullmq/index.js +55 -0
- package/src/bullmq/jobs.js +454 -0
- package/src/bullmq/processor.js +608 -0
- package/src/bullmq/producer.js +424 -0
- package/src/bullmq/service.js +653 -0
- package/src/bullmq/wait.js +124 -0
- package/src/cli/args.js +12 -2
- package/src/cli/commands/baseline.js +45 -0
- package/src/cli/commands/converge.js +160 -0
- package/src/cli/commands/down.js +2 -0
- package/src/cli/commands/lock.js +2 -1
- package/src/cli/commands/redo.js +8 -1
- package/src/cli/commands/unlock.js +12 -2
- package/src/cli/commands/up.js +14 -1
- package/src/cli/exit-codes.js +10 -2
- package/src/cli/index.js +4 -0
- package/src/cli/shared.js +29 -7
- package/src/cli/table.js +105 -0
- package/src/core/audit.js +17 -3
- package/src/core/baseline.js +80 -0
- package/src/core/changelog.js +140 -24
- package/src/core/collections.js +372 -0
- package/src/core/config.js +125 -27
- package/src/core/converge-log.js +47 -0
- package/src/core/converge-plan.js +483 -0
- package/src/core/converge.js +867 -0
- package/src/core/import-runner.js +34 -6
- package/src/core/import.js +14 -7
- package/src/core/index-spec.js +496 -0
- package/src/core/lock-wait.js +260 -0
- package/src/core/lock.js +71 -20
- package/src/core/migrator.js +805 -304
- package/src/core/options.js +251 -0
- package/src/core/run-recorder.js +157 -0
- package/src/core/run.js +71 -71
- package/src/core/runner.js +70 -20
- package/src/core/sequence.js +134 -0
- package/src/errors/index.js +71 -1
- package/src/index.js +16 -0
- package/src/utils/actor.js +48 -0
- package/src/utils/canonical.js +179 -0
- package/src/utils/collection-name.js +21 -0
- package/src/utils/error.js +18 -1
- package/src/utils/id.js +77 -0
- package/src/utils/loader.js +39 -21
- package/src/utils/logger.js +30 -12
- package/src/utils/migration-name.js +32 -0
- package/src/utils/redact.js +57 -4
- package/src/utils/sanitize.js +8 -3
- package/src/utils/telemetry.js +393 -0
- package/src/utils/template.js +60 -12
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
const { setTimeout: delay } = require('node:timers/promises');
|
|
2
|
+
const {
|
|
3
|
+
ConfigInvalidError,
|
|
4
|
+
LockAlreadyHeldError,
|
|
5
|
+
MigronautError,
|
|
6
|
+
RunAbortedError,
|
|
7
|
+
} = require('../errors/index.js');
|
|
8
|
+
const { errorText } = require('../utils/error.js');
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Longer than the default 60s lock TTL on purpose: with a shorter budget, a peer
|
|
12
|
+
* migration that outlives it makes every waiting instance fail to boot, even
|
|
13
|
+
* though the peer is healthy and still holding a valid lock.
|
|
14
|
+
*/
|
|
15
|
+
const DEFAULT_LOCK_WAIT_TIMEOUT_MS = 90_000;
|
|
16
|
+
const DEFAULT_LOCK_POLL_INTERVAL_MS = 500;
|
|
17
|
+
/**
|
|
18
|
+
* The longest a waiter sleeps between two polls, however long it has waited.
|
|
19
|
+
* Polls back off from `lockPollIntervalMs` up to this, so a hundred instances
|
|
20
|
+
* waiting out a long deploy do not hammer the one lock document — while a
|
|
21
|
+
* released lock is still noticed within a few seconds.
|
|
22
|
+
*/
|
|
23
|
+
const MAX_LOCK_POLL_INTERVAL_MS = 5000;
|
|
24
|
+
/** The longest delay a Node timer honours — anything above it fires after 1ms */
|
|
25
|
+
const MAX_TIMER_MS = 2 ** 31 - 1;
|
|
26
|
+
/** ±25% jitter so N instances booting together stop polling in lockstep */
|
|
27
|
+
const POLL_JITTER_RATIO = 0.25;
|
|
28
|
+
/**
|
|
29
|
+
* How much longer than the holder's TTL a waiter is patient by default: the
|
|
30
|
+
* holder's heartbeat moves `lockedAt` every TTL/2, and a crashed holder's lock
|
|
31
|
+
* is only reclaimable after a full TTL — a default below that would give up
|
|
32
|
+
* on healthy holders, or before a dead one could be replaced.
|
|
33
|
+
*/
|
|
34
|
+
const TTL_PATIENCE_FACTOR = 1.5;
|
|
35
|
+
|
|
36
|
+
function jitteredDelay(baseMs) {
|
|
37
|
+
const spread = baseMs * POLL_JITTER_RATIO;
|
|
38
|
+
return Math.max(1, Math.round(baseMs - spread + Math.random() * spread * 2));
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Polls that must fit in one wait budget: enough to see a live holder's heartbeat move */
|
|
42
|
+
const POLLS_PER_BUDGET = 4;
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* The sleep before poll number `attempt` (1-based): doubling from `baseMs`,
|
|
46
|
+
* jittered, and capped — at `MAX_LOCK_POLL_INTERVAL_MS`, and at a quarter of
|
|
47
|
+
* the wait budget, so a long sleep can never outlast the budget and make a
|
|
48
|
+
* live holder look stalled. Never below `baseMs`.
|
|
49
|
+
*/
|
|
50
|
+
function backoffDelay(baseMs, attempt, budgetMs = Number.POSITIVE_INFINITY) {
|
|
51
|
+
const cap = Math.max(baseMs, Math.min(MAX_LOCK_POLL_INTERVAL_MS, budgetMs / POLLS_PER_BUDGET));
|
|
52
|
+
const grown = Math.min(cap, baseMs * 2 ** Math.min(Math.max(attempt - 1, 0), 30));
|
|
53
|
+
return Math.min(jitteredDelay(grown), MAX_TIMER_MS);
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Validate the wait options. A NaN here (or any non-positive value) disables
|
|
58
|
+
* every deadline comparison in the loop — `NaN > deadline` is always false —
|
|
59
|
+
* turning the wait into an unbounded retry storm against the lock collection
|
|
60
|
+
* that never returns and never surfaces the real error. An interval above the
|
|
61
|
+
* largest timer would fire after 1ms: the same storm. `lockWaitTimeoutMs` may
|
|
62
|
+
* be left out — the default then follows the holder's TTL.
|
|
63
|
+
*/
|
|
64
|
+
function assertLockWaitOptions({
|
|
65
|
+
onLockHeld,
|
|
66
|
+
lockWaitTimeoutMs,
|
|
67
|
+
lockPollIntervalMs = DEFAULT_LOCK_POLL_INTERVAL_MS,
|
|
68
|
+
} = {}) {
|
|
69
|
+
if (onLockHeld !== undefined && onLockHeld !== 'wait' && onLockHeld !== 'throw') {
|
|
70
|
+
throw new ConfigInvalidError("onLockHeld must be 'wait' or 'throw'", { onLockHeld });
|
|
71
|
+
}
|
|
72
|
+
if (
|
|
73
|
+
lockWaitTimeoutMs !== undefined &&
|
|
74
|
+
(!Number.isFinite(lockWaitTimeoutMs) || lockWaitTimeoutMs <= 0)
|
|
75
|
+
) {
|
|
76
|
+
throw new ConfigInvalidError('lockWaitTimeoutMs must be a positive finite number', {
|
|
77
|
+
lockWaitTimeoutMs,
|
|
78
|
+
});
|
|
79
|
+
}
|
|
80
|
+
if (
|
|
81
|
+
!Number.isFinite(lockPollIntervalMs) ||
|
|
82
|
+
lockPollIntervalMs <= 0 ||
|
|
83
|
+
lockPollIntervalMs > MAX_TIMER_MS
|
|
84
|
+
) {
|
|
85
|
+
throw new ConfigInvalidError(
|
|
86
|
+
`lockPollIntervalMs must be a positive number of milliseconds, at most ${MAX_TIMER_MS}`,
|
|
87
|
+
{ lockPollIntervalMs },
|
|
88
|
+
);
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** The error an aborted wait rejects with — the signal's own typed reason when it has one */
|
|
93
|
+
function abortedError(signal, extra) {
|
|
94
|
+
const reason = signal.reason;
|
|
95
|
+
if (reason instanceof MigronautError) return reason;
|
|
96
|
+
return new RunAbortedError('Lock wait aborted', {
|
|
97
|
+
reason: errorText(reason ?? 'aborted'),
|
|
98
|
+
...extra,
|
|
99
|
+
});
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* The wait budget: the caller's, or — left out — the default, stretched to
|
|
104
|
+
* outlast the holder's heartbeat and its stale-reclaim window. The holder's
|
|
105
|
+
* TTL comes from its lock document (written since 2.1), else this process's.
|
|
106
|
+
*/
|
|
107
|
+
function waitBudget(lockWaitTimeoutMs, error) {
|
|
108
|
+
if (lockWaitTimeoutMs !== undefined) return lockWaitTimeoutMs;
|
|
109
|
+
const ttlMs = error.context?.holder?.ttlMs ?? error.context?.ttlMs;
|
|
110
|
+
return Number.isFinite(ttlMs)
|
|
111
|
+
? Math.max(DEFAULT_LOCK_WAIT_TIMEOUT_MS, Math.round(ttlMs * TTL_PATIENCE_FACTOR))
|
|
112
|
+
: DEFAULT_LOCK_WAIT_TIMEOUT_MS;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Run `attempt()`; while it rejects with LockAlreadyHeldError and `onLockHeld`
|
|
117
|
+
* is `'wait'`, poll again. Resolves `{ result, waited, waitedMs, attempts }`.
|
|
118
|
+
*
|
|
119
|
+
* The one wait-for-the-lock loop, shared by `runMigrations` (instances booting
|
|
120
|
+
* together) and the queue processor (a job that finds a CLI run or a peer
|
|
121
|
+
* worker holding the lock). `signal` aborts the wait *between* attempts — it
|
|
122
|
+
* never interrupts an attempt in progress — and `onWait` fires before each
|
|
123
|
+
* sleep, for callers that report progress. `isTransient(error)` widens what is
|
|
124
|
+
* waited out beyond a held lock (the queue's "an earlier job is still in
|
|
125
|
+
* flight"); `onSettle({ waitedMs, attempts, outcome })` fires once when a wait
|
|
126
|
+
* that happened ends — `'acquired'`, `'timeout'` or `'aborted'`.
|
|
127
|
+
*/
|
|
128
|
+
async function withLockWait(attempt, options = {}) {
|
|
129
|
+
const {
|
|
130
|
+
onLockHeld = 'throw',
|
|
131
|
+
lockWaitTimeoutMs,
|
|
132
|
+
lockPollIntervalMs = DEFAULT_LOCK_POLL_INTERVAL_MS,
|
|
133
|
+
logger,
|
|
134
|
+
signal,
|
|
135
|
+
onWait,
|
|
136
|
+
onSettle,
|
|
137
|
+
isTransient = (error) => error instanceof LockAlreadyHeldError,
|
|
138
|
+
} = options;
|
|
139
|
+
let waited = false;
|
|
140
|
+
let attempts = 0;
|
|
141
|
+
// The clock starts at the first contention, not before the first attempt —
|
|
142
|
+
// otherwise a slow initial attempt eats the whole waiting budget.
|
|
143
|
+
let firstRefusalAt;
|
|
144
|
+
let deadline;
|
|
145
|
+
// The holder's lockedAt from the last refusal: its heartbeat advances it
|
|
146
|
+
// every TTL/2, so a change between polls is proof of a live, progressing
|
|
147
|
+
// peer.
|
|
148
|
+
let lastHolderLockedAt;
|
|
149
|
+
let warnedShortBudget = false;
|
|
150
|
+
const waitedMs = () => (firstRefusalAt === undefined ? 0 : Date.now() - firstRefusalAt);
|
|
151
|
+
const settle = (outcome) => {
|
|
152
|
+
if (!waited) return;
|
|
153
|
+
try {
|
|
154
|
+
onSettle?.({ waitedMs: waitedMs(), attempts, outcome });
|
|
155
|
+
} catch {
|
|
156
|
+
// A reporting callback must never change how the wait ends.
|
|
157
|
+
}
|
|
158
|
+
};
|
|
159
|
+
|
|
160
|
+
for (;;) {
|
|
161
|
+
if (signal?.aborted) {
|
|
162
|
+
settle('aborted');
|
|
163
|
+
throw abortedError(signal, { attempts, waitedMs: waitedMs() });
|
|
164
|
+
}
|
|
165
|
+
try {
|
|
166
|
+
attempts += 1;
|
|
167
|
+
const result = await attempt(attempts);
|
|
168
|
+
const total = waitedMs();
|
|
169
|
+
settle('acquired');
|
|
170
|
+
return { result, waited, waitedMs: total, attempts };
|
|
171
|
+
} catch (error) {
|
|
172
|
+
if (onLockHeld !== 'wait' || !isTransient(error)) {
|
|
173
|
+
throw error;
|
|
174
|
+
}
|
|
175
|
+
firstRefusalAt ??= Date.now();
|
|
176
|
+
const budget = waitBudget(lockWaitTimeoutMs, error);
|
|
177
|
+
const holderTtlMs = error.context?.holder?.ttlMs ?? error.context?.ttlMs;
|
|
178
|
+
if (
|
|
179
|
+
!warnedShortBudget &&
|
|
180
|
+
lockWaitTimeoutMs !== undefined &&
|
|
181
|
+
Number.isFinite(holderTtlMs) &&
|
|
182
|
+
lockWaitTimeoutMs <= holderTtlMs / 2
|
|
183
|
+
) {
|
|
184
|
+
warnedShortBudget = true;
|
|
185
|
+
logger?.warn(
|
|
186
|
+
`⚠ lockWaitTimeoutMs (${lockWaitTimeoutMs}ms) is no longer than the holder's heartbeat ` +
|
|
187
|
+
`(${holderTtlMs / 2}ms) — a healthy holder may look stalled and the wait give up`,
|
|
188
|
+
{ lockWaitTimeoutMs, holderTtlMs },
|
|
189
|
+
);
|
|
190
|
+
}
|
|
191
|
+
// The timeout bounds *stall* time, not total wait: while the holder's
|
|
192
|
+
// heartbeat visibly advances, it is healthy and working through its
|
|
193
|
+
// backlog — timing out then would crash-loop every waiting instance
|
|
194
|
+
// on exactly the deploys (a large first backlog) that take longest.
|
|
195
|
+
// Only a holder that stops renewing runs the deadline down.
|
|
196
|
+
const holderLockedAt = error.context?.holder?.lockedAt?.getTime?.();
|
|
197
|
+
const holderAdvanced =
|
|
198
|
+
holderLockedAt !== undefined &&
|
|
199
|
+
lastHolderLockedAt !== undefined &&
|
|
200
|
+
holderLockedAt > lastHolderLockedAt;
|
|
201
|
+
if (deadline === undefined || holderAdvanced) {
|
|
202
|
+
deadline = Date.now() + budget;
|
|
203
|
+
}
|
|
204
|
+
if (holderLockedAt !== undefined) lastHolderLockedAt = holderLockedAt;
|
|
205
|
+
const nextDelay = backoffDelay(lockPollIntervalMs, attempts, budget);
|
|
206
|
+
if (Date.now() + nextDelay > deadline) {
|
|
207
|
+
settle('timeout');
|
|
208
|
+
if (error instanceof MigronautError) {
|
|
209
|
+
// Copy-on-write: the context may be shared with whoever threw it.
|
|
210
|
+
error.context = { ...error.context, attempts, waitedMs: waitedMs(), timedOut: true };
|
|
211
|
+
}
|
|
212
|
+
throw error;
|
|
213
|
+
}
|
|
214
|
+
if (!waited) {
|
|
215
|
+
logger?.info(
|
|
216
|
+
error instanceof LockAlreadyHeldError
|
|
217
|
+
? 'Migration lock held by another process — waiting for it to release…'
|
|
218
|
+
: `Waiting: ${errorText(error)}`,
|
|
219
|
+
);
|
|
220
|
+
}
|
|
221
|
+
waited = true;
|
|
222
|
+
logger?.debug('Migration lock still held — retrying', {
|
|
223
|
+
attempts,
|
|
224
|
+
waitedMs: waitedMs(),
|
|
225
|
+
nextDelayMs: nextDelay,
|
|
226
|
+
});
|
|
227
|
+
try {
|
|
228
|
+
onWait?.({
|
|
229
|
+
attempts,
|
|
230
|
+
waitedMs: waitedMs(),
|
|
231
|
+
nextDelayMs: nextDelay,
|
|
232
|
+
holder: error.context?.holder,
|
|
233
|
+
code: error.code,
|
|
234
|
+
});
|
|
235
|
+
} catch {
|
|
236
|
+
// Progress reporting must never end the wait.
|
|
237
|
+
}
|
|
238
|
+
try {
|
|
239
|
+
await delay(nextDelay, undefined, signal ? { signal } : undefined);
|
|
240
|
+
} catch (delayError) {
|
|
241
|
+
if (signal?.aborted) {
|
|
242
|
+
settle('aborted');
|
|
243
|
+
throw abortedError(signal, { attempts, waitedMs: waitedMs() });
|
|
244
|
+
}
|
|
245
|
+
throw delayError;
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
module.exports = {
|
|
252
|
+
DEFAULT_LOCK_POLL_INTERVAL_MS,
|
|
253
|
+
DEFAULT_LOCK_WAIT_TIMEOUT_MS,
|
|
254
|
+
MAX_LOCK_POLL_INTERVAL_MS,
|
|
255
|
+
assertLockWaitOptions,
|
|
256
|
+
backoffDelay,
|
|
257
|
+
jitteredDelay,
|
|
258
|
+
waitBudget,
|
|
259
|
+
withLockWait,
|
|
260
|
+
};
|
package/src/core/lock.js
CHANGED
|
@@ -1,4 +1,3 @@
|
|
|
1
|
-
const { randomUUID } = require('node:crypto');
|
|
2
1
|
const os = require('node:os');
|
|
3
2
|
const {
|
|
4
3
|
LockAlreadyHeldError,
|
|
@@ -6,6 +5,7 @@ const {
|
|
|
6
5
|
LockReleaseFailedError,
|
|
7
6
|
} = require('../errors/index.js');
|
|
8
7
|
const { errorText } = require('../utils/error.js');
|
|
8
|
+
const { randomId } = require('../utils/id.js');
|
|
9
9
|
const { safeUsername } = require('../utils/user.js');
|
|
10
10
|
|
|
11
11
|
/** Fixed `_id` of the singleton lock document */
|
|
@@ -17,13 +17,23 @@ function isDuplicateKeyError(error) {
|
|
|
17
17
|
}
|
|
18
18
|
|
|
19
19
|
/**
|
|
20
|
-
* Map a raw lock document to the public LockInfo shape.
|
|
21
|
-
*
|
|
22
|
-
*
|
|
20
|
+
* Map a raw lock document to the public LockInfo shape. The `nonce` — what
|
|
21
|
+
* proves ownership — never leaves this module. The `owner` is the holder's run
|
|
22
|
+
* id, the same value its events, log lines and changelog records carry, so it
|
|
23
|
+
* is reported as `runId`: "which run holds the lock?" is the first question
|
|
24
|
+
* when one is stuck. `ttlMs` is the holder's own TTL, which paces its
|
|
25
|
+
* heartbeat (written since 2.1; absent from an older holder's document).
|
|
23
26
|
*/
|
|
24
27
|
function toLockInfo(doc) {
|
|
25
28
|
if (!doc) return null;
|
|
26
|
-
return {
|
|
29
|
+
return {
|
|
30
|
+
lockedAt: doc.lockedAt,
|
|
31
|
+
pid: doc.pid,
|
|
32
|
+
host: doc.host,
|
|
33
|
+
executedBy: doc.executedBy,
|
|
34
|
+
...(typeof doc.owner === 'string' ? { runId: doc.owner } : {}),
|
|
35
|
+
...(typeof doc.ttlMs === 'number' ? { ttlMs: doc.ttlMs } : {}),
|
|
36
|
+
};
|
|
27
37
|
}
|
|
28
38
|
|
|
29
39
|
/**
|
|
@@ -35,8 +45,15 @@ class MigrationLock {
|
|
|
35
45
|
#db;
|
|
36
46
|
#collectionName;
|
|
37
47
|
#ttlSeconds;
|
|
38
|
-
/** Token
|
|
48
|
+
/** Token naming this instance as the current holder; set on acquire, cleared on release */
|
|
39
49
|
#owner;
|
|
50
|
+
/**
|
|
51
|
+
* Minted here on every acquire and matched alongside `owner`. The owner token
|
|
52
|
+
* is the run id, whose format — and therefore whose uniqueness — a user's
|
|
53
|
+
* `generateId` decides; the nonce is what keeps two holders apart even when
|
|
54
|
+
* that generator hands both the same id.
|
|
55
|
+
*/
|
|
56
|
+
#nonce;
|
|
40
57
|
|
|
41
58
|
constructor(db, collectionName, ttlSeconds) {
|
|
42
59
|
this.#db = db;
|
|
@@ -68,7 +85,8 @@ class MigrationLock {
|
|
|
68
85
|
const collection = this.#db.collection(this.#collectionName);
|
|
69
86
|
// The caller may supply the run id so the lock document, the changelog
|
|
70
87
|
// records and the log lines of one run all carry the same token.
|
|
71
|
-
const owner = token ??
|
|
88
|
+
const owner = token ?? randomId();
|
|
89
|
+
const nonce = randomId();
|
|
72
90
|
// The replacement document for the taken branch. $literal guards the
|
|
73
91
|
// strings: a pipeline expression would otherwise interpret a leading `$`
|
|
74
92
|
// in a value as a field path.
|
|
@@ -79,15 +97,18 @@ class MigrationLock {
|
|
|
79
97
|
host: { $literal: os.hostname() },
|
|
80
98
|
executedBy: { $literal: safeUsername() },
|
|
81
99
|
owner: { $literal: owner },
|
|
100
|
+
nonce: { $literal: nonce },
|
|
101
|
+
ttlMs: { $literal: this.ttlMs },
|
|
82
102
|
};
|
|
83
103
|
|
|
104
|
+
let result;
|
|
84
105
|
try {
|
|
85
106
|
// Upsert on plain `_id` — upserts reject `$expr` filters (server error
|
|
86
107
|
// 224), so the staleness decision lives in the pipeline instead, still
|
|
87
108
|
// in server time: take the lock when no `lockedAt` exists (fresh insert)
|
|
88
109
|
// or the holder is stale; otherwise keep the current document untouched.
|
|
89
110
|
// The read-back below tells those outcomes apart.
|
|
90
|
-
await collection.updateOne(
|
|
111
|
+
result = await collection.updateOne(
|
|
91
112
|
{ _id: LOCK_ID },
|
|
92
113
|
[
|
|
93
114
|
{
|
|
@@ -113,29 +134,43 @@ class MigrationLock {
|
|
|
113
134
|
const holder = await collection.findOne({ _id: LOCK_ID });
|
|
114
135
|
throw new LockAlreadyHeldError('Migration lock is already held', {
|
|
115
136
|
holder: toLockInfo(holder) ?? undefined,
|
|
137
|
+
ttlMs: this.ttlMs,
|
|
116
138
|
});
|
|
117
139
|
}
|
|
118
140
|
throw error;
|
|
119
141
|
}
|
|
120
142
|
|
|
143
|
+
// A fresh upsert-insert on the unique `_id` already proves ownership — no
|
|
144
|
+
// other outcome can insert — and it is the common uncontended path, since
|
|
145
|
+
// release() deletes the document after every clean run. Skipping the
|
|
146
|
+
// read-back halves that path's round trips.
|
|
147
|
+
if (result.upsertedCount === 1) {
|
|
148
|
+
this.#owner = owner;
|
|
149
|
+
this.#nonce = nonce;
|
|
150
|
+
return;
|
|
151
|
+
}
|
|
152
|
+
|
|
121
153
|
// Confirm we are the holder. A fresh lock left the document untouched, and
|
|
122
154
|
// if two processes raced to reclaim the same stale lock only the last
|
|
123
|
-
// writer's
|
|
124
|
-
// and backs off instead of running concurrently.
|
|
155
|
+
// writer's document wins; either way the loser reads a different one here
|
|
156
|
+
// and backs off instead of running concurrently. The nonce is what makes
|
|
157
|
+
// that hold when both carry the same owner token.
|
|
125
158
|
const current = await collection.findOne({ _id: LOCK_ID });
|
|
126
|
-
if (!current || current.owner !== owner) {
|
|
159
|
+
if (!current || current.owner !== owner || current.nonce !== nonce) {
|
|
127
160
|
throw new LockAlreadyHeldError('Migration lock is already held', {
|
|
128
161
|
holder: toLockInfo(current) ?? undefined,
|
|
162
|
+
ttlMs: this.ttlMs,
|
|
129
163
|
});
|
|
130
164
|
}
|
|
131
165
|
this.#owner = owner;
|
|
166
|
+
this.#nonce = nonce;
|
|
132
167
|
}
|
|
133
168
|
|
|
134
169
|
/**
|
|
135
170
|
* Refresh `lockedAt` so a long-running migration's lock never goes stale and
|
|
136
|
-
* gets reclaimed mid-run. Scoped to our `owner` token, so it is a
|
|
137
|
-
* lock was already lost. Server time, for the same reason as
|
|
138
|
-
* Returns true while we still hold the lock.
|
|
171
|
+
* gets reclaimed mid-run. Scoped to our `owner` token and nonce, so it is a
|
|
172
|
+
* no-op if the lock was already lost. Server time, for the same reason as
|
|
173
|
+
* acquire(). Returns true while we still hold the lock.
|
|
139
174
|
*/
|
|
140
175
|
async renew() {
|
|
141
176
|
if (!this.#owner) {
|
|
@@ -143,7 +178,9 @@ class MigrationLock {
|
|
|
143
178
|
}
|
|
144
179
|
const result = await this.#db
|
|
145
180
|
.collection(this.#collectionName)
|
|
146
|
-
.updateOne({ _id: LOCK_ID, owner: this.#owner
|
|
181
|
+
.updateOne({ _id: LOCK_ID, owner: this.#owner, nonce: this.#nonce }, [
|
|
182
|
+
{ $set: { lockedAt: '$$NOW' } },
|
|
183
|
+
]);
|
|
147
184
|
return result.matchedCount === 1;
|
|
148
185
|
}
|
|
149
186
|
|
|
@@ -166,15 +203,21 @@ class MigrationLock {
|
|
|
166
203
|
|
|
167
204
|
/**
|
|
168
205
|
* Release the lock by deleting the lock document. Scoped to our `owner` token
|
|
169
|
-
*
|
|
170
|
-
*
|
|
206
|
+
* and nonce so we never delete a lock that has since been reclaimed by another
|
|
207
|
+
* process — not even one that was handed the same owner token.
|
|
208
|
+
* With no token held this is a no-op — an unscoped delete here would be
|
|
209
|
+
* `forceRelease()` without its deliberate opt-in, and a future caller
|
|
210
|
+
* releasing twice (or before acquiring) must not silently steal a peer's
|
|
211
|
+
* live lock.
|
|
171
212
|
* @throws {LockReleaseFailedError} when the delete operation fails
|
|
172
213
|
*/
|
|
173
214
|
async release() {
|
|
174
|
-
|
|
215
|
+
if (!this.#owner) return;
|
|
216
|
+
const filter = { _id: LOCK_ID, owner: this.#owner, nonce: this.#nonce };
|
|
175
217
|
try {
|
|
176
218
|
await this.#db.collection(this.#collectionName).deleteOne(filter);
|
|
177
219
|
this.#owner = undefined;
|
|
220
|
+
this.#nonce = undefined;
|
|
178
221
|
} catch (error) {
|
|
179
222
|
throw new LockReleaseFailedError(
|
|
180
223
|
'Failed to release migration lock',
|
|
@@ -332,8 +375,16 @@ async function runWithLock(lock, options, fn) {
|
|
|
332
375
|
} catch (releaseError) {
|
|
333
376
|
// Never let a release failure replace the reason the run failed: that would
|
|
334
377
|
// report "Failed to release migration lock" instead of the actual migration
|
|
335
|
-
// error. When the run succeeded, the release failure is the only news
|
|
336
|
-
|
|
378
|
+
// error. When the run succeeded, the release failure is the only news —
|
|
379
|
+
// but the migrations DID apply, and that list must survive onto the error,
|
|
380
|
+
// or "everything applied, only the lock cleanup failed" (the lock document
|
|
381
|
+
// self-heals via its TTL) is indistinguishable from a failed run.
|
|
382
|
+
if (!failed) {
|
|
383
|
+
if (releaseError instanceof LockReleaseFailedError && Array.isArray(result)) {
|
|
384
|
+
releaseError.context = { ...releaseError.context, results: [...result] };
|
|
385
|
+
}
|
|
386
|
+
throw releaseError;
|
|
387
|
+
}
|
|
337
388
|
const message = errorText(releaseError);
|
|
338
389
|
options.logger.warn(`⚠ Failed to release the migration lock: ${message}`, {
|
|
339
390
|
event: 'lock:release-failed',
|