@alexify/migronaut 2.0.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +320 -0
- package/README.md +208 -6
- package/bullmq.d.ts +845 -0
- package/bullmq.js +1 -0
- package/index.d.ts +634 -18
- package/migronaut.schema.json +182 -1
- package/package.json +21 -5
- package/src/bullmq/index.js +55 -0
- package/src/bullmq/jobs.js +454 -0
- package/src/bullmq/processor.js +608 -0
- package/src/bullmq/producer.js +424 -0
- package/src/bullmq/service.js +653 -0
- package/src/bullmq/wait.js +124 -0
- package/src/cli/args.js +12 -2
- package/src/cli/commands/converge.js +160 -0
- package/src/cli/commands/down.js +2 -0
- package/src/cli/commands/lock.js +2 -1
- package/src/cli/commands/redo.js +8 -1
- package/src/cli/commands/up.js +14 -1
- package/src/cli/exit-codes.js +9 -2
- package/src/cli/index.js +2 -0
- package/src/cli/shared.js +14 -4
- package/src/cli/table.js +105 -0
- package/src/core/changelog.js +71 -6
- package/src/core/collections.js +372 -0
- package/src/core/config.js +100 -25
- package/src/core/converge-log.js +47 -0
- package/src/core/converge-plan.js +483 -0
- package/src/core/converge.js +867 -0
- package/src/core/index-spec.js +496 -0
- package/src/core/lock-wait.js +260 -0
- package/src/core/lock.js +45 -16
- package/src/core/migrator.js +563 -283
- package/src/core/options.js +251 -0
- package/src/core/run-recorder.js +157 -0
- package/src/core/run.js +58 -90
- package/src/core/sequence.js +134 -0
- package/src/errors/index.js +56 -0
- package/src/index.js +8 -0
- package/src/utils/actor.js +48 -0
- package/src/utils/canonical.js +179 -0
- package/src/utils/collection-name.js +21 -0
- package/src/utils/error.js +18 -1
- package/src/utils/id.js +77 -0
- package/src/utils/loader.js +39 -21
- package/src/utils/migration-name.js +32 -0
- package/src/utils/redact.js +21 -1
- package/src/utils/telemetry.js +393 -0
- package/src/utils/template.js +36 -2
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
const { setTimeout: delay } = require('node:timers/promises');
|
|
2
|
+
const {
|
|
3
|
+
ConfigInvalidError,
|
|
4
|
+
LockAlreadyHeldError,
|
|
5
|
+
MigronautError,
|
|
6
|
+
RunAbortedError,
|
|
7
|
+
} = require('../errors/index.js');
|
|
8
|
+
const { errorText } = require('../utils/error.js');
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Longer than the default 60s lock TTL on purpose: with a shorter budget, a peer
|
|
12
|
+
* migration that outlives it makes every waiting instance fail to boot, even
|
|
13
|
+
* though the peer is healthy and still holding a valid lock.
|
|
14
|
+
*/
|
|
15
|
+
const DEFAULT_LOCK_WAIT_TIMEOUT_MS = 90_000;
|
|
16
|
+
const DEFAULT_LOCK_POLL_INTERVAL_MS = 500;
|
|
17
|
+
/**
|
|
18
|
+
* The longest a waiter sleeps between two polls, however long it has waited.
|
|
19
|
+
* Polls back off from `lockPollIntervalMs` up to this, so a hundred instances
|
|
20
|
+
* waiting out a long deploy do not hammer the one lock document — while a
|
|
21
|
+
* released lock is still noticed within a few seconds.
|
|
22
|
+
*/
|
|
23
|
+
const MAX_LOCK_POLL_INTERVAL_MS = 5000;
|
|
24
|
+
/** The longest delay a Node timer honours — anything above it fires after 1ms */
|
|
25
|
+
const MAX_TIMER_MS = 2 ** 31 - 1;
|
|
26
|
+
/** ±25% jitter so N instances booting together stop polling in lockstep */
|
|
27
|
+
const POLL_JITTER_RATIO = 0.25;
|
|
28
|
+
/**
|
|
29
|
+
* How much longer than the holder's TTL a waiter is patient by default: the
|
|
30
|
+
* holder's heartbeat moves `lockedAt` every TTL/2, and a crashed holder's lock
|
|
31
|
+
* is only reclaimable after a full TTL — a default below that would give up
|
|
32
|
+
* on healthy holders, or before a dead one could be replaced.
|
|
33
|
+
*/
|
|
34
|
+
const TTL_PATIENCE_FACTOR = 1.5;
|
|
35
|
+
|
|
36
|
+
function jitteredDelay(baseMs) {
|
|
37
|
+
const spread = baseMs * POLL_JITTER_RATIO;
|
|
38
|
+
return Math.max(1, Math.round(baseMs - spread + Math.random() * spread * 2));
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Polls that must fit in one wait budget: enough to see a live holder's heartbeat move */
|
|
42
|
+
const POLLS_PER_BUDGET = 4;
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* The sleep before poll number `attempt` (1-based): doubling from `baseMs`,
|
|
46
|
+
* jittered, and capped — at `MAX_LOCK_POLL_INTERVAL_MS`, and at a quarter of
|
|
47
|
+
* the wait budget, so a long sleep can never outlast the budget and make a
|
|
48
|
+
* live holder look stalled. Never below `baseMs`.
|
|
49
|
+
*/
|
|
50
|
+
function backoffDelay(baseMs, attempt, budgetMs = Number.POSITIVE_INFINITY) {
|
|
51
|
+
const cap = Math.max(baseMs, Math.min(MAX_LOCK_POLL_INTERVAL_MS, budgetMs / POLLS_PER_BUDGET));
|
|
52
|
+
const grown = Math.min(cap, baseMs * 2 ** Math.min(Math.max(attempt - 1, 0), 30));
|
|
53
|
+
return Math.min(jitteredDelay(grown), MAX_TIMER_MS);
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Validate the wait options. A NaN here (or any non-positive value) disables
|
|
58
|
+
* every deadline comparison in the loop — `NaN > deadline` is always false —
|
|
59
|
+
* turning the wait into an unbounded retry storm against the lock collection
|
|
60
|
+
* that never returns and never surfaces the real error. An interval above the
|
|
61
|
+
* largest timer would fire after 1ms: the same storm. `lockWaitTimeoutMs` may
|
|
62
|
+
* be left out — the default then follows the holder's TTL.
|
|
63
|
+
*/
|
|
64
|
+
function assertLockWaitOptions({
|
|
65
|
+
onLockHeld,
|
|
66
|
+
lockWaitTimeoutMs,
|
|
67
|
+
lockPollIntervalMs = DEFAULT_LOCK_POLL_INTERVAL_MS,
|
|
68
|
+
} = {}) {
|
|
69
|
+
if (onLockHeld !== undefined && onLockHeld !== 'wait' && onLockHeld !== 'throw') {
|
|
70
|
+
throw new ConfigInvalidError("onLockHeld must be 'wait' or 'throw'", { onLockHeld });
|
|
71
|
+
}
|
|
72
|
+
if (
|
|
73
|
+
lockWaitTimeoutMs !== undefined &&
|
|
74
|
+
(!Number.isFinite(lockWaitTimeoutMs) || lockWaitTimeoutMs <= 0)
|
|
75
|
+
) {
|
|
76
|
+
throw new ConfigInvalidError('lockWaitTimeoutMs must be a positive finite number', {
|
|
77
|
+
lockWaitTimeoutMs,
|
|
78
|
+
});
|
|
79
|
+
}
|
|
80
|
+
if (
|
|
81
|
+
!Number.isFinite(lockPollIntervalMs) ||
|
|
82
|
+
lockPollIntervalMs <= 0 ||
|
|
83
|
+
lockPollIntervalMs > MAX_TIMER_MS
|
|
84
|
+
) {
|
|
85
|
+
throw new ConfigInvalidError(
|
|
86
|
+
`lockPollIntervalMs must be a positive number of milliseconds, at most ${MAX_TIMER_MS}`,
|
|
87
|
+
{ lockPollIntervalMs },
|
|
88
|
+
);
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** The error an aborted wait rejects with — the signal's own typed reason when it has one */
|
|
93
|
+
function abortedError(signal, extra) {
|
|
94
|
+
const reason = signal.reason;
|
|
95
|
+
if (reason instanceof MigronautError) return reason;
|
|
96
|
+
return new RunAbortedError('Lock wait aborted', {
|
|
97
|
+
reason: errorText(reason ?? 'aborted'),
|
|
98
|
+
...extra,
|
|
99
|
+
});
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* The wait budget: the caller's, or — left out — the default, stretched to
|
|
104
|
+
* outlast the holder's heartbeat and its stale-reclaim window. The holder's
|
|
105
|
+
* TTL comes from its lock document (written since 2.1), else this process's.
|
|
106
|
+
*/
|
|
107
|
+
function waitBudget(lockWaitTimeoutMs, error) {
|
|
108
|
+
if (lockWaitTimeoutMs !== undefined) return lockWaitTimeoutMs;
|
|
109
|
+
const ttlMs = error.context?.holder?.ttlMs ?? error.context?.ttlMs;
|
|
110
|
+
return Number.isFinite(ttlMs)
|
|
111
|
+
? Math.max(DEFAULT_LOCK_WAIT_TIMEOUT_MS, Math.round(ttlMs * TTL_PATIENCE_FACTOR))
|
|
112
|
+
: DEFAULT_LOCK_WAIT_TIMEOUT_MS;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Run `attempt()`; while it rejects with LockAlreadyHeldError and `onLockHeld`
|
|
117
|
+
* is `'wait'`, poll again. Resolves `{ result, waited, waitedMs, attempts }`.
|
|
118
|
+
*
|
|
119
|
+
* The one wait-for-the-lock loop, shared by `runMigrations` (instances booting
|
|
120
|
+
* together) and the queue processor (a job that finds a CLI run or a peer
|
|
121
|
+
* worker holding the lock). `signal` aborts the wait *between* attempts — it
|
|
122
|
+
* never interrupts an attempt in progress — and `onWait` fires before each
|
|
123
|
+
* sleep, for callers that report progress. `isTransient(error)` widens what is
|
|
124
|
+
* waited out beyond a held lock (the queue's "an earlier job is still in
|
|
125
|
+
* flight"); `onSettle({ waitedMs, attempts, outcome })` fires once when a wait
|
|
126
|
+
* that happened ends — `'acquired'`, `'timeout'` or `'aborted'`.
|
|
127
|
+
*/
|
|
128
|
+
async function withLockWait(attempt, options = {}) {
|
|
129
|
+
const {
|
|
130
|
+
onLockHeld = 'throw',
|
|
131
|
+
lockWaitTimeoutMs,
|
|
132
|
+
lockPollIntervalMs = DEFAULT_LOCK_POLL_INTERVAL_MS,
|
|
133
|
+
logger,
|
|
134
|
+
signal,
|
|
135
|
+
onWait,
|
|
136
|
+
onSettle,
|
|
137
|
+
isTransient = (error) => error instanceof LockAlreadyHeldError,
|
|
138
|
+
} = options;
|
|
139
|
+
let waited = false;
|
|
140
|
+
let attempts = 0;
|
|
141
|
+
// The clock starts at the first contention, not before the first attempt —
|
|
142
|
+
// otherwise a slow initial attempt eats the whole waiting budget.
|
|
143
|
+
let firstRefusalAt;
|
|
144
|
+
let deadline;
|
|
145
|
+
// The holder's lockedAt from the last refusal: its heartbeat advances it
|
|
146
|
+
// every TTL/2, so a change between polls is proof of a live, progressing
|
|
147
|
+
// peer.
|
|
148
|
+
let lastHolderLockedAt;
|
|
149
|
+
let warnedShortBudget = false;
|
|
150
|
+
const waitedMs = () => (firstRefusalAt === undefined ? 0 : Date.now() - firstRefusalAt);
|
|
151
|
+
const settle = (outcome) => {
|
|
152
|
+
if (!waited) return;
|
|
153
|
+
try {
|
|
154
|
+
onSettle?.({ waitedMs: waitedMs(), attempts, outcome });
|
|
155
|
+
} catch {
|
|
156
|
+
// A reporting callback must never change how the wait ends.
|
|
157
|
+
}
|
|
158
|
+
};
|
|
159
|
+
|
|
160
|
+
for (;;) {
|
|
161
|
+
if (signal?.aborted) {
|
|
162
|
+
settle('aborted');
|
|
163
|
+
throw abortedError(signal, { attempts, waitedMs: waitedMs() });
|
|
164
|
+
}
|
|
165
|
+
try {
|
|
166
|
+
attempts += 1;
|
|
167
|
+
const result = await attempt(attempts);
|
|
168
|
+
const total = waitedMs();
|
|
169
|
+
settle('acquired');
|
|
170
|
+
return { result, waited, waitedMs: total, attempts };
|
|
171
|
+
} catch (error) {
|
|
172
|
+
if (onLockHeld !== 'wait' || !isTransient(error)) {
|
|
173
|
+
throw error;
|
|
174
|
+
}
|
|
175
|
+
firstRefusalAt ??= Date.now();
|
|
176
|
+
const budget = waitBudget(lockWaitTimeoutMs, error);
|
|
177
|
+
const holderTtlMs = error.context?.holder?.ttlMs ?? error.context?.ttlMs;
|
|
178
|
+
if (
|
|
179
|
+
!warnedShortBudget &&
|
|
180
|
+
lockWaitTimeoutMs !== undefined &&
|
|
181
|
+
Number.isFinite(holderTtlMs) &&
|
|
182
|
+
lockWaitTimeoutMs <= holderTtlMs / 2
|
|
183
|
+
) {
|
|
184
|
+
warnedShortBudget = true;
|
|
185
|
+
logger?.warn(
|
|
186
|
+
`⚠ lockWaitTimeoutMs (${lockWaitTimeoutMs}ms) is no longer than the holder's heartbeat ` +
|
|
187
|
+
`(${holderTtlMs / 2}ms) — a healthy holder may look stalled and the wait give up`,
|
|
188
|
+
{ lockWaitTimeoutMs, holderTtlMs },
|
|
189
|
+
);
|
|
190
|
+
}
|
|
191
|
+
// The timeout bounds *stall* time, not total wait: while the holder's
|
|
192
|
+
// heartbeat visibly advances, it is healthy and working through its
|
|
193
|
+
// backlog — timing out then would crash-loop every waiting instance
|
|
194
|
+
// on exactly the deploys (a large first backlog) that take longest.
|
|
195
|
+
// Only a holder that stops renewing runs the deadline down.
|
|
196
|
+
const holderLockedAt = error.context?.holder?.lockedAt?.getTime?.();
|
|
197
|
+
const holderAdvanced =
|
|
198
|
+
holderLockedAt !== undefined &&
|
|
199
|
+
lastHolderLockedAt !== undefined &&
|
|
200
|
+
holderLockedAt > lastHolderLockedAt;
|
|
201
|
+
if (deadline === undefined || holderAdvanced) {
|
|
202
|
+
deadline = Date.now() + budget;
|
|
203
|
+
}
|
|
204
|
+
if (holderLockedAt !== undefined) lastHolderLockedAt = holderLockedAt;
|
|
205
|
+
const nextDelay = backoffDelay(lockPollIntervalMs, attempts, budget);
|
|
206
|
+
if (Date.now() + nextDelay > deadline) {
|
|
207
|
+
settle('timeout');
|
|
208
|
+
if (error instanceof MigronautError) {
|
|
209
|
+
// Copy-on-write: the context may be shared with whoever threw it.
|
|
210
|
+
error.context = { ...error.context, attempts, waitedMs: waitedMs(), timedOut: true };
|
|
211
|
+
}
|
|
212
|
+
throw error;
|
|
213
|
+
}
|
|
214
|
+
if (!waited) {
|
|
215
|
+
logger?.info(
|
|
216
|
+
error instanceof LockAlreadyHeldError
|
|
217
|
+
? 'Migration lock held by another process — waiting for it to release…'
|
|
218
|
+
: `Waiting: ${errorText(error)}`,
|
|
219
|
+
);
|
|
220
|
+
}
|
|
221
|
+
waited = true;
|
|
222
|
+
logger?.debug('Migration lock still held — retrying', {
|
|
223
|
+
attempts,
|
|
224
|
+
waitedMs: waitedMs(),
|
|
225
|
+
nextDelayMs: nextDelay,
|
|
226
|
+
});
|
|
227
|
+
try {
|
|
228
|
+
onWait?.({
|
|
229
|
+
attempts,
|
|
230
|
+
waitedMs: waitedMs(),
|
|
231
|
+
nextDelayMs: nextDelay,
|
|
232
|
+
holder: error.context?.holder,
|
|
233
|
+
code: error.code,
|
|
234
|
+
});
|
|
235
|
+
} catch {
|
|
236
|
+
// Progress reporting must never end the wait.
|
|
237
|
+
}
|
|
238
|
+
try {
|
|
239
|
+
await delay(nextDelay, undefined, signal ? { signal } : undefined);
|
|
240
|
+
} catch (delayError) {
|
|
241
|
+
if (signal?.aborted) {
|
|
242
|
+
settle('aborted');
|
|
243
|
+
throw abortedError(signal, { attempts, waitedMs: waitedMs() });
|
|
244
|
+
}
|
|
245
|
+
throw delayError;
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
module.exports = {
|
|
252
|
+
DEFAULT_LOCK_POLL_INTERVAL_MS,
|
|
253
|
+
DEFAULT_LOCK_WAIT_TIMEOUT_MS,
|
|
254
|
+
MAX_LOCK_POLL_INTERVAL_MS,
|
|
255
|
+
assertLockWaitOptions,
|
|
256
|
+
backoffDelay,
|
|
257
|
+
jitteredDelay,
|
|
258
|
+
waitBudget,
|
|
259
|
+
withLockWait,
|
|
260
|
+
};
|
package/src/core/lock.js
CHANGED
|
@@ -1,4 +1,3 @@
|
|
|
1
|
-
const { randomUUID } = require('node:crypto');
|
|
2
1
|
const os = require('node:os');
|
|
3
2
|
const {
|
|
4
3
|
LockAlreadyHeldError,
|
|
@@ -6,6 +5,7 @@ const {
|
|
|
6
5
|
LockReleaseFailedError,
|
|
7
6
|
} = require('../errors/index.js');
|
|
8
7
|
const { errorText } = require('../utils/error.js');
|
|
8
|
+
const { randomId } = require('../utils/id.js');
|
|
9
9
|
const { safeUsername } = require('../utils/user.js');
|
|
10
10
|
|
|
11
11
|
/** Fixed `_id` of the singleton lock document */
|
|
@@ -17,13 +17,23 @@ function isDuplicateKeyError(error) {
|
|
|
17
17
|
}
|
|
18
18
|
|
|
19
19
|
/**
|
|
20
|
-
* Map a raw lock document to the public LockInfo shape.
|
|
21
|
-
*
|
|
22
|
-
*
|
|
20
|
+
* Map a raw lock document to the public LockInfo shape. The `nonce` — what
|
|
21
|
+
* proves ownership — never leaves this module. The `owner` is the holder's run
|
|
22
|
+
* id, the same value its events, log lines and changelog records carry, so it
|
|
23
|
+
* is reported as `runId`: "which run holds the lock?" is the first question
|
|
24
|
+
* when one is stuck. `ttlMs` is the holder's own TTL, which paces its
|
|
25
|
+
* heartbeat (written since 2.1; absent from an older holder's document).
|
|
23
26
|
*/
|
|
24
27
|
function toLockInfo(doc) {
|
|
25
28
|
if (!doc) return null;
|
|
26
|
-
return {
|
|
29
|
+
return {
|
|
30
|
+
lockedAt: doc.lockedAt,
|
|
31
|
+
pid: doc.pid,
|
|
32
|
+
host: doc.host,
|
|
33
|
+
executedBy: doc.executedBy,
|
|
34
|
+
...(typeof doc.owner === 'string' ? { runId: doc.owner } : {}),
|
|
35
|
+
...(typeof doc.ttlMs === 'number' ? { ttlMs: doc.ttlMs } : {}),
|
|
36
|
+
};
|
|
27
37
|
}
|
|
28
38
|
|
|
29
39
|
/**
|
|
@@ -35,8 +45,15 @@ class MigrationLock {
|
|
|
35
45
|
#db;
|
|
36
46
|
#collectionName;
|
|
37
47
|
#ttlSeconds;
|
|
38
|
-
/** Token
|
|
48
|
+
/** Token naming this instance as the current holder; set on acquire, cleared on release */
|
|
39
49
|
#owner;
|
|
50
|
+
/**
|
|
51
|
+
* Minted here on every acquire and matched alongside `owner`. The owner token
|
|
52
|
+
* is the run id, whose format — and therefore whose uniqueness — a user's
|
|
53
|
+
* `generateId` decides; the nonce is what keeps two holders apart even when
|
|
54
|
+
* that generator hands both the same id.
|
|
55
|
+
*/
|
|
56
|
+
#nonce;
|
|
40
57
|
|
|
41
58
|
constructor(db, collectionName, ttlSeconds) {
|
|
42
59
|
this.#db = db;
|
|
@@ -68,7 +85,8 @@ class MigrationLock {
|
|
|
68
85
|
const collection = this.#db.collection(this.#collectionName);
|
|
69
86
|
// The caller may supply the run id so the lock document, the changelog
|
|
70
87
|
// records and the log lines of one run all carry the same token.
|
|
71
|
-
const owner = token ??
|
|
88
|
+
const owner = token ?? randomId();
|
|
89
|
+
const nonce = randomId();
|
|
72
90
|
// The replacement document for the taken branch. $literal guards the
|
|
73
91
|
// strings: a pipeline expression would otherwise interpret a leading `$`
|
|
74
92
|
// in a value as a field path.
|
|
@@ -79,6 +97,8 @@ class MigrationLock {
|
|
|
79
97
|
host: { $literal: os.hostname() },
|
|
80
98
|
executedBy: { $literal: safeUsername() },
|
|
81
99
|
owner: { $literal: owner },
|
|
100
|
+
nonce: { $literal: nonce },
|
|
101
|
+
ttlMs: { $literal: this.ttlMs },
|
|
82
102
|
};
|
|
83
103
|
|
|
84
104
|
let result;
|
|
@@ -114,6 +134,7 @@ class MigrationLock {
|
|
|
114
134
|
const holder = await collection.findOne({ _id: LOCK_ID });
|
|
115
135
|
throw new LockAlreadyHeldError('Migration lock is already held', {
|
|
116
136
|
holder: toLockInfo(holder) ?? undefined,
|
|
137
|
+
ttlMs: this.ttlMs,
|
|
117
138
|
});
|
|
118
139
|
}
|
|
119
140
|
throw error;
|
|
@@ -125,27 +146,31 @@ class MigrationLock {
|
|
|
125
146
|
// read-back halves that path's round trips.
|
|
126
147
|
if (result.upsertedCount === 1) {
|
|
127
148
|
this.#owner = owner;
|
|
149
|
+
this.#nonce = nonce;
|
|
128
150
|
return;
|
|
129
151
|
}
|
|
130
152
|
|
|
131
153
|
// Confirm we are the holder. A fresh lock left the document untouched, and
|
|
132
154
|
// if two processes raced to reclaim the same stale lock only the last
|
|
133
|
-
// writer's
|
|
134
|
-
// and backs off instead of running concurrently.
|
|
155
|
+
// writer's document wins; either way the loser reads a different one here
|
|
156
|
+
// and backs off instead of running concurrently. The nonce is what makes
|
|
157
|
+
// that hold when both carry the same owner token.
|
|
135
158
|
const current = await collection.findOne({ _id: LOCK_ID });
|
|
136
|
-
if (!current || current.owner !== owner) {
|
|
159
|
+
if (!current || current.owner !== owner || current.nonce !== nonce) {
|
|
137
160
|
throw new LockAlreadyHeldError('Migration lock is already held', {
|
|
138
161
|
holder: toLockInfo(current) ?? undefined,
|
|
162
|
+
ttlMs: this.ttlMs,
|
|
139
163
|
});
|
|
140
164
|
}
|
|
141
165
|
this.#owner = owner;
|
|
166
|
+
this.#nonce = nonce;
|
|
142
167
|
}
|
|
143
168
|
|
|
144
169
|
/**
|
|
145
170
|
* Refresh `lockedAt` so a long-running migration's lock never goes stale and
|
|
146
|
-
* gets reclaimed mid-run. Scoped to our `owner` token, so it is a
|
|
147
|
-
* lock was already lost. Server time, for the same reason as
|
|
148
|
-
* Returns true while we still hold the lock.
|
|
171
|
+
* gets reclaimed mid-run. Scoped to our `owner` token and nonce, so it is a
|
|
172
|
+
* no-op if the lock was already lost. Server time, for the same reason as
|
|
173
|
+
* acquire(). Returns true while we still hold the lock.
|
|
149
174
|
*/
|
|
150
175
|
async renew() {
|
|
151
176
|
if (!this.#owner) {
|
|
@@ -153,7 +178,9 @@ class MigrationLock {
|
|
|
153
178
|
}
|
|
154
179
|
const result = await this.#db
|
|
155
180
|
.collection(this.#collectionName)
|
|
156
|
-
.updateOne({ _id: LOCK_ID, owner: this.#owner
|
|
181
|
+
.updateOne({ _id: LOCK_ID, owner: this.#owner, nonce: this.#nonce }, [
|
|
182
|
+
{ $set: { lockedAt: '$$NOW' } },
|
|
183
|
+
]);
|
|
157
184
|
return result.matchedCount === 1;
|
|
158
185
|
}
|
|
159
186
|
|
|
@@ -176,7 +203,8 @@ class MigrationLock {
|
|
|
176
203
|
|
|
177
204
|
/**
|
|
178
205
|
* Release the lock by deleting the lock document. Scoped to our `owner` token
|
|
179
|
-
* so we never delete a lock that has since been reclaimed by another
|
|
206
|
+
* and nonce so we never delete a lock that has since been reclaimed by another
|
|
207
|
+
* process — not even one that was handed the same owner token.
|
|
180
208
|
* With no token held this is a no-op — an unscoped delete here would be
|
|
181
209
|
* `forceRelease()` without its deliberate opt-in, and a future caller
|
|
182
210
|
* releasing twice (or before acquiring) must not silently steal a peer's
|
|
@@ -185,10 +213,11 @@ class MigrationLock {
|
|
|
185
213
|
*/
|
|
186
214
|
async release() {
|
|
187
215
|
if (!this.#owner) return;
|
|
188
|
-
const filter = { _id: LOCK_ID, owner: this.#owner };
|
|
216
|
+
const filter = { _id: LOCK_ID, owner: this.#owner, nonce: this.#nonce };
|
|
189
217
|
try {
|
|
190
218
|
await this.#db.collection(this.#collectionName).deleteOne(filter);
|
|
191
219
|
this.#owner = undefined;
|
|
220
|
+
this.#nonce = undefined;
|
|
192
221
|
} catch (error) {
|
|
193
222
|
throw new LockReleaseFailedError(
|
|
194
223
|
'Failed to release migration lock',
|