@memberjunction/scheduling-engine 5.38.0 → 5.39.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +93 -5
- package/dist/ScheduledJobEngine.d.ts +233 -29
- package/dist/ScheduledJobEngine.d.ts.map +1 -1
- package/dist/ScheduledJobEngine.js +611 -229
- package/dist/ScheduledJobEngine.js.map +1 -1
- package/dist/drivers/AgentRunSweepScheduledJobDriver.d.ts +35 -0
- package/dist/drivers/AgentRunSweepScheduledJobDriver.d.ts.map +1 -0
- package/dist/drivers/AgentRunSweepScheduledJobDriver.js +86 -0
- package/dist/drivers/AgentRunSweepScheduledJobDriver.js.map +1 -0
- package/dist/drivers/index.d.ts +1 -0
- package/dist/drivers/index.d.ts.map +1 -1
- package/dist/drivers/index.js +1 -0
- package/dist/drivers/index.js.map +1 -1
- package/package.json +14 -13
|
@@ -3,7 +3,8 @@
|
|
|
3
3
|
* @module @memberjunction/scheduling-engine
|
|
4
4
|
*/
|
|
5
5
|
import os from 'os';
|
|
6
|
-
import {
|
|
6
|
+
import { v4 as uuidv4 } from 'uuid';
|
|
7
|
+
import { Metadata, LogError, LogStatusEx, LocalCacheManager, RunView } from '@memberjunction/core';
|
|
7
8
|
import { BaseSingleton, MJGlobal, UUIDsEqual } from '@memberjunction/global';
|
|
8
9
|
import { SchedulingEngineBase } from '@memberjunction/scheduling-engine-base';
|
|
9
10
|
import { BaseScheduledJob } from './BaseScheduledJob.js';
|
|
@@ -38,6 +39,52 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
38
39
|
this.hasInitialized = false;
|
|
39
40
|
/** Job IDs we have already warned about for sub-threshold run frequency. */
|
|
40
41
|
this.highFrequencyWarnedJobIds = new Set();
|
|
42
|
+
// ========================================================================
|
|
43
|
+
// DECOUPLING STATE (added in v5.39 — see plans/scheduled-job-engine-decoupling.md)
|
|
44
|
+
// ========================================================================
|
|
45
|
+
/**
|
|
46
|
+
* Maximum concurrent scheduled jobs on this engine instance. Default 5.
|
|
47
|
+
* Configurable via MJServer's `scheduledJobs.maxConcurrentJobs` config.
|
|
48
|
+
*
|
|
49
|
+
* SOFT CAP: under overlapping poll bodies the cap may be transiently
|
|
50
|
+
* exceeded by a small amount bounded by overlap count. See README
|
|
51
|
+
* "Cap and lease semantics" for tuning guidance.
|
|
52
|
+
*/
|
|
53
|
+
this._maxConcurrentJobs = 5;
|
|
54
|
+
/**
|
|
55
|
+
* Lock lease duration. Default 10 minutes. Configurable via
|
|
56
|
+
* mj.config.cjs `scheduling.leaseTimeoutMinutes`.
|
|
57
|
+
*
|
|
58
|
+
* Public API uses minutes (operational readability); internal computations
|
|
59
|
+
* use _leaseTimeoutMs for precision and testability. Tests inject sub-second
|
|
60
|
+
* leases via `_setLeaseTimeoutMsForTest`.
|
|
61
|
+
*
|
|
62
|
+
* Constraint: must be > maximum expected runtime of any job. Setting too
|
|
63
|
+
* low causes healthy long jobs to be reclaimed and re-dispatched.
|
|
64
|
+
*/
|
|
65
|
+
this._leaseTimeoutMs = 10 * 60 * 1000;
|
|
66
|
+
/**
|
|
67
|
+
* Promises for jobs currently dispatched but not yet settled. Keyed by job ID.
|
|
68
|
+
*
|
|
69
|
+
* Three purposes:
|
|
70
|
+
* 1. Bounded concurrency — DispatchScheduledJobs checks size vs MaxConcurrentJobs.
|
|
71
|
+
* 2. Sweep untracking — sweepStaleInflightJobs deletes by ID for jobs whose
|
|
72
|
+
* lease has expired, freeing the cap slot even though the JS promise leaks
|
|
73
|
+
* (see README "Leaked promise behavior").
|
|
74
|
+
* 3. Graceful shutdown — StopPolling can await all in-flight via .values().
|
|
75
|
+
*
|
|
76
|
+
* Self-cleans via identity-checked .finally() on each dispatched promise
|
|
77
|
+
* (no-op if a sweep + re-dispatch already replaced the entry).
|
|
78
|
+
*
|
|
79
|
+
* NOT used for double-dispatch prevention — that's the atomic lock sproc's job.
|
|
80
|
+
*/
|
|
81
|
+
this.inflightJobPromises = new Map();
|
|
82
|
+
/**
|
|
83
|
+
* When false, DispatchScheduledJobs becomes a no-op. Set false in StopPolling
|
|
84
|
+
* BEFORE snapshotting inflightJobPromises for shutdown drain, so no new
|
|
85
|
+
* entries sneak in during the shutdown window.
|
|
86
|
+
*/
|
|
87
|
+
this.acceptingDispatches = true;
|
|
41
88
|
}
|
|
42
89
|
/**
|
|
43
90
|
* Get singleton instance
|
|
@@ -72,6 +119,52 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
72
119
|
get ActivePollingInterval() {
|
|
73
120
|
return this.Base.ActivePollingInterval;
|
|
74
121
|
}
|
|
122
|
+
/**
|
|
123
|
+
* Maximum concurrent scheduled jobs on this engine instance. Default 5.
|
|
124
|
+
* Configurable via MJServer's `scheduledJobs.maxConcurrentJobs` config.
|
|
125
|
+
*/
|
|
126
|
+
get MaxConcurrentJobs() {
|
|
127
|
+
return this._maxConcurrentJobs;
|
|
128
|
+
}
|
|
129
|
+
set MaxConcurrentJobs(value) {
|
|
130
|
+
if (!Number.isInteger(value) || value < 1) {
|
|
131
|
+
throw new Error(`MaxConcurrentJobs must be a positive integer, got ${value}`);
|
|
132
|
+
}
|
|
133
|
+
const old = this._maxConcurrentJobs;
|
|
134
|
+
this._maxConcurrentJobs = value;
|
|
135
|
+
this.log(`MaxConcurrentJobs changed from ${old} to ${value}`);
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* Lock lease duration in milliseconds. Default 600000 (10 minutes).
|
|
139
|
+
* Production callers should use this setter — matches the ms unit of
|
|
140
|
+
* MJServer's `scheduledJobs.defaultLockTimeout` config.
|
|
141
|
+
*/
|
|
142
|
+
get LeaseTimeoutMs() {
|
|
143
|
+
return this._leaseTimeoutMs;
|
|
144
|
+
}
|
|
145
|
+
set LeaseTimeoutMs(value) {
|
|
146
|
+
if (!Number.isFinite(value) || value <= 0) {
|
|
147
|
+
throw new Error(`LeaseTimeoutMs must be a positive number, got ${value}`);
|
|
148
|
+
}
|
|
149
|
+
const old = this._leaseTimeoutMs;
|
|
150
|
+
this._leaseTimeoutMs = value;
|
|
151
|
+
this.log(`LeaseTimeoutMs changed from ${old} to ${value}`);
|
|
152
|
+
}
|
|
153
|
+
/**
|
|
154
|
+
* Convenience accessor — lease duration as integer minutes. Production
|
|
155
|
+
* code may use either this or `LeaseTimeoutMs`. The setter validates
|
|
156
|
+
* positive integer minutes (no fractional minutes via this path; use
|
|
157
|
+
* `LeaseTimeoutMs` for sub-minute precision, including tests).
|
|
158
|
+
*/
|
|
159
|
+
get LeaseTimeoutMinutes() {
|
|
160
|
+
return Math.round(this._leaseTimeoutMs / 60_000);
|
|
161
|
+
}
|
|
162
|
+
set LeaseTimeoutMinutes(value) {
|
|
163
|
+
if (!Number.isInteger(value) || value < 1) {
|
|
164
|
+
throw new Error(`LeaseTimeoutMinutes must be a positive integer, got ${value}`);
|
|
165
|
+
}
|
|
166
|
+
this.LeaseTimeoutMs = value * 60 * 1000;
|
|
167
|
+
}
|
|
75
168
|
/** Find a job type by name. */
|
|
76
169
|
GetJobTypeByName(name) {
|
|
77
170
|
return this.Base.GetJobTypeByName(name);
|
|
@@ -106,76 +199,110 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
106
199
|
// EXECUTION METHODS
|
|
107
200
|
// ========================================================================
|
|
108
201
|
/**
|
|
109
|
-
* Start continuous polling for scheduled jobs
|
|
110
|
-
*
|
|
202
|
+
* Start continuous polling for scheduled jobs.
|
|
203
|
+
*
|
|
204
|
+
* Async (changed in v5.39) because upfront work — Config, initial-NextRunAt
|
|
205
|
+
* seeding, stale-lock cleanup, permission probe — runs ONCE before the
|
|
206
|
+
* first poll fires. Subsequent polls assume that work is complete.
|
|
207
|
+
*
|
|
208
|
+
* The poll callback re-arms its timer FIRST, before any awaited work, so
|
|
209
|
+
* that any hang downstream (Config, DispatchScheduledJobs, etc.) cannot
|
|
210
|
+
* prevent the next poll from firing on schedule. This is the load-bearing
|
|
211
|
+
* invariant of the decoupling fix (see plans/scheduled-job-engine-decoupling.md).
|
|
111
212
|
*
|
|
112
213
|
* @param contextUser - User context for execution
|
|
113
214
|
*/
|
|
114
|
-
StartPolling(contextUser) {
|
|
215
|
+
async StartPolling(contextUser) {
|
|
115
216
|
if (this.isPolling) {
|
|
116
217
|
this.log('Polling already started');
|
|
117
218
|
return;
|
|
118
219
|
}
|
|
220
|
+
// Upfront work — done ONCE before any poll fires.
|
|
221
|
+
// Order: Config first (so this.ScheduledJobs is populated), then
|
|
222
|
+
// initializeNextRunTimes / cleanupStaleLocks / permission probe.
|
|
223
|
+
await this.Config(false, contextUser);
|
|
224
|
+
await this.initializeNextRunTimes(contextUser);
|
|
225
|
+
await this.cleanupStaleLocks(contextUser);
|
|
226
|
+
await this.probeLockSprocPermissions();
|
|
227
|
+
if (this.ScheduledJobs.length === 0) {
|
|
228
|
+
console.log(`📅 Scheduled Jobs: No active jobs found, polling not started`);
|
|
229
|
+
return;
|
|
230
|
+
}
|
|
231
|
+
this.warnAboutHighFrequencyJobs();
|
|
119
232
|
this.isPolling = true;
|
|
233
|
+
this.acceptingDispatches = true;
|
|
234
|
+
this.hasInitialized = true;
|
|
120
235
|
const poll = async () => {
|
|
121
|
-
//
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
if (this.ScheduledJobs.length === 0) {
|
|
131
|
-
console.log(`📅 Scheduled Jobs: No active jobs found, stopping polling`);
|
|
132
|
-
this.StopPolling();
|
|
236
|
+
// Re-arm IMMEDIATELY — load-bearing invariant. The next poll fires
|
|
237
|
+
// on schedule regardless of any await downstream.
|
|
238
|
+
if (this.isPolling) {
|
|
239
|
+
const interval = this.ActivePollingInterval;
|
|
240
|
+
if (interval === null) {
|
|
241
|
+
console.log(`📅 Scheduled Jobs: All jobs removed, stopping polling`);
|
|
242
|
+
// Schedule the cleanup via the same async path so we still
|
|
243
|
+
// return cleanly from this poll body.
|
|
244
|
+
this.StopPolling().catch(err => this.logError('Error during StopPolling', err));
|
|
133
245
|
return;
|
|
134
246
|
}
|
|
135
|
-
this.
|
|
247
|
+
this.pollingTimer = setTimeout(poll, interval);
|
|
136
248
|
}
|
|
137
249
|
try {
|
|
138
|
-
const
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
// Schedule next poll based on current ActivePollingInterval
|
|
144
|
-
if (this.isPolling) {
|
|
145
|
-
const interval = this.ActivePollingInterval;
|
|
146
|
-
// If interval is null (no jobs), stop polling
|
|
147
|
-
if (interval === null) {
|
|
148
|
-
console.log(`📅 Scheduled Jobs: All jobs removed, stopping polling`);
|
|
149
|
-
this.StopPolling();
|
|
150
|
-
return;
|
|
151
|
-
}
|
|
152
|
-
this.pollingTimer = setTimeout(poll, interval);
|
|
250
|
+
const result = await this.DispatchScheduledJobs(contextUser);
|
|
251
|
+
if (result.swept > 0 || result.dispatched > 0 || result.lockedOut > 0 || result.skippedAtCapacity > 0) {
|
|
252
|
+
console.log(`📅 Scheduled Jobs: swept=${result.swept}, dispatched=${result.dispatched}, ` +
|
|
253
|
+
`lockedOut=${result.lockedOut}, skippedAtCapacity=${result.skippedAtCapacity}, ` +
|
|
254
|
+
`inflight=${this.inflightJobPromises.size}/${this.MaxConcurrentJobs}`);
|
|
153
255
|
}
|
|
154
256
|
}
|
|
155
257
|
catch (error) {
|
|
156
|
-
this.logError('Error during
|
|
157
|
-
//
|
|
158
|
-
|
|
159
|
-
this.pollingTimer = setTimeout(poll, 60000); // Fallback to 1 minute
|
|
160
|
-
}
|
|
258
|
+
this.logError('Error during DispatchScheduledJobs', error);
|
|
259
|
+
// No fallback timer needed — re-arm at top of function already
|
|
260
|
+
// scheduled the next poll before any await could fire.
|
|
161
261
|
}
|
|
162
262
|
};
|
|
163
|
-
//
|
|
164
|
-
poll
|
|
263
|
+
// First poll fires on the timer (not immediate) so observers can stop
|
|
264
|
+
// us between StartPolling returning and the first poll firing.
|
|
265
|
+
this.pollingTimer = setTimeout(poll, this.ActivePollingInterval ?? 60_000);
|
|
266
|
+
this.log('Started scheduled job polling');
|
|
165
267
|
}
|
|
166
268
|
/**
|
|
167
|
-
* Stop continuous polling
|
|
269
|
+
* Stop continuous polling.
|
|
270
|
+
*
|
|
271
|
+
* Async (changed in v5.39). With opts.waitForInflight=true, awaits all
|
|
272
|
+
* currently-dispatched jobs to settle before returning. With opts.maxWaitMs,
|
|
273
|
+
* bounds that wait so a zombie can't make shutdown hang indefinitely.
|
|
274
|
+
*
|
|
275
|
+
* Order matters: sets acceptingDispatches=false FIRST so no new entries
|
|
276
|
+
* can be added to inflightJobPromises during the snapshot for allSettled.
|
|
277
|
+
*
|
|
278
|
+
* @param opts.waitForInflight - Await dispatched jobs before returning
|
|
279
|
+
* @param opts.maxWaitMs - Bound the wait (only meaningful with waitForInflight)
|
|
168
280
|
*/
|
|
169
|
-
StopPolling() {
|
|
170
|
-
if (!this.isPolling)
|
|
281
|
+
async StopPolling(opts) {
|
|
282
|
+
if (!this.isPolling)
|
|
171
283
|
return;
|
|
172
|
-
|
|
284
|
+
// Order matters: block new dispatches BEFORE snapshotting inflight.
|
|
285
|
+
this.acceptingDispatches = false;
|
|
173
286
|
this.isPolling = false;
|
|
174
287
|
if (this.pollingTimer) {
|
|
175
288
|
clearTimeout(this.pollingTimer);
|
|
176
289
|
this.pollingTimer = undefined;
|
|
177
290
|
}
|
|
178
291
|
this.log('Stopped scheduled job polling');
|
|
292
|
+
if (opts?.waitForInflight && this.inflightJobPromises.size > 0) {
|
|
293
|
+
const promises = [...this.inflightJobPromises.values()];
|
|
294
|
+
this.log(`Waiting for ${promises.length} in-flight job(s) to settle...`);
|
|
295
|
+
if (opts.maxWaitMs) {
|
|
296
|
+
await Promise.race([
|
|
297
|
+
Promise.allSettled(promises),
|
|
298
|
+
new Promise(resolve => setTimeout(resolve, opts.maxWaitMs))
|
|
299
|
+
]);
|
|
300
|
+
}
|
|
301
|
+
else {
|
|
302
|
+
await Promise.allSettled(promises);
|
|
303
|
+
}
|
|
304
|
+
this.log(`Shutdown wait complete`);
|
|
305
|
+
}
|
|
179
306
|
}
|
|
180
307
|
/**
|
|
181
308
|
* Check if polling is currently active
|
|
@@ -312,14 +439,30 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
312
439
|
for (const job of this.ScheduledJobs) {
|
|
313
440
|
console.log(` - ${job.Name}: NextRunAt=${job.NextRunAt?.toISOString() || 'NULL'}, Status=${job.Status}`);
|
|
314
441
|
if (this.isJobDue(job, evalTime)) {
|
|
315
|
-
console.log(` ✓ Job is due,
|
|
442
|
+
console.log(` ✓ Job is due, attempting to acquire lock...`);
|
|
316
443
|
try {
|
|
317
|
-
const
|
|
318
|
-
if (
|
|
319
|
-
|
|
444
|
+
const lockResult = await this.tryAcquireLock(job.ID);
|
|
445
|
+
if (!lockResult.acquired) {
|
|
446
|
+
if (job.ConcurrencyMode === 'Queue') {
|
|
447
|
+
this.log(`Job ${job.Name} is locked, queueing (ConcurrencyMode=Queue)`);
|
|
448
|
+
const queuedRun = await this.createQueuedJobRun(job, contextUser);
|
|
449
|
+
runs.push(queuedRun);
|
|
450
|
+
}
|
|
451
|
+
else if (job.ConcurrencyMode === 'Skip') {
|
|
452
|
+
console.log(` ⊘ Job is locked, skipping (ConcurrencyMode=Skip)`);
|
|
453
|
+
}
|
|
454
|
+
else {
|
|
455
|
+
// Concurrent mode: proceed without lock
|
|
456
|
+
console.log(` ↪ Concurrent mode, proceeding without lock`);
|
|
457
|
+
const run = await this.executeJobWithLock(job, null, contextUser);
|
|
458
|
+
if (run)
|
|
459
|
+
runs.push(run);
|
|
460
|
+
}
|
|
320
461
|
}
|
|
321
462
|
else {
|
|
322
|
-
|
|
463
|
+
const run = await this.executeJobWithLock(job, lockResult.token, contextUser);
|
|
464
|
+
if (run)
|
|
465
|
+
runs.push(run);
|
|
323
466
|
}
|
|
324
467
|
}
|
|
325
468
|
catch (error) {
|
|
@@ -348,7 +491,96 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
348
491
|
if (!job) {
|
|
349
492
|
throw new Error(`Scheduled job ${jobId} not found or not active`);
|
|
350
493
|
}
|
|
351
|
-
|
|
494
|
+
const lockResult = await this.tryAcquireLock(job.ID);
|
|
495
|
+
if (!lockResult.acquired) {
|
|
496
|
+
throw new Error(`Could not acquire lock for job ${jobId} — held by another holder`);
|
|
497
|
+
}
|
|
498
|
+
const run = await this.executeJobWithLock(job, lockResult.token, contextUser);
|
|
499
|
+
if (!run) {
|
|
500
|
+
throw new Error(`Job execution returned null for ${jobId}`);
|
|
501
|
+
}
|
|
502
|
+
return run;
|
|
503
|
+
}
|
|
504
|
+
/**
|
|
505
|
+
* Dispatch all currently-due scheduled jobs WITHOUT awaiting their completion.
|
|
506
|
+
*
|
|
507
|
+
* This is the polling-path entry point introduced in v5.39 as part of the
|
|
508
|
+
* scheduler decoupling fix (GH #2736). The poll loop calls this and re-arms
|
|
509
|
+
* its timer based on the synchronous-portion return; jobs run in the background.
|
|
510
|
+
*
|
|
511
|
+
* Two phases:
|
|
512
|
+
*
|
|
513
|
+
* PHASE 1 — Stale-inflight sweep (decoupled from isJobDue AND from the cap):
|
|
514
|
+
* Walks inflightJobPromises looking for jobs whose DB lease has expired.
|
|
515
|
+
* Untracks each, frees its cap slot, and fire-and-forget marks any
|
|
516
|
+
* orphaned `Status='Running'` run records as abandoned. Runs first so
|
|
517
|
+
* it can free slots BEFORE the cap check throttles dispatch.
|
|
518
|
+
*
|
|
519
|
+
* PHASE 2 — Cap-bounded dispatch loop:
|
|
520
|
+
* For each due job, atomically acquire its lock via spAcquireScheduledJobLock.
|
|
521
|
+
* Only jobs whose lock was acquired count against MaxConcurrentJobs.
|
|
522
|
+
* Lock-failed jobs are reported via `lockedOut` counter.
|
|
523
|
+
* If at-cap, remaining due jobs counted via `skippedAtCapacity` and
|
|
524
|
+
* picked up by subsequent polls as slots free (no in-memory queueing).
|
|
525
|
+
*
|
|
526
|
+
* Same-instance double-dispatch is structurally prevented by the atomic
|
|
527
|
+
* lock sproc — its WHERE clause filters held-and-not-stale locks, so any
|
|
528
|
+
* second attempt against the same job ID returns Acquired=0.
|
|
529
|
+
*
|
|
530
|
+
* In-flight dispatched promises are tracked in `inflightJobPromises` so
|
|
531
|
+
* `StopPolling({ waitForInflight: true })` can perform graceful shutdown.
|
|
532
|
+
*
|
|
533
|
+
* @returns Counters for observability.
|
|
534
|
+
*/
|
|
535
|
+
async DispatchScheduledJobs(contextUser, evalTime = new Date()) {
|
|
536
|
+
if (!this.acceptingDispatches) {
|
|
537
|
+
return { swept: 0, dispatched: 0, lockedOut: 0, skippedAtCapacity: 0 };
|
|
538
|
+
}
|
|
539
|
+
// PHASE 1: stale-inflight sweep. Decoupled from isJobDue and cap.
|
|
540
|
+
const swept = await this.sweepStaleInflightJobs(contextUser);
|
|
541
|
+
// PHASE 2: cap-bounded dispatch.
|
|
542
|
+
let dispatched = 0;
|
|
543
|
+
let lockedOut = 0;
|
|
544
|
+
let skippedAtCapacity = 0;
|
|
545
|
+
for (const job of this.ScheduledJobs) {
|
|
546
|
+
if (!this.isJobDue(job, evalTime))
|
|
547
|
+
continue;
|
|
548
|
+
if (this.inflightJobPromises.size >= this.MaxConcurrentJobs) {
|
|
549
|
+
skippedAtCapacity++;
|
|
550
|
+
continue;
|
|
551
|
+
}
|
|
552
|
+
const lockResult = await this.tryAcquireLock(job.ID);
|
|
553
|
+
if (!lockResult.acquired) {
|
|
554
|
+
if (job.ConcurrencyMode === 'Queue') {
|
|
555
|
+
await this.createQueuedJobRun(job, contextUser);
|
|
556
|
+
}
|
|
557
|
+
lockedOut++;
|
|
558
|
+
continue;
|
|
559
|
+
}
|
|
560
|
+
// TDZ-safe identity tracking: set Map entry SYNCHRONOUSLY before
|
|
561
|
+
// attaching catch/finally. A synchronous throw in executeJobWithLock
|
|
562
|
+
// (e.g. ClassFactory.CreateInstance failing on a missing DriverClass)
|
|
563
|
+
// would otherwise fire .finally before .set runs and orphan the entry.
|
|
564
|
+
const promise = this.executeJobWithLock(job, lockResult.token, contextUser);
|
|
565
|
+
this.inflightJobPromises.set(job.ID, promise);
|
|
566
|
+
promise
|
|
567
|
+
.catch(error => {
|
|
568
|
+
this.logError(`Unexpected throw escaping executeJobWithLock for job ${job.Name} ` +
|
|
569
|
+
`(indicates a bug in the engine itself, not a plugin)`, error);
|
|
570
|
+
return null;
|
|
571
|
+
})
|
|
572
|
+
.finally(() => {
|
|
573
|
+
// Identity check: only delete if the entry still refers to
|
|
574
|
+
// OUR promise. If a sweep + re-dispatch already replaced it,
|
|
575
|
+
// leave the new entry alone.
|
|
576
|
+
if (this.inflightJobPromises.get(job.ID) === promise) {
|
|
577
|
+
this.inflightJobPromises.delete(job.ID);
|
|
578
|
+
}
|
|
579
|
+
});
|
|
580
|
+
dispatched++;
|
|
581
|
+
}
|
|
582
|
+
this.UpdatePollingInterval();
|
|
583
|
+
return { swept, dispatched, lockedOut, skippedAtCapacity };
|
|
352
584
|
}
|
|
353
585
|
/**
|
|
354
586
|
* Determine if a job is currently due for execution
|
|
@@ -376,30 +608,31 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
376
608
|
/**
|
|
377
609
|
* Execute a single scheduled job
|
|
378
610
|
*
|
|
379
|
-
*
|
|
380
|
-
*
|
|
381
|
-
*
|
|
611
|
+
* Execute a single scheduled job WITH a pre-acquired lock token.
|
|
612
|
+
*
|
|
613
|
+
* Caller (DispatchScheduledJobs / ExecuteScheduledJob / ExecuteScheduledJobs)
|
|
614
|
+
* is responsible for acquiring the lock atomically via tryAcquireLock and
|
|
615
|
+
* passing the resulting token. This method owns the lock's lifecycle from
|
|
616
|
+
* this point forward: every exit path (success, failure, exception)
|
|
617
|
+
* releases the lock via releaseLockIfTokenMatches.
|
|
618
|
+
*
|
|
619
|
+
* If lockToken is null, the job is running in `ConcurrencyMode='Concurrent'`
|
|
620
|
+
* (no lock acquired) — finally simply skips the release.
|
|
621
|
+
*
|
|
622
|
+
* @param job - The job entity. READ-ONLY from this method's perspective —
|
|
623
|
+
* do not mutate or call Save on it. The shared entity in
|
|
624
|
+
* this.ScheduledJobs must not be touched here.
|
|
625
|
+
* @param lockToken - The token returned by tryAcquireLock, or null for
|
|
626
|
+
* Concurrent mode where no lock was acquired.
|
|
627
|
+
* @param contextUser - User context for execution.
|
|
628
|
+
* @returns The created run record (Completed or Failed), or null if a
|
|
629
|
+
* synchronous setup error prevented run creation.
|
|
382
630
|
* @private
|
|
383
631
|
*/
|
|
384
|
-
async
|
|
385
|
-
|
|
386
|
-
const lockAcquired = await this.tryAcquireLock(job);
|
|
387
|
-
if (!lockAcquired) {
|
|
388
|
-
// Handle based on concurrency mode
|
|
389
|
-
if (job.ConcurrencyMode === 'Skip') {
|
|
390
|
-
this.log(`Job ${job.Name} is locked, skipping (ConcurrencyMode=Skip)`);
|
|
391
|
-
return null; // Skip this execution
|
|
392
|
-
}
|
|
393
|
-
else if (job.ConcurrencyMode === 'Queue') {
|
|
394
|
-
this.log(`Job ${job.Name} is locked, queueing (ConcurrencyMode=Queue)`);
|
|
395
|
-
// Create a queued run record for future processing
|
|
396
|
-
return await this.createQueuedJobRun(job, contextUser);
|
|
397
|
-
}
|
|
398
|
-
// Concurrent mode: proceed without lock
|
|
399
|
-
}
|
|
400
|
-
// Create run record
|
|
401
|
-
const run = await this.createJobRun(job, contextUser);
|
|
632
|
+
async executeJobWithLock(job, lockToken, contextUser) {
|
|
633
|
+
let run = null;
|
|
402
634
|
try {
|
|
635
|
+
run = await this.createJobRun(job, contextUser);
|
|
403
636
|
// Get job type
|
|
404
637
|
const jobType = this.ScheduledJobTypes.find(t => UUIDsEqual(t.ID, job.JobTypeID));
|
|
405
638
|
if (!jobType) {
|
|
@@ -426,7 +659,13 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
426
659
|
run.Success = result.Success;
|
|
427
660
|
run.ErrorMessage = result.ErrorMessage || null;
|
|
428
661
|
run.Details = result.Details ? JSON.stringify(result.Details) : null;
|
|
429
|
-
await run.Save();
|
|
662
|
+
const runSaved = await run.Save();
|
|
663
|
+
if (!runSaved) {
|
|
664
|
+
this.logError(`Failed to save run record for job ${job.Name} (run ${run.ID}): ` +
|
|
665
|
+
`${run.LatestResult?.CompleteMessage ?? 'unknown'}`, null);
|
|
666
|
+
// Continue — stats update + release are still important even if
|
|
667
|
+
// the run record save failed (best-effort persistence).
|
|
668
|
+
}
|
|
430
669
|
// Update job statistics
|
|
431
670
|
await this.updateJobStatistics(job, result.Success, run.ID);
|
|
432
671
|
// Send notifications if configured
|
|
@@ -438,21 +677,28 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
438
677
|
return run;
|
|
439
678
|
}
|
|
440
679
|
catch (error) {
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
680
|
+
if (run) {
|
|
681
|
+
run.CompletedAt = new Date();
|
|
682
|
+
run.Status = 'Failed';
|
|
683
|
+
run.Success = false;
|
|
684
|
+
run.ErrorMessage = error instanceof Error ? error.message : 'Unknown error';
|
|
685
|
+
const runSaved = await run.Save();
|
|
686
|
+
if (!runSaved) {
|
|
687
|
+
this.logError(`Failed to save failed-run record for job ${job.Name} (run ${run.ID}): ` +
|
|
688
|
+
`${run.LatestResult?.CompleteMessage ?? 'unknown'}`, null);
|
|
689
|
+
// Continue — stats update + release still need to happen.
|
|
690
|
+
}
|
|
691
|
+
await this.updateJobStatistics(job, false, run.ID);
|
|
692
|
+
}
|
|
449
693
|
this.logError(`Job failed: ${job.Name}`, error);
|
|
450
694
|
return run;
|
|
451
695
|
}
|
|
452
696
|
finally {
|
|
453
|
-
//
|
|
454
|
-
if (
|
|
455
|
-
|
|
697
|
+
// Token-checked release — safe under lease-expiry races.
|
|
698
|
+
// Sproc no-ops if our token no longer matches (another holder reclaimed).
|
|
699
|
+
// Skip entirely for Concurrent mode (no lock to release).
|
|
700
|
+
if (lockToken) {
|
|
701
|
+
await this.releaseLockIfTokenMatches(job.ID, lockToken);
|
|
456
702
|
}
|
|
457
703
|
}
|
|
458
704
|
}
|
|
@@ -474,7 +720,11 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
474
720
|
* Update job statistics after execution
|
|
475
721
|
* @private
|
|
476
722
|
*/
|
|
477
|
-
async updateJobStatistics(job, success,
|
|
723
|
+
async updateJobStatistics(job, success, _runId) {
|
|
724
|
+
const now = new Date();
|
|
725
|
+
const nextRun = CronExpressionHelper.GetNextRunTime(job.CronExpression, job.Timezone);
|
|
726
|
+
// Update in-memory entity so callers reading job.RunCount/job.NextRunAt
|
|
727
|
+
// immediately after see the new values without a Load round-trip.
|
|
478
728
|
job.RunCount++;
|
|
479
729
|
if (success) {
|
|
480
730
|
job.SuccessCount++;
|
|
@@ -482,9 +732,49 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
482
732
|
else {
|
|
483
733
|
job.FailureCount++;
|
|
484
734
|
}
|
|
485
|
-
job.LastRunAt =
|
|
486
|
-
job.NextRunAt =
|
|
487
|
-
|
|
735
|
+
job.LastRunAt = now;
|
|
736
|
+
job.NextRunAt = nextRun;
|
|
737
|
+
// Persist via the targeted sproc — touches ONLY the 5 stats columns.
|
|
738
|
+
// We deliberately do NOT call job.Save() here: a full-entity Save would
|
|
739
|
+
// overwrite the DB's lock columns with the entity's stale in-memory
|
|
740
|
+
// values (the entity was loaded BEFORE the atomic lock sproc set
|
|
741
|
+
// LockToken), blowing away the live lock. That manifested as a
|
|
742
|
+
// false-positive token mismatch on every successful job completion.
|
|
743
|
+
// See plans/scheduled-job-engine-decoupling.md and the migration
|
|
744
|
+
// V202606022027__v5.39.x__Scheduling_Engine_Atomic_Stats_Update.sql.
|
|
745
|
+
// MJ pattern: positional placeholders; see tryAcquireLock for rationale.
|
|
746
|
+
const provider = this.Base.ProviderToUse;
|
|
747
|
+
const schema = provider.MJCoreSchemaName;
|
|
748
|
+
await provider.ExecuteSQL(`EXEC [${schema}].[spUpdateScheduledJobStatistics] ` +
|
|
749
|
+
`@JobID=@p0, @Success=@p1, @LastRunAt=@p2, @NextRunAt=@p3`, [job.ID, success ? 1 : 0, now, nextRun], { isMutation: true, description: 'spUpdateScheduledJobStatistics' }, this.Base.ContextUser);
|
|
750
|
+
// Direct SQL bypasses BaseEntity.Save(), so it skips the save event that
|
|
751
|
+
// would normally drive LocalCacheManager invalidation. Without this call,
|
|
752
|
+
// any cached RunView for 'MJ: Scheduled Jobs' (e.g., the Scheduling
|
|
753
|
+
// Dashboard's filtered views) would keep showing stale RunCount /
|
|
754
|
+
// SuccessCount / NextRunAt until the cache TTL expired.
|
|
755
|
+
//
|
|
756
|
+
// We use full invalidation rather than in-place upsert (UpsertSingleEntity)
|
|
757
|
+
// because the dashboard reads through filtered/sorted views — MJ can't
|
|
758
|
+
// safely patch those in JS (it would need to evaluate the SQL filter to
|
|
759
|
+
// know whether the updated row still belongs in each cached result).
|
|
760
|
+
//
|
|
761
|
+
// See guides/CACHING_AND_PUBSUB_GUIDE.md and CLAUDE.md §"Server-Side
|
|
762
|
+
// Caching" — `BypassCache: true` is the documented opposite escape hatch
|
|
763
|
+
// for readers; this is the producer-side counterpart.
|
|
764
|
+
//
|
|
765
|
+
// Wrapped in try/catch: cache invalidation is best-effort observability
|
|
766
|
+
// hygiene, not load-bearing for job execution. If LocalCacheManager throws
|
|
767
|
+
// (e.g., uninitialized in a test environment, transient storage error),
|
|
768
|
+
// the stats UPDATE has already persisted — the worst case is one cached
|
|
769
|
+
// dashboard view stays stale until its TTL.
|
|
770
|
+
try {
|
|
771
|
+
if (LocalCacheManager.Instance.IsInitialized) {
|
|
772
|
+
await LocalCacheManager.Instance.InvalidateEntityCaches('MJ: Scheduled Jobs');
|
|
773
|
+
}
|
|
774
|
+
}
|
|
775
|
+
catch (cacheError) {
|
|
776
|
+
this.logError('Cache invalidation after stats update failed (non-fatal)', cacheError);
|
|
777
|
+
}
|
|
488
778
|
}
|
|
489
779
|
/**
|
|
490
780
|
* Send notifications if configured
|
|
@@ -515,118 +805,196 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
515
805
|
* Try to acquire a lock for job execution
|
|
516
806
|
* @private
|
|
517
807
|
*/
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
808
|
+
/**
|
|
809
|
+
* Atomically acquire a lock on a job via spAcquireScheduledJobLock.
|
|
810
|
+
* The sproc's WHERE clause handles both the free-lock and stale-lease cases
|
|
811
|
+
* in a single statement — no TOCTOU window between check and write.
|
|
812
|
+
*
|
|
813
|
+
* Operates only on lock columns; never mutates the shared entity in
|
|
814
|
+
* this.ScheduledJobs. Caller passes jobId (string), not the entity object.
|
|
815
|
+
*
|
|
816
|
+
* @returns { acquired: true, token } on success; { acquired: false } otherwise.
|
|
817
|
+
* @private
|
|
818
|
+
*/
|
|
819
|
+
async tryAcquireLock(jobId) {
|
|
820
|
+
// uuidv4 (cryptographic) — Math.random()-based GUIDs would risk collision
|
|
821
|
+
// between concurrent engine instances and break the lost-mutex protection
|
|
822
|
+
// that releaseLockIfTokenMatches relies on (token identity = execution identity).
|
|
823
|
+
const token = uuidv4();
|
|
824
|
+
const instance = this.getInstanceIdentifier();
|
|
825
|
+
const expectedCompletion = new Date(Date.now() + this._leaseTimeoutMs);
|
|
826
|
+
const provider = this.Base.ProviderToUse;
|
|
827
|
+
const schema = provider.MJCoreSchemaName;
|
|
828
|
+
// MJ pattern: positional parameter array with @p0/@p1/... placeholders.
|
|
829
|
+
// SQLServerDataProvider binds positional params by index (request.input('p0', val)).
|
|
830
|
+
// The sproc's own parameters are bound by name via `@SprocParam=@p0` syntax.
|
|
831
|
+
// See packages/SchemaEngine/src/RuntimeSchemaManager.ts for the canonical example.
|
|
832
|
+
const rows = await provider.ExecuteSQL(`EXEC [${schema}].[spAcquireScheduledJobLock] ` +
|
|
833
|
+
`@JobID=@p0, @Token=@p1, @Instance=@p2, @ExpectedCompletionAt=@p3`, [jobId, token, instance, expectedCompletion], { isMutation: true, description: 'spAcquireScheduledJobLock' }, this.Base.ContextUser);
|
|
834
|
+
const acquired = rows?.[0]?.Acquired === 1;
|
|
835
|
+
return acquired ? { acquired: true, token } : { acquired: false };
|
|
836
|
+
}
|
|
837
|
+
/**
|
|
838
|
+
* Atomically release a lock IF AND ONLY IF the current DB token matches
|
|
839
|
+
* expectedToken. Prevents the lost-mutex hazard under lease-expiry races:
|
|
840
|
+
* if a stale holder's execution eventually settles after the lease was
|
|
841
|
+
* reclaimed by a fresh holder, this no-ops (token mismatch).
|
|
842
|
+
*
|
|
843
|
+
* Idempotent — safe to call on an already-released lock (returns false).
|
|
844
|
+
*
|
|
845
|
+
* @returns true if released, false if token mismatch / already released
|
|
846
|
+
* @private
|
|
847
|
+
*/
|
|
848
|
+
async releaseLockIfTokenMatches(jobId, expectedToken) {
|
|
849
|
+
const provider = this.Base.ProviderToUse;
|
|
850
|
+
const schema = provider.MJCoreSchemaName;
|
|
851
|
+
// MJ pattern: positional placeholders; see tryAcquireLock for rationale.
|
|
852
|
+
const rows = await provider.ExecuteSQL(`EXEC [${schema}].[spReleaseScheduledJobLockIfTokenMatches] ` +
|
|
853
|
+
`@JobID=@p0, @ExpectedToken=@p1`, [jobId, expectedToken], { isMutation: true, description: 'spReleaseScheduledJobLockIfTokenMatches' }, this.Base.ContextUser);
|
|
854
|
+
const released = rows?.[0]?.Released === 1;
|
|
855
|
+
if (!released) {
|
|
856
|
+
this.log(`Lock for job ${jobId.substring(0, 8)} was not released ` +
|
|
857
|
+
`(token mismatch — reclaimed by another holder, or already released)`);
|
|
546
858
|
}
|
|
547
|
-
return
|
|
859
|
+
return released;
|
|
548
860
|
}
|
|
549
861
|
/**
|
|
550
|
-
*
|
|
551
|
-
*
|
|
552
|
-
*
|
|
862
|
+
* Pre-flight: verify EXECUTE permission on lock sprocs. Fails LOUDLY at boot
|
|
863
|
+
* if the engine's DB principal lacks grants — much better than a silent
|
|
864
|
+
* runtime failure the next time a job tries to dispatch.
|
|
865
|
+
*
|
|
866
|
+
* Wrapped in try/catch: probe failure (e.g., non-SQL-Server provider where
|
|
867
|
+
* `sys.fn_my_permissions` doesn't exist) must NOT crash boot. We log and
|
|
868
|
+
* continue; any actual permission issue will surface at first sproc call.
|
|
869
|
+
*
|
|
553
870
|
* @private
|
|
554
871
|
*/
|
|
555
|
-
async
|
|
556
|
-
const lockToken = this.generateGuid();
|
|
557
|
-
const instanceId = this.getInstanceIdentifier();
|
|
558
|
-
const expectedCompletion = new Date(Date.now() + 10 * 60 * 1000);
|
|
872
|
+
async probeLockSprocPermissions() {
|
|
559
873
|
try {
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
job.ExpectedCompletionAt = expectedCompletion;
|
|
570
|
-
console.log(` → Attempting to save with lock: ${lockToken.substring(0, 8)}...`);
|
|
571
|
-
const saveResult = await job.Save();
|
|
572
|
-
if (saveResult) {
|
|
573
|
-
console.log(` ✅ Lock acquired successfully!`);
|
|
574
|
-
return true;
|
|
874
|
+
const provider = this.Base.ProviderToUse;
|
|
875
|
+
const schema = provider.MJCoreSchemaName;
|
|
876
|
+
const sql = `SELECT permission_name FROM sys.fn_my_permissions(` +
|
|
877
|
+
`'${schema}.spAcquireScheduledJobLock', 'OBJECT') WHERE permission_name = 'EXECUTE'`;
|
|
878
|
+
const rows = await provider.ExecuteSQL(sql, [], { isMutation: false, description: 'Scheduling engine permission probe' }, this.Base.ContextUser);
|
|
879
|
+
if (!rows || rows.length === 0) {
|
|
880
|
+
this.logError(`⚠️ Scheduling engine DB principal lacks EXECUTE on ` +
|
|
881
|
+
`${schema}.spAcquireScheduledJobLock. Job dispatch WILL fail. ` +
|
|
882
|
+
`Grant cdp_Developer or cdp_Integration role to the principal and restart.`, null);
|
|
575
883
|
}
|
|
576
884
|
else {
|
|
577
|
-
|
|
578
|
-
this.clearInMemoryLockFields(job);
|
|
579
|
-
return false;
|
|
885
|
+
this.log(`Lock sproc permission check OK`);
|
|
580
886
|
}
|
|
581
887
|
}
|
|
582
|
-
catch (
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
888
|
+
catch (probeError) {
|
|
889
|
+
// Probe itself failed (non-SQL-Server provider, or unexpected error).
|
|
890
|
+
// Don't crash boot. Log and continue.
|
|
891
|
+
this.log(`Permission probe skipped (provider may not support sys.fn_my_permissions): ${probeError}`);
|
|
586
892
|
}
|
|
587
893
|
}
|
|
588
894
|
/**
|
|
589
|
-
*
|
|
590
|
-
*
|
|
591
|
-
*
|
|
895
|
+
* Sweep stale inflight jobs. Runs unconditionally at top of every poll.
|
|
896
|
+
*
|
|
897
|
+
* SINGLE BATCH QUERY (not N round-trips). Returns only jobs whose lease has
|
|
898
|
+
* expired OR whose lock has already been cleared. In steady-state (no zombies)
|
|
899
|
+
* the query matches zero rows and the sweep is essentially free.
|
|
900
|
+
*
|
|
901
|
+
* For each stale entry:
|
|
902
|
+
* - Untrack the leaked promise from inflightJobPromises (frees cap slot).
|
|
903
|
+
* - FIRE-AND-FORGET abandon any orphaned `Status='Running'` run records.
|
|
904
|
+
* NOT awaited because cleanup must not delay dispatch under a fleet-wide
|
|
905
|
+
* hang event where the sweep finds many zombies at once.
|
|
906
|
+
*
|
|
907
|
+
* Decoupled from:
|
|
908
|
+
* - isJobDue — irrelevant; we care about lease state, not cron.
|
|
909
|
+
* - MaxConcurrentJobs — the sweep IS what frees the cap when saturated by hangs.
|
|
910
|
+
*
|
|
911
|
+
* See plans/scheduled-job-engine-decoupling.md for the full rationale.
|
|
912
|
+
*
|
|
913
|
+
* @returns count of inflight entries swept
|
|
592
914
|
* @private
|
|
593
915
|
*/
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
916
|
+
async sweepStaleInflightJobs(contextUser) {
|
|
917
|
+
if (this.inflightJobPromises.size === 0)
|
|
918
|
+
return 0;
|
|
919
|
+
const trackedIds = [...this.inflightJobPromises.keys()];
|
|
920
|
+
// ID values interpolated below are engine-generated GUIDs from
|
|
921
|
+
// this.inflightJobPromises keys (originally from this.ScheduledJobs[].ID),
|
|
922
|
+
// never user input. No SQL-injection vector. RunViewParams.ExtraFilter
|
|
923
|
+
// does not support parameterized binding in current MJCore.
|
|
924
|
+
const idList = trackedIds.map(id => `'${id}'`).join(',');
|
|
925
|
+
const nowIso = new Date().toISOString();
|
|
926
|
+
const rv = new RunView(this.Base.RunViewProviderToUse);
|
|
927
|
+
const result = await rv.RunView({
|
|
928
|
+
EntityName: 'MJ: Scheduled Jobs',
|
|
929
|
+
ExtraFilter: `ID IN (${idList}) AND (LockToken IS NULL OR ExpectedCompletionAt IS NULL OR ExpectedCompletionAt < '${nowIso}')`,
|
|
930
|
+
Fields: ['ID', 'Name', 'ExpectedCompletionAt'],
|
|
931
|
+
ResultType: 'simple',
|
|
932
|
+
}, contextUser);
|
|
933
|
+
if (!result.Success || result.Results.length === 0)
|
|
934
|
+
return 0;
|
|
935
|
+
let swept = 0;
|
|
936
|
+
for (const row of result.Results) {
|
|
937
|
+
const jobName = row.Name ?? row.ID;
|
|
938
|
+
this.log(`[sweep] Untracking inflight job ${jobName}: ` +
|
|
939
|
+
`lease=${row.ExpectedCompletionAt?.toISOString() ?? 'NULL'}, now=${nowIso}. ` +
|
|
940
|
+
`Original execution presumed hung. See README "Leaked promise behavior".`);
|
|
941
|
+
this.inflightJobPromises.delete(row.ID);
|
|
942
|
+
swept++;
|
|
943
|
+
// FIRE-AND-FORGET: cleanup must not block dispatch.
|
|
944
|
+
this.abandonOrphanedRunRecords(row.ID, contextUser).catch(err => this.logError(`Background abandon-orphaned-runs failed for ${row.ID}`, err));
|
|
945
|
+
}
|
|
946
|
+
return swept;
|
|
599
947
|
}
|
|
600
948
|
/**
|
|
601
|
-
*
|
|
949
|
+
* Mark any Running run records for the given job as Failed/abandoned.
|
|
950
|
+
*
|
|
951
|
+
* IMPORTANT: the `Status='Running'` filter is LOAD-BEARING — not just for
|
|
952
|
+
* finding zombies. It also protects against a sweep/release race:
|
|
953
|
+
*
|
|
954
|
+
* - Job completes normally.
|
|
955
|
+
* - executeJobWithLock's finally calls releaseLockIfTokenMatches (clears LockToken).
|
|
956
|
+
* - BEFORE that completes, a poll's sweep query sees LockToken IS NULL
|
|
957
|
+
* and classifies the just-completed job as a zombie.
|
|
958
|
+
* - But its run record is already Status='Completed' (set inside the try block,
|
|
959
|
+
* before the finally), so THIS FILTER excludes it from abandonment.
|
|
960
|
+
*
|
|
961
|
+
* Removing or relaxing this filter would corrupt completed run records.
|
|
962
|
+
* If "optimizing" this method, preserve the Status='Running' filter.
|
|
963
|
+
*
|
|
602
964
|
* @private
|
|
603
965
|
*/
|
|
604
|
-
async
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
966
|
+
async abandonOrphanedRunRecords(jobId, contextUser) {
|
|
967
|
+
// jobId is an engine-supplied GUID from the sweep's RunView result row,
|
|
968
|
+
// not user input. No SQL-injection vector.
|
|
969
|
+
const rv = new RunView(this.Base.RunViewProviderToUse);
|
|
970
|
+
const result = await rv.RunView({
|
|
971
|
+
EntityName: 'MJ: Scheduled Job Runs',
|
|
972
|
+
ExtraFilter: `ScheduledJobID='${jobId}' AND Status='Running'`,
|
|
973
|
+
ResultType: 'entity_object',
|
|
974
|
+
}, contextUser);
|
|
975
|
+
if (!result.Success || result.Results.length === 0)
|
|
976
|
+
return;
|
|
977
|
+
const now = new Date();
|
|
978
|
+
for (const run of result.Results) {
|
|
979
|
+
run.CompletedAt = now;
|
|
980
|
+
run.Status = 'Failed';
|
|
981
|
+
run.Success = false;
|
|
982
|
+
run.ErrorMessage =
|
|
983
|
+
`Execution abandoned by scheduling engine sweep: lease (started ` +
|
|
984
|
+
`${run.StartedAt?.toISOString()}) expired and the original execution ` +
|
|
985
|
+
`never settled. The hung promise was untracked so its concurrency slot ` +
|
|
986
|
+
`could be reused. See packages/Scheduling/engine/README.md ` +
|
|
987
|
+
`"Leaked promise behavior" for details.`;
|
|
988
|
+
const saved = await run.Save();
|
|
611
989
|
if (!saved) {
|
|
612
|
-
this.logError(`Failed to
|
|
990
|
+
this.logError(`Failed to abandon orphaned run ${run.ID}: ` +
|
|
991
|
+
`${run.LatestResult?.CompleteMessage ?? 'unknown'}`, null);
|
|
992
|
+
}
|
|
993
|
+
else {
|
|
994
|
+
this.log(`[sweep] Abandoned orphaned run ${run.ID} for job ${jobId}`);
|
|
613
995
|
}
|
|
614
|
-
return saved;
|
|
615
|
-
}
|
|
616
|
-
catch (error) {
|
|
617
|
-
this.logError(`Failed to release lock for job ${job.Name}`, error);
|
|
618
|
-
return false;
|
|
619
996
|
}
|
|
620
997
|
}
|
|
621
|
-
/**
|
|
622
|
-
* Clean up a stale lock. Returns true if the lock was successfully
|
|
623
|
-
* cleared in the database, false if the save failed.
|
|
624
|
-
* @private
|
|
625
|
-
*/
|
|
626
|
-
async cleanupStaleLock(job) {
|
|
627
|
-
this.log(`Cleaning up stale lock on job ${job.Name} (locked by ${job.LockedByInstance})`);
|
|
628
|
-
return this.releaseLock(job);
|
|
629
|
-
}
|
|
630
998
|
/**
|
|
631
999
|
* Create a queued job run for later execution
|
|
632
1000
|
* @private
|
|
@@ -651,25 +1019,33 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
651
1019
|
return `${os.hostname()}-${process.pid}`;
|
|
652
1020
|
}
|
|
653
1021
|
/**
|
|
654
|
-
*
|
|
655
|
-
*
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
});
|
|
663
|
-
}
|
|
664
|
-
/**
|
|
665
|
-
* Initialize NextRunAt for jobs that don't have it set
|
|
1022
|
+
* Initialize NextRunAt for jobs that don't have it set.
|
|
1023
|
+
*
|
|
1024
|
+
* If a job has `RunImmediatelyIfNeverRun = true` AND has never run
|
|
1025
|
+
* (`LastRunAt IS NULL`), `NextRunAt` is set to `now()` so the job
|
|
1026
|
+
* executes on the next polling cycle instead of waiting for the next
|
|
1027
|
+
* cron tick. Useful for freshly-seeded jobs that should not wait up
|
|
1028
|
+
* to a full cron interval (e.g. 24h for a daily job) for their first run.
|
|
1029
|
+
*
|
|
666
1030
|
* @private
|
|
667
1031
|
*/
|
|
668
1032
|
async initializeNextRunTimes(contextUser) {
|
|
669
1033
|
for (const job of this.ScheduledJobs) {
|
|
670
1034
|
if (!job.NextRunAt) {
|
|
671
|
-
|
|
1035
|
+
if (job.RunImmediatelyIfNeverRun && !job.LastRunAt) {
|
|
1036
|
+
job.NextRunAt = new Date();
|
|
1037
|
+
console.log(` ⏱️ Job ${job.Name} flagged RunImmediatelyIfNeverRun — scheduling for immediate execution`);
|
|
1038
|
+
}
|
|
1039
|
+
else {
|
|
1040
|
+
job.NextRunAt = CronExpressionHelper.GetNextRunTime(job.CronExpression, job.Timezone);
|
|
1041
|
+
}
|
|
672
1042
|
try {
|
|
1043
|
+
// SAFE: this Save runs in StartPolling's upfront block, BEFORE
|
|
1044
|
+
// isPolling=true is set. No locks can have been acquired yet,
|
|
1045
|
+
// so the full-entity Save cannot clobber any live lock state.
|
|
1046
|
+
// If you ever move this call site outside the upfront block,
|
|
1047
|
+
// refactor to a targeted UPDATE sproc — see
|
|
1048
|
+
// updateJobStatistics for the pattern.
|
|
673
1049
|
await job.Save();
|
|
674
1050
|
console.log(` ⚙️ Initialized NextRunAt for ${job.Name} -> ${job.NextRunAt.toISOString()}`);
|
|
675
1051
|
}
|
|
@@ -680,56 +1056,62 @@ export class SchedulingEngine extends BaseSingleton {
|
|
|
680
1056
|
}
|
|
681
1057
|
}
|
|
682
1058
|
/**
|
|
683
|
-
* Clean up stale locks on startup
|
|
1059
|
+
* Clean up stale locks on startup using atomic sprocs.
|
|
1060
|
+
*
|
|
1061
|
+
* For each job whose DB shows a stale lock (ExpectedCompletionAt < now OR
|
|
1062
|
+
* ExpectedCompletionAt IS NULL while LockToken IS NOT NULL):
|
|
1063
|
+
* 1. Atomically acquire the stale lock with a fresh token (sproc's WHERE
|
|
1064
|
+
* handles the stale-detection in a single statement).
|
|
1065
|
+
* 2. Immediately release it with that same token.
|
|
1066
|
+
*
|
|
1067
|
+
* Net effect: stale lock cleared atomically with zero TOCTOU window. Uses
|
|
1068
|
+
* the new sproc-backed pattern instead of load-compare-save on shared
|
|
1069
|
+
* this.ScheduledJobs entities (see plans/scheduled-job-engine-decoupling.md
|
|
1070
|
+
* for why the old pattern was unsafe once polling became concurrent).
|
|
1071
|
+
*
|
|
684
1072
|
* @private
|
|
685
1073
|
*/
|
|
686
|
-
async cleanupStaleLocks(
|
|
1074
|
+
async cleanupStaleLocks(_contextUser) {
|
|
687
1075
|
const now = new Date();
|
|
688
|
-
let cleanedCount = 0;
|
|
689
1076
|
console.log(` 🔍 Checking for stale locks (current time: ${now.toISOString()})...`);
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
1077
|
+
// Use the engine's already-loaded cache instead of round-tripping to
|
|
1078
|
+
// the DB — this method runs in StartPolling's upfront block, IMMEDIATELY
|
|
1079
|
+
// after Config() loads this.Base.ScheduledJobs, so the cached lock-column
|
|
1080
|
+
// values are current (no stale-state risk that the sweep path has). Saves
|
|
1081
|
+
// one query + silences the "already loaded by SchedulingEngineBase"
|
|
1082
|
+
// telemetry warning.
|
|
1083
|
+
//
|
|
1084
|
+
// Caveat: only safe HERE because the cache was just loaded. The sweep
|
|
1085
|
+
// path (sweepStaleInflightJobs) MUST hit the DB because by then the
|
|
1086
|
+
// cache's lock columns are stale (atomic sprocs bypass the entity cache).
|
|
1087
|
+
const stale = this.Base.ScheduledJobs.filter(job => job.LockToken != null &&
|
|
1088
|
+
(job.ExpectedCompletionAt == null || job.ExpectedCompletionAt < now));
|
|
1089
|
+
if (stale.length === 0) {
|
|
1090
|
+
console.log(` ✓ No stale locks found`);
|
|
1091
|
+
return;
|
|
1092
|
+
}
|
|
1093
|
+
let cleanedCount = 0;
|
|
1094
|
+
for (const job of stale) {
|
|
1095
|
+
try {
|
|
1096
|
+
// Atomic reclaim: acquire returns Acquired=1 if the lock was
|
|
1097
|
+
// stale (per the sproc's WHERE clause). Then immediately release
|
|
1098
|
+
// with the new token to leave the lock free.
|
|
1099
|
+
const lockResult = await this.tryAcquireLock(job.ID);
|
|
1100
|
+
if (lockResult.acquired) {
|
|
1101
|
+
await this.releaseLockIfTokenMatches(job.ID, lockResult.token);
|
|
1102
|
+
console.log(` 🔓 Cleared stale lock on "${job.Name}" (was held by ${job.LockedByInstance})`);
|
|
1103
|
+
cleanedCount++;
|
|
710
1104
|
}
|
|
711
1105
|
else {
|
|
712
|
-
|
|
713
|
-
job.
|
|
714
|
-
job.LockedAt = null;
|
|
715
|
-
job.LockedByInstance = null;
|
|
716
|
-
job.ExpectedCompletionAt = null;
|
|
717
|
-
try {
|
|
718
|
-
await job.Save();
|
|
719
|
-
cleanedCount++;
|
|
720
|
-
}
|
|
721
|
-
catch (error) {
|
|
722
|
-
this.logError(`Failed to clean stale lock for job ${job.Name}`, error);
|
|
723
|
-
}
|
|
1106
|
+
// Another instance acquired between our cache load and acquire.
|
|
1107
|
+
console.log(` ℹ️ Stale lock on "${job.Name}" was cleared by another holder`);
|
|
724
1108
|
}
|
|
725
1109
|
}
|
|
1110
|
+
catch (error) {
|
|
1111
|
+
this.logError(`Failed to clean stale lock for job ${job.Name}`, error);
|
|
1112
|
+
}
|
|
726
1113
|
}
|
|
727
|
-
|
|
728
|
-
console.log(` ✅ Cleaned ${cleanedCount} stale lock(s)`);
|
|
729
|
-
}
|
|
730
|
-
else {
|
|
731
|
-
console.log(` ✓ No stale locks found`);
|
|
732
|
-
}
|
|
1114
|
+
console.log(` ✅ Cleaned ${cleanedCount} stale lock(s)`);
|
|
733
1115
|
}
|
|
734
1116
|
log(message) {
|
|
735
1117
|
LogStatusEx({
|