@memberjunction/scheduling-engine 5.38.0 → 5.39.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,7 +3,8 @@
3
3
  * @module @memberjunction/scheduling-engine
4
4
  */
5
5
  import os from 'os';
6
- import { Metadata, LogError, LogStatusEx } from '@memberjunction/core';
6
+ import { v4 as uuidv4 } from 'uuid';
7
+ import { Metadata, LogError, LogStatusEx, LocalCacheManager, RunView } from '@memberjunction/core';
7
8
  import { BaseSingleton, MJGlobal, UUIDsEqual } from '@memberjunction/global';
8
9
  import { SchedulingEngineBase } from '@memberjunction/scheduling-engine-base';
9
10
  import { BaseScheduledJob } from './BaseScheduledJob.js';
@@ -38,6 +39,52 @@ export class SchedulingEngine extends BaseSingleton {
38
39
  this.hasInitialized = false;
39
40
  /** Job IDs we have already warned about for sub-threshold run frequency. */
40
41
  this.highFrequencyWarnedJobIds = new Set();
42
+ // ========================================================================
43
+ // DECOUPLING STATE (added in v5.39 — see plans/scheduled-job-engine-decoupling.md)
44
+ // ========================================================================
45
+ /**
46
+ * Maximum concurrent scheduled jobs on this engine instance. Default 5.
47
+ * Configurable via MJServer's `scheduledJobs.maxConcurrentJobs` config.
48
+ *
49
+ * SOFT CAP: under overlapping poll bodies the cap may be transiently
50
+ * exceeded by a small amount bounded by overlap count. See README
51
+ * "Cap and lease semantics" for tuning guidance.
52
+ */
53
+ this._maxConcurrentJobs = 5;
54
+ /**
55
+ * Lock lease duration. Default 10 minutes. Configurable via
56
+ * mj.config.cjs `scheduling.leaseTimeoutMinutes`.
57
+ *
58
+ * Public API uses minutes (operational readability); internal computations
59
+ * use _leaseTimeoutMs for precision and testability. Tests inject sub-second
60
+ * leases via `_setLeaseTimeoutMsForTest`.
61
+ *
62
+ * Constraint: must be > maximum expected runtime of any job. Setting too
63
+ * low causes healthy long jobs to be reclaimed and re-dispatched.
64
+ */
65
+ this._leaseTimeoutMs = 10 * 60 * 1000;
66
+ /**
67
+ * Promises for jobs currently dispatched but not yet settled. Keyed by job ID.
68
+ *
69
+ * Three purposes:
70
+ * 1. Bounded concurrency — DispatchScheduledJobs checks size vs MaxConcurrentJobs.
71
+ * 2. Sweep untracking — sweepStaleInflightJobs deletes by ID for jobs whose
72
+ * lease has expired, freeing the cap slot even though the JS promise leaks
73
+ * (see README "Leaked promise behavior").
74
+ * 3. Graceful shutdown — StopPolling can await all in-flight via .values().
75
+ *
76
+ * Self-cleans via identity-checked .finally() on each dispatched promise
77
+ * (no-op if a sweep + re-dispatch already replaced the entry).
78
+ *
79
+ * NOT used for double-dispatch prevention — that's the atomic lock sproc's job.
80
+ */
81
+ this.inflightJobPromises = new Map();
82
+ /**
83
+ * When false, DispatchScheduledJobs becomes a no-op. Set false in StopPolling
84
+ * BEFORE snapshotting inflightJobPromises for shutdown drain, so no new
85
+ * entries sneak in during the shutdown window.
86
+ */
87
+ this.acceptingDispatches = true;
41
88
  }
42
89
  /**
43
90
  * Get singleton instance
@@ -72,6 +119,52 @@ export class SchedulingEngine extends BaseSingleton {
72
119
  get ActivePollingInterval() {
73
120
  return this.Base.ActivePollingInterval;
74
121
  }
122
+ /**
123
+ * Maximum concurrent scheduled jobs on this engine instance. Default 5.
124
+ * Configurable via MJServer's `scheduledJobs.maxConcurrentJobs` config.
125
+ */
126
+ get MaxConcurrentJobs() {
127
+ return this._maxConcurrentJobs;
128
+ }
129
+ set MaxConcurrentJobs(value) {
130
+ if (!Number.isInteger(value) || value < 1) {
131
+ throw new Error(`MaxConcurrentJobs must be a positive integer, got ${value}`);
132
+ }
133
+ const old = this._maxConcurrentJobs;
134
+ this._maxConcurrentJobs = value;
135
+ this.log(`MaxConcurrentJobs changed from ${old} to ${value}`);
136
+ }
137
+ /**
138
+ * Lock lease duration in milliseconds. Default 600000 (10 minutes).
139
+ * Production callers should use this setter — matches the ms unit of
140
+ * MJServer's `scheduledJobs.defaultLockTimeout` config.
141
+ */
142
+ get LeaseTimeoutMs() {
143
+ return this._leaseTimeoutMs;
144
+ }
145
+ set LeaseTimeoutMs(value) {
146
+ if (!Number.isFinite(value) || value <= 0) {
147
+ throw new Error(`LeaseTimeoutMs must be a positive number, got ${value}`);
148
+ }
149
+ const old = this._leaseTimeoutMs;
150
+ this._leaseTimeoutMs = value;
151
+ this.log(`LeaseTimeoutMs changed from ${old} to ${value}`);
152
+ }
153
+ /**
154
+ * Convenience accessor — lease duration as integer minutes. Production
155
+ * code may use either this or `LeaseTimeoutMs`. The setter validates
156
+ * positive integer minutes (no fractional minutes via this path; use
157
+ * `LeaseTimeoutMs` for sub-minute precision, including tests).
158
+ */
159
+ get LeaseTimeoutMinutes() {
160
+ return Math.round(this._leaseTimeoutMs / 60_000);
161
+ }
162
+ set LeaseTimeoutMinutes(value) {
163
+ if (!Number.isInteger(value) || value < 1) {
164
+ throw new Error(`LeaseTimeoutMinutes must be a positive integer, got ${value}`);
165
+ }
166
+ this.LeaseTimeoutMs = value * 60 * 1000;
167
+ }
75
168
  /** Find a job type by name. */
76
169
  GetJobTypeByName(name) {
77
170
  return this.Base.GetJobTypeByName(name);
@@ -106,76 +199,110 @@ export class SchedulingEngine extends BaseSingleton {
106
199
  // EXECUTION METHODS
107
200
  // ========================================================================
108
201
  /**
109
- * Start continuous polling for scheduled jobs
110
- * Uses adaptive interval based on ActivePollingInterval
202
+ * Start continuous polling for scheduled jobs.
203
+ *
204
+ * Async (changed in v5.39) because upfront work — Config, initial-NextRunAt
205
+ * seeding, stale-lock cleanup, permission probe — runs ONCE before the
206
+ * first poll fires. Subsequent polls assume that work is complete.
207
+ *
208
+ * The poll callback re-arms its timer FIRST, before any awaited work, so
209
+ * that any hang downstream (Config, DispatchScheduledJobs, etc.) cannot
210
+ * prevent the next poll from firing on schedule. This is the load-bearing
211
+ * invariant of the decoupling fix (see plans/scheduled-job-engine-decoupling.md).
111
212
  *
112
213
  * @param contextUser - User context for execution
113
214
  */
114
- StartPolling(contextUser) {
215
+ async StartPolling(contextUser) {
115
216
  if (this.isPolling) {
116
217
  this.log('Polling already started');
117
218
  return;
118
219
  }
220
+ // Upfront work — done ONCE before any poll fires.
221
+ // Order: Config first (so this.ScheduledJobs is populated), then
222
+ // initializeNextRunTimes / cleanupStaleLocks / permission probe.
223
+ await this.Config(false, contextUser);
224
+ await this.initializeNextRunTimes(contextUser);
225
+ await this.cleanupStaleLocks(contextUser);
226
+ await this.probeLockSprocPermissions();
227
+ if (this.ScheduledJobs.length === 0) {
228
+ console.log(`📅 Scheduled Jobs: No active jobs found, polling not started`);
229
+ return;
230
+ }
231
+ this.warnAboutHighFrequencyJobs();
119
232
  this.isPolling = true;
233
+ this.acceptingDispatches = true;
234
+ this.hasInitialized = true;
120
235
  const poll = async () => {
121
- // Initialize NextRunAt and clean up stale locks on first poll only
122
- if (!this.hasInitialized) {
123
- await this.initializeNextRunTimes(contextUser);
124
- await this.cleanupStaleLocks(contextUser);
125
- // No need to force-reload: initializeNextRunTimes and cleanupStaleLocks
126
- // modify and save the in-memory entity objects directly, so the cache
127
- // already reflects the current DB state.
128
- this.hasInitialized = true;
129
- // Check if there are no jobs after initialization
130
- if (this.ScheduledJobs.length === 0) {
131
- console.log(`📅 Scheduled Jobs: No active jobs found, stopping polling`);
132
- this.StopPolling();
236
+ // Re-arm IMMEDIATELY — load-bearing invariant. The next poll fires
237
+ // on schedule regardless of any await downstream.
238
+ if (this.isPolling) {
239
+ const interval = this.ActivePollingInterval;
240
+ if (interval === null) {
241
+ console.log(`📅 Scheduled Jobs: All jobs removed, stopping polling`);
242
+ // Schedule the cleanup via the same async path so we still
243
+ // return cleanly from this poll body.
244
+ this.StopPolling().catch(err => this.logError('Error during StopPolling', err));
133
245
  return;
134
246
  }
135
- this.warnAboutHighFrequencyJobs();
247
+ this.pollingTimer = setTimeout(poll, interval);
136
248
  }
137
249
  try {
138
- const runs = await this.ExecuteScheduledJobs(contextUser);
139
- // Only log if jobs were actually executed
140
- if (runs.length > 0) {
141
- console.log(`📅 Scheduled Jobs: Executed ${runs.length} job(s)`);
142
- }
143
- // Schedule next poll based on current ActivePollingInterval
144
- if (this.isPolling) {
145
- const interval = this.ActivePollingInterval;
146
- // If interval is null (no jobs), stop polling
147
- if (interval === null) {
148
- console.log(`📅 Scheduled Jobs: All jobs removed, stopping polling`);
149
- this.StopPolling();
150
- return;
151
- }
152
- this.pollingTimer = setTimeout(poll, interval);
250
+ const result = await this.DispatchScheduledJobs(contextUser);
251
+ if (result.swept > 0 || result.dispatched > 0 || result.lockedOut > 0 || result.skippedAtCapacity > 0) {
252
+ console.log(`📅 Scheduled Jobs: swept=${result.swept}, dispatched=${result.dispatched}, ` +
253
+ `lockedOut=${result.lockedOut}, skippedAtCapacity=${result.skippedAtCapacity}, ` +
254
+ `inflight=${this.inflightJobPromises.size}/${this.MaxConcurrentJobs}`);
153
255
  }
154
256
  }
155
257
  catch (error) {
156
- this.logError('Error during polling', error);
157
- // Continue polling even after errors
158
- if (this.isPolling) {
159
- this.pollingTimer = setTimeout(poll, 60000); // Fallback to 1 minute
160
- }
258
+ this.logError('Error during DispatchScheduledJobs', error);
259
+ // No fallback timer needed — re-arm at top of function already
260
+ // scheduled the next poll before any await could fire.
161
261
  }
162
262
  };
163
- // Start first poll immediately
164
- poll();
263
+ // First poll fires on the timer (not immediate) so observers can stop
264
+ // us between StartPolling returning and the first poll firing.
265
+ this.pollingTimer = setTimeout(poll, this.ActivePollingInterval ?? 60_000);
266
+ this.log('Started scheduled job polling');
165
267
  }
166
268
  /**
167
- * Stop continuous polling
269
+ * Stop continuous polling.
270
+ *
271
+ * Async (changed in v5.39). With opts.waitForInflight=true, awaits all
272
+ * currently-dispatched jobs to settle before returning. With opts.maxWaitMs,
273
+ * bounds that wait so a zombie can't make shutdown hang indefinitely.
274
+ *
275
+ * Order matters: sets acceptingDispatches=false FIRST so no new entries
276
+ * can be added to inflightJobPromises during the snapshot for allSettled.
277
+ *
278
+ * @param opts.waitForInflight - Await dispatched jobs before returning
279
+ * @param opts.maxWaitMs - Bound the wait (only meaningful with waitForInflight)
168
280
  */
169
- StopPolling() {
170
- if (!this.isPolling) {
281
+ async StopPolling(opts) {
282
+ if (!this.isPolling)
171
283
  return;
172
- }
284
+ // Order matters: block new dispatches BEFORE snapshotting inflight.
285
+ this.acceptingDispatches = false;
173
286
  this.isPolling = false;
174
287
  if (this.pollingTimer) {
175
288
  clearTimeout(this.pollingTimer);
176
289
  this.pollingTimer = undefined;
177
290
  }
178
291
  this.log('Stopped scheduled job polling');
292
+ if (opts?.waitForInflight && this.inflightJobPromises.size > 0) {
293
+ const promises = [...this.inflightJobPromises.values()];
294
+ this.log(`Waiting for ${promises.length} in-flight job(s) to settle...`);
295
+ if (opts.maxWaitMs) {
296
+ await Promise.race([
297
+ Promise.allSettled(promises),
298
+ new Promise(resolve => setTimeout(resolve, opts.maxWaitMs))
299
+ ]);
300
+ }
301
+ else {
302
+ await Promise.allSettled(promises);
303
+ }
304
+ this.log(`Shutdown wait complete`);
305
+ }
179
306
  }
180
307
  /**
181
308
  * Check if polling is currently active
@@ -312,14 +439,30 @@ export class SchedulingEngine extends BaseSingleton {
312
439
  for (const job of this.ScheduledJobs) {
313
440
  console.log(` - ${job.Name}: NextRunAt=${job.NextRunAt?.toISOString() || 'NULL'}, Status=${job.Status}`);
314
441
  if (this.isJobDue(job, evalTime)) {
315
- console.log(` ✓ Job is due, executing...`);
442
+ console.log(` ✓ Job is due, attempting to acquire lock...`);
316
443
  try {
317
- const run = await this.executeJob(job, contextUser);
318
- if (run) { // null if skipped
319
- runs.push(run);
444
+ const lockResult = await this.tryAcquireLock(job.ID);
445
+ if (!lockResult.acquired) {
446
+ if (job.ConcurrencyMode === 'Queue') {
447
+ this.log(`Job ${job.Name} is locked, queueing (ConcurrencyMode=Queue)`);
448
+ const queuedRun = await this.createQueuedJobRun(job, contextUser);
449
+ runs.push(queuedRun);
450
+ }
451
+ else if (job.ConcurrencyMode === 'Skip') {
452
+ console.log(` ⊘ Job is locked, skipping (ConcurrencyMode=Skip)`);
453
+ }
454
+ else {
455
+ // Concurrent mode: proceed without lock
456
+ console.log(` ↪ Concurrent mode, proceeding without lock`);
457
+ const run = await this.executeJobWithLock(job, null, contextUser);
458
+ if (run)
459
+ runs.push(run);
460
+ }
320
461
  }
321
462
  else {
322
- console.log(` ⊘ Job was skipped (locked or queued)`);
463
+ const run = await this.executeJobWithLock(job, lockResult.token, contextUser);
464
+ if (run)
465
+ runs.push(run);
323
466
  }
324
467
  }
325
468
  catch (error) {
@@ -348,7 +491,96 @@ export class SchedulingEngine extends BaseSingleton {
348
491
  if (!job) {
349
492
  throw new Error(`Scheduled job ${jobId} not found or not active`);
350
493
  }
351
- return await this.executeJob(job, contextUser);
494
+ const lockResult = await this.tryAcquireLock(job.ID);
495
+ if (!lockResult.acquired) {
496
+ throw new Error(`Could not acquire lock for job ${jobId} — held by another holder`);
497
+ }
498
+ const run = await this.executeJobWithLock(job, lockResult.token, contextUser);
499
+ if (!run) {
500
+ throw new Error(`Job execution returned null for ${jobId}`);
501
+ }
502
+ return run;
503
+ }
504
+ /**
505
+ * Dispatch all currently-due scheduled jobs WITHOUT awaiting their completion.
506
+ *
507
+ * This is the polling-path entry point introduced in v5.39 as part of the
508
+ * scheduler decoupling fix (GH #2736). The poll loop calls this and re-arms
509
+ * its timer based on the synchronous-portion return; jobs run in the background.
510
+ *
511
+ * Two phases:
512
+ *
513
+ * PHASE 1 — Stale-inflight sweep (decoupled from isJobDue AND from the cap):
514
+ * Walks inflightJobPromises looking for jobs whose DB lease has expired.
515
+ * Untracks each, frees its cap slot, and fire-and-forget marks any
516
+ * orphaned `Status='Running'` run records as abandoned. Runs first so
517
+ * it can free slots BEFORE the cap check throttles dispatch.
518
+ *
519
+ * PHASE 2 — Cap-bounded dispatch loop:
520
+ * For each due job, atomically acquire its lock via spAcquireScheduledJobLock.
521
+ * Only jobs whose lock was acquired count against MaxConcurrentJobs.
522
+ * Lock-failed jobs are reported via `lockedOut` counter.
523
+ * If at-cap, remaining due jobs counted via `skippedAtCapacity` and
524
+ * picked up by subsequent polls as slots free (no in-memory queueing).
525
+ *
526
+ * Same-instance double-dispatch is structurally prevented by the atomic
527
+ * lock sproc — its WHERE clause filters held-and-not-stale locks, so any
528
+ * second attempt against the same job ID returns Acquired=0.
529
+ *
530
+ * In-flight dispatched promises are tracked in `inflightJobPromises` so
531
+ * `StopPolling({ waitForInflight: true })` can perform graceful shutdown.
532
+ *
533
+ * @returns Counters for observability.
534
+ */
535
+ async DispatchScheduledJobs(contextUser, evalTime = new Date()) {
536
+ if (!this.acceptingDispatches) {
537
+ return { swept: 0, dispatched: 0, lockedOut: 0, skippedAtCapacity: 0 };
538
+ }
539
+ // PHASE 1: stale-inflight sweep. Decoupled from isJobDue and cap.
540
+ const swept = await this.sweepStaleInflightJobs(contextUser);
541
+ // PHASE 2: cap-bounded dispatch.
542
+ let dispatched = 0;
543
+ let lockedOut = 0;
544
+ let skippedAtCapacity = 0;
545
+ for (const job of this.ScheduledJobs) {
546
+ if (!this.isJobDue(job, evalTime))
547
+ continue;
548
+ if (this.inflightJobPromises.size >= this.MaxConcurrentJobs) {
549
+ skippedAtCapacity++;
550
+ continue;
551
+ }
552
+ const lockResult = await this.tryAcquireLock(job.ID);
553
+ if (!lockResult.acquired) {
554
+ if (job.ConcurrencyMode === 'Queue') {
555
+ await this.createQueuedJobRun(job, contextUser);
556
+ }
557
+ lockedOut++;
558
+ continue;
559
+ }
560
+ // TDZ-safe identity tracking: set Map entry SYNCHRONOUSLY before
561
+ // attaching catch/finally. A synchronous throw in executeJobWithLock
562
+ // (e.g. ClassFactory.CreateInstance failing on a missing DriverClass)
563
+ // would otherwise fire .finally before .set runs and orphan the entry.
564
+ const promise = this.executeJobWithLock(job, lockResult.token, contextUser);
565
+ this.inflightJobPromises.set(job.ID, promise);
566
+ promise
567
+ .catch(error => {
568
+ this.logError(`Unexpected throw escaping executeJobWithLock for job ${job.Name} ` +
569
+ `(indicates a bug in the engine itself, not a plugin)`, error);
570
+ return null;
571
+ })
572
+ .finally(() => {
573
+ // Identity check: only delete if the entry still refers to
574
+ // OUR promise. If a sweep + re-dispatch already replaced it,
575
+ // leave the new entry alone.
576
+ if (this.inflightJobPromises.get(job.ID) === promise) {
577
+ this.inflightJobPromises.delete(job.ID);
578
+ }
579
+ });
580
+ dispatched++;
581
+ }
582
+ this.UpdatePollingInterval();
583
+ return { swept, dispatched, lockedOut, skippedAtCapacity };
352
584
  }
353
585
  /**
354
586
  * Determine if a job is currently due for execution
@@ -376,30 +608,31 @@ export class SchedulingEngine extends BaseSingleton {
376
608
  /**
377
609
  * Execute a single scheduled job
378
610
  *
379
- * @param job - The job to execute
380
- * @param contextUser - User context
381
- * @returns The created job run record
611
+ * Execute a single scheduled job WITH a pre-acquired lock token.
612
+ *
613
+ * Caller (DispatchScheduledJobs / ExecuteScheduledJob / ExecuteScheduledJobs)
614
+ * is responsible for acquiring the lock atomically via tryAcquireLock and
615
+ * passing the resulting token. This method owns the lock's lifecycle from
616
+ * this point forward: every exit path (success, failure, exception)
617
+ * releases the lock via releaseLockIfTokenMatches.
618
+ *
619
+ * If lockToken is null, the job is running in `ConcurrencyMode='Concurrent'`
620
+ * (no lock acquired) — finally simply skips the release.
621
+ *
622
+ * @param job - The job entity. READ-ONLY from this method's perspective —
623
+ * do not mutate or call Save on it. The shared entity in
624
+ * this.ScheduledJobs must not be touched here.
625
+ * @param lockToken - The token returned by tryAcquireLock, or null for
626
+ * Concurrent mode where no lock was acquired.
627
+ * @param contextUser - User context for execution.
628
+ * @returns The created run record (Completed or Failed), or null if a
629
+ * synchronous setup error prevented run creation.
382
630
  * @private
383
631
  */
384
- async executeJob(job, contextUser) {
385
- // Try to acquire lock for this job
386
- const lockAcquired = await this.tryAcquireLock(job);
387
- if (!lockAcquired) {
388
- // Handle based on concurrency mode
389
- if (job.ConcurrencyMode === 'Skip') {
390
- this.log(`Job ${job.Name} is locked, skipping (ConcurrencyMode=Skip)`);
391
- return null; // Skip this execution
392
- }
393
- else if (job.ConcurrencyMode === 'Queue') {
394
- this.log(`Job ${job.Name} is locked, queueing (ConcurrencyMode=Queue)`);
395
- // Create a queued run record for future processing
396
- return await this.createQueuedJobRun(job, contextUser);
397
- }
398
- // Concurrent mode: proceed without lock
399
- }
400
- // Create run record
401
- const run = await this.createJobRun(job, contextUser);
632
+ async executeJobWithLock(job, lockToken, contextUser) {
633
+ let run = null;
402
634
  try {
635
+ run = await this.createJobRun(job, contextUser);
403
636
  // Get job type
404
637
  const jobType = this.ScheduledJobTypes.find(t => UUIDsEqual(t.ID, job.JobTypeID));
405
638
  if (!jobType) {
@@ -426,7 +659,13 @@ export class SchedulingEngine extends BaseSingleton {
426
659
  run.Success = result.Success;
427
660
  run.ErrorMessage = result.ErrorMessage || null;
428
661
  run.Details = result.Details ? JSON.stringify(result.Details) : null;
429
- await run.Save();
662
+ const runSaved = await run.Save();
663
+ if (!runSaved) {
664
+ this.logError(`Failed to save run record for job ${job.Name} (run ${run.ID}): ` +
665
+ `${run.LatestResult?.CompleteMessage ?? 'unknown'}`, null);
666
+ // Continue — stats update + release are still important even if
667
+ // the run record save failed (best-effort persistence).
668
+ }
430
669
  // Update job statistics
431
670
  await this.updateJobStatistics(job, result.Success, run.ID);
432
671
  // Send notifications if configured
@@ -438,21 +677,28 @@ export class SchedulingEngine extends BaseSingleton {
438
677
  return run;
439
678
  }
440
679
  catch (error) {
441
- // Update run with failure
442
- run.CompletedAt = new Date();
443
- run.Status = 'Failed';
444
- run.Success = false;
445
- run.ErrorMessage = error instanceof Error ? error.message : 'Unknown error';
446
- await run.Save();
447
- // Update job failure count
448
- await this.updateJobStatistics(job, false, run.ID);
680
+ if (run) {
681
+ run.CompletedAt = new Date();
682
+ run.Status = 'Failed';
683
+ run.Success = false;
684
+ run.ErrorMessage = error instanceof Error ? error.message : 'Unknown error';
685
+ const runSaved = await run.Save();
686
+ if (!runSaved) {
687
+ this.logError(`Failed to save failed-run record for job ${job.Name} (run ${run.ID}): ` +
688
+ `${run.LatestResult?.CompleteMessage ?? 'unknown'}`, null);
689
+ // Continue — stats update + release still need to happen.
690
+ }
691
+ await this.updateJobStatistics(job, false, run.ID);
692
+ }
449
693
  this.logError(`Job failed: ${job.Name}`, error);
450
694
  return run;
451
695
  }
452
696
  finally {
453
- // Release lock if we acquired it
454
- if (lockAcquired) {
455
- await this.releaseLock(job);
697
+ // Token-checked release — safe under lease-expiry races.
698
+ // Sproc no-ops if our token no longer matches (another holder reclaimed).
699
+ // Skip entirely for Concurrent mode (no lock to release).
700
+ if (lockToken) {
701
+ await this.releaseLockIfTokenMatches(job.ID, lockToken);
456
702
  }
457
703
  }
458
704
  }
@@ -474,7 +720,11 @@ export class SchedulingEngine extends BaseSingleton {
474
720
  * Update job statistics after execution
475
721
  * @private
476
722
  */
477
- async updateJobStatistics(job, success, runId) {
723
+ async updateJobStatistics(job, success, _runId) {
724
+ const now = new Date();
725
+ const nextRun = CronExpressionHelper.GetNextRunTime(job.CronExpression, job.Timezone);
726
+ // Update in-memory entity so callers reading job.RunCount/job.NextRunAt
727
+ // immediately after see the new values without a Load round-trip.
478
728
  job.RunCount++;
479
729
  if (success) {
480
730
  job.SuccessCount++;
@@ -482,9 +732,49 @@ export class SchedulingEngine extends BaseSingleton {
482
732
  else {
483
733
  job.FailureCount++;
484
734
  }
485
- job.LastRunAt = new Date();
486
- job.NextRunAt = CronExpressionHelper.GetNextRunTime(job.CronExpression, job.Timezone);
487
- await job.Save();
735
+ job.LastRunAt = now;
736
+ job.NextRunAt = nextRun;
737
+ // Persist via the targeted sproc — touches ONLY the 5 stats columns.
738
+ // We deliberately do NOT call job.Save() here: a full-entity Save would
739
+ // overwrite the DB's lock columns with the entity's stale in-memory
740
+ // values (the entity was loaded BEFORE the atomic lock sproc set
741
+ // LockToken), blowing away the live lock. That manifested as a
742
+ // false-positive token mismatch on every successful job completion.
743
+ // See plans/scheduled-job-engine-decoupling.md and the migration
744
+ // V202606022027__v5.39.x__Scheduling_Engine_Atomic_Stats_Update.sql.
745
+ // MJ pattern: positional placeholders; see tryAcquireLock for rationale.
746
+ const provider = this.Base.ProviderToUse;
747
+ const schema = provider.MJCoreSchemaName;
748
+ await provider.ExecuteSQL(`EXEC [${schema}].[spUpdateScheduledJobStatistics] ` +
749
+ `@JobID=@p0, @Success=@p1, @LastRunAt=@p2, @NextRunAt=@p3`, [job.ID, success ? 1 : 0, now, nextRun], { isMutation: true, description: 'spUpdateScheduledJobStatistics' }, this.Base.ContextUser);
750
+ // Direct SQL bypasses BaseEntity.Save(), so it skips the save event that
751
+ // would normally drive LocalCacheManager invalidation. Without this call,
752
+ // any cached RunView for 'MJ: Scheduled Jobs' (e.g., the Scheduling
753
+ // Dashboard's filtered views) would keep showing stale RunCount /
754
+ // SuccessCount / NextRunAt until the cache TTL expired.
755
+ //
756
+ // We use full invalidation rather than in-place upsert (UpsertSingleEntity)
757
+ // because the dashboard reads through filtered/sorted views — MJ can't
758
+ // safely patch those in JS (it would need to evaluate the SQL filter to
759
+ // know whether the updated row still belongs in each cached result).
760
+ //
761
+ // See guides/CACHING_AND_PUBSUB_GUIDE.md and CLAUDE.md §"Server-Side
762
+ // Caching" — `BypassCache: true` is the documented opposite escape hatch
763
+ // for readers; this is the producer-side counterpart.
764
+ //
765
+ // Wrapped in try/catch: cache invalidation is best-effort observability
766
+ // hygiene, not load-bearing for job execution. If LocalCacheManager throws
767
+ // (e.g., uninitialized in a test environment, transient storage error),
768
+ // the stats UPDATE has already persisted — the worst case is one cached
769
+ // dashboard view stays stale until its TTL.
770
+ try {
771
+ if (LocalCacheManager.Instance.IsInitialized) {
772
+ await LocalCacheManager.Instance.InvalidateEntityCaches('MJ: Scheduled Jobs');
773
+ }
774
+ }
775
+ catch (cacheError) {
776
+ this.logError('Cache invalidation after stats update failed (non-fatal)', cacheError);
777
+ }
488
778
  }
489
779
  /**
490
780
  * Send notifications if configured
@@ -515,118 +805,196 @@ export class SchedulingEngine extends BaseSingleton {
515
805
  * Try to acquire a lock for job execution
516
806
  * @private
517
807
  */
518
- async tryAcquireLock(job) {
519
- console.log(` 🔒 tryAcquireLock: job.LockToken=${job.LockToken?.substring(0, 8) || 'NULL'}, ExpectedCompletionAt=${job.ExpectedCompletionAt?.toISOString() || 'NULL'}`);
520
- if (job.LockToken != null) {
521
- const now = new Date();
522
- console.log(` Lock exists! Checking if stale: ExpectedCompletionAt=${job.ExpectedCompletionAt?.toISOString()}, now=${now.toISOString()}`);
523
- if (job.ExpectedCompletionAt && now > job.ExpectedCompletionAt) {
524
- console.log(` → Lock is STALE, cleaning up...`);
525
- this.log(`Detected stale lock on job ${job.Name}, cleaning up`);
526
- const cleaned = await this.cleanupStaleLock(job);
527
- if (!cleaned) {
528
- console.log(` ❌ Failed to clean up stale lock, skipping`);
529
- return false;
530
- }
531
- // Reload from DB to verify cleanup succeeded
532
- await job.Load(job.ID);
533
- if (job.LockToken != null) {
534
- console.log(` ❌ Lock still present after cleanup (re-acquired by ${job.LockedByInstance}), skipping`);
535
- return false;
536
- }
537
- console.log(` ✓ Stale lock cleaned successfully`);
538
- }
539
- else {
540
- console.log(` → Lock is ACTIVE (not stale), returning false`);
541
- return false;
542
- }
543
- }
544
- else {
545
- console.log(` → No lock exists, will try to acquire`);
808
+ /**
809
+ * Atomically acquire a lock on a job via spAcquireScheduledJobLock.
810
+ * The sproc's WHERE clause handles both the free-lock and stale-lease cases
811
+ * in a single statement — no TOCTOU window between check and write.
812
+ *
813
+ * Operates only on lock columns; never mutates the shared entity in
814
+ * this.ScheduledJobs. Caller passes jobId (string), not the entity object.
815
+ *
816
+ * @returns { acquired: true, token } on success; { acquired: false } otherwise.
817
+ * @private
818
+ */
819
+ async tryAcquireLock(jobId) {
820
+ // uuidv4 (cryptographic) — Math.random()-based GUIDs would risk collision
821
+ // between concurrent engine instances and break the lost-mutex protection
822
+ // that releaseLockIfTokenMatches relies on (token identity = execution identity).
823
+ const token = uuidv4();
824
+ const instance = this.getInstanceIdentifier();
825
+ const expectedCompletion = new Date(Date.now() + this._leaseTimeoutMs);
826
+ const provider = this.Base.ProviderToUse;
827
+ const schema = provider.MJCoreSchemaName;
828
+ // MJ pattern: positional parameter array with @p0/@p1/... placeholders.
829
+ // SQLServerDataProvider binds positional params by index (request.input('p0', val)).
830
+ // The sproc's own parameters are bound by name via `@SprocParam=@p0` syntax.
831
+ // See packages/SchemaEngine/src/RuntimeSchemaManager.ts for the canonical example.
832
+ const rows = await provider.ExecuteSQL(`EXEC [${schema}].[spAcquireScheduledJobLock] ` +
833
+ `@JobID=@p0, @Token=@p1, @Instance=@p2, @ExpectedCompletionAt=@p3`, [jobId, token, instance, expectedCompletion], { isMutation: true, description: 'spAcquireScheduledJobLock' }, this.Base.ContextUser);
834
+ const acquired = rows?.[0]?.Acquired === 1;
835
+ return acquired ? { acquired: true, token } : { acquired: false };
836
+ }
837
+ /**
838
+ * Atomically release a lock IF AND ONLY IF the current DB token matches
839
+ * expectedToken. Prevents the lost-mutex hazard under lease-expiry races:
840
+ * if a stale holder's execution eventually settles after the lease was
841
+ * reclaimed by a fresh holder, this no-ops (token mismatch).
842
+ *
843
+ * Idempotent — safe to call on an already-released lock (returns false).
844
+ *
845
+ * @returns true if released, false if token mismatch / already released
846
+ * @private
847
+ */
848
+ async releaseLockIfTokenMatches(jobId, expectedToken) {
849
+ const provider = this.Base.ProviderToUse;
850
+ const schema = provider.MJCoreSchemaName;
851
+ // MJ pattern: positional placeholders; see tryAcquireLock for rationale.
852
+ const rows = await provider.ExecuteSQL(`EXEC [${schema}].[spReleaseScheduledJobLockIfTokenMatches] ` +
853
+ `@JobID=@p0, @ExpectedToken=@p1`, [jobId, expectedToken], { isMutation: true, description: 'spReleaseScheduledJobLockIfTokenMatches' }, this.Base.ContextUser);
854
+ const released = rows?.[0]?.Released === 1;
855
+ if (!released) {
856
+ this.log(`Lock for job ${jobId.substring(0, 8)} was not released ` +
857
+ `(token mismatch — reclaimed by another holder, or already released)`);
546
858
  }
547
- return this.attemptLockAcquisition(job);
859
+ return released;
548
860
  }
549
861
  /**
550
- * Attempt to acquire a fresh lock on a job that is currently unlocked.
551
- * Separated from tryAcquireLock to keep the stale-cleanup and acquisition
552
- * paths distinct and easier to follow.
862
+ * Pre-flight: verify EXECUTE permission on lock sprocs. Fails LOUDLY at boot
863
+ * if the engine's DB principal lacks grants — much better than a silent
864
+ * runtime failure the next time a job tries to dispatch.
865
+ *
866
+ * Wrapped in try/catch: probe failure (e.g., non-SQL-Server provider where
867
+ * `sys.fn_my_permissions` doesn't exist) must NOT crash boot. We log and
868
+ * continue; any actual permission issue will surface at first sproc call.
869
+ *
553
870
  * @private
554
871
  */
555
- async attemptLockAcquisition(job) {
556
- const lockToken = this.generateGuid();
557
- const instanceId = this.getInstanceIdentifier();
558
- const expectedCompletion = new Date(Date.now() + 10 * 60 * 1000);
872
+ async probeLockSprocPermissions() {
559
873
  try {
560
- // Reload to get latest DB state right before acquiring
561
- await job.Load(job.ID);
562
- if (job.LockToken != null) {
563
- console.log(` ❌ Lock was acquired by another process during reload`);
564
- return false;
565
- }
566
- job.LockToken = lockToken;
567
- job.LockedAt = new Date();
568
- job.LockedByInstance = instanceId;
569
- job.ExpectedCompletionAt = expectedCompletion;
570
- console.log(` → Attempting to save with lock: ${lockToken.substring(0, 8)}...`);
571
- const saveResult = await job.Save();
572
- if (saveResult) {
573
- console.log(` ✅ Lock acquired successfully!`);
574
- return true;
874
+ const provider = this.Base.ProviderToUse;
875
+ const schema = provider.MJCoreSchemaName;
876
+ const sql = `SELECT permission_name FROM sys.fn_my_permissions(` +
877
+ `'${schema}.spAcquireScheduledJobLock', 'OBJECT') WHERE permission_name = 'EXECUTE'`;
878
+ const rows = await provider.ExecuteSQL(sql, [], { isMutation: false, description: 'Scheduling engine permission probe' }, this.Base.ContextUser);
879
+ if (!rows || rows.length === 0) {
880
+ this.logError(`⚠️ Scheduling engine DB principal lacks EXECUTE on ` +
881
+ `${schema}.spAcquireScheduledJobLock. Job dispatch WILL fail. ` +
882
+ `Grant cdp_Developer or cdp_Integration role to the principal and restart.`, null);
575
883
  }
576
884
  else {
577
- console.log(` ❌ Save failed: ${job.LatestResult?.CompleteMessage ?? 'unknown'}`);
578
- this.clearInMemoryLockFields(job);
579
- return false;
885
+ this.log(`Lock sproc permission check OK`);
580
886
  }
581
887
  }
582
- catch (error) {
583
- this.logError(`Failed to acquire lock for job ${job.Name}`, error);
584
- this.clearInMemoryLockFields(job);
585
- return false;
888
+ catch (probeError) {
889
+ // Probe itself failed (non-SQL-Server provider, or unexpected error).
890
+ // Don't crash boot. Log and continue.
891
+ this.log(`Permission probe skipped (provider may not support sys.fn_my_permissions): ${probeError}`);
586
892
  }
587
893
  }
588
894
  /**
589
- * Clear in-memory lock fields without saving — used when a save attempt
590
- * fails and we need the in-memory object to reflect "unlocked" so the
591
- * next poll cycle doesn't see a phantom lock.
895
+ * Sweep stale inflight jobs. Runs unconditionally at top of every poll.
896
+ *
897
+ * SINGLE BATCH QUERY (not N round-trips). Returns only jobs whose lease has
898
+ * expired OR whose lock has already been cleared. In steady-state (no zombies)
899
+ * the query matches zero rows and the sweep is essentially free.
900
+ *
901
+ * For each stale entry:
902
+ * - Untrack the leaked promise from inflightJobPromises (frees cap slot).
903
+ * - FIRE-AND-FORGET abandon any orphaned `Status='Running'` run records.
904
+ * NOT awaited because cleanup must not delay dispatch under a fleet-wide
905
+ * hang event where the sweep finds many zombies at once.
906
+ *
907
+ * Decoupled from:
908
+ * - isJobDue — irrelevant; we care about lease state, not cron.
909
+ * - MaxConcurrentJobs — the sweep IS what frees the cap when saturated by hangs.
910
+ *
911
+ * See plans/scheduled-job-engine-decoupling.md for the full rationale.
912
+ *
913
+ * @returns count of inflight entries swept
592
914
  * @private
593
915
  */
594
- clearInMemoryLockFields(job) {
595
- job.LockToken = null;
596
- job.LockedAt = null;
597
- job.LockedByInstance = null;
598
- job.ExpectedCompletionAt = null;
916
+ async sweepStaleInflightJobs(contextUser) {
917
+ if (this.inflightJobPromises.size === 0)
918
+ return 0;
919
+ const trackedIds = [...this.inflightJobPromises.keys()];
920
+ // ID values interpolated below are engine-generated GUIDs from
921
+ // this.inflightJobPromises keys (originally from this.ScheduledJobs[].ID),
922
+ // never user input. No SQL-injection vector. RunViewParams.ExtraFilter
923
+ // does not support parameterized binding in current MJCore.
924
+ const idList = trackedIds.map(id => `'${id}'`).join(',');
925
+ const nowIso = new Date().toISOString();
926
+ const rv = new RunView(this.Base.RunViewProviderToUse);
927
+ const result = await rv.RunView({
928
+ EntityName: 'MJ: Scheduled Jobs',
929
+ ExtraFilter: `ID IN (${idList}) AND (LockToken IS NULL OR ExpectedCompletionAt IS NULL OR ExpectedCompletionAt < '${nowIso}')`,
930
+ Fields: ['ID', 'Name', 'ExpectedCompletionAt'],
931
+ ResultType: 'simple',
932
+ }, contextUser);
933
+ if (!result.Success || result.Results.length === 0)
934
+ return 0;
935
+ let swept = 0;
936
+ for (const row of result.Results) {
937
+ const jobName = row.Name ?? row.ID;
938
+ this.log(`[sweep] Untracking inflight job ${jobName}: ` +
939
+ `lease=${row.ExpectedCompletionAt?.toISOString() ?? 'NULL'}, now=${nowIso}. ` +
940
+ `Original execution presumed hung. See README "Leaked promise behavior".`);
941
+ this.inflightJobPromises.delete(row.ID);
942
+ swept++;
943
+ // FIRE-AND-FORGET: cleanup must not block dispatch.
944
+ this.abandonOrphanedRunRecords(row.ID, contextUser).catch(err => this.logError(`Background abandon-orphaned-runs failed for ${row.ID}`, err));
945
+ }
946
+ return swept;
599
947
  }
600
948
  /**
601
- * Release a lock after job execution
949
+ * Mark any Running run records for the given job as Failed/abandoned.
950
+ *
951
+ * IMPORTANT: the `Status='Running'` filter is LOAD-BEARING — not just for
952
+ * finding zombies. It also protects against a sweep/release race:
953
+ *
954
+ * - Job completes normally.
955
+ * - executeJobWithLock's finally calls releaseLockIfTokenMatches (clears LockToken).
956
+ * - BEFORE that completes, a poll's sweep query sees LockToken IS NULL
957
+ * and classifies the just-completed job as a zombie.
958
+ * - But its run record is already Status='Completed' (set inside the try block,
959
+ * before the finally), so THIS FILTER excludes it from abandonment.
960
+ *
961
+ * Removing or relaxing this filter would corrupt completed run records.
962
+ * If "optimizing" this method, preserve the Status='Running' filter.
963
+ *
602
964
  * @private
603
965
  */
604
- async releaseLock(job) {
605
- try {
606
- job.LockToken = null;
607
- job.LockedAt = null;
608
- job.LockedByInstance = null;
609
- job.ExpectedCompletionAt = null;
610
- const saved = await job.Save();
966
+ async abandonOrphanedRunRecords(jobId, contextUser) {
967
+ // jobId is an engine-supplied GUID from the sweep's RunView result row,
968
+ // not user input. No SQL-injection vector.
969
+ const rv = new RunView(this.Base.RunViewProviderToUse);
970
+ const result = await rv.RunView({
971
+ EntityName: 'MJ: Scheduled Job Runs',
972
+ ExtraFilter: `ScheduledJobID='${jobId}' AND Status='Running'`,
973
+ ResultType: 'entity_object',
974
+ }, contextUser);
975
+ if (!result.Success || result.Results.length === 0)
976
+ return;
977
+ const now = new Date();
978
+ for (const run of result.Results) {
979
+ run.CompletedAt = now;
980
+ run.Status = 'Failed';
981
+ run.Success = false;
982
+ run.ErrorMessage =
983
+ `Execution abandoned by scheduling engine sweep: lease (started ` +
984
+ `${run.StartedAt?.toISOString()}) expired and the original execution ` +
985
+ `never settled. The hung promise was untracked so its concurrency slot ` +
986
+ `could be reused. See packages/Scheduling/engine/README.md ` +
987
+ `"Leaked promise behavior" for details.`;
988
+ const saved = await run.Save();
611
989
  if (!saved) {
612
- this.logError(`Failed to release lock for job ${job.Name}: ${job.LatestResult?.CompleteMessage ?? 'Save returned false'}`);
990
+ this.logError(`Failed to abandon orphaned run ${run.ID}: ` +
991
+ `${run.LatestResult?.CompleteMessage ?? 'unknown'}`, null);
992
+ }
993
+ else {
994
+ this.log(`[sweep] Abandoned orphaned run ${run.ID} for job ${jobId}`);
613
995
  }
614
- return saved;
615
- }
616
- catch (error) {
617
- this.logError(`Failed to release lock for job ${job.Name}`, error);
618
- return false;
619
996
  }
620
997
  }
621
- /**
622
- * Clean up a stale lock. Returns true if the lock was successfully
623
- * cleared in the database, false if the save failed.
624
- * @private
625
- */
626
- async cleanupStaleLock(job) {
627
- this.log(`Cleaning up stale lock on job ${job.Name} (locked by ${job.LockedByInstance})`);
628
- return this.releaseLock(job);
629
- }
630
998
  /**
631
999
  * Create a queued job run for later execution
632
1000
  * @private
@@ -651,25 +1019,33 @@ export class SchedulingEngine extends BaseSingleton {
651
1019
  return `${os.hostname()}-${process.pid}`;
652
1020
  }
653
1021
  /**
654
- * Generate a GUID for lock tokens
655
- * @private
656
- */
657
- generateGuid() {
658
- return 'xxxxxxxx-xxxx-4xxx-yxxx-xxxxxxxxxxxx'.replace(/[xy]/g, (c) => {
659
- const r = (Math.random() * 16) | 0;
660
- const v = c === 'x' ? r : (r & 0x3) | 0x8;
661
- return v.toString(16);
662
- });
663
- }
664
- /**
665
- * Initialize NextRunAt for jobs that don't have it set
1022
+ * Initialize NextRunAt for jobs that don't have it set.
1023
+ *
1024
+ * If a job has `RunImmediatelyIfNeverRun = true` AND has never run
1025
+ * (`LastRunAt IS NULL`), `NextRunAt` is set to `now()` so the job
1026
+ * executes on the next polling cycle instead of waiting for the next
1027
+ * cron tick. Useful for freshly-seeded jobs that should not wait up
1028
+ * to a full cron interval (e.g. 24h for a daily job) for their first run.
1029
+ *
666
1030
  * @private
667
1031
  */
668
1032
  async initializeNextRunTimes(contextUser) {
669
1033
  for (const job of this.ScheduledJobs) {
670
1034
  if (!job.NextRunAt) {
671
- job.NextRunAt = CronExpressionHelper.GetNextRunTime(job.CronExpression, job.Timezone);
1035
+ if (job.RunImmediatelyIfNeverRun && !job.LastRunAt) {
1036
+ job.NextRunAt = new Date();
1037
+ console.log(` ⏱️ Job ${job.Name} flagged RunImmediatelyIfNeverRun — scheduling for immediate execution`);
1038
+ }
1039
+ else {
1040
+ job.NextRunAt = CronExpressionHelper.GetNextRunTime(job.CronExpression, job.Timezone);
1041
+ }
672
1042
  try {
1043
+ // SAFE: this Save runs in StartPolling's upfront block, BEFORE
1044
+ // isPolling=true is set. No locks can have been acquired yet,
1045
+ // so the full-entity Save cannot clobber any live lock state.
1046
+ // If you ever move this call site outside the upfront block,
1047
+ // refactor to a targeted UPDATE sproc — see
1048
+ // updateJobStatistics for the pattern.
673
1049
  await job.Save();
674
1050
  console.log(` ⚙️ Initialized NextRunAt for ${job.Name} -> ${job.NextRunAt.toISOString()}`);
675
1051
  }
@@ -680,56 +1056,62 @@ export class SchedulingEngine extends BaseSingleton {
680
1056
  }
681
1057
  }
682
1058
  /**
683
- * Clean up stale locks on startup
1059
+ * Clean up stale locks on startup using atomic sprocs.
1060
+ *
1061
+ * For each job whose DB shows a stale lock (ExpectedCompletionAt < now OR
1062
+ * ExpectedCompletionAt IS NULL while LockToken IS NOT NULL):
1063
+ * 1. Atomically acquire the stale lock with a fresh token (sproc's WHERE
1064
+ * handles the stale-detection in a single statement).
1065
+ * 2. Immediately release it with that same token.
1066
+ *
1067
+ * Net effect: stale lock cleared atomically with zero TOCTOU window. Uses
1068
+ * the new sproc-backed pattern instead of load-compare-save on shared
1069
+ * this.ScheduledJobs entities (see plans/scheduled-job-engine-decoupling.md
1070
+ * for why the old pattern was unsafe once polling became concurrent).
1071
+ *
684
1072
  * @private
685
1073
  */
686
- async cleanupStaleLocks(contextUser) {
1074
+ async cleanupStaleLocks(_contextUser) {
687
1075
  const now = new Date();
688
- let cleanedCount = 0;
689
1076
  console.log(` 🔍 Checking for stale locks (current time: ${now.toISOString()})...`);
690
- for (const job of this.ScheduledJobs) {
691
- if (job.LockToken) {
692
- console.log(` Job "${job.Name}": LockToken=${job.LockToken?.substring(0, 8)}..., ExpectedCompletionAt=${job.ExpectedCompletionAt?.toISOString() || 'NULL'}`);
693
- if (job.ExpectedCompletionAt) {
694
- const isStale = job.ExpectedCompletionAt < now;
695
- console.log(` → Is stale? ${isStale} (${job.ExpectedCompletionAt.getTime()} < ${now.getTime()} = ${job.ExpectedCompletionAt.getTime() < now.getTime()})`);
696
- if (isStale) {
697
- console.log(` 🔓 Cleaning stale lock (locked by ${job.LockedByInstance})`);
698
- job.LockToken = null;
699
- job.LockedAt = null;
700
- job.LockedByInstance = null;
701
- job.ExpectedCompletionAt = null;
702
- try {
703
- await job.Save();
704
- cleanedCount++;
705
- }
706
- catch (error) {
707
- this.logError(`Failed to clean stale lock for job ${job.Name}`, error);
708
- }
709
- }
1077
+ // Use the engine's already-loaded cache instead of round-tripping to
1078
+ // the DB — this method runs in StartPolling's upfront block, IMMEDIATELY
1079
+ // after Config() loads this.Base.ScheduledJobs, so the cached lock-column
1080
+ // values are current (no stale-state risk that the sweep path has). Saves
1081
+ // one query + silences the "already loaded by SchedulingEngineBase"
1082
+ // telemetry warning.
1083
+ //
1084
+ // Caveat: only safe HERE because the cache was just loaded. The sweep
1085
+ // path (sweepStaleInflightJobs) MUST hit the DB because by then the
1086
+ // cache's lock columns are stale (atomic sprocs bypass the entity cache).
1087
+ const stale = this.Base.ScheduledJobs.filter(job => job.LockToken != null &&
1088
+ (job.ExpectedCompletionAt == null || job.ExpectedCompletionAt < now));
1089
+ if (stale.length === 0) {
1090
+ console.log(` ✓ No stale locks found`);
1091
+ return;
1092
+ }
1093
+ let cleanedCount = 0;
1094
+ for (const job of stale) {
1095
+ try {
1096
+ // Atomic reclaim: acquire returns Acquired=1 if the lock was
1097
+ // stale (per the sproc's WHERE clause). Then immediately release
1098
+ // with the new token to leave the lock free.
1099
+ const lockResult = await this.tryAcquireLock(job.ID);
1100
+ if (lockResult.acquired) {
1101
+ await this.releaseLockIfTokenMatches(job.ID, lockResult.token);
1102
+ console.log(` 🔓 Cleared stale lock on "${job.Name}" (was held by ${job.LockedByInstance})`);
1103
+ cleanedCount++;
710
1104
  }
711
1105
  else {
712
- console.log(` ⚠️ Lock exists but no ExpectedCompletionAt - clearing anyway`);
713
- job.LockToken = null;
714
- job.LockedAt = null;
715
- job.LockedByInstance = null;
716
- job.ExpectedCompletionAt = null;
717
- try {
718
- await job.Save();
719
- cleanedCount++;
720
- }
721
- catch (error) {
722
- this.logError(`Failed to clean stale lock for job ${job.Name}`, error);
723
- }
1106
+ // Another instance acquired between our cache load and acquire.
1107
+ console.log(` ℹ️ Stale lock on "${job.Name}" was cleared by another holder`);
724
1108
  }
725
1109
  }
1110
+ catch (error) {
1111
+ this.logError(`Failed to clean stale lock for job ${job.Name}`, error);
1112
+ }
726
1113
  }
727
- if (cleanedCount > 0) {
728
- console.log(` ✅ Cleaned ${cleanedCount} stale lock(s)`);
729
- }
730
- else {
731
- console.log(` ✓ No stale locks found`);
732
- }
1114
+ console.log(` ✅ Cleaned ${cleanedCount} stale lock(s)`);
733
1115
  }
734
1116
  log(message) {
735
1117
  LogStatusEx({