@stonyx/cron 0.2.1-alpha.26 → 0.2.1-alpha.27

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -35,6 +35,14 @@ When a job is executed, its next trigger time is updated, and it is re-inserted
35
35
  | `register` | `key: string, callback: Function, interval: number, runOnInit?: boolean` | Register a new job with a given interval in seconds. If `runOnInit` is true, the job runs immediately upon registration. |
36
36
  | `unregister` | `key: string` | Remove a previously registered job. |
37
37
 
38
+ > **Callback semantics.** Callbacks are invoked fire-and-forget: `Cron` never waits for one to settle, and reschedules a job *before* invoking it. Two *different* jobs that fall due on the same tick may therefore overlap.
39
+ >
40
+ > A job that is still running when it next falls due is skipped — and **keeps** being skipped until that invocation settles. One warning is logged per stuck run (not per tick), including how long the invocation has been running. `Cron` provides no timeout by design, so **bounding your own callback is your responsibility**: a promise that never settles means that job never runs again for the lifetime of the process. Other jobs are unaffected.
41
+ >
42
+ > Synchronous throws and asynchronous rejections are both caught and reported through `log.error`, with the error's stack interpolated into the message. Neither can stop the scheduler.
43
+ >
44
+ > `interval` is **whole seconds, as a string**, and the value must be *wholly* numeric. `register` throws a `TypeError` on anything else — including **partially** numeric values: `'1h'`, `'30s'` and `'5m'` are rejected outright, **not** read as 1, 30 and 5 seconds. Cron expressions are rejected for the same reason; use `CronService` (`@stonyx/cron/service`) for those, and an empty string is rejected too. The interval is read with `Number()`, so exponent notation and surrounding whitespace resolve at full value (`'1e3'` is 1000 seconds, `' 60 '` is 60). A value that *is* wholly numeric but below `1` (`'0'`, `'-5'`) is interpretable as “as often as possible” and is clamped to `1` second with a warning rather than rejected.
45
+
38
46
  > `MinHeap` is also exported as a public subpath (`@stonyx/cron/min-heap`) and can be imported directly for advanced usage.
39
47
 
40
48
  ## Configuration
package/dist/main.d.ts CHANGED
@@ -3,6 +3,17 @@ interface CronJob extends HeapItem {
3
3
  callback: () => void | Promise<void>;
4
4
  interval: string;
5
5
  key: string;
6
+ /**
7
+ * Timestamp (ms) at which the current invocation started; `undefined` when the
8
+ * job is idle. Mirrors `job.state.runningAtMs` in the service tier
9
+ * (`src/job.ts` `markRunning`/`applyResult`/`isDue`).
10
+ */
11
+ runningAtMs?: number;
12
+ /**
13
+ * True once a skip has been reported for the *current* invocation. Bounds the
14
+ * still-running warning to one line per stuck run instead of one per tick.
15
+ */
16
+ skipReported?: boolean;
6
17
  }
7
18
  export default class Cron {
8
19
  static instance: Cron | null;
@@ -13,8 +24,55 @@ export default class Cron {
13
24
  init(): Promise<void>;
14
25
  scheduleNextRun(): void;
15
26
  runDueJobs(): Promise<void>;
27
+ /**
28
+ * The one safe way this class invokes a consumer callback.
29
+ *
30
+ * Never blocks the caller, catches synchronous throws and asynchronous
31
+ * rejections alike, and skips the invocation entirely when the job's previous
32
+ * invocation has not settled yet (fire-and-forget would otherwise let a slow
33
+ * job stack invocations on itself).
34
+ */
35
+ safeInvoke(job: CronJob, runOnInit?: boolean): void;
36
+ /**
37
+ * Report a scheduler-level message without ever letting the logger's own
38
+ * failure reach the caller.
39
+ *
40
+ * `@stonyx/logs` convenience methods return a promise and write to disk
41
+ * through an unguarded `mkdirSync` + `fsp.appendFile`. On a read-only or full
42
+ * log volume that promise rejects; an unobserved rejection raised from inside
43
+ * the handler that exists to prevent unhandled rejections would re-create
44
+ * exactly the defect this class was fixed for (measured: exit code 1).
45
+ */
46
+ report(level: 'error' | 'warn', message: string): void;
47
+ /** Release a job's in-flight guard. Only ever called for the job it belongs to. */
48
+ release(job: CronJob): void;
16
49
  register(key: string, callback: () => void | Promise<void>, interval: string, runOnInit?: boolean): void;
17
50
  unregister(key: string): void;
51
+ /**
52
+ * Read a job interval (whole seconds, as a string) as a number, WITHOUT
53
+ * applying the floor. Returns `null` when the value is not wholly numeric.
54
+ *
55
+ * `Number()` rather than `parseInt`, deliberately. `parseInt` stops at the
56
+ * first non-numeric character and so fails in the dangerous direction: it
57
+ * reads `'1h'` as 1, `'30s'` as 30 and `'5m'` as 5 — intervals 3600x, 120x and
58
+ * 60x faster than written, scheduled with no error attached to them. A `NaN`
59
+ * check catches a cron expression but not those, and those are the likelier
60
+ * typo: `stonyx-orm` hands `DB_SAVE_INTERVAL` straight through from the
61
+ * environment as a string. `Number()` reads the whole value or none of it, and
62
+ * also gets `'1e3'` (1000, not 1) and `' 60 '` (60) right.
63
+ *
64
+ * An empty or whitespace-only string is rejected rather than read as
65
+ * `Number('')` === 0, so a missing value is a loud error and not a job silently
66
+ * clamped to the floor.
67
+ */
68
+ toSeconds(interval: string): number | null;
69
+ /**
70
+ * Parse a job interval into a positive integer at or above the floor.
71
+ *
72
+ * Returns `null` when the value cannot be parsed at all, so callers can choose
73
+ * between failing fast (`register`) and falling back (`setNextTrigger`).
74
+ */
75
+ parseInterval(interval: string): number | null;
18
76
  setNextTrigger(job: CronJob): void;
19
77
  log(text: string, key?: string | null): void;
20
78
  }
package/dist/main.js CHANGED
@@ -17,6 +17,28 @@ import config from 'stonyx/config';
17
17
  import log from 'stonyx/log';
18
18
  import { getTimestamp } from '@stonyx/utils/date';
19
19
  import MinHeap from './min-heap.js';
20
+ /**
21
+ * Floor for a job interval, in whole seconds.
22
+ *
23
+ * `runDueJobs` no longer awaits the callback, so `next.nextTrigger > now` is the
24
+ * drain loop's only exit condition *and* the loop has no suspension point left.
25
+ * An interval that fails to advance `nextTrigger` therefore spins the loop
26
+ * forever and blocks the event loop, rather than merely scheduling too often.
27
+ */
28
+ const MIN_INTERVAL_SECONDS = 1;
29
+ /**
30
+ * Render an unknown thrown value as log text.
31
+ *
32
+ * `@stonyx/logs` reads a second argument as `logToFile`, not as a format
33
+ * argument, so `log.error(message, err)` discards the error entirely *and*
34
+ * forces a disk write on every failure. The error has to be interpolated into
35
+ * the message instead — the shape `CronService.executeJob` already uses.
36
+ */
37
+ function describeError(err) {
38
+ if (err instanceof Error)
39
+ return err.stack ?? `${err.name}: ${err.message}`;
40
+ return String(err);
41
+ }
20
42
  export default class Cron {
21
43
  static instance;
22
44
  jobs = {};
@@ -55,18 +77,118 @@ export default class Cron {
55
77
  const job = heap.pop();
56
78
  if (config.debug)
57
79
  this.log('job has been triggered', job.key);
58
- try {
59
- await job.callback();
60
- }
61
- catch (err) {
62
- log.error(`Cron job "${job.key}" failed:`, err);
63
- }
80
+ // Reschedule *before* invoking. The callback's result is not used by this
81
+ // class (`runDueJobs` returns void), so awaiting it bought nothing and
82
+ // cost the scheduler: a callback that never settled left the job absent
83
+ // from the heap and stopped the timer from ever re-arming.
64
84
  this.setNextTrigger(job);
65
85
  heap.push(job);
86
+ this.safeInvoke(job);
66
87
  }
67
88
  this.scheduleNextRun();
68
89
  }
90
+ /**
91
+ * The one safe way this class invokes a consumer callback.
92
+ *
93
+ * Never blocks the caller, catches synchronous throws and asynchronous
94
+ * rejections alike, and skips the invocation entirely when the job's previous
95
+ * invocation has not settled yet (fire-and-forget would otherwise let a slow
96
+ * job stack invocations on itself).
97
+ */
98
+ safeInvoke(job, runOnInit = false) {
99
+ const { key } = job;
100
+ const context = runOnInit ? 'failed on init:' : 'failed:';
101
+ // The in-flight guard lives on the job object, not in a module-level set
102
+ // keyed by string. That is what gives each invocation an identity: the only
103
+ // thing that ever clears the flag is the settle handler of the invocation
104
+ // that set it, and that handler closes over this exact job object. A stale
105
+ // handler therefore cannot release a *later* invocation's guard. It also
106
+ // matches the in-repo idiom one tier up (`job.state.runningAtMs`).
107
+ //
108
+ // `unregister` needs no explicit clear as a result: the flag is dropped with
109
+ // the job object, so a re-registered key gets a fresh object and runs
110
+ // immediately, while the abandoned invocation can only ever release itself.
111
+ if (job.runningAtMs !== undefined) {
112
+ // Bounded: one line per stuck run, not one per tick. A permanently hung
113
+ // job is re-pushed and re-skipped every interval forever, which at the
114
+ // 1s interval this class's own tests use is ~86k log lines a day, per job
115
+ // — a disk-fill and ingest-cost vector on any deployment capturing stdout.
116
+ if (!job.skipReported) {
117
+ job.skipReported = true;
118
+ const runningForSeconds = Math.max(0, Math.round((Date.now() - job.runningAtMs) / 1000));
119
+ this.report('warn', `Cron job ${JSON.stringify(key)} is still running after ${runningForSeconds}s; skipping this `
120
+ + 'tick and any further ticks until it settles (this warning is not repeated for this run)');
121
+ }
122
+ return;
123
+ }
124
+ job.runningAtMs = Date.now();
125
+ job.skipReported = false;
126
+ try {
127
+ const result = job.callback();
128
+ if (result && typeof result.then === 'function') {
129
+ Promise.resolve(result)
130
+ .catch((err) => {
131
+ // Braces matter: returning `report`'s value would put it back into
132
+ // the chain, and `.finally` passes a rejection straight through.
133
+ this.report('error', `Cron job ${JSON.stringify(key)} ${context} ${describeError(err)}`);
134
+ })
135
+ .finally(() => { this.release(job); })
136
+ // Backstop: a throw inside the error handler or the release must not
137
+ // re-create the unhandled rejection this helper exists to prevent.
138
+ .catch(() => { });
139
+ return;
140
+ }
141
+ this.release(job);
142
+ }
143
+ catch (err) {
144
+ this.release(job);
145
+ this.report('error', `Cron job ${JSON.stringify(key)} ${context} ${describeError(err)}`);
146
+ }
147
+ }
148
+ /**
149
+ * Report a scheduler-level message without ever letting the logger's own
150
+ * failure reach the caller.
151
+ *
152
+ * `@stonyx/logs` convenience methods return a promise and write to disk
153
+ * through an unguarded `mkdirSync` + `fsp.appendFile`. On a read-only or full
154
+ * log volume that promise rejects; an unobserved rejection raised from inside
155
+ * the handler that exists to prevent unhandled rejections would re-create
156
+ * exactly the defect this class was fixed for (measured: exit code 1).
157
+ */
158
+ report(level, message) {
159
+ try {
160
+ const result = level === 'error' ? log.error(message) : log.warn(message);
161
+ void Promise.resolve(result).catch(() => { });
162
+ }
163
+ catch {
164
+ // Nowhere left to report to; the logger must never stop the scheduler.
165
+ }
166
+ }
167
+ /** Release a job's in-flight guard. Only ever called for the job it belongs to. */
168
+ release(job) {
169
+ job.runningAtMs = undefined;
170
+ job.skipReported = false;
171
+ }
69
172
  register(key, callback, interval, runOnInit = false) {
173
+ const seconds = this.toSeconds(interval);
174
+ // Fail fast rather than clamp. An interval that is not wholly a number is a
175
+ // programming error — a cron expression handed to the legacy class, or a
176
+ // duration with a unit on it (`'1h'`, `'30s'`) — and clamping or truncating
177
+ // it would silently run a job intended for every hour once per second,
178
+ // hammering whatever the callback talks to. Throwing surfaces it at the call
179
+ // site, at boot, before anything is scheduled. A degenerate-but-numeric
180
+ // interval (`'0'`, `'-5'`) is a different case: it is interpretable as "as
181
+ // often as possible" and is clamped to the floor with one warning.
182
+ if (seconds === null) {
183
+ throw new TypeError(`Cron job ${JSON.stringify(key)} has an invalid interval ${JSON.stringify(interval)}: `
184
+ + 'expected a value that is wholly a whole-second count (e.g. \'30\'). Units are not '
185
+ + 'accepted — \'1h\' is rejected, not read as 1. The legacy Cron class does not accept '
186
+ + 'cron expressions — use CronService for those.');
187
+ }
188
+ if (seconds < MIN_INTERVAL_SECONDS) {
189
+ this.report('warn', `Cron job ${JSON.stringify(key)} interval ${JSON.stringify(interval)} is below the `
190
+ + `${MIN_INTERVAL_SECONDS}s floor; clamping to ${MIN_INTERVAL_SECONDS}s`);
191
+ }
70
192
  const job = { callback, interval, key, nextTrigger: 0 };
71
193
  this.jobs[key] = job;
72
194
  this.setNextTrigger(job);
@@ -74,14 +196,8 @@ export default class Cron {
74
196
  if (config.debug) {
75
197
  this.log(`job has been registered with interval: ${interval}`, key);
76
198
  }
77
- if (runOnInit) {
78
- try {
79
- callback();
80
- }
81
- catch (err) {
82
- log.error(`Cron job "${key}" failed on init:`, err);
83
- }
84
- }
199
+ if (runOnInit)
200
+ this.safeInvoke(job, true);
85
201
  this.scheduleNextRun();
86
202
  }
87
203
  unregister(key) {
@@ -95,8 +211,51 @@ export default class Cron {
95
211
  this.log('job has been unregistered', key);
96
212
  this.scheduleNextRun();
97
213
  }
214
+ /**
215
+ * Read a job interval (whole seconds, as a string) as a number, WITHOUT
216
+ * applying the floor. Returns `null` when the value is not wholly numeric.
217
+ *
218
+ * `Number()` rather than `parseInt`, deliberately. `parseInt` stops at the
219
+ * first non-numeric character and so fails in the dangerous direction: it
220
+ * reads `'1h'` as 1, `'30s'` as 30 and `'5m'` as 5 — intervals 3600x, 120x and
221
+ * 60x faster than written, scheduled with no error attached to them. A `NaN`
222
+ * check catches a cron expression but not those, and those are the likelier
223
+ * typo: `stonyx-orm` hands `DB_SAVE_INTERVAL` straight through from the
224
+ * environment as a string. `Number()` reads the whole value or none of it, and
225
+ * also gets `'1e3'` (1000, not 1) and `' 60 '` (60) right.
226
+ *
227
+ * An empty or whitespace-only string is rejected rather than read as
228
+ * `Number('')` === 0, so a missing value is a loud error and not a job silently
229
+ * clamped to the floor.
230
+ */
231
+ toSeconds(interval) {
232
+ const trimmed = String(interval ?? '').trim();
233
+ if (!trimmed)
234
+ return null;
235
+ const seconds = Number(trimmed);
236
+ if (!Number.isFinite(seconds))
237
+ return null;
238
+ // Whole seconds; a fractional value truncates as it always has.
239
+ return Math.trunc(seconds);
240
+ }
241
+ /**
242
+ * Parse a job interval into a positive integer at or above the floor.
243
+ *
244
+ * Returns `null` when the value cannot be parsed at all, so callers can choose
245
+ * between failing fast (`register`) and falling back (`setNextTrigger`).
246
+ */
247
+ parseInterval(interval) {
248
+ const seconds = this.toSeconds(interval);
249
+ if (seconds === null)
250
+ return null;
251
+ return Math.max(MIN_INTERVAL_SECONDS, seconds);
252
+ }
98
253
  setNextTrigger(job) {
99
- job.nextTrigger = getTimestamp() + parseInt(job.interval, 10);
254
+ // `register` rejects an unparseable interval up front; this floor is the
255
+ // backstop for a job object mutated after registration (`cron.jobs` is
256
+ // public, mutable state) and is what actually guarantees the drain loop
257
+ // terminates. Never let `nextTrigger` land on `NaN` or on `now`.
258
+ job.nextTrigger = getTimestamp() + (this.parseInterval(job.interval) ?? MIN_INTERVAL_SECONDS);
100
259
  }
101
260
  log(text, key = null) {
102
261
  if (!config.cron?.log)
package/dist/service.d.ts CHANGED
@@ -15,8 +15,7 @@ interface ExecuteResult {
15
15
  summary?: string;
16
16
  durationMs?: number;
17
17
  deleted?: boolean;
18
- /** Only set when `status` is `'skipped'`. */
19
- reason?: 'not due' | 'already running' | 'removed';
18
+ reason?: string;
20
19
  }
21
20
  interface ServiceStatus {
22
21
  started: boolean;
@@ -28,7 +27,6 @@ interface ListOptions {
28
27
  }
29
28
  type OnJobDueCallback = (job: Job) => Promise<JobDueResult | void> | JobDueResult | void;
30
29
  export default class CronService {
31
- #private;
32
30
  jobs: Map<string, Job>;
33
31
  heap: MinHeap<HeapEntry>;
34
32
  timer: ReturnType<typeof setTimeout> | null;
@@ -71,25 +69,6 @@ export default class CronService {
71
69
  remove(id: string): Promise<void>;
72
70
  /**
73
71
  * Manually trigger a job.
74
- *
75
- * Returns `{ status: 'skipped', reason }` without invoking the callback when
76
- * the job is not due (`mode: 'due'`), is already in flight
77
- * (`'already running'`), or was removed before the claim landed
78
- * (`'removed'`). Before the phase split a forced run against an in-flight job
79
- * launched a second concurrent invocation; refusing it is AC4 of #34.
80
- *
81
- * CONCURRENCY: the same job is bounded to one in-flight invocation on every
82
- * path, and the timer path invokes due jobs one at a time. `run()` fan-out
83
- * across DIFFERENT jobs is deliberately unbounded - N concurrent `run()`
84
- * calls produce N concurrent consumer callbacks. Before the phase split
85
- * these serialized behind the module-global lock; that serialization was the
86
- * bug, not the feature (one hung callback wedged every other caller), so it
87
- * is not being restored here. The fan-out is caller-driven: it is bounded by
88
- * how many times the consumer chooses to call `run()`, exactly like any other
89
- * async API, and the scheduler never produces it on its own. A consumer that
90
- * exposes `run()` over HTTP or a CLI owns that bound the same way it owns
91
- * request concurrency for every other handler. A per-invoke bound inside the
92
- * service is tracked separately (#35).
93
72
  */
94
73
  run(id: string, mode?: 'due' | 'force'): Promise<ExecuteResult>;
95
74
  /**
@@ -99,46 +78,7 @@ export default class CronService {
99
78
  armTimer(): void;
100
79
  onTimer(): Promise<void>;
101
80
  findDueJobs(nowMs: number): Job[];
102
- /**
103
- * Execute a job in three phases:
104
- *
105
- * 1. claim (locked) - take ownership of the job, detach it from the heap
106
- * 2. invoke (UNLOCKED) - await the consumer callback
107
- * 3. settle (locked) - apply the result, log it, re-insert into the heap
108
- *
109
- * The critical section deliberately excludes phase 2. `onJobDue` is
110
- * arbitrary, unbounded consumer code; awaiting it under the module-global
111
- * lock is what wedged every subsequent `locked()` call (add/update/remove)
112
- * when a callback never settled.
113
- *
114
- * `onTimer` performs the batch claim (findDueJobs + markRunning) for all due
115
- * jobs under a single lock, then enters at phase 2 via `#executeClaimed`.
116
- * That entry point is a `#private` method rather than a parameter on this
117
- * one: as a published `alreadyClaimed` boolean it was a supported way for a
118
- * consumer to skip phase 1 entirely, which defeats the claim guard AC4 asks
119
- * for and allows concurrent `onJobDue` invocations for the same job.
120
- */
121
81
  executeJob(job: Job): Promise<ExecuteResult>;
122
- /**
123
- * Phase 1 - claim. Must be called while holding the lock (`locked()`, whose
124
- * chain is module-global and therefore shared across CronService instances).
125
- *
126
- * Returns `null` on a successful claim, or the reason the claim was refused.
127
- * "already running" is what makes a second `run()` report a skip instead of
128
- * launching a concurrent invocation. "removed" covers the job being deleted
129
- * between `run()`'s unlocked lookup and this lock turn - claiming then would
130
- * `markRunning` an orphan and, worse, `removeFromHeap` an id that may now
131
- * belong to a replacement.
132
- *
133
- * Detaching from the heap here (rather than relying on phase 3 to push a
134
- * fresh entry) is what keeps manual runs from permanently duplicating heap
135
- * entries.
136
- */
137
- claimJob(job: Job): 'already running' | 'removed' | null;
138
- /**
139
- * Phase 3 - settle. Must be called while holding the lock.
140
- */
141
- settleJob(job: Job, status: string, error: string | undefined, summary: string | undefined, startMs: number, durationMs: number): ExecuteResult;
142
82
  removeFromHeap(id: string): void;
143
83
  log(message: string): void;
144
84
  }
package/dist/service.js CHANGED
@@ -12,24 +12,6 @@ import { locked } from './locked.js';
12
12
  import { normalizeJobInput, recoverFlatParams } from './normalize.js';
13
13
  import RunLog from './run-log.js';
14
14
  const MAX_TIMER_DELAY_MS = 60_000;
15
- /**
16
- * Describe a thrown value without ever throwing.
17
- *
18
- * `String(err)` is not total: a null-prototype object, or any object whose
19
- * `toString`/`Symbol.toPrimitive` throws, raises "Cannot convert object to
20
- * primitive value". Consumer callbacks throw arbitrary values, so the error
21
- * handler itself must not be a second failure source.
22
- */
23
- function describeError(err) {
24
- if (err instanceof Error)
25
- return err.message;
26
- try {
27
- return String(err);
28
- }
29
- catch {
30
- return 'unknown error';
31
- }
32
- }
33
15
  export default class CronService {
34
16
  jobs;
35
17
  heap;
@@ -154,25 +136,6 @@ export default class CronService {
154
136
  }
155
137
  /**
156
138
  * Manually trigger a job.
157
- *
158
- * Returns `{ status: 'skipped', reason }` without invoking the callback when
159
- * the job is not due (`mode: 'due'`), is already in flight
160
- * (`'already running'`), or was removed before the claim landed
161
- * (`'removed'`). Before the phase split a forced run against an in-flight job
162
- * launched a second concurrent invocation; refusing it is AC4 of #34.
163
- *
164
- * CONCURRENCY: the same job is bounded to one in-flight invocation on every
165
- * path, and the timer path invokes due jobs one at a time. `run()` fan-out
166
- * across DIFFERENT jobs is deliberately unbounded - N concurrent `run()`
167
- * calls produce N concurrent consumer callbacks. Before the phase split
168
- * these serialized behind the module-global lock; that serialization was the
169
- * bug, not the feature (one hung callback wedged every other caller), so it
170
- * is not being restored here. The fan-out is caller-driven: it is bounded by
171
- * how many times the consumer chooses to call `run()`, exactly like any other
172
- * async API, and the scheduler never produces it on its own. A consumer that
173
- * exposes `run()` over HTTP or a CLI owns that bound the same way it owns
174
- * request concurrency for every other handler. A per-invoke bound inside the
175
- * service is tracked separately (#35).
176
139
  */
177
140
  async run(id, mode = 'force') {
178
141
  const job = this.jobs.get(id);
@@ -181,10 +144,6 @@ export default class CronService {
181
144
  if (mode === 'due' && !isDue(job, Date.now())) {
182
145
  return { status: 'skipped', reason: 'not due' };
183
146
  }
184
- // Deliberately NOT wrapped in locked(): executeJob takes the lock itself
185
- // for its claim and settle phases only. Wrapping here would re-create the
186
- // wedge through a second door, since the callback would again be awaited
187
- // while a lock is held.
188
147
  return this.executeJob(job);
189
148
  }
190
149
  /**
@@ -213,42 +172,16 @@ export default class CronService {
213
172
  }
214
173
  this.running = true;
215
174
  try {
216
- // Phase 1 - claim (locked). Collecting due jobs pops them off the heap
217
- // and marking them running makes them un-collectable by anyone else, so
218
- // both must happen under the same lock.
219
- const dueJobs = await locked(() => {
175
+ await locked(async () => {
220
176
  const nowMs = Date.now();
221
- const due = this.findDueJobs(nowMs);
222
- for (const job of due) {
177
+ const dueJobs = this.findDueJobs(nowMs);
178
+ for (const job of dueJobs) {
223
179
  markRunning(job);
224
180
  }
225
- return due;
226
- });
227
- // Phases 2 and 3 run outside the claim lock. The consumer callback is
228
- // awaited here holding no lock at all, so a callback that never settles
229
- // cannot poison the lock chain.
230
- for (const job of dueJobs) {
231
- try {
232
- await this.#executeClaimed(job);
233
- }
234
- catch (err) {
235
- // One job's unexpected throw must not abort the batch. Every job in
236
- // `dueJobs` is already claimed - marked running and detached from the
237
- // heap - and only its own settle releases it, so aborting here would
238
- // strand every sibling permanently un-due.
239
- //
240
- // This is the outermost handler on the timer path, so it is the one
241
- // that must not be able to throw. `log()` is public, overridable and
242
- // can reach a file transport, so its own failure is swallowed here
243
- // rather than being allowed to take the batch down.
244
- try {
245
- this.log(`Job "${job.name}" (${job.id}) execution failed unexpectedly: ${describeError(err)}`);
246
- }
247
- catch {
248
- // Nothing left to report to.
249
- }
181
+ for (const job of dueJobs) {
182
+ await this.executeJob(job);
250
183
  }
251
- }
184
+ });
252
185
  }
253
186
  finally {
254
187
  this.running = false;
@@ -269,171 +202,50 @@ export default class CronService {
269
202
  }
270
203
  return due;
271
204
  }
272
- /**
273
- * Execute a job in three phases:
274
- *
275
- * 1. claim (locked) - take ownership of the job, detach it from the heap
276
- * 2. invoke (UNLOCKED) - await the consumer callback
277
- * 3. settle (locked) - apply the result, log it, re-insert into the heap
278
- *
279
- * The critical section deliberately excludes phase 2. `onJobDue` is
280
- * arbitrary, unbounded consumer code; awaiting it under the module-global
281
- * lock is what wedged every subsequent `locked()` call (add/update/remove)
282
- * when a callback never settled.
283
- *
284
- * `onTimer` performs the batch claim (findDueJobs + markRunning) for all due
285
- * jobs under a single lock, then enters at phase 2 via `#executeClaimed`.
286
- * That entry point is a `#private` method rather than a parameter on this
287
- * one: as a published `alreadyClaimed` boolean it was a supported way for a
288
- * consumer to skip phase 1 entirely, which defeats the claim guard AC4 asks
289
- * for and allows concurrent `onJobDue` invocations for the same job.
290
- */
291
205
  async executeJob(job) {
292
- // -- Phase 1: claim (locked) --
293
- const refusal = await locked(() => this.claimJob(job));
294
- if (refusal)
295
- return { status: 'skipped', reason: refusal };
296
- return this.#executeClaimed(job);
297
- }
298
- /**
299
- * Phases 2 and 3 for a job that has already been claimed - either by
300
- * `executeJob` above or by `onTimer`'s batch claim.
301
- *
302
- * Private: reaching this without a claim would run the consumer callback for
303
- * a job nobody owns.
304
- */
305
- async #executeClaimed(job) {
306
- // Membership re-check. The claim and the invoke are no longer in the same
307
- // critical section, and sibling callbacks run unlocked, so a `remove()` can
308
- // land in between AND RESOLVE - it used to deadlock. A resolved `remove()`
309
- // must keep meaning "this callback will not fire": settleJob's identity
310
- // guard only cleans up afterwards, by which point the side effect has
311
- // already happened. Identity, not id, so a removed-then-replaced key is
312
- // caught too. Deliberately synchronous with the `onJobDue` call below -
313
- // nothing can interleave between this check and the invocation.
314
- //
315
- // Returning here without a settle is NOT the claim-without-settle hazard
316
- // the try/finally below exists for: the job is already out of `this.jobs`,
317
- // so the object holding `runningAtMs` is unreachable, its heap entry was
318
- // removed by `remove()`, and re-inserting or run-logging it is exactly the
319
- // resurrection `settleJob`'s identity guard refuses.
320
- if (this.jobs.get(job.id) !== job)
321
- return { status: 'skipped', reason: 'removed' };
322
206
  const startMs = Date.now();
323
207
  let status = 'ok';
324
208
  let error;
325
209
  let summary;
326
- let settled;
327
- // The claim above marked the job running and detached it from the heap.
328
- // Phase 3 is the ONLY thing that undoes either, so it must survive every
329
- // non-local exit from phase 2 - including a throw from the catch handler
330
- // itself. A claim with no matching settle is not a degraded state, it is a
331
- // permanently dead job: `runningAtMs` set, no heap entry, `isDue` false
332
- // forever and `run()` refused forever.
333
210
  try {
334
- // -- Phase 2: invoke (NOT locked) --
335
- try {
336
- if (this.onJobDue) {
337
- const result = await this.onJobDue(job);
338
- if (result) {
339
- status = result.status || 'ok';
340
- error = result.error;
341
- summary = result.summary;
342
- }
211
+ if (this.onJobDue) {
212
+ const result = await this.onJobDue(job);
213
+ if (result) {
214
+ status = result.status || 'ok';
215
+ error = result.error;
216
+ summary = result.summary;
343
217
  }
344
218
  }
345
- catch (err) {
346
- status = 'error';
347
- error = describeError(err);
348
- this.log(`Job "${job.name}" (${job.id}) failed: ${error}`);
349
- }
350
219
  }
351
- finally {
352
- // -- Phase 3: settle (locked) --
353
- settled = await locked(() => this.settleJob(job, status, error, summary, startMs, Date.now() - startMs));
220
+ catch (err) {
221
+ status = 'error';
222
+ error = err instanceof Error ? err.message : String(err);
223
+ this.log(`Job "${job.name}" (${job.id}) failed: ${error}`);
354
224
  }
355
- return settled;
356
- }
357
- /**
358
- * Phase 1 - claim. Must be called while holding the lock (`locked()`, whose
359
- * chain is module-global and therefore shared across CronService instances).
360
- *
361
- * Returns `null` on a successful claim, or the reason the claim was refused.
362
- * "already running" is what makes a second `run()` report a skip instead of
363
- * launching a concurrent invocation. "removed" covers the job being deleted
364
- * between `run()`'s unlocked lookup and this lock turn - claiming then would
365
- * `markRunning` an orphan and, worse, `removeFromHeap` an id that may now
366
- * belong to a replacement.
367
- *
368
- * Detaching from the heap here (rather than relying on phase 3 to push a
369
- * fresh entry) is what keeps manual runs from permanently duplicating heap
370
- * entries.
371
- */
372
- claimJob(job) {
373
- if (this.jobs.get(job.id) !== job)
374
- return 'removed';
375
- if (job.state.runningAtMs)
376
- return 'already running';
377
- markRunning(job);
378
- this.removeFromHeap(job.id);
379
- return null;
380
- }
381
- /**
382
- * Phase 3 - settle. Must be called while holding the lock.
383
- */
384
- settleJob(job, status, error, summary, startMs, durationMs) {
385
- try {
386
- const validStatus = (status === 'ok' || status === 'error' || status === 'skipped') ? status : 'error';
387
- applyResult(job, validStatus, error, durationMs);
388
- // The callback ran unlocked, so this job may have been removed - or
389
- // removed and re-registered under the same id (the shape
390
- // `start(initialJobs)` uses) - while it was in flight. Identity, not id.
391
- //
392
- // Deliberately touch NOTHING here. The claim phase already detached this
393
- // job's own heap entry and nothing re-added it, so there is nothing to
394
- // clean up; any entry now filed under this id belongs to the
395
- // replacement, and removing it by id would silently unschedule a live
396
- // job. Do not resurrect a removed job's heap entry or run log either.
397
- if (this.jobs.get(job.id) !== job) {
398
- return { status, error, summary, durationMs };
399
- }
400
- // Log the run
401
- this.runLog.record({
402
- jobId: job.id,
403
- status,
404
- error,
405
- summary,
406
- runAtMs: startMs,
407
- durationMs,
408
- nextRunAtMs: job.state.nextRunAtMs,
409
- });
410
- // Handle one-shot auto-delete. The callback ran unlocked and may have
411
- // pushed a heap entry for this job via update(), so drop it - the job is
412
- // about to stop existing.
413
- if (job.deleteAfterRun && status === 'ok' && !job.enabled) {
414
- this.jobs.delete(job.id);
415
- this.removeFromHeap(job.id);
416
- this.runLog.removeJob(job.id);
417
- return { status, summary, deleted: true };
418
- }
419
- // Re-insert into heap if still active. The callback ran unlocked, so it
420
- // may itself have added a heap entry for this job (via add/update); drop
421
- // any such entry first to preserve one-entry-per-key.
422
- this.removeFromHeap(job.id);
423
- if (job.enabled && job.state.nextRunAtMs) {
424
- this.heap.push({ key: job.id, nextTrigger: job.state.nextRunAtMs });
425
- }
426
- return { status, error, summary, durationMs };
225
+ const durationMs = Date.now() - startMs;
226
+ const validStatus = (status === 'ok' || status === 'error' || status === 'skipped') ? status : 'error';
227
+ applyResult(job, validStatus, error, durationMs);
228
+ // Log the run
229
+ this.runLog.record({
230
+ jobId: job.id,
231
+ status,
232
+ error,
233
+ summary,
234
+ runAtMs: startMs,
235
+ durationMs,
236
+ nextRunAtMs: job.state.nextRunAtMs,
237
+ });
238
+ // Handle one-shot auto-delete
239
+ if (job.deleteAfterRun && status === 'ok' && !job.enabled) {
240
+ this.jobs.delete(job.id);
241
+ this.runLog.removeJob(job.id);
242
+ return { status, summary, deleted: true };
427
243
  }
428
- finally {
429
- // One re-arm covering every exit, rather than one per branch. The claim
430
- // phase detached this job from the heap, so a timer that fired during the
431
- // unlocked invoke would have found an empty heap and armed nothing -
432
- // and `run()` has no `finally { armTimer() }` of its own the way
433
- // `onTimer` does. Without this a manual run() can leave the scheduler
434
- // with no pending wake at all.
435
- this.armTimer();
244
+ // Re-insert into heap if still active
245
+ if (job.enabled && job.state.nextRunAtMs) {
246
+ this.heap.push({ key: job.id, nextTrigger: job.state.nextRunAtMs });
436
247
  }
248
+ return { status, error, summary, durationMs };
437
249
  }
438
250
  // -- Helpers ---------------------------------------------------------
439
251
  removeFromHeap(id) {
@@ -454,19 +266,6 @@ export default class CronService {
454
266
  log(message) {
455
267
  if (!config.cron?.log)
456
268
  return;
457
- // `log.cron` is created by `log.defineType`, which runs in `Cron.init()`
458
- // (src/main.ts) - a DIFFERENT class. A consumer wiring CronService directly
459
- // never runs it, while `config/environment.js` defaults `cron.log` to true,
460
- // so an unguarded call throws `log.cron is not a function`. That throw
461
- // escapes executeJob's catch, and the error-reporting path must never be
462
- // the thing that kills the scheduler. `src/types/stonyx.d.ts:19` declares
463
- // `cron()` unconditionally, so the type system will not catch this.
464
- const { logColor = '#888', logMethod = 'cron' } = config.cron ?? {};
465
- if (typeof log[logMethod] !== 'function')
466
- log.defineType(logMethod, logColor);
467
- const method = log[logMethod];
468
- if (typeof method !== 'function')
469
- return;
470
- method.call(log, `Cron — ${message}`);
269
+ log.cron(`Cron ${message}`);
471
270
  }
472
271
  }
package/package.json CHANGED
@@ -3,7 +3,7 @@
3
3
  "keywords": [
4
4
  "stonyx-module"
5
5
  ],
6
- "version": "0.2.1-alpha.26",
6
+ "version": "0.2.1-alpha.27",
7
7
  "description": "Cron/job scheduler for Stonyx framework",
8
8
  "main": "dist/main.js",
9
9
  "types": "dist/main.d.ts",