@stonyx/cron 0.2.1-alpha.45 → 0.2.1-alpha.46
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -16
- package/dist/main.d.ts +0 -45
- package/dist/main.js +25 -147
- package/dist/service.d.ts +37 -1
- package/dist/service.js +281 -38
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -35,34 +35,20 @@ When a job is executed, its next trigger time is updated, and it is re-inserted
|
|
|
35
35
|
| `register` | `key: string, callback: Function, interval: number, runOnInit?: boolean` | Register a new job with a given interval in seconds. If `runOnInit` is true, the job runs immediately upon registration. |
|
|
36
36
|
| `unregister` | `key: string` | Remove a previously registered job. |
|
|
37
37
|
|
|
38
|
-
> **Callback semantics.** Callbacks are invoked fire-and-forget: `Cron` never waits for one to settle, and reschedules a job *before* invoking it. Two *different* jobs that fall due on the same tick may therefore overlap.
|
|
39
|
-
>
|
|
40
|
-
> A job that is still running when it next falls due is skipped — and **keeps** being skipped until that invocation settles. `Cron` provides no timeout by design, so **bounding your own callback is your responsibility**: a promise that never settles means that job never runs again for the lifetime of the process, even though the scheduler stays healthy and the job stays visible in `jobs` and in the heap. Other jobs are unaffected.
|
|
41
|
-
>
|
|
42
|
-
> One warning is emitted per stuck run (not per tick), including how long the invocation has been running. That warning goes to `log.warn` and is **not** gated by `config.cron.log` — a dropped execution reported on a channel a config flag can silence would be indistinguishable from a healthy scheduler.
|
|
43
|
-
>
|
|
44
|
-
> The same-job guarantee holds for the lifetime of a **registration**, not of a key: `unregister` followed by `register` on a key whose invocation is still in flight builds a fresh job object with a fresh guard, so the replacement can run alongside the abandoned invocation. That is also the only way to recover a permanently stuck job.
|
|
45
|
-
>
|
|
46
|
-
> Synchronous throws and asynchronous rejections are both caught and reported through `log.error`, with the error's stack interpolated into the message. Neither can stop the scheduler. Note that a rejection which previously escaped `register()` as an unhandled rejection — process-fatal under Node's default — is now swallowed into `log.error`.
|
|
47
|
-
|
|
48
38
|
> `MinHeap` is also exported as a public subpath (`@stonyx/cron/min-heap`) and can be imported directly for advanced usage.
|
|
49
39
|
|
|
50
40
|
## Configuration
|
|
51
41
|
|
|
52
|
-
Optionally,
|
|
42
|
+
Optionally, logging and debugging can be enabled through `config.cron`:
|
|
53
43
|
|
|
54
44
|
```js
|
|
55
45
|
config.cron = {
|
|
56
|
-
log: true //
|
|
46
|
+
log: true // enable cron job logs
|
|
57
47
|
};
|
|
58
48
|
|
|
59
49
|
config.debug = true; // optional: debug logs for job registration and execution
|
|
60
50
|
```
|
|
61
51
|
|
|
62
|
-
`config.cron.log` gates **informational** messages only. Error reports and the
|
|
63
|
-
stuck-job warning described above are never gated by it, so setting it to `false`
|
|
64
|
-
cannot make a dropped execution silent.
|
|
65
|
-
|
|
66
52
|
## License
|
|
67
53
|
|
|
68
54
|
Apache — do what you want, just keep attribution.
|
package/dist/main.d.ts
CHANGED
|
@@ -3,24 +3,6 @@ interface CronJob extends HeapItem {
|
|
|
3
3
|
callback: () => void | Promise<void>;
|
|
4
4
|
interval: string;
|
|
5
5
|
key: string;
|
|
6
|
-
/**
|
|
7
|
-
* Timestamp (ms) at which the current invocation started; `undefined` when the
|
|
8
|
-
* job is idle. Optional so the emitted `CronJob` stays assignable from a job
|
|
9
|
-
* object built by a consumer — `jobs`, `heap` and `setNextTrigger` all expose
|
|
10
|
-
* this interface structurally, so a required field is a breaking type change.
|
|
11
|
-
*
|
|
12
|
-
* A timestamp rather than a boolean, mirroring `job.state.runningAtMs` in the
|
|
13
|
-
* service tier (`markRunning` / `applyResult` / `isDue` in `src/job.ts`), and
|
|
14
|
-
* carrying the one fact a stuck-job warning needs: how long it has been stuck.
|
|
15
|
-
* `CronService.running` is a class-level re-entrancy flag and a different
|
|
16
|
-
* concept; reusing that word here would collide.
|
|
17
|
-
*/
|
|
18
|
-
runningAtMs?: number;
|
|
19
|
-
/**
|
|
20
|
-
* True once a skip has been reported for the *current* invocation. Bounds the
|
|
21
|
-
* still-running warning to one line per stuck run instead of one per tick.
|
|
22
|
-
*/
|
|
23
|
-
skipReported?: boolean;
|
|
24
6
|
}
|
|
25
7
|
export default class Cron {
|
|
26
8
|
static instance: Cron | null;
|
|
@@ -33,33 +15,6 @@ export default class Cron {
|
|
|
33
15
|
runDueJobs(): Promise<void>;
|
|
34
16
|
register(key: string, callback: () => void | Promise<void>, interval: string, runOnInit?: boolean): void;
|
|
35
17
|
unregister(key: string): void;
|
|
36
|
-
/**
|
|
37
|
-
* The one place this class invokes a consumer callback.
|
|
38
|
-
*
|
|
39
|
-
* Never blocks the caller, catches synchronous throws and asynchronous
|
|
40
|
-
* rejections identically, and skips the invocation entirely while the job's
|
|
41
|
-
* previous invocation has not settled (fire-and-forget would otherwise let a
|
|
42
|
-
* slow job stack invocations on itself).
|
|
43
|
-
*
|
|
44
|
-
* Everything that touches the callback — including the thenable probe and the
|
|
45
|
-
* handler attachment — is inside the `try`. A callback may return an object
|
|
46
|
-
* whose `then` is a throwing getter, and reading it outside the guard would
|
|
47
|
-
* abort the drain loop before `scheduleNextRun()`, which is defect #36 again.
|
|
48
|
-
*/
|
|
49
|
-
invokeJob(job: CronJob, runOnInit?: boolean): void;
|
|
50
|
-
/**
|
|
51
|
-
* Report a scheduler-level message without ever letting the logger's own
|
|
52
|
-
* failure reach the caller.
|
|
53
|
-
*
|
|
54
|
-
* `@stonyx/logs` convenience methods return a promise and write to disk
|
|
55
|
-
* through an unguarded `mkdirSync` + `fsp.appendFile`. On a read-only or full
|
|
56
|
-
* log volume that promise rejects; an unobserved rejection raised from inside
|
|
57
|
-
* the handler that exists to prevent unhandled rejections would re-create
|
|
58
|
-
* exactly the defect this class was fixed for.
|
|
59
|
-
*/
|
|
60
|
-
report(level: 'error' | 'warn', message: string): void;
|
|
61
|
-
/** Release a job's in-flight guard. Only ever called for the job it belongs to. */
|
|
62
|
-
release(job: CronJob): void;
|
|
63
18
|
setNextTrigger(job: CronJob): void;
|
|
64
19
|
log(text: string, key?: string | null): void;
|
|
65
20
|
}
|
package/dist/main.js
CHANGED
|
@@ -17,30 +17,6 @@ import config from 'stonyx/config';
|
|
|
17
17
|
import log from 'stonyx/log';
|
|
18
18
|
import { getTimestamp } from '@stonyx/utils/date';
|
|
19
19
|
import MinHeap from './min-heap.js';
|
|
20
|
-
/**
|
|
21
|
-
* Render an unknown thrown value as log text.
|
|
22
|
-
*
|
|
23
|
-
* `@stonyx/logs` reads a second argument as `logToFile`, not as a format
|
|
24
|
-
* argument, so `log.error(message, err)` discards the error entirely *and*
|
|
25
|
-
* forces a disk write on every failure. The error has to be interpolated into
|
|
26
|
-
* the message instead — the shape `CronService.executeJob` already uses.
|
|
27
|
-
*/
|
|
28
|
-
function describeError(err) {
|
|
29
|
-
if (err instanceof Error)
|
|
30
|
-
return err.stack ?? `${err.name}: ${err.message}`;
|
|
31
|
-
return String(err);
|
|
32
|
-
}
|
|
33
|
-
/**
|
|
34
|
-
* Render a consumer-supplied job key safely for log output.
|
|
35
|
-
*
|
|
36
|
-
* Keys reach the log verbatim, so a key containing a newline can forge a
|
|
37
|
-
* complete, well-formed log line (`'a:\n[FORGED] Cron::admin - all jobs
|
|
38
|
-
* healthy'`). `JSON.stringify` quotes the value and escapes the control
|
|
39
|
-
* characters, which is also how the key is rendered one tier up.
|
|
40
|
-
*/
|
|
41
|
-
function describeKey(key) {
|
|
42
|
-
return JSON.stringify(key);
|
|
43
|
-
}
|
|
44
20
|
export default class Cron {
|
|
45
21
|
static instance;
|
|
46
22
|
jobs = {};
|
|
@@ -67,41 +43,28 @@ export default class Cron {
|
|
|
67
43
|
if (!nextJob)
|
|
68
44
|
return;
|
|
69
45
|
const delay = Math.max(0, nextJob.nextTrigger - getTimestamp()) * 1000;
|
|
70
|
-
|
|
71
|
-
// otherwise become an unhandled rejection raised from a bare timer callback
|
|
72
|
-
// — the very failure mode this class was fixed for.
|
|
73
|
-
this.timer = setTimeout(() => {
|
|
74
|
-
this.runDueJobs().catch((err) => {
|
|
75
|
-
this.report('error', `Cron scheduler tick failed: ${describeError(err)}`);
|
|
76
|
-
});
|
|
77
|
-
}, delay);
|
|
46
|
+
this.timer = setTimeout(() => this.runDueJobs(), delay);
|
|
78
47
|
}
|
|
79
48
|
async runDueJobs() {
|
|
80
49
|
const now = getTimestamp();
|
|
81
50
|
const { heap } = this;
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
const job = heap.pop();
|
|
92
|
-
if (config.debug)
|
|
93
|
-
this.log('job has been triggered', job.key);
|
|
94
|
-
// Reschedule before invoking: a consumer callback is never awaited here,
|
|
95
|
-
// so a callback that hangs or rejects can no longer starve the drain loop
|
|
96
|
-
// or leave the job orphaned outside the heap.
|
|
97
|
-
this.setNextTrigger(job);
|
|
98
|
-
heap.push(job);
|
|
99
|
-
this.invokeJob(job);
|
|
51
|
+
while (!heap.isEmpty()) {
|
|
52
|
+
const next = heap.peek();
|
|
53
|
+
if (!next || next.nextTrigger > now)
|
|
54
|
+
break;
|
|
55
|
+
const job = heap.pop();
|
|
56
|
+
if (config.debug)
|
|
57
|
+
this.log('job has been triggered', job.key);
|
|
58
|
+
try {
|
|
59
|
+
await job.callback();
|
|
100
60
|
}
|
|
61
|
+
catch (err) {
|
|
62
|
+
log.error(`Cron job "${job.key}" failed:`, err);
|
|
63
|
+
}
|
|
64
|
+
this.setNextTrigger(job);
|
|
65
|
+
heap.push(job);
|
|
101
66
|
}
|
|
102
|
-
|
|
103
|
-
this.scheduleNextRun();
|
|
104
|
-
}
|
|
67
|
+
this.scheduleNextRun();
|
|
105
68
|
}
|
|
106
69
|
register(key, callback, interval, runOnInit = false) {
|
|
107
70
|
const job = { callback, interval, key, nextTrigger: 0 };
|
|
@@ -111,8 +74,14 @@ export default class Cron {
|
|
|
111
74
|
if (config.debug) {
|
|
112
75
|
this.log(`job has been registered with interval: ${interval}`, key);
|
|
113
76
|
}
|
|
114
|
-
if (runOnInit)
|
|
115
|
-
|
|
77
|
+
if (runOnInit) {
|
|
78
|
+
try {
|
|
79
|
+
callback();
|
|
80
|
+
}
|
|
81
|
+
catch (err) {
|
|
82
|
+
log.error(`Cron job "${key}" failed on init:`, err);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
116
85
|
this.scheduleNextRun();
|
|
117
86
|
}
|
|
118
87
|
unregister(key) {
|
|
@@ -126,104 +95,13 @@ export default class Cron {
|
|
|
126
95
|
this.log('job has been unregistered', key);
|
|
127
96
|
this.scheduleNextRun();
|
|
128
97
|
}
|
|
129
|
-
/**
|
|
130
|
-
* The one place this class invokes a consumer callback.
|
|
131
|
-
*
|
|
132
|
-
* Never blocks the caller, catches synchronous throws and asynchronous
|
|
133
|
-
* rejections identically, and skips the invocation entirely while the job's
|
|
134
|
-
* previous invocation has not settled (fire-and-forget would otherwise let a
|
|
135
|
-
* slow job stack invocations on itself).
|
|
136
|
-
*
|
|
137
|
-
* Everything that touches the callback — including the thenable probe and the
|
|
138
|
-
* handler attachment — is inside the `try`. A callback may return an object
|
|
139
|
-
* whose `then` is a throwing getter, and reading it outside the guard would
|
|
140
|
-
* abort the drain loop before `scheduleNextRun()`, which is defect #36 again.
|
|
141
|
-
*/
|
|
142
|
-
invokeJob(job, runOnInit = false) {
|
|
143
|
-
const { key } = job;
|
|
144
|
-
const context = runOnInit ? 'failed on init:' : 'failed:';
|
|
145
|
-
// The in-flight guard lives on the job object, not in a module-level set
|
|
146
|
-
// keyed by string. Object identity is invocation identity: the only thing
|
|
147
|
-
// that clears the guard is the settle handler of the invocation that set it,
|
|
148
|
-
// and that handler closes over this exact job object, so a stale handler can
|
|
149
|
-
// never release a later invocation's guard.
|
|
150
|
-
if (job.runningAtMs !== undefined) {
|
|
151
|
-
// Bounded to one line per stuck run, not one per tick. A permanently hung
|
|
152
|
-
// job is re-pushed and re-skipped every interval forever; at the 1s
|
|
153
|
-
// interval this class's own tests use that measures 43,200 lines/day per
|
|
154
|
-
// job — a disk-fill and ingest-cost vector whose natural operator response
|
|
155
|
-
// is to silence the only signal that the job is dead.
|
|
156
|
-
if (!job.skipReported) {
|
|
157
|
-
job.skipReported = true;
|
|
158
|
-
const runningForSeconds = Math.max(0, Math.round((Date.now() - job.runningAtMs) / 1000));
|
|
159
|
-
// Ungated, deliberately, matching the sibling `CronService` handler. A
|
|
160
|
-
// skipped run is a *lost* execution, and `runDueJobs`/`register` both
|
|
161
|
-
// return `void`, so this is the legacy class's only wedged-job channel.
|
|
162
|
-
// Routing it through `this.log` would put it behind `config.cron.log`,
|
|
163
|
-
// where a permanently dead job is indistinguishable from a healthy one.
|
|
164
|
-
this.report('warn', `Cron job ${describeKey(key)} is still running after ${runningForSeconds}s; skipping this `
|
|
165
|
-
+ 'tick and any further ticks until it settles (this warning is not repeated for this run)');
|
|
166
|
-
}
|
|
167
|
-
return;
|
|
168
|
-
}
|
|
169
|
-
job.runningAtMs = Date.now();
|
|
170
|
-
job.skipReported = false;
|
|
171
|
-
try {
|
|
172
|
-
const result = job.callback();
|
|
173
|
-
if (result && typeof result.then === 'function') {
|
|
174
|
-
Promise.resolve(result)
|
|
175
|
-
.catch((err) => {
|
|
176
|
-
// Braces matter: returning `report`'s value would put it back into
|
|
177
|
-
// the chain, and `.finally` passes a rejection straight through.
|
|
178
|
-
this.report('error', `Cron job ${describeKey(key)} ${context} ${describeError(err)}`);
|
|
179
|
-
})
|
|
180
|
-
.finally(() => { this.release(job); })
|
|
181
|
-
// Backstop: a throw inside the error handler or the release must not
|
|
182
|
-
// re-create the unhandled rejection this helper exists to prevent.
|
|
183
|
-
.catch(() => { });
|
|
184
|
-
return;
|
|
185
|
-
}
|
|
186
|
-
this.release(job);
|
|
187
|
-
}
|
|
188
|
-
catch (err) {
|
|
189
|
-
this.release(job);
|
|
190
|
-
this.report('error', `Cron job ${describeKey(key)} ${context} ${describeError(err)}`);
|
|
191
|
-
}
|
|
192
|
-
}
|
|
193
|
-
/**
|
|
194
|
-
* Report a scheduler-level message without ever letting the logger's own
|
|
195
|
-
* failure reach the caller.
|
|
196
|
-
*
|
|
197
|
-
* `@stonyx/logs` convenience methods return a promise and write to disk
|
|
198
|
-
* through an unguarded `mkdirSync` + `fsp.appendFile`. On a read-only or full
|
|
199
|
-
* log volume that promise rejects; an unobserved rejection raised from inside
|
|
200
|
-
* the handler that exists to prevent unhandled rejections would re-create
|
|
201
|
-
* exactly the defect this class was fixed for.
|
|
202
|
-
*/
|
|
203
|
-
report(level, message) {
|
|
204
|
-
try {
|
|
205
|
-
const result = level === 'error' ? log.error(message) : log.warn(message);
|
|
206
|
-
void Promise.resolve(result).catch(() => { });
|
|
207
|
-
}
|
|
208
|
-
catch {
|
|
209
|
-
// Nowhere left to report to; the logger must never stop the scheduler.
|
|
210
|
-
}
|
|
211
|
-
}
|
|
212
|
-
/** Release a job's in-flight guard. Only ever called for the job it belongs to. */
|
|
213
|
-
release(job) {
|
|
214
|
-
job.runningAtMs = undefined;
|
|
215
|
-
job.skipReported = false;
|
|
216
|
-
}
|
|
217
98
|
setNextTrigger(job) {
|
|
218
99
|
job.nextTrigger = getTimestamp() + parseInt(job.interval, 10);
|
|
219
100
|
}
|
|
220
101
|
log(text, key = null) {
|
|
221
102
|
if (!config.cron?.log)
|
|
222
103
|
return;
|
|
223
|
-
|
|
224
|
-
// line terminators so a key cannot forge a second, well-formed log line;
|
|
225
|
-
// the surrounding format is unchanged.
|
|
226
|
-
const tag = key ? `Cron::${key.replace(/[\r\n]+/g, ' ')}` : `Cron`;
|
|
104
|
+
const tag = key ? `Cron::${key}` : `Cron`;
|
|
227
105
|
log.cron(`${tag} - ${text}:`);
|
|
228
106
|
}
|
|
229
107
|
}
|
package/dist/service.d.ts
CHANGED
|
@@ -15,7 +15,8 @@ interface ExecuteResult {
|
|
|
15
15
|
summary?: string;
|
|
16
16
|
durationMs?: number;
|
|
17
17
|
deleted?: boolean;
|
|
18
|
-
|
|
18
|
+
/** Only set when `status` is `'skipped'`. */
|
|
19
|
+
reason?: 'not due' | 'already running' | 'removed';
|
|
19
20
|
}
|
|
20
21
|
interface ServiceStatus {
|
|
21
22
|
started: boolean;
|
|
@@ -27,6 +28,7 @@ interface ListOptions {
|
|
|
27
28
|
}
|
|
28
29
|
type OnJobDueCallback = (job: Job) => Promise<JobDueResult | void> | JobDueResult | void;
|
|
29
30
|
export default class CronService {
|
|
31
|
+
#private;
|
|
30
32
|
jobs: Map<string, Job>;
|
|
31
33
|
heap: MinHeap<HeapEntry>;
|
|
32
34
|
timer: ReturnType<typeof setTimeout> | null;
|
|
@@ -69,6 +71,21 @@ export default class CronService {
|
|
|
69
71
|
remove(id: string): Promise<void>;
|
|
70
72
|
/**
|
|
71
73
|
* Manually trigger a job.
|
|
74
|
+
*
|
|
75
|
+
* Returns `{ status: 'skipped', reason }` without invoking the callback when
|
|
76
|
+
* the job is not due (`mode: 'due'`), is already in flight
|
|
77
|
+
* (`'already running'`), or was removed before the claim landed
|
|
78
|
+
* (`'removed'`). Before the phase split, a forced run against an in-flight
|
|
79
|
+
* job launched a second concurrent invocation.
|
|
80
|
+
*
|
|
81
|
+
* CONCURRENCY: the same job is bounded to one in-flight invocation on every
|
|
82
|
+
* path, and the timer path invokes due jobs one at a time. `run()` fan-out
|
|
83
|
+
* across DIFFERENT jobs is deliberately unbounded — N concurrent `run()`
|
|
84
|
+
* calls produce N concurrent consumer callbacks. Before the phase split
|
|
85
|
+
* these serialized behind the module-global lock; that serialization was the
|
|
86
|
+
* bug rather than the feature (one hung callback wedged every other caller),
|
|
87
|
+
* so it is not restored here. The fan-out is caller-driven and the scheduler
|
|
88
|
+
* never produces it on its own.
|
|
72
89
|
*/
|
|
73
90
|
run(id: string, mode?: 'due' | 'force'): Promise<ExecuteResult>;
|
|
74
91
|
/**
|
|
@@ -78,6 +95,25 @@ export default class CronService {
|
|
|
78
95
|
armTimer(): void;
|
|
79
96
|
onTimer(): Promise<void>;
|
|
80
97
|
findDueJobs(nowMs: number): Job[];
|
|
98
|
+
/**
|
|
99
|
+
* Execute a job in three phases:
|
|
100
|
+
*
|
|
101
|
+
* 1. claim (locked) — take ownership of the job, detach it from the heap
|
|
102
|
+
* 2. invoke (UNLOCKED) — await the consumer callback
|
|
103
|
+
* 3. settle (locked) — apply the result, log it, re-insert into the heap
|
|
104
|
+
*
|
|
105
|
+
* The critical section deliberately excludes phase 2. `onJobDue` is
|
|
106
|
+
* arbitrary, unbounded consumer code; awaiting it under the module-global
|
|
107
|
+
* lock is what wedged every subsequent `locked()` call (add/update/remove)
|
|
108
|
+
* when a callback never settled.
|
|
109
|
+
*
|
|
110
|
+
* `onTimer` performs the batch claim (`findDueJobs` + `markRunning`) for all
|
|
111
|
+
* due jobs under a single lock and then enters at phase 2 via
|
|
112
|
+
* `#executeClaimed`. That entry point is `#private` rather than a parameter
|
|
113
|
+
* on this method: as a published `alreadyClaimed` flag it would be a
|
|
114
|
+
* supported way to skip phase 1 entirely, defeating the claim guard and
|
|
115
|
+
* allowing concurrent `onJobDue` invocations for the same job.
|
|
116
|
+
*/
|
|
81
117
|
executeJob(job: Job): Promise<ExecuteResult>;
|
|
82
118
|
removeFromHeap(id: string): void;
|
|
83
119
|
log(message: string): void;
|
package/dist/service.js
CHANGED
|
@@ -12,6 +12,24 @@ import { locked } from './locked.js';
|
|
|
12
12
|
import { normalizeJobInput, recoverFlatParams } from './normalize.js';
|
|
13
13
|
import RunLog from './run-log.js';
|
|
14
14
|
const MAX_TIMER_DELAY_MS = 60_000;
|
|
15
|
+
/**
|
|
16
|
+
* Describe a thrown value without ever throwing.
|
|
17
|
+
*
|
|
18
|
+
* `String(err)` is not total: a null-prototype object, or any object whose
|
|
19
|
+
* `toString`/`Symbol.toPrimitive` throws, raises "Cannot convert object to
|
|
20
|
+
* primitive value". Consumer callbacks throw arbitrary values, so the error
|
|
21
|
+
* handler must not become a second failure source of its own.
|
|
22
|
+
*/
|
|
23
|
+
function describeError(err) {
|
|
24
|
+
if (err instanceof Error)
|
|
25
|
+
return err.message;
|
|
26
|
+
try {
|
|
27
|
+
return String(err);
|
|
28
|
+
}
|
|
29
|
+
catch {
|
|
30
|
+
return 'unknown error';
|
|
31
|
+
}
|
|
32
|
+
}
|
|
15
33
|
export default class CronService {
|
|
16
34
|
jobs;
|
|
17
35
|
heap;
|
|
@@ -40,6 +58,23 @@ export default class CronService {
|
|
|
40
58
|
this.started = true;
|
|
41
59
|
if (initialJobs) {
|
|
42
60
|
for (const job of initialJobs) {
|
|
61
|
+
// A `runningAtMs` on a rehydrated job is always stale. The claim it
|
|
62
|
+
// records was taken by a process that is gone, so nothing will ever
|
|
63
|
+
// settle it, and nothing reaps it — there is no lease on the field
|
|
64
|
+
// (tracked on #35). Left in place it is a permanently dead job that
|
|
65
|
+
// still reports healthy: `isDue` returns false forever because of the
|
|
66
|
+
// flag, `run()` answers `'already running'` forever, `update()` never
|
|
67
|
+
// touches `state.runningAtMs`, and `status()` counts it like any other.
|
|
68
|
+
// The consumer's only recovery would be remove() + add(), losing the
|
|
69
|
+
// job id and its run history.
|
|
70
|
+
//
|
|
71
|
+
// Same hazard, same treatment as the hand-release on the `'removed'`
|
|
72
|
+
// path in `#executeClaimed`: a claim with no reachable settle must be
|
|
73
|
+
// released. Assigned directly rather than via `applyResult` for the same
|
|
74
|
+
// reason — this releases the claim and nothing else. The job did not
|
|
75
|
+
// run, so it gets no run-log row, no `lastStatus`, and no recomputed
|
|
76
|
+
// `nextRunAtMs`; it is rescheduled from the store's own value below.
|
|
77
|
+
job.state.runningAtMs = undefined;
|
|
43
78
|
this.jobs.set(job.id, job);
|
|
44
79
|
if (job.enabled && job.state.nextRunAtMs) {
|
|
45
80
|
this.heap.push({ key: job.id, nextTrigger: job.state.nextRunAtMs });
|
|
@@ -136,6 +171,21 @@ export default class CronService {
|
|
|
136
171
|
}
|
|
137
172
|
/**
|
|
138
173
|
* Manually trigger a job.
|
|
174
|
+
*
|
|
175
|
+
* Returns `{ status: 'skipped', reason }` without invoking the callback when
|
|
176
|
+
* the job is not due (`mode: 'due'`), is already in flight
|
|
177
|
+
* (`'already running'`), or was removed before the claim landed
|
|
178
|
+
* (`'removed'`). Before the phase split, a forced run against an in-flight
|
|
179
|
+
* job launched a second concurrent invocation.
|
|
180
|
+
*
|
|
181
|
+
* CONCURRENCY: the same job is bounded to one in-flight invocation on every
|
|
182
|
+
* path, and the timer path invokes due jobs one at a time. `run()` fan-out
|
|
183
|
+
* across DIFFERENT jobs is deliberately unbounded — N concurrent `run()`
|
|
184
|
+
* calls produce N concurrent consumer callbacks. Before the phase split
|
|
185
|
+
* these serialized behind the module-global lock; that serialization was the
|
|
186
|
+
* bug rather than the feature (one hung callback wedged every other caller),
|
|
187
|
+
* so it is not restored here. The fan-out is caller-driven and the scheduler
|
|
188
|
+
* never produces it on its own.
|
|
139
189
|
*/
|
|
140
190
|
async run(id, mode = 'force') {
|
|
141
191
|
const job = this.jobs.get(id);
|
|
@@ -144,6 +194,10 @@ export default class CronService {
|
|
|
144
194
|
if (mode === 'due' && !isDue(job, Date.now())) {
|
|
145
195
|
return { status: 'skipped', reason: 'not due' };
|
|
146
196
|
}
|
|
197
|
+
// Deliberately NOT wrapped in locked(): executeJob takes the lock itself,
|
|
198
|
+
// for its claim and settle phases only. Wrapping here would re-create the
|
|
199
|
+
// wedge through a second door, because the consumer callback would once
|
|
200
|
+
// again be awaited while a lock is held.
|
|
147
201
|
return this.executeJob(job);
|
|
148
202
|
}
|
|
149
203
|
/**
|
|
@@ -172,16 +226,65 @@ export default class CronService {
|
|
|
172
226
|
}
|
|
173
227
|
this.running = true;
|
|
174
228
|
try {
|
|
175
|
-
|
|
229
|
+
// -- Phase 1: claim (locked), batched --
|
|
230
|
+
// Collecting due jobs pops them off the heap, and marking them running
|
|
231
|
+
// makes them un-claimable by anyone else. Both must happen under the
|
|
232
|
+
// same lock turn, or a concurrent run() could claim a job this batch has
|
|
233
|
+
// already detached.
|
|
234
|
+
const dueJobs = await locked(() => {
|
|
176
235
|
const nowMs = Date.now();
|
|
177
|
-
const
|
|
178
|
-
for (const job of
|
|
236
|
+
const due = this.findDueJobs(nowMs);
|
|
237
|
+
for (const job of due) {
|
|
179
238
|
markRunning(job);
|
|
180
239
|
}
|
|
181
|
-
|
|
182
|
-
await this.executeJob(job);
|
|
183
|
-
}
|
|
240
|
+
return due;
|
|
184
241
|
});
|
|
242
|
+
// Phases 2 and 3 run OUTSIDE the claim lock. The consumer callback is
|
|
243
|
+
// awaited here holding no lock at all, so a callback that never settles
|
|
244
|
+
// cannot poison the lock chain and wedge add/update/remove.
|
|
245
|
+
for (const job of dueJobs) {
|
|
246
|
+
try {
|
|
247
|
+
await this.#executeClaimed(job);
|
|
248
|
+
}
|
|
249
|
+
catch (err) {
|
|
250
|
+
// One job's unexpected throw must not abort the batch. Every job in
|
|
251
|
+
// `dueJobs` is already claimed — marked running and detached from
|
|
252
|
+
// the heap — and only its own settle releases it, so aborting here
|
|
253
|
+
// would strand every sibling permanently un-due.
|
|
254
|
+
//
|
|
255
|
+
// Reported on an UNGATED channel. `this.log()` returns early when
|
|
256
|
+
// `config.cron.log` is false, which is a supported production
|
|
257
|
+
// setting, and a failure here permanently unschedules a job while
|
|
258
|
+
// `status()` keeps reporting the service healthy. Silent-and-healthy
|
|
259
|
+
// is exactly the failure class this split exists to remove.
|
|
260
|
+
//
|
|
261
|
+
// This is the outermost handler on the timer path, so it is the one
|
|
262
|
+
// that must not be able to throw: `log` is a shared singleton whose
|
|
263
|
+
// transports can reach the filesystem, so its own failure is
|
|
264
|
+
// swallowed rather than allowed to take the batch down.
|
|
265
|
+
//
|
|
266
|
+
// BOTH halves of that failure have to be caught, and they are caught
|
|
267
|
+
// by different constructs. `log.error` is a chronicle convenience
|
|
268
|
+
// method that returns `logAction(...)` -> `async log(...)`, so its
|
|
269
|
+
// console write, colour lookup, `mkdirSync` and `appendFile` all
|
|
270
|
+
// surface as REJECTIONS, never as synchronous throws. A bare call
|
|
271
|
+
// here escapes this `catch` entirely and terminates the process under
|
|
272
|
+
// Node's default `--unhandled-rejections=throw` — the handler written
|
|
273
|
+
// so it "must not be able to throw" would be the one taking the
|
|
274
|
+
// daemon down. The `try` covers the synchronous half (evaluating the
|
|
275
|
+
// template literal); `Promise.resolve(...).catch()` covers the async
|
|
276
|
+
// half. Deliberately not awaited: the batch must not block on a log
|
|
277
|
+
// transport, and `void` marks the floated promise as intentional.
|
|
278
|
+
try {
|
|
279
|
+
void Promise.resolve(log.error(`Cron — Job "${job.name}" (${job.id}) execution failed unexpectedly: ${describeError(err)}`)).catch(() => {
|
|
280
|
+
// Nothing left to report to.
|
|
281
|
+
});
|
|
282
|
+
}
|
|
283
|
+
catch {
|
|
284
|
+
// Nothing left to report to.
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
}
|
|
185
288
|
}
|
|
186
289
|
finally {
|
|
187
290
|
this.running = false;
|
|
@@ -202,50 +305,190 @@ export default class CronService {
|
|
|
202
305
|
}
|
|
203
306
|
return due;
|
|
204
307
|
}
|
|
308
|
+
/**
|
|
309
|
+
* Execute a job in three phases:
|
|
310
|
+
*
|
|
311
|
+
* 1. claim (locked) — take ownership of the job, detach it from the heap
|
|
312
|
+
* 2. invoke (UNLOCKED) — await the consumer callback
|
|
313
|
+
* 3. settle (locked) — apply the result, log it, re-insert into the heap
|
|
314
|
+
*
|
|
315
|
+
* The critical section deliberately excludes phase 2. `onJobDue` is
|
|
316
|
+
* arbitrary, unbounded consumer code; awaiting it under the module-global
|
|
317
|
+
* lock is what wedged every subsequent `locked()` call (add/update/remove)
|
|
318
|
+
* when a callback never settled.
|
|
319
|
+
*
|
|
320
|
+
* `onTimer` performs the batch claim (`findDueJobs` + `markRunning`) for all
|
|
321
|
+
* due jobs under a single lock and then enters at phase 2 via
|
|
322
|
+
* `#executeClaimed`. That entry point is `#private` rather than a parameter
|
|
323
|
+
* on this method: as a published `alreadyClaimed` flag it would be a
|
|
324
|
+
* supported way to skip phase 1 entirely, defeating the claim guard and
|
|
325
|
+
* allowing concurrent `onJobDue` invocations for the same job.
|
|
326
|
+
*/
|
|
205
327
|
async executeJob(job) {
|
|
328
|
+
// -- Phase 1: claim (locked) --
|
|
329
|
+
const refusal = await locked(() => this.#claimJob(job));
|
|
330
|
+
if (refusal)
|
|
331
|
+
return { status: 'skipped', reason: refusal };
|
|
332
|
+
return this.#executeClaimed(job);
|
|
333
|
+
}
|
|
334
|
+
/**
|
|
335
|
+
* Phases 2 and 3 for a job that has already been claimed — either by
|
|
336
|
+
* `executeJob` above or by `onTimer`'s batch claim.
|
|
337
|
+
*
|
|
338
|
+
* Private: reaching this without a claim would run the consumer callback for
|
|
339
|
+
* a job nobody owns, and would leave nothing to release the claim.
|
|
340
|
+
*/
|
|
341
|
+
async #executeClaimed(job) {
|
|
342
|
+
// Membership re-check. The claim and the invoke are no longer in the same
|
|
343
|
+
// critical section, and sibling callbacks run unlocked, so a `remove()` can
|
|
344
|
+
// now land in between AND RESOLVE — it used to deadlock. A resolved
|
|
345
|
+
// `remove()` must keep meaning "this callback will not fire"; the identity
|
|
346
|
+
// guard in `#settleJob` only cleans up afterwards, by which point the side
|
|
347
|
+
// effect has already happened. Identity, not id, so a removed-then-replaced
|
|
348
|
+
// key is caught too. Deliberately synchronous with the `onJobDue` call
|
|
349
|
+
// below — nothing can interleave between this check and the invocation.
|
|
350
|
+
//
|
|
351
|
+
// This is the one early return after a claim, so it is the one that has to
|
|
352
|
+
// release the claim by hand. Skipping settle is right — re-inserting or
|
|
353
|
+
// run-logging a removed job is the resurrection `#settleJob` refuses, and
|
|
354
|
+
// the heap entry is already gone. But the claim must still come off,
|
|
355
|
+
// because the detached object is NOT unreachable: it is the object `add()`
|
|
356
|
+
// returned and `get()`/`list()` hand out, and `start(initialJobs)`
|
|
357
|
+
// re-registers those objects verbatim, `state` included. A leftover
|
|
358
|
+
// `runningAtMs` rehydrates a permanently dead job — `isDue` false forever,
|
|
359
|
+
// `run()` refused forever, `status()` reporting it healthy.
|
|
360
|
+
//
|
|
361
|
+
// Assigned directly rather than via `applyResult`: this releases the claim
|
|
362
|
+
// and nothing else. No run-log row, no heap entry, no `lastStatus`, no
|
|
363
|
+
// recomputed `nextRunAtMs` — the job did not run.
|
|
364
|
+
if (this.jobs.get(job.id) !== job) {
|
|
365
|
+
job.state.runningAtMs = undefined;
|
|
366
|
+
return { status: 'skipped', reason: 'removed' };
|
|
367
|
+
}
|
|
206
368
|
const startMs = Date.now();
|
|
207
369
|
let status = 'ok';
|
|
208
370
|
let error;
|
|
209
371
|
let summary;
|
|
372
|
+
let settled;
|
|
373
|
+
// The claim marked the job running and detached it from the heap. Phase 3
|
|
374
|
+
// is the ONLY thing that undoes either, so it must survive every non-local
|
|
375
|
+
// exit from phase 2 — including a throw from the catch handler itself
|
|
376
|
+
// (`this.log` is public and overridable and reaches a transport). A claim
|
|
377
|
+
// with no matching settle is not a degraded state, it is a permanently
|
|
378
|
+
// dead job.
|
|
210
379
|
try {
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
if (
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
380
|
+
// -- Phase 2: invoke (NOT locked) --
|
|
381
|
+
try {
|
|
382
|
+
if (this.onJobDue) {
|
|
383
|
+
const result = await this.onJobDue(job);
|
|
384
|
+
if (result) {
|
|
385
|
+
status = result.status || 'ok';
|
|
386
|
+
error = result.error;
|
|
387
|
+
summary = result.summary;
|
|
388
|
+
}
|
|
217
389
|
}
|
|
218
390
|
}
|
|
391
|
+
catch (err) {
|
|
392
|
+
status = 'error';
|
|
393
|
+
error = describeError(err);
|
|
394
|
+
this.log(`Job "${job.name}" (${job.id}) failed: ${error}`);
|
|
395
|
+
}
|
|
219
396
|
}
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
this.log(`Job "${job.name}" (${job.id}) failed: ${error}`);
|
|
397
|
+
finally {
|
|
398
|
+
// -- Phase 3: settle (locked) --
|
|
399
|
+
settled = await locked(() => this.#settleJob(job, status, error, summary, startMs, Date.now() - startMs));
|
|
224
400
|
}
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
401
|
+
return settled;
|
|
402
|
+
}
|
|
403
|
+
/**
|
|
404
|
+
* Phase 1 — claim. Must be called while holding the lock (`locked()`, whose
|
|
405
|
+
* chain is module-global and therefore shared across CronService instances).
|
|
406
|
+
*
|
|
407
|
+
* Returns `null` on a successful claim, or the reason the claim was refused.
|
|
408
|
+
* `'already running'` is what makes a second `run()` report a skip instead of
|
|
409
|
+
* launching a concurrent invocation. `'removed'` covers the job being deleted
|
|
410
|
+
* between `run()`'s unlocked lookup and this lock turn — claiming then would
|
|
411
|
+
* `markRunning` an orphan and, worse, `removeFromHeap` an id that may now
|
|
412
|
+
* belong to a replacement.
|
|
413
|
+
*
|
|
414
|
+
* Detaching from the heap here — rather than relying on phase 3 to push a
|
|
415
|
+
* fresh entry — is what stops manual runs permanently duplicating entries.
|
|
416
|
+
*
|
|
417
|
+
* `#private`: published, this would be a supported call performing
|
|
418
|
+
* `markRunning` + `removeFromHeap` with no guaranteed settle and no lease on
|
|
419
|
+
* `runningAtMs`, so a single such call would strand the job forever. The
|
|
420
|
+
* lock-held precondition cannot be expressed in the type system, so the
|
|
421
|
+
* method must not be reachable from outside the class body.
|
|
422
|
+
*/
|
|
423
|
+
#claimJob(job) {
|
|
424
|
+
if (this.jobs.get(job.id) !== job)
|
|
425
|
+
return 'removed';
|
|
426
|
+
if (job.state.runningAtMs)
|
|
427
|
+
return 'already running';
|
|
428
|
+
markRunning(job);
|
|
429
|
+
this.removeFromHeap(job.id);
|
|
430
|
+
return null;
|
|
431
|
+
}
|
|
432
|
+
/**
|
|
433
|
+
* Phase 3 — settle. Must be called while holding the lock.
|
|
434
|
+
*
|
|
435
|
+
* `#private` for the same reason as `#claimJob`: unlocked it would run
|
|
436
|
+
* `applyResult`, a `runLog.record`, a full `removeFromHeap` rebuild, a
|
|
437
|
+
* `heap.push` and an `armTimer` with no mutual exclusion — exactly the
|
|
438
|
+
* corruption `locked()` exists to prevent.
|
|
439
|
+
*/
|
|
440
|
+
#settleJob(job, status, error, summary, startMs, durationMs) {
|
|
441
|
+
try {
|
|
442
|
+
const validStatus = (status === 'ok' || status === 'error' || status === 'skipped') ? status : 'error';
|
|
443
|
+
applyResult(job, validStatus, error, durationMs);
|
|
444
|
+
// The callback ran unlocked, so this job may have been removed — or
|
|
445
|
+
// removed and re-registered under the same id, the shape
|
|
446
|
+
// `start(initialJobs)` uses — while it was in flight. Identity, not id.
|
|
447
|
+
//
|
|
448
|
+
// Deliberately touch NOTHING here. The claim already detached this job's
|
|
449
|
+
// heap entry and nothing re-added it, so there is nothing to clean up;
|
|
450
|
+
// any entry now filed under this id belongs to the replacement, and
|
|
451
|
+
// removing it by id would silently unschedule a live job. Do not
|
|
452
|
+
// resurrect a removed job's heap entry or run log either.
|
|
453
|
+
if (this.jobs.get(job.id) !== job) {
|
|
454
|
+
return { status, error, summary, durationMs };
|
|
455
|
+
}
|
|
456
|
+
// Log the run
|
|
457
|
+
this.runLog.record({
|
|
458
|
+
jobId: job.id,
|
|
459
|
+
status,
|
|
460
|
+
error,
|
|
461
|
+
summary,
|
|
462
|
+
runAtMs: startMs,
|
|
463
|
+
durationMs,
|
|
464
|
+
nextRunAtMs: job.state.nextRunAtMs,
|
|
465
|
+
});
|
|
466
|
+
// Handle one-shot auto-delete. The callback ran unlocked and may have
|
|
467
|
+
// pushed a heap entry for this job via add()/update(), so drop it — the
|
|
468
|
+
// job is about to stop existing.
|
|
469
|
+
if (job.deleteAfterRun && status === 'ok' && !job.enabled) {
|
|
470
|
+
this.jobs.delete(job.id);
|
|
471
|
+
this.removeFromHeap(job.id);
|
|
472
|
+
this.runLog.removeJob(job.id);
|
|
473
|
+
return { status, summary, deleted: true };
|
|
474
|
+
}
|
|
475
|
+
// Re-insert into the heap if still active. Same reason as above: drop any
|
|
476
|
+
// entry the unlocked callback added for this job first, to preserve
|
|
477
|
+
// one-entry-per-key.
|
|
478
|
+
this.removeFromHeap(job.id);
|
|
479
|
+
if (job.enabled && job.state.nextRunAtMs) {
|
|
480
|
+
this.heap.push({ key: job.id, nextTrigger: job.state.nextRunAtMs });
|
|
481
|
+
}
|
|
482
|
+
return { status, error, summary, durationMs };
|
|
243
483
|
}
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
this
|
|
484
|
+
finally {
|
|
485
|
+
// One re-arm covering every exit, rather than one per branch. The claim
|
|
486
|
+
// detached this job from the heap, so a timer that fired during the
|
|
487
|
+
// unlocked invoke would have found nothing to arm — and `run()` has no
|
|
488
|
+
// `finally { armTimer() }` of its own the way `onTimer` does. Without
|
|
489
|
+
// this, a manual run() can leave the scheduler with no pending wake.
|
|
490
|
+
this.armTimer();
|
|
247
491
|
}
|
|
248
|
-
return { status, error, summary, durationMs };
|
|
249
492
|
}
|
|
250
493
|
// -- Helpers ---------------------------------------------------------
|
|
251
494
|
removeFromHeap(id) {
|
package/package.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"keywords": [
|
|
4
4
|
"stonyx-module"
|
|
5
5
|
],
|
|
6
|
-
"version": "0.2.1-alpha.
|
|
6
|
+
"version": "0.2.1-alpha.46",
|
|
7
7
|
"description": "Cron/job scheduler for Stonyx framework",
|
|
8
8
|
"main": "dist/main.js",
|
|
9
9
|
"types": "dist/main.d.ts",
|
|
@@ -79,7 +79,7 @@
|
|
|
79
79
|
"typescript": "^5.8.3"
|
|
80
80
|
},
|
|
81
81
|
"dependencies": {
|
|
82
|
-
"stonyx": "0.2.3-beta.
|
|
82
|
+
"stonyx": "0.2.3-beta.81"
|
|
83
83
|
},
|
|
84
84
|
"scripts": {
|
|
85
85
|
"build": "tsc",
|