@ultimat3/jobs 1.2.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +567 -0
- package/README.md +355 -16
- package/package.json +7 -5
- package/src/backfill-gate.ts +97 -0
- package/src/backfill-inspect.ts +73 -0
- package/src/backfill-ledger.ts +183 -0
- package/src/backfill-pass.ts +276 -0
- package/src/backfill-pending.ts +131 -0
- package/src/backfill-rate.ts +109 -0
- package/src/backfill-registry.ts +108 -0
- package/src/backfill-scope.ts +70 -0
- package/src/backfill.ts +213 -0
- package/src/driver-memory.ts +61 -9
- package/src/driver-nats.ts +2 -1
- package/src/driver-pg-ddl.ts +180 -0
- package/src/driver-pg-rows.ts +123 -0
- package/src/driver-pg-sql.ts +251 -55
- package/src/driver-pg.ts +138 -92
- package/src/driver-redis.ts +2 -1
- package/src/driver.ts +91 -7
- package/src/errors.ts +290 -4
- package/src/events-pg.ts +121 -0
- package/src/events.ts +7 -1
- package/src/execute.ts +289 -0
- package/src/heartbeat.ts +146 -0
- package/src/index.ts +121 -27
- package/src/inspect.ts +43 -2
- package/src/job.ts +72 -2
- package/src/leases.ts +90 -0
- package/src/limits.ts +0 -0
- package/src/metrics.ts +35 -0
- package/src/outbox-pg.ts +137 -0
- package/src/outbox.ts +115 -53
- package/src/register.ts +1 -1
- package/src/retry.ts +6 -1
- package/src/run-signal.ts +50 -0
- package/src/scheduler-pg.ts +103 -0
- package/src/scheduler.ts +159 -245
- package/src/steps.ts +155 -31
- package/src/task.ts +229 -0
- package/src/tenant.ts +61 -0
- package/src/worker-fleet-slots.ts +124 -0
- package/src/worker-run.ts +132 -0
- package/src/worker.ts +207 -190
package/src/scheduler.ts
CHANGED
|
@@ -1,90 +1,30 @@
|
|
|
1
|
-
// The `
|
|
2
|
-
//
|
|
3
|
-
// re-invented per cron.
|
|
4
|
-
//
|
|
5
|
-
// `tz` is required by the type. A cron without a timezone is a bug waiting for March: `0 3 *
|
|
6
|
-
// * *` in a DST-observing zone runs twice or zero times on the switch day, and "the nightly
|
|
7
|
-
// digest went out at 2am and again at 3am" is not a mystery anyone should have to debug.
|
|
1
|
+
// The `scheduler` role: one node walks every registered task's cron, dispatches the occurrences
|
|
2
|
+
// it owes and enqueues their jobs. The `task` primitive it reads lives in `task.ts`.
|
|
8
3
|
//
|
|
9
4
|
// Exactly one node dispatches per tick, enforced by a Postgres advisory lock. Two schedulers
|
|
10
5
|
// double-enqueue every task; the idempotency key would absorb it, but leader election means
|
|
11
|
-
// the queue never sees the duplicate at all.
|
|
6
|
+
// the queue never sees the duplicate at all. One ROUND at a time is the same rule inside one
|
|
7
|
+
// process: the loop re-arms on the round it just finished, and any other caller joins that
|
|
8
|
+
// round rather than opening a second one over the same `lastFiredAt`.
|
|
12
9
|
|
|
13
10
|
import type { Clock } from '@ultimat3/core';
|
|
14
|
-
import {
|
|
11
|
+
import { isUltimateError, logger, onShutdown } from '@ultimat3/core';
|
|
15
12
|
import { instant, nextCronOccurrence } from '@ultimat3/time';
|
|
16
13
|
import { nowMs } from './clock';
|
|
17
|
-
import type {
|
|
18
|
-
import {
|
|
19
|
-
import
|
|
20
|
-
import type { EnqueueOptions } from './outbox';
|
|
21
|
-
|
|
22
|
-
/** `[[sendDigest, {}]]` — a job handle plus its input. */
|
|
23
|
-
export type TaskEnqueueEntry = readonly [AnyJobHandle, unknown];
|
|
14
|
+
import type { JobDriver } from './driver';
|
|
15
|
+
import type { TaskHandle, TaskJobResult } from './task';
|
|
16
|
+
import { registeredTasks } from './task';
|
|
24
17
|
|
|
25
18
|
/**
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
19
|
+
* A round that failed, as log fields. `message` alone throws away the half of an `UltimateError`
|
|
20
|
+
* that makes it actionable — the operator reading `jobs.scheduler.tick-failed` at 3am needs the
|
|
21
|
+
* stable code to search on and the `fix:` to run, not a sentence.
|
|
29
22
|
*/
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
readonly name?: string;
|
|
36
|
-
readonly cron: string;
|
|
37
|
-
/** REQUIRED IANA zone, e.g. `'UTC'`, `'America/New_York'`. */
|
|
38
|
-
readonly tz: string;
|
|
39
|
-
/**
|
|
40
|
-
* Builds the entries for ONE occurrence, given that occurrence's instant in epoch ms.
|
|
41
|
-
*
|
|
42
|
-
* The argument exists because catch-up does: a tick dispatched late, or replayed for a
|
|
43
|
-
* missed occurrence, has a wall clock that no longer matches the occurrence being fired.
|
|
44
|
-
* A payload derived from `Date.now()` there is silently for the wrong day — and the
|
|
45
|
-
* scheduler's own key is occurrence-scoped, so nothing downstream catches it.
|
|
46
|
-
*/
|
|
47
|
-
enqueue: (occurrenceMs: number) => readonly TaskEnqueueEntry[];
|
|
48
|
-
readonly catchUp?: CatchUpPolicy;
|
|
49
|
-
readonly maxCatchUp?: number;
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
/** One entry's outcome, from a scheduled dispatch or a manual `task.enqueue()` alike. */
|
|
53
|
-
export interface TaskJobResult {
|
|
54
|
-
readonly job: string;
|
|
55
|
-
readonly result: EnqueueResult;
|
|
56
|
-
}
|
|
57
|
-
|
|
58
|
-
/** JSON-safe view of a task for the manifest, `/_x` and the MCP dev server. */
|
|
59
|
-
export interface TaskDescriptor {
|
|
60
|
-
readonly kind: 'task';
|
|
61
|
-
readonly name: string;
|
|
62
|
-
readonly cron: string;
|
|
63
|
-
readonly tz: string;
|
|
64
|
-
readonly catchUp: CatchUpPolicy;
|
|
65
|
-
readonly maxCatchUp: number;
|
|
66
|
-
readonly jobs: readonly string[];
|
|
67
|
-
}
|
|
68
|
-
|
|
69
|
-
export interface TaskHandle {
|
|
70
|
-
readonly kind: 'task';
|
|
71
|
-
readonly name: string;
|
|
72
|
-
readonly cron: string;
|
|
73
|
-
readonly tz: string;
|
|
74
|
-
readonly catchUp: CatchUpPolicy;
|
|
75
|
-
readonly maxCatchUp: number;
|
|
76
|
-
/**
|
|
77
|
-
* Entries for `occurrenceMs`. Defaults to now, which is the honest answer for the two
|
|
78
|
-
* callers that have no occurrence: a manual `task.enqueue()` and `describe()`, which only
|
|
79
|
-
* wants the job names.
|
|
80
|
-
*/
|
|
81
|
-
entries(occurrenceMs?: number): readonly TaskEnqueueEntry[];
|
|
82
|
-
/**
|
|
83
|
-
* Fire this task's declared entries now, through the same facade `JobHandle.enqueue` uses —
|
|
84
|
-
* the backfill and "run it again" path, with no scheduler and no leader involved.
|
|
85
|
-
*/
|
|
86
|
-
enqueue(options?: EnqueueOptions): Promise<readonly TaskJobResult[]>;
|
|
87
|
-
describe(): TaskDescriptor;
|
|
23
|
+
function failureFields(error: unknown): Record<string, unknown> {
|
|
24
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
25
|
+
return isUltimateError(error)
|
|
26
|
+
? { error: message, code: error.code, cause: error.cause, fix: error.fix }
|
|
27
|
+
: { error: message };
|
|
88
28
|
}
|
|
89
29
|
|
|
90
30
|
/** Resolves the next fire time. Injected so scheduling logic is testable without a cron impl. */
|
|
@@ -94,154 +34,13 @@ const defaultCronResolver: CronResolver = (cron, options) =>
|
|
|
94
34
|
// Instant is a branded Date, so it satisfies CronResolver's Date return directly.
|
|
95
35
|
nextCronOccurrence(cron, options.tz, instant(options.from));
|
|
96
36
|
|
|
97
|
-
const registry = new Map<string, TaskHandle>();
|
|
98
|
-
let anonymous = 0;
|
|
99
|
-
|
|
100
|
-
/** Job's store, for tasks: proof `task()` built the handle, plus whether it named itself. */
|
|
101
|
-
interface TaskOrigin {
|
|
102
|
-
readonly declaredName: boolean;
|
|
103
|
-
/** The export name already stamped, once one has been. `undefined` while still provisional. */
|
|
104
|
-
readonly exportName?: string;
|
|
105
|
-
}
|
|
106
|
-
|
|
107
|
-
const origin = new WeakMap<object, TaskOrigin>();
|
|
108
|
-
|
|
109
|
-
/**
|
|
110
|
-
* `Intl` carries the runtime's copy of the tz database and rejects anything not in it with a
|
|
111
|
-
* `RangeError`, so it is the only check that can tell `America/Bogota` from `Bogota`.
|
|
112
|
-
*/
|
|
113
|
-
function isIanaZone(tz: string): boolean {
|
|
114
|
-
try {
|
|
115
|
-
return new Intl.DateTimeFormat('en-US', { timeZone: tz }).resolvedOptions().timeZone.length > 0;
|
|
116
|
-
} catch {
|
|
117
|
-
return false;
|
|
118
|
-
}
|
|
119
|
-
}
|
|
120
|
-
|
|
121
|
-
export function task(definition: TaskDefinition): TaskHandle {
|
|
122
|
-
anonymous += 1;
|
|
123
|
-
const name = definition.name ?? `anonymous-task-${anonymous}`;
|
|
124
|
-
// Runtime backstop; the type already makes an omitted tz a build error.
|
|
125
|
-
assert(
|
|
126
|
-
typeof definition.tz === 'string' && definition.tz.length > 0,
|
|
127
|
-
`task "${name}" needs an explicit IANA tz — a cron without a timezone is a bug`,
|
|
128
|
-
`add tz to task("${name}"), e.g. tz: 'UTC' — an unzoned cron silently drifts by an hour at every DST transition`,
|
|
129
|
-
);
|
|
130
|
-
// A non-empty string is not a timezone. `tz: 'Bogota'` would otherwise resolve every
|
|
131
|
-
// occurrence in UTC and the cron would run five hours off, silently, forever.
|
|
132
|
-
assert(
|
|
133
|
-
isIanaZone(definition.tz),
|
|
134
|
-
`task "${name}" has tz "${definition.tz}", which is not a zone in the IANA tz database`,
|
|
135
|
-
`use the full zone id on task("${name}"), e.g. tz: 'America/Bogota' — list the valid ones with: bun -e "console.log(Intl.supportedValuesOf('timeZone').join('\\n'))"`,
|
|
136
|
-
);
|
|
137
|
-
|
|
138
|
-
const handle: TaskHandle = {
|
|
139
|
-
kind: 'task',
|
|
140
|
-
name,
|
|
141
|
-
cron: definition.cron,
|
|
142
|
-
tz: definition.tz,
|
|
143
|
-
catchUp: definition.catchUp ?? 'skip',
|
|
144
|
-
maxCatchUp: definition.maxCatchUp ?? 10,
|
|
145
|
-
// `nowMs()` and not `Date.now()`: every reading of time in this package goes through a
|
|
146
|
-
// Clock so a frozen one cannot be bypassed.
|
|
147
|
-
entries: (occurrenceMs: number = nowMs()) => definition.enqueue(occurrenceMs),
|
|
148
|
-
async enqueue(options?: EnqueueOptions): Promise<readonly TaskJobResult[]> {
|
|
149
|
-
const fired: TaskJobResult[] = [];
|
|
150
|
-
for (const [handleForJob, input] of handle.entries()) {
|
|
151
|
-
// The job's PLAIN key, deliberately not `dispatch()`'s `task:occurrence:key`: that one
|
|
152
|
-
// is occurrence-scoped so two schedulers cannot double-fire the same tick, and reusing
|
|
153
|
-
// it here would make a manual run dedupe against whichever occurrence it landed in.
|
|
154
|
-
fired.push({ job: handleForJob.name, result: await handleForJob.enqueue(input, options) });
|
|
155
|
-
}
|
|
156
|
-
return fired;
|
|
157
|
-
},
|
|
158
|
-
// Reads `handle`, never the captured `name`: `nameTasks()` rebinds the property in place.
|
|
159
|
-
describe(): TaskDescriptor {
|
|
160
|
-
return {
|
|
161
|
-
kind: 'task',
|
|
162
|
-
name: handle.name,
|
|
163
|
-
cron: handle.cron,
|
|
164
|
-
tz: handle.tz,
|
|
165
|
-
catchUp: handle.catchUp,
|
|
166
|
-
maxCatchUp: handle.maxCatchUp,
|
|
167
|
-
// Declaration order: a task's entries are a sequence, not a set.
|
|
168
|
-
jobs: handle.entries().map(([entry]) => entry.name),
|
|
169
|
-
};
|
|
170
|
-
},
|
|
171
|
-
};
|
|
172
|
-
origin.set(handle, { declaredName: definition.name !== undefined });
|
|
173
|
-
// Refused here, not at `registerTask`: a second `task({ name: 'nightly' })` would otherwise
|
|
174
|
-
// replace the seated handle, and the scheduler's persisted `lastFiredAt` — keyed by that name —
|
|
175
|
-
// would silently start driving a different cron. The anonymous names cannot collide.
|
|
176
|
-
if (registry.has(name)) throw new JobNameTakenError({ kind: 'task', name });
|
|
177
|
-
registry.set(name, handle);
|
|
178
|
-
return handle;
|
|
179
|
-
}
|
|
180
|
-
|
|
181
|
-
/** Structural, exactly as `isJobHandle` is: only a handle `task()` built has a cron behind it. */
|
|
182
|
-
export function isTaskHandle(value: unknown): value is TaskHandle {
|
|
183
|
-
return (
|
|
184
|
-
typeof value === 'object' &&
|
|
185
|
-
value !== null &&
|
|
186
|
-
(value as { kind?: unknown }).kind === 'task' &&
|
|
187
|
-
origin.has(value)
|
|
188
|
-
);
|
|
189
|
-
}
|
|
190
|
-
|
|
191
|
-
/**
|
|
192
|
-
* Register `target` under `name`, stamped onto the handle the module exported — the scheduler's
|
|
193
|
-
* occurrence key is `task:occurrenceMs:jobKey`, so the task's name is what stops two nodes
|
|
194
|
-
* double-firing a tick, and a copy under a second name would defeat it.
|
|
195
|
-
*
|
|
196
|
-
* A definition that supplied its own `name` keeps it, for the same reason a job's does: the
|
|
197
|
-
* scheduler's persisted `lastFiredAt` is keyed by that name.
|
|
198
|
-
*/
|
|
199
|
-
export function registerTask(name: string, target: TaskHandle): TaskHandle {
|
|
200
|
-
const source = origin.get(target);
|
|
201
|
-
const key = source?.declaredName === true ? target.name : name;
|
|
202
|
-
const seated = registry.get(key);
|
|
203
|
-
// The same handle under the same name is one registration seen twice — `defineApi` and the
|
|
204
|
-
// framework's module scan both reach the same declaration file. A DIFFERENT task under a taken
|
|
205
|
-
// name is the ambiguity to refuse.
|
|
206
|
-
if (seated !== undefined) {
|
|
207
|
-
if (seated !== target) throw new JobNameTakenError({ kind: 'task', name: key });
|
|
208
|
-
return target;
|
|
209
|
-
}
|
|
210
|
-
// One handle exported under two names: the rebind below is in place, so the second alias would
|
|
211
|
-
// move the occurrence key the scheduler dedupes ticks on.
|
|
212
|
-
if (source?.exportName !== undefined && source.exportName !== key)
|
|
213
|
-
throw new JobNameTakenError({ kind: 'task', name: key });
|
|
214
|
-
registry.delete(target.name);
|
|
215
|
-
Object.defineProperty(target, 'name', { value: key, configurable: true });
|
|
216
|
-
if (source !== undefined) origin.set(target, { ...source, exportName: key });
|
|
217
|
-
registry.set(key, target);
|
|
218
|
-
return target;
|
|
219
|
-
}
|
|
220
|
-
|
|
221
|
-
/** `registerTasks(module)` is the call app code makes; this is the same rules over a record. */
|
|
222
|
-
export function nameTasks(record: Readonly<Record<string, TaskHandle>>): void {
|
|
223
|
-
for (const [exportName, handle] of Object.entries(record)) registerTask(exportName, handle);
|
|
224
|
-
}
|
|
225
|
-
|
|
226
|
-
export function registeredTasks(): readonly TaskHandle[] {
|
|
227
|
-
return [...registry.values()].sort((a, b) => a.name.localeCompare(b.name));
|
|
228
|
-
}
|
|
229
|
-
|
|
230
|
-
export function getTask(name: string): TaskHandle | undefined {
|
|
231
|
-
return registry.get(name);
|
|
232
|
-
}
|
|
233
|
-
|
|
234
|
-
export function resetTasks(): void {
|
|
235
|
-
registry.clear();
|
|
236
|
-
anonymous = 0;
|
|
237
|
-
}
|
|
238
|
-
|
|
239
37
|
export interface LeaderElection {
|
|
240
38
|
acquire(): Promise<boolean>;
|
|
241
39
|
release(): Promise<void>;
|
|
242
40
|
}
|
|
243
41
|
|
|
244
|
-
/** Single-node default: always the leader. Multi-node uses `
|
|
42
|
+
/** Single-node default: always the leader. Multi-node uses `createPgLeaseLeader()` — never
|
|
43
|
+
* `createPgLeader()`, whose advisory lock dies with the pooled connection that took it. */
|
|
245
44
|
export function soleLeader(): LeaderElection {
|
|
246
45
|
return {
|
|
247
46
|
acquire: () => Promise.resolve(true),
|
|
@@ -272,9 +71,12 @@ export interface SchedulerOptions {
|
|
|
272
71
|
readonly leader?: LeaderElection;
|
|
273
72
|
readonly state?: SchedulerState;
|
|
274
73
|
readonly cron?: CronResolver;
|
|
74
|
+
/** Gap between the end of one dispatch round and the start of the next. Default 1s. */
|
|
275
75
|
readonly tickIntervalMs?: number;
|
|
276
76
|
/** Defaults to every registered task. */
|
|
277
77
|
readonly tasks?: readonly TaskHandle[];
|
|
78
|
+
/** Default true. Registers a SIGTERM drain via `onShutdown`, exactly as the worker does. */
|
|
79
|
+
readonly drainOnShutdown?: boolean;
|
|
278
80
|
}
|
|
279
81
|
|
|
280
82
|
export interface DispatchedOccurrence {
|
|
@@ -286,19 +88,31 @@ export interface DispatchedOccurrence {
|
|
|
286
88
|
|
|
287
89
|
export interface Scheduler {
|
|
288
90
|
start(): void;
|
|
289
|
-
|
|
290
|
-
|
|
91
|
+
/** Stop dispatching, wait out the round in flight, then hand the leader lock back. */
|
|
92
|
+
stop(reason?: string): Promise<void>;
|
|
93
|
+
/**
|
|
94
|
+
* One dispatch round. Returns what it enqueued — tests call this, not the timer. A call
|
|
95
|
+
* landing on a round already in flight JOINS that round; there is never a second one.
|
|
96
|
+
*/
|
|
291
97
|
tick(): Promise<readonly DispatchedOccurrence[]>;
|
|
292
98
|
nextRunFor(handle: TaskHandle, from?: Date): Date;
|
|
293
99
|
}
|
|
294
100
|
|
|
295
101
|
export function createScheduler(options: SchedulerOptions): Scheduler {
|
|
296
|
-
const
|
|
102
|
+
const schedulerState = options.state ?? createMemorySchedulerState();
|
|
297
103
|
const resolveCron = options.cron ?? defaultCronResolver;
|
|
298
104
|
const tickIntervalMs = options.tickIntervalMs ?? 1_000;
|
|
299
105
|
const leader = options.leader ?? soleLeader();
|
|
300
|
-
let timer: ReturnType<typeof
|
|
106
|
+
let timer: ReturnType<typeof setTimeout> | undefined;
|
|
301
107
|
let isLeader = false;
|
|
108
|
+
/** The drain's state, keyed on the same four values the worker's is. */
|
|
109
|
+
let state: 'idle' | 'running' | 'draining' | 'stopped' = 'idle';
|
|
110
|
+
/** The dispatch round in flight — what a second caller joins and what `stop()` waits out. */
|
|
111
|
+
let round: Promise<readonly DispatchedOccurrence[]> | undefined;
|
|
112
|
+
/** The `onShutdown` registration this scheduler holds while it runs. Handed back by `stop()`. */
|
|
113
|
+
let releaseShutdownHook: (() => void) | undefined;
|
|
114
|
+
/** The teardown in flight, so a SIGTERM landing on a manual stop joins it. */
|
|
115
|
+
let stopping: Promise<void> | undefined;
|
|
302
116
|
|
|
303
117
|
const nextRunFor = (handle: TaskHandle, from?: Date): Date =>
|
|
304
118
|
resolveCron(handle.cron, { tz: handle.tz, from: from ?? new Date(nowMs(options.clock)) });
|
|
@@ -343,7 +157,7 @@ export function createScheduler(options: SchedulerOptions): Scheduler {
|
|
|
343
157
|
});
|
|
344
158
|
jobs.push({ job: handleForJob.name, result });
|
|
345
159
|
}
|
|
346
|
-
await
|
|
160
|
+
await schedulerState.markFired(handle.name, occurrenceMs);
|
|
347
161
|
logger.info('jobs.scheduler.dispatched', {
|
|
348
162
|
task: handle.name,
|
|
349
163
|
occurrence: new Date(occurrenceMs).toISOString(),
|
|
@@ -354,21 +168,41 @@ export function createScheduler(options: SchedulerOptions): Scheduler {
|
|
|
354
168
|
return { task: handle.name, occurrenceMs, jobs, catchUp };
|
|
355
169
|
};
|
|
356
170
|
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
171
|
+
/** The drain's one question: may this scheduler still dispatch an occurrence? */
|
|
172
|
+
const dispatching = (): boolean => state !== 'draining' && state !== 'stopped';
|
|
173
|
+
|
|
174
|
+
const runRound = async (): Promise<readonly DispatchedOccurrence[]> => {
|
|
175
|
+
// Never take the lock a drain is on its way to releasing: a round that acquired it here
|
|
176
|
+
// would still be enqueueing after `stop()` handed the occurrence to the next node.
|
|
177
|
+
if (!dispatching()) return [];
|
|
178
|
+
// Asked EVERY round, not only while `isLeader` is false. A lease-backed election (the one a
|
|
179
|
+
// pooled executor can use — `createPgLeaseLeader`) expires, so `acquire()` is also its
|
|
180
|
+
// renewal, and a node that cached `isLeader = true` would keep dispatching past a lease
|
|
181
|
+
// another node has already taken. `soleLeader` answers true every time, and `createPgLeader`
|
|
182
|
+
// holds its grant internally, so this is a no-op for both.
|
|
183
|
+
const held = await leader.acquire();
|
|
184
|
+
if (!held) {
|
|
185
|
+
// Demoted, or never elected. Nothing to release — a lease we no longer hold is not ours to
|
|
186
|
+
// hand back, and `teardown` reads this same flag before it calls `release()`.
|
|
187
|
+
if (isLeader) logger.warn('jobs.scheduler.leadership-lost', { at: nowMs(options.clock) });
|
|
188
|
+
isLeader = false;
|
|
189
|
+
return [];
|
|
361
190
|
}
|
|
191
|
+
isLeader = true;
|
|
362
192
|
|
|
363
193
|
const at = nowMs(options.clock);
|
|
364
194
|
const tasks = options.tasks ?? registeredTasks();
|
|
365
195
|
const dispatched: DispatchedOccurrence[] = [];
|
|
366
196
|
|
|
367
197
|
for (const handle of tasks) {
|
|
368
|
-
|
|
198
|
+
// Re-read per task, not once on entry: a `stop()` between two tasks means stop now, not
|
|
199
|
+
// at the next round. A task not reached simply fires next time — its `lastFiredAt` is
|
|
200
|
+
// untouched — while the occurrence this round already began is the one `stop()` waits for.
|
|
201
|
+
if (!dispatching()) break;
|
|
202
|
+
const last = await schedulerState.lastFiredAt(handle.name);
|
|
369
203
|
if (last === undefined) {
|
|
370
204
|
// First sight of this task: arm it, never fire retroactively for all of history.
|
|
371
|
-
await
|
|
205
|
+
await schedulerState.markFired(handle.name, nextRunFor(handle, new Date(at)).getTime() - 1);
|
|
372
206
|
continue;
|
|
373
207
|
}
|
|
374
208
|
|
|
@@ -382,7 +216,17 @@ export function createScheduler(options: SchedulerOptions): Scheduler {
|
|
|
382
216
|
}
|
|
383
217
|
if (handle.catchUp === 'run-once') {
|
|
384
218
|
const first = due[0];
|
|
385
|
-
if (first !== undefined)
|
|
219
|
+
if (first !== undefined) {
|
|
220
|
+
dispatched.push(await dispatch(handle, first, due.length > 1));
|
|
221
|
+
// `dispatch` leaves the watermark on the occurrence it RAN — the earliest missed one
|
|
222
|
+
// here — so the next round found occurrences 2..n still due and fired the second, then
|
|
223
|
+
// the third, one per tick until the backlog drained: 24 nightly digests a second apart
|
|
224
|
+
// after a day down. "One catch-up" means the rest are DROPPED, and dropping an
|
|
225
|
+
// occurrence is moving the watermark past it. `at` rather than the last element of
|
|
226
|
+
// `due`, which `maxCatchUp` truncates: every occurrence at or before `at` is missed by
|
|
227
|
+
// definition, and this policy fires none of them.
|
|
228
|
+
if (first !== at) await schedulerState.markFired(handle.name, at);
|
|
229
|
+
}
|
|
386
230
|
continue;
|
|
387
231
|
}
|
|
388
232
|
for (const occurrence of due) {
|
|
@@ -393,24 +237,94 @@ export function createScheduler(options: SchedulerOptions): Scheduler {
|
|
|
393
237
|
return dispatched;
|
|
394
238
|
};
|
|
395
239
|
|
|
240
|
+
/**
|
|
241
|
+
* One round, and never two at once. A round slower than `tickIntervalMs` used to leave the
|
|
242
|
+
* timer starting a second one over the same `lastFiredAt`: both read the same watermark, both
|
|
243
|
+
* walked the same occurrences and both dispatched them. The occurrence key deduped the JOBS,
|
|
244
|
+
* so nothing downstream showed it — but the loser re-marked `lastFiredAt`, reported
|
|
245
|
+
* occurrences it never enqueued, and under `run-all` interleaved a catch-up sequence with
|
|
246
|
+
* itself. Joining also gives `stop()` the one promise it has to wait out.
|
|
247
|
+
*/
|
|
248
|
+
const tick = (): Promise<readonly DispatchedOccurrence[]> => {
|
|
249
|
+
round ??= runRound().finally(() => {
|
|
250
|
+
round = undefined;
|
|
251
|
+
});
|
|
252
|
+
return round;
|
|
253
|
+
};
|
|
254
|
+
|
|
255
|
+
/**
|
|
256
|
+
* Re-arms on the round it just finished, never on a fixed period — the interval is the GAP
|
|
257
|
+
* between rounds, which is what makes overlap impossible at the source rather than caught by
|
|
258
|
+
* a guard. Cron accuracy does not pay for it: an occurrence is computed from the clock, so a
|
|
259
|
+
* few ms of drift between rounds moves nothing.
|
|
260
|
+
*/
|
|
261
|
+
const schedule = (): void => {
|
|
262
|
+
timer = setTimeout(() => {
|
|
263
|
+
void tick()
|
|
264
|
+
.catch((error: unknown) => {
|
|
265
|
+
logger.error('jobs.scheduler.tick-failed', failureFields(error));
|
|
266
|
+
})
|
|
267
|
+
.finally(() => {
|
|
268
|
+
if (state === 'running') schedule();
|
|
269
|
+
});
|
|
270
|
+
}, tickIntervalMs);
|
|
271
|
+
};
|
|
272
|
+
|
|
273
|
+
const teardown = async (reason: string): Promise<void> => {
|
|
274
|
+
state = 'draining';
|
|
275
|
+
if (timer !== undefined) clearTimeout(timer);
|
|
276
|
+
timer = undefined;
|
|
277
|
+
logger.info('jobs.scheduler.draining', { reason, dispatching: round !== undefined });
|
|
278
|
+
try {
|
|
279
|
+
// The round this stop races runs to the end first. Releasing the advisory lock under a
|
|
280
|
+
// live dispatch hands the next node a task this one is still enqueueing for, and both
|
|
281
|
+
// then own the same occurrence — the exact double-fire leader election exists to prevent.
|
|
282
|
+
// Settled, not awaited: a round that failed is its own caller's to see, and the lock still
|
|
283
|
+
// has to go back.
|
|
284
|
+
await Promise.allSettled([round]);
|
|
285
|
+
if (isLeader) await leader.release();
|
|
286
|
+
} finally {
|
|
287
|
+
// Whatever the release did, this scheduler is done. `isLeader` false because a lock this
|
|
288
|
+
// process no longer holds — or failed to hand back — must never be re-used as if it did,
|
|
289
|
+
// and the hook goes back: one left registered dispatches through a stopped scheduler on
|
|
290
|
+
// the next process-wide drain, and keeps this closure and its driver alive with it.
|
|
291
|
+
isLeader = false;
|
|
292
|
+
state = 'stopped';
|
|
293
|
+
releaseShutdownHook?.();
|
|
294
|
+
releaseShutdownHook = undefined;
|
|
295
|
+
}
|
|
296
|
+
};
|
|
297
|
+
|
|
298
|
+
const stop = async (reason = 'stop'): Promise<void> => {
|
|
299
|
+
if (state === 'stopped') return;
|
|
300
|
+
// One teardown, joined rather than repeated — the worker's rule, for the same reason: a
|
|
301
|
+
// SIGTERM landing on a manual stop must wait out the same round, not release the lock a
|
|
302
|
+
// second time behind it. Cleared as it settles, so a scheduler started again stops again.
|
|
303
|
+
stopping ??= teardown(reason).finally(() => {
|
|
304
|
+
stopping = undefined;
|
|
305
|
+
});
|
|
306
|
+
await stopping;
|
|
307
|
+
};
|
|
308
|
+
|
|
396
309
|
return {
|
|
397
310
|
start() {
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
311
|
+
// Only from a standstill. A start mid-drain would re-arm the loop on a lock the drain is
|
|
312
|
+
// about to release, and stack a second shutdown hook on the one still running.
|
|
313
|
+
if (state !== 'idle' && state !== 'stopped') return;
|
|
314
|
+
state = 'running';
|
|
315
|
+
// 'accept' phase: stop dispatching before core waits on in-flight work — an occurrence
|
|
316
|
+
// enqueued during the drain is work nothing in this process is left to run. The
|
|
317
|
+
// unregister is kept, never discarded: `stop()` hands it back, so start -> stop -> start
|
|
318
|
+
// holds ONE hook rather than one per start.
|
|
319
|
+
if (options.drainOnShutdown !== false) {
|
|
320
|
+
releaseShutdownHook = onShutdown('jobs.scheduler', () => stop('SIGTERM'), {
|
|
321
|
+
phase: 'accept',
|
|
404
322
|
});
|
|
405
|
-
}
|
|
323
|
+
}
|
|
324
|
+
schedule();
|
|
406
325
|
logger.info('jobs.scheduler.started', { tasks: (options.tasks ?? registeredTasks()).length });
|
|
407
326
|
},
|
|
408
|
-
|
|
409
|
-
if (timer !== undefined) clearInterval(timer);
|
|
410
|
-
timer = undefined;
|
|
411
|
-
if (isLeader) await leader.release();
|
|
412
|
-
isLeader = false;
|
|
413
|
-
},
|
|
327
|
+
stop,
|
|
414
328
|
tick,
|
|
415
329
|
nextRunFor,
|
|
416
330
|
};
|