@ultimat3/jobs 1.2.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +660 -0
- package/README.md +432 -17
- package/package.json +7 -5
- package/src/backfill-gate.ts +97 -0
- package/src/backfill-inspect.ts +73 -0
- package/src/backfill-ledger.ts +183 -0
- package/src/backfill-pass.ts +276 -0
- package/src/backfill-pending.ts +131 -0
- package/src/backfill-rate.ts +109 -0
- package/src/backfill-registry.ts +108 -0
- package/src/backfill-scope.ts +70 -0
- package/src/backfill.ts +213 -0
- package/src/driver-memory.ts +61 -9
- package/src/driver-nats.ts +2 -1
- package/src/driver-pg-ddl.ts +191 -0
- package/src/driver-pg-rows.ts +123 -0
- package/src/driver-pg-sql.ts +312 -55
- package/src/driver-pg.ts +138 -92
- package/src/driver-redis.ts +2 -1
- package/src/driver.ts +91 -7
- package/src/errors.ts +314 -5
- package/src/events-pg.ts +121 -0
- package/src/events.ts +7 -1
- package/src/execute.ts +308 -0
- package/src/heartbeat.ts +148 -0
- package/src/index.ts +128 -27
- package/src/inspect.ts +43 -2
- package/src/job.ts +127 -3
- package/src/leases.ts +90 -0
- package/src/limits.ts +0 -0
- package/src/metrics.ts +35 -0
- package/src/outbox-lease.ts +29 -0
- package/src/outbox-pg.ts +188 -0
- package/src/outbox.ts +204 -59
- package/src/register.ts +1 -1
- package/src/renewal-timer.ts +35 -0
- package/src/retry-classification.ts +112 -0
- package/src/retry.ts +6 -1
- package/src/run-signal.ts +50 -0
- package/src/scheduler-pg.ts +103 -0
- package/src/scheduler.ts +159 -245
- package/src/steps.ts +155 -31
- package/src/task.ts +239 -0
- package/src/tenant.ts +61 -0
- package/src/worker-fleet-slots.ts +129 -0
- package/src/worker-run.ts +132 -0
- package/src/worker.ts +207 -190
package/src/worker.ts
CHANGED
|
@@ -4,19 +4,19 @@
|
|
|
4
4
|
// deploy turns "at least once" into "always twice", so draining is on by default.
|
|
5
5
|
|
|
6
6
|
import type { Clock, Ctx } from '@ultimat3/core';
|
|
7
|
-
import { logger, onShutdown, recordQueueDepth, uuid
|
|
7
|
+
import { logger, onShutdown, recordJob, recordQueueDepth, uuid } from '@ultimat3/core';
|
|
8
8
|
import { nowMs } from './clock';
|
|
9
9
|
import type { ClaimedJob, JobDriver, QueueStats } from './driver';
|
|
10
10
|
import { DEFAULT_QUEUE, DEFAULT_VISIBILITY_TIMEOUT_MS } from './driver';
|
|
11
|
-
import {
|
|
12
|
-
import {
|
|
13
|
-
import
|
|
14
|
-
import { getJob } from './job';
|
|
11
|
+
import { ConcurrencyUnenforceableError } from './errors';
|
|
12
|
+
import type { JobExecution, JobOutcome } from './execute';
|
|
13
|
+
import { getJob, registeredJobs } from './job';
|
|
15
14
|
import type { Limiter } from './limits';
|
|
16
15
|
import { createLimiter } from './limits';
|
|
17
|
-
import {
|
|
18
|
-
import type { EventLookup
|
|
19
|
-
import {
|
|
16
|
+
import { recordQueueDeadJobs, recordQueueOldestReady } from './metrics';
|
|
17
|
+
import type { EventLookup } from './steps';
|
|
18
|
+
import { createFleetSlots } from './worker-fleet-slots';
|
|
19
|
+
import { runClaimedJob } from './worker-run';
|
|
20
20
|
|
|
21
21
|
/**
|
|
22
22
|
* How often the claim loop republishes `queue_depth`. Its own interval, not `pollIntervalMs`:
|
|
@@ -26,135 +26,19 @@ import { createStepRunner, isStepSuspension } from './steps';
|
|
|
26
26
|
*/
|
|
27
27
|
const QUEUE_DEPTH_INTERVAL_MS = 15_000;
|
|
28
28
|
|
|
29
|
-
export type JobOutcome = 'completed' | 'suspended' | 'retried' | 'dead-lettered';
|
|
30
|
-
|
|
31
|
-
export interface JobExecution {
|
|
32
|
-
readonly outcome: JobOutcome;
|
|
33
|
-
readonly jobId: string;
|
|
34
|
-
readonly job: string;
|
|
35
|
-
readonly attempt: number;
|
|
36
|
-
readonly durationMs: number;
|
|
37
|
-
readonly resumeAt?: number;
|
|
38
|
-
readonly error?: string;
|
|
39
|
-
readonly steps: readonly StepRecord[];
|
|
40
|
-
readonly replayed: readonly string[];
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
export interface ExecuteJobOptions {
|
|
44
|
-
readonly driver: JobDriver;
|
|
45
|
-
readonly claimed: ClaimedJob;
|
|
46
|
-
readonly handle: AnyJobHandle;
|
|
47
|
-
readonly ctx: Ctx;
|
|
48
|
-
readonly clock?: Clock;
|
|
49
|
-
readonly events?: EventLookup;
|
|
50
|
-
}
|
|
51
|
-
|
|
52
29
|
/**
|
|
53
|
-
*
|
|
54
|
-
*
|
|
30
|
+
* `JobOutcome` -> the `jobs_total` label, and `null` for the outcome that is not one. `suspended`
|
|
31
|
+
* is deliberately unmapped: parking a run is control flow, so counting it would make every
|
|
32
|
+
* `step.sleep` read as a finished job and make the failure ratio meaningless.
|
|
55
33
|
*/
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
store: driver.steps,
|
|
63
|
-
...(options.clock === undefined ? {} : { clock: options.clock }),
|
|
64
|
-
events: options.events ?? eventBus(),
|
|
34
|
+
const JOB_OUTCOME_LABELS: Readonly<Record<JobOutcome, 'ok' | 'failed' | 'dead' | null>> =
|
|
35
|
+
Object.freeze({
|
|
36
|
+
completed: 'ok',
|
|
37
|
+
suspended: null,
|
|
38
|
+
retried: 'failed',
|
|
39
|
+
'dead-lettered': 'dead',
|
|
65
40
|
});
|
|
66
41
|
|
|
67
|
-
const settle = async (outcome: JobExecution): Promise<JobExecution> => {
|
|
68
|
-
const steps = await driver.steps.list(claimed.runId);
|
|
69
|
-
return { ...outcome, steps, replayed: runner.replayedNames() };
|
|
70
|
-
};
|
|
71
|
-
|
|
72
|
-
try {
|
|
73
|
-
const input = handle.parse(claimed.input);
|
|
74
|
-
const work = handle.run({
|
|
75
|
-
input,
|
|
76
|
-
step: runner.step,
|
|
77
|
-
ctx: options.ctx,
|
|
78
|
-
attempt: claimed.attempt,
|
|
79
|
-
jobId: claimed.id,
|
|
80
|
-
runId: claimed.runId,
|
|
81
|
-
});
|
|
82
|
-
|
|
83
|
-
await (handle.timeoutMs === undefined
|
|
84
|
-
? work
|
|
85
|
-
: raceTimeout(work, handle.timeoutMs, handle.name));
|
|
86
|
-
|
|
87
|
-
await driver.ack(claimed.id);
|
|
88
|
-
return settle({
|
|
89
|
-
outcome: 'completed',
|
|
90
|
-
jobId: claimed.id,
|
|
91
|
-
job: handle.name,
|
|
92
|
-
attempt: claimed.attempt,
|
|
93
|
-
durationMs: nowMs(options.clock) - startedAt,
|
|
94
|
-
steps: [],
|
|
95
|
-
replayed: [],
|
|
96
|
-
});
|
|
97
|
-
} catch (error) {
|
|
98
|
-
if (isStepSuspension(error)) {
|
|
99
|
-
const delayMs = Math.max(0, error.resumeAt - nowMs(options.clock));
|
|
100
|
-
// countsAsAttempt: false — parking a run is not a failure.
|
|
101
|
-
await driver.nack(claimed.id, { delayMs, countsAsAttempt: false });
|
|
102
|
-
return settle({
|
|
103
|
-
outcome: 'suspended',
|
|
104
|
-
jobId: claimed.id,
|
|
105
|
-
job: handle.name,
|
|
106
|
-
attempt: claimed.attempt,
|
|
107
|
-
durationMs: nowMs(options.clock) - startedAt,
|
|
108
|
-
resumeAt: error.resumeAt,
|
|
109
|
-
steps: [],
|
|
110
|
-
replayed: [],
|
|
111
|
-
});
|
|
112
|
-
}
|
|
113
|
-
|
|
114
|
-
const message = error instanceof Error ? error.message : String(error);
|
|
115
|
-
const decision = nextRetry(handle.retry, claimed.attempt);
|
|
116
|
-
await driver.nack(claimed.id, {
|
|
117
|
-
delayMs: decision.delayMs,
|
|
118
|
-
error: message,
|
|
119
|
-
countsAsAttempt: true,
|
|
120
|
-
deadLetter: !decision.retry && decision.deadLetter,
|
|
121
|
-
});
|
|
122
|
-
logger.warn('jobs.attempt.failed', {
|
|
123
|
-
job: handle.name,
|
|
124
|
-
jobId: claimed.id,
|
|
125
|
-
attempt: claimed.attempt,
|
|
126
|
-
retry: decision.retry,
|
|
127
|
-
error: message,
|
|
128
|
-
});
|
|
129
|
-
return settle({
|
|
130
|
-
outcome: decision.retry ? 'retried' : 'dead-lettered',
|
|
131
|
-
jobId: claimed.id,
|
|
132
|
-
job: handle.name,
|
|
133
|
-
attempt: claimed.attempt,
|
|
134
|
-
durationMs: nowMs(options.clock) - startedAt,
|
|
135
|
-
error: message,
|
|
136
|
-
steps: [],
|
|
137
|
-
replayed: [],
|
|
138
|
-
});
|
|
139
|
-
}
|
|
140
|
-
}
|
|
141
|
-
|
|
142
|
-
function raceTimeout(work: Promise<unknown>, timeoutMs: number, job: string): Promise<unknown> {
|
|
143
|
-
return new Promise((resolve, reject) => {
|
|
144
|
-
const timer = setTimeout(() => reject(new JobTimeoutError({ job, timeoutMs })), timeoutMs);
|
|
145
|
-
work.then(
|
|
146
|
-
(value) => {
|
|
147
|
-
clearTimeout(timer);
|
|
148
|
-
resolve(value);
|
|
149
|
-
},
|
|
150
|
-
(error) => {
|
|
151
|
-
clearTimeout(timer);
|
|
152
|
-
reject(error);
|
|
153
|
-
},
|
|
154
|
-
);
|
|
155
|
-
});
|
|
156
|
-
}
|
|
157
|
-
|
|
158
42
|
export interface WorkerOptions {
|
|
159
43
|
readonly driver: JobDriver;
|
|
160
44
|
/** Queues this process serves. Default `['default']`. */
|
|
@@ -205,10 +89,28 @@ export function createWorker(options: WorkerOptions): Worker {
|
|
|
205
89
|
? options.concurrency
|
|
206
90
|
: (options.concurrency?.[queue] ?? 5);
|
|
207
91
|
const limiter = options.limiter ?? createLimiter({});
|
|
92
|
+
const driverLeases = options.driver.leases;
|
|
93
|
+
// `job.concurrency`, held as a row every replica sees. The TTL is the visibility timeout and the
|
|
94
|
+
// renewal rides the lease heartbeat's interval — `worker-fleet-slots.ts` says why both.
|
|
95
|
+
const fleetSlots = createFleetSlots({
|
|
96
|
+
leases: driverLeases,
|
|
97
|
+
workerId,
|
|
98
|
+
ttlMs: visibilityTimeoutMs,
|
|
99
|
+
renewIntervalMs: heartbeatIntervalMs,
|
|
100
|
+
});
|
|
208
101
|
|
|
209
102
|
const inFlight = new Set<Promise<unknown>>();
|
|
103
|
+
/**
|
|
104
|
+
* Claim rounds in flight. Jobs land in `inFlight` mid-round, so a drain that waited only on
|
|
105
|
+
* `inFlight` waited on a set the round it was racing had not finished filling.
|
|
106
|
+
*/
|
|
107
|
+
const rounds = new Set<Promise<unknown>>();
|
|
210
108
|
let state: WorkerStats['state'] = 'idle';
|
|
211
109
|
let loop: ReturnType<typeof setTimeout> | undefined;
|
|
110
|
+
/** The `onShutdown` registration this worker holds while it runs. Handed back by `stop()`. */
|
|
111
|
+
let releaseShutdownHook: (() => void) | undefined;
|
|
112
|
+
/** The teardown in flight, so a second `stop()` joins it instead of running a second one. */
|
|
113
|
+
let stopping: Promise<void> | undefined;
|
|
212
114
|
let processed = 0;
|
|
213
115
|
let failed = 0;
|
|
214
116
|
let suspended = 0;
|
|
@@ -227,7 +129,15 @@ export function createWorker(options: WorkerOptions): Worker {
|
|
|
227
129
|
if (now - depthPublishedAt < QUEUE_DEPTH_INTERVAL_MS) return;
|
|
228
130
|
depthPublishedAt = now;
|
|
229
131
|
try {
|
|
230
|
-
for (const stat of await options.driver.stats())
|
|
132
|
+
for (const stat of await options.driver.stats()) {
|
|
133
|
+
recordQueueDepth(stat.queue, stat.ready);
|
|
134
|
+
// Depth alone is not alertable: it cannot tell "10 jobs stuck for an hour" from "10 jobs
|
|
135
|
+
// enqueued a second ago", and `jobs_total{outcome="dead"}` is a rate, so a dead-letter
|
|
136
|
+
// queue that filled overnight and stopped growing pages nobody. Both numbers are already
|
|
137
|
+
// in `stats()` — this queries nothing new.
|
|
138
|
+
recordQueueOldestReady(stat.queue, stat.oldestReadyMs);
|
|
139
|
+
recordQueueDeadJobs(stat.queue, stat.dead);
|
|
140
|
+
}
|
|
231
141
|
} catch (error) {
|
|
232
142
|
// Instrumentation never costs a tick: a queue that cannot be measured must still be worked.
|
|
233
143
|
logger.warn('jobs.worker.depth-failed', {
|
|
@@ -237,53 +147,39 @@ export function createWorker(options: WorkerOptions): Worker {
|
|
|
237
147
|
}
|
|
238
148
|
};
|
|
239
149
|
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
attempt: claimed.attempt,
|
|
254
|
-
durationMs: 0,
|
|
255
|
-
error: `no job registered as "${claimed.name}"`,
|
|
256
|
-
steps: [],
|
|
257
|
-
replayed: [],
|
|
258
|
-
};
|
|
259
|
-
}
|
|
150
|
+
/** One claimed job, run under its lease, its slot and its span. `worker-run.ts` owns the wiring. */
|
|
151
|
+
const runClaimed = (claimed: ClaimedJob): Promise<JobExecution> =>
|
|
152
|
+
runClaimedJob({
|
|
153
|
+
driver: options.driver,
|
|
154
|
+
claimed,
|
|
155
|
+
context: options.context,
|
|
156
|
+
fleetSlots,
|
|
157
|
+
workerId,
|
|
158
|
+
visibilityTimeoutMs,
|
|
159
|
+
heartbeatIntervalMs,
|
|
160
|
+
...(options.clock === undefined ? {} : { clock: options.clock }),
|
|
161
|
+
...(options.events === undefined ? {} : { events: options.events }),
|
|
162
|
+
});
|
|
260
163
|
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
}, heartbeatIntervalMs);
|
|
164
|
+
/** The drain's one question: may this worker still take work off the queue? */
|
|
165
|
+
const claiming = (): boolean => state !== 'draining' && state !== 'stopped';
|
|
264
166
|
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
...(options.events === undefined ? {} : { events: options.events }),
|
|
274
|
-
}),
|
|
275
|
-
);
|
|
276
|
-
} finally {
|
|
277
|
-
clearInterval(heartbeat);
|
|
278
|
-
}
|
|
279
|
-
};
|
|
280
|
-
|
|
281
|
-
const tick = async (): Promise<readonly JobExecution[]> => {
|
|
282
|
-
if (state === 'draining' || state === 'stopped') return [];
|
|
167
|
+
/**
|
|
168
|
+
* One claim pass: every queue asked once, and everything it hands back STARTED. It does not wait
|
|
169
|
+
* for the jobs — a slot is free again the moment its own job settles and the limiter releases
|
|
170
|
+
* the lease, so the next pass refills exactly that slot. Waiting for the whole batch made a pool
|
|
171
|
+
* as slow as its slowest member and, because the pass walks every queue before it waits, left
|
|
172
|
+
* every OTHER queue idle behind one long-running job too.
|
|
173
|
+
*/
|
|
174
|
+
const claimRound = async (): Promise<readonly Promise<JobExecution>[]> => {
|
|
283
175
|
await publishQueueDepth();
|
|
284
|
-
const
|
|
176
|
+
const started: Promise<JobExecution>[] = [];
|
|
285
177
|
|
|
286
178
|
for (const queue of queues) {
|
|
179
|
+
// Re-read per queue, not once at the top: a `stop()` between two queues means "stop
|
|
180
|
+
// claiming" now, not at the next tick. What this round already holds still runs to the end
|
|
181
|
+
// — that is the drain, and `stop()` waits for it.
|
|
182
|
+
if (!claiming()) break;
|
|
287
183
|
const free = Math.max(0, slotsFor(queue) - limiter.inFlight({ queue }));
|
|
288
184
|
if (free === 0) continue;
|
|
289
185
|
|
|
@@ -309,31 +205,106 @@ export function createWorker(options: WorkerOptions): Worker {
|
|
|
309
205
|
continue;
|
|
310
206
|
}
|
|
311
207
|
|
|
208
|
+
// `job.concurrency`, at last enforced. The limiter above counts slots in THIS heap, which
|
|
209
|
+
// twenty pods multiply by twenty; this one is a row every replica sees. Taken after the
|
|
210
|
+
// in-process lease so the cheap refusal happens first, and released in the same `finally`.
|
|
211
|
+
//
|
|
212
|
+
// The `try` is the whole of a bug this had: taking a fleet slot is a WRITE to
|
|
213
|
+
// `x_job_leases`, so a failover, a pool timeout or a `57P01` REJECTS here — between the
|
|
214
|
+
// in-process lease above and the `.finally` below that gives it back. The slot was burned
|
|
215
|
+
// permanently, and four of them on a concurrency-4 worker is the whole role dead, silent
|
|
216
|
+
// but for `jobs.worker.tick-failed` and a queue depth that climbs forever.
|
|
217
|
+
let granted: boolean;
|
|
218
|
+
try {
|
|
219
|
+
granted = await fleetSlots.acquire(job);
|
|
220
|
+
} catch (error) {
|
|
221
|
+
lease.release();
|
|
222
|
+
throw error;
|
|
223
|
+
}
|
|
224
|
+
if (!granted) {
|
|
225
|
+
lease.release();
|
|
226
|
+
await options.driver.nack(job.id, {
|
|
227
|
+
delayMs: pollIntervalMs,
|
|
228
|
+
countsAsAttempt: false,
|
|
229
|
+
error: `limited: job concurrency (${getJob(job.name)?.concurrency ?? 0})`,
|
|
230
|
+
});
|
|
231
|
+
continue;
|
|
232
|
+
}
|
|
233
|
+
|
|
312
234
|
const running = runClaimed(job)
|
|
313
235
|
.then((execution) => {
|
|
314
|
-
results.push(execution);
|
|
315
236
|
if (execution.outcome === 'completed') processed += 1;
|
|
316
237
|
else if (execution.outcome === 'suspended') suspended += 1;
|
|
317
238
|
else if (execution.outcome === 'retried') failed += 1;
|
|
318
239
|
else deadLettered += 1;
|
|
240
|
+
// The other half of this package's metrics contract: `queue_depth` says how much work
|
|
241
|
+
// is waiting, `jobs_total` says whether any of it is succeeding. Depth alone cannot
|
|
242
|
+
// tell a drained queue from a queue nothing ever claimed. Labelled by QUEUE and
|
|
243
|
+
// OUTCOME only — a label per job name is unbounded in an app's own vocabulary.
|
|
244
|
+
const label = JOB_OUTCOME_LABELS[execution.outcome];
|
|
245
|
+
if (label !== null) recordJob(queue, label);
|
|
319
246
|
return execution;
|
|
320
247
|
})
|
|
321
248
|
.finally(() => {
|
|
322
249
|
lease.release();
|
|
250
|
+
void fleetSlots.release(job.id);
|
|
323
251
|
});
|
|
324
252
|
|
|
253
|
+
started.push(running);
|
|
325
254
|
inFlight.add(running);
|
|
326
|
-
|
|
255
|
+
// The claim loop no longer awaits these, so this is the one place a rejection is observed:
|
|
256
|
+
// unobserved it is an unhandled rejection, which on Bun's default is the whole process.
|
|
257
|
+
// `executeJob` settles the job itself, so reaching here means the driver could not be
|
|
258
|
+
// told how it ended — the lease will lapse and the queue will deliver it again.
|
|
259
|
+
void running.then(
|
|
260
|
+
() => {
|
|
261
|
+
inFlight.delete(running);
|
|
262
|
+
},
|
|
263
|
+
(error: unknown) => {
|
|
264
|
+
inFlight.delete(running);
|
|
265
|
+
logger.error('jobs.worker.settle-failed', {
|
|
266
|
+
workerId,
|
|
267
|
+
job: job.name,
|
|
268
|
+
jobId: job.id,
|
|
269
|
+
error: error instanceof Error ? error.message : String(error),
|
|
270
|
+
});
|
|
271
|
+
},
|
|
272
|
+
);
|
|
327
273
|
}
|
|
328
274
|
}
|
|
329
275
|
|
|
330
|
-
|
|
331
|
-
|
|
276
|
+
return started;
|
|
277
|
+
};
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* One claim pass, tracked. The guard and the registration are one synchronous step — no await
|
|
281
|
+
* between them — so a pass is either refused by a drain already under way or visible to every
|
|
282
|
+
* drain that starts after it. A pass that reached `claim()` first is the one `stop()` must
|
|
283
|
+
* wait out: it is still adding to `inFlight`.
|
|
284
|
+
*/
|
|
285
|
+
const round = (): Promise<readonly Promise<JobExecution>[]> => {
|
|
286
|
+
if (!claiming()) return Promise.resolve([]);
|
|
287
|
+
const pass = claimRound().finally(() => {
|
|
288
|
+
rounds.delete(pass);
|
|
289
|
+
});
|
|
290
|
+
rounds.add(pass);
|
|
291
|
+
return pass;
|
|
292
|
+
};
|
|
293
|
+
|
|
294
|
+
/** One claim+run round: the pass, then the jobs THIS pass started — never the whole pool. */
|
|
295
|
+
const tick = async (): Promise<readonly JobExecution[]> => {
|
|
296
|
+
const settled = await Promise.allSettled(await round());
|
|
297
|
+
return settled.flatMap((result) => (result.status === 'fulfilled' ? [result.value] : []));
|
|
332
298
|
};
|
|
333
299
|
|
|
300
|
+
/**
|
|
301
|
+
* The claim loop re-arms on the PASS, not on the jobs: polling is how a free slot gets refilled,
|
|
302
|
+
* and a loop that waited for the last job of the previous pass could not refill one until the
|
|
303
|
+
* whole batch was done.
|
|
304
|
+
*/
|
|
334
305
|
const schedule = (): void => {
|
|
335
306
|
loop = setTimeout(() => {
|
|
336
|
-
void
|
|
307
|
+
void round()
|
|
337
308
|
.catch((error: unknown) => {
|
|
338
309
|
logger.error('jobs.worker.tick-failed', {
|
|
339
310
|
workerId,
|
|
@@ -346,26 +317,72 @@ export function createWorker(options: WorkerOptions): Worker {
|
|
|
346
317
|
}, pollIntervalMs);
|
|
347
318
|
};
|
|
348
319
|
|
|
349
|
-
const
|
|
350
|
-
if (state === 'stopped') return;
|
|
320
|
+
const teardown = async (reason: string): Promise<void> => {
|
|
351
321
|
state = 'draining';
|
|
352
322
|
if (loop !== undefined) clearTimeout(loop);
|
|
353
323
|
loop = undefined;
|
|
354
324
|
logger.info('jobs.worker.draining', { workerId, reason, inFlight: inFlight.size });
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
325
|
+
try {
|
|
326
|
+
// Stop claiming, finish what we hold, then close. Anything else re-runs work on deploy.
|
|
327
|
+
// Rounds first: one that passed the guard before the flag flipped is still awaiting its
|
|
328
|
+
// `claim()`, and the jobs it starts join `inFlight` after any snapshot taken here — so a
|
|
329
|
+
// drain that waited on `inFlight` alone closed the driver under a job that had just begun.
|
|
330
|
+
await Promise.allSettled([...rounds]);
|
|
331
|
+
await Promise.allSettled([...inFlight]);
|
|
332
|
+
await options.driver.close?.();
|
|
333
|
+
} finally {
|
|
334
|
+
// Whatever the close did, this worker is done: a state left at 'draining' is a drain that
|
|
335
|
+
// is not happening — `start()` refuses it for the rest of the process and `stats()` reports
|
|
336
|
+
// a worker still finishing work it finished. And the hook goes back. It exists only to call
|
|
337
|
+
// this, so one left registered drains a stopped worker on the next process-wide shutdown —
|
|
338
|
+
// through a driver already closed — and keeps this closure, its driver and its in-flight
|
|
339
|
+
// set alive with it.
|
|
340
|
+
state = 'stopped';
|
|
341
|
+
releaseShutdownHook?.();
|
|
342
|
+
releaseShutdownHook = undefined;
|
|
343
|
+
}
|
|
344
|
+
};
|
|
345
|
+
|
|
346
|
+
const stop = async (reason = 'stop'): Promise<void> => {
|
|
347
|
+
if (state === 'stopped') return;
|
|
348
|
+
// One teardown, joined rather than repeated: a SIGTERM landing on a manual stop must wait out
|
|
349
|
+
// the same in-flight work, not close the driver a second time underneath it. Cleared as it
|
|
350
|
+
// settles, so a worker that started again tears down again instead of joining a promise that
|
|
351
|
+
// settled a lifetime ago. A close that threw still stopped this worker — the failure is the
|
|
352
|
+
// caller's to see on the promise it awaited, not a teardown to run twice.
|
|
353
|
+
stopping ??= teardown(reason).finally(() => {
|
|
354
|
+
stopping = undefined;
|
|
355
|
+
});
|
|
356
|
+
await stopping;
|
|
359
357
|
};
|
|
360
358
|
|
|
361
359
|
return {
|
|
362
360
|
start() {
|
|
363
|
-
|
|
361
|
+
// Only from a standstill. A start mid-drain would put the claim loop back on a driver the
|
|
362
|
+
// drain is about to close, and stack a second shutdown hook on the one still running.
|
|
363
|
+
if (state !== 'idle' && state !== 'stopped') return;
|
|
364
|
+
// Refused HERE, at the earliest decidable point, and refused rather than logged: an agent
|
|
365
|
+
// reads "max in-flight runs of THIS job across the fleet", writes `concurrency: 1` on
|
|
366
|
+
// `rebuildSearchIndex`, ships, and two workers run it on the first deploy — while
|
|
367
|
+
// `x jobs show` and the manifest both confirm a guarantee that does not exist. A driver
|
|
368
|
+
// with no `leases` can only hold the cap per process, so it does not get to claim it.
|
|
369
|
+
if (driverLeases === undefined) {
|
|
370
|
+
const capped = registeredJobs()
|
|
371
|
+
.filter((handle) => handle.concurrency !== undefined)
|
|
372
|
+
.map((handle) => handle.name);
|
|
373
|
+
if (capped.length > 0) {
|
|
374
|
+
throw new ConcurrencyUnenforceableError({ driver: options.driver.name, jobs: capped });
|
|
375
|
+
}
|
|
376
|
+
}
|
|
364
377
|
state = 'running';
|
|
365
378
|
logger.info('jobs.worker.started', { workerId, queues });
|
|
366
|
-
// 'accept' phase: stop claiming before core waits on in-flight jobs.
|
|
379
|
+
// 'accept' phase: stop claiming before core waits on in-flight jobs. The unregister is
|
|
380
|
+
// kept, never discarded: `stop()` hands it back, so start -> stop -> start holds ONE hook
|
|
381
|
+
// rather than one per start, each retaining the driver of a worker that is already gone.
|
|
367
382
|
if (options.drainOnShutdown !== false) {
|
|
368
|
-
onShutdown(`jobs.worker.${workerId}`, () => stop('SIGTERM'), {
|
|
383
|
+
releaseShutdownHook = onShutdown(`jobs.worker.${workerId}`, () => stop('SIGTERM'), {
|
|
384
|
+
phase: 'accept',
|
|
385
|
+
});
|
|
369
386
|
}
|
|
370
387
|
schedule();
|
|
371
388
|
},
|