@edgehero/pi-dispatch 3.0.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +34 -0
- package/package.json +6 -2
- package/src/backend-local.mjs +69 -0
- package/src/backend-podman.mjs +44 -13
- package/src/config.mjs +39 -1
- package/src/container-spec.mjs +70 -7
- package/src/cpu-reserve.mjs +344 -0
- package/src/daemon-facts.mjs +58 -0
- package/src/docker-run.mjs +83 -6
- package/src/doctor.mjs +631 -21
- package/src/env-allowlist.mjs +23 -5
- package/src/host-budget.mjs +736 -0
- package/src/host-pi.mjs +1 -1
- package/src/index.mjs +237 -65
- package/src/job-size.mjs +286 -0
- package/src/job-user.mjs +66 -5
- package/src/live-probes.mjs +150 -25
- package/src/model-catalog.mjs +1 -1
- package/src/model-endpoints.mjs +22 -0
- package/src/models-json.mjs +10 -4
- package/src/output-cap.mjs +3 -3
- package/src/prepare.mjs +7 -3
- package/src/processor.mjs +91 -14
- package/src/reserved-env.mjs +3 -2
- package/src/run-container.mjs +62 -8
- package/src/run-history.mjs +204 -117
- package/src/sandbox-store.mjs +5 -1
- package/src/sandbox.mjs +48 -7
- package/src/scoped-limits.mjs +94 -10
- package/src/size-records.mjs +80 -0
- package/src/size-suggest.mjs +441 -0
- package/src/start.mjs +133 -9
- package/src/triggers.mjs +8 -5
package/src/host-pi.mjs
CHANGED
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* What it must never become. Nothing on the worker's BOOT path may import this file. It reads host paths and
|
|
16
16
|
* may spawn a package manager, and neither belongs anywhere near `start.mjs`.
|
|
17
17
|
*
|
|
18
|
-
* Everything here MIRRORS a private detail of the pinned pi (0.
|
|
18
|
+
* Everything here MIRRORS a private detail of the pinned pi (1.0.3) rather than calling it: pi exports no
|
|
19
19
|
* public answer to "where is this package installed" or "is this resource enabled", and importing the whole
|
|
20
20
|
* coding-agent SDK to read two well-known paths is not worth the weight. That mirroring is a real risk --
|
|
21
21
|
* pi could change the grammar and we would silently start staging something the operator turned off -- so it
|
package/src/index.mjs
CHANGED
|
@@ -16,6 +16,8 @@ import { effectiveCostCapMicros } from "./money.mjs";
|
|
|
16
16
|
import { dollarWindowCaps } from "./dollar-budget.mjs";
|
|
17
17
|
import { concurrencyFor, dollarCapsFor, makeInFlight, modelDollarRows, projectDollarCapsFor, projectRowFor, rowScopeFor, scopedLedgers } from "./scoped-limits.mjs";
|
|
18
18
|
import { memberScopeOf, projectOf } from "./projects.mjs";
|
|
19
|
+
import { resolveJobSize } from "./job-size.mjs";
|
|
20
|
+
import { BUDGET_RECHECK_MS, HOST_BUDGET_TICK_MS, NEVER_FITS_RECHECK_MS, makeHostBudget } from "./host-budget.mjs";
|
|
19
21
|
import { governedDollars } from "./allocation.mjs";
|
|
20
22
|
import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
|
|
21
23
|
import { makeWaitState } from "./wait-state.mjs";
|
|
@@ -90,7 +92,7 @@ const ABORT_GRACE_MS = 30_000;
|
|
|
90
92
|
* container produces, so nothing downstream needs to know the difference -- the processor's abort
|
|
91
93
|
* classification, the run record and the refund all behave exactly as they do for a stop that worked.
|
|
92
94
|
*/
|
|
93
|
-
function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
|
|
95
|
+
function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS, onStopDidNotTake = () => {}) {
|
|
94
96
|
if (!signal) return run;
|
|
95
97
|
return new Promise((resolve, reject) => {
|
|
96
98
|
let timer = null;
|
|
@@ -106,6 +108,8 @@ function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
|
|
|
106
108
|
if (settled) return;
|
|
107
109
|
settled = true;
|
|
108
110
|
log("stop_did_not_take", { job: job.id, graceMs });
|
|
111
|
+
// Issue #596, phase 2: the container may still run, so its host budget hold must outlive this job.
|
|
112
|
+
onStopDidNotTake();
|
|
109
113
|
resolve({ code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null, exitReason: null });
|
|
110
114
|
}, graceMs);
|
|
111
115
|
// A boot-blocking handle is not wanted here: the worker should be able to exit if everything else
|
|
@@ -186,7 +190,7 @@ export function effectiveJobOf(data, settings, allowedModels = null, log = () =>
|
|
|
186
190
|
};
|
|
187
191
|
}
|
|
188
192
|
|
|
189
|
-
export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], projects = () => [], allocation = null, inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels = () => null, endpointSetFor = mainModelEndpoints, deps, recordRun = () => {}, settledRecord = null, timeoutMs = JOB_TIMEOUT_MS, cancelPollMs = 2_000, cancelStopBoundMs = CANCEL_STOP_BOUND_MS, hostName = "", now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
193
|
+
export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, hostBudget = null, abortGraceMs = ABORT_GRACE_MS, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], projects = () => [], jobSizeEnv = {}, allocation = null, inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels = () => null, endpointSetFor = mainModelEndpoints, deps, recordRun = () => {}, settledRecord = null, timeoutMs = JOB_TIMEOUT_MS, cancelPollMs = 2_000, cancelStopBoundMs = CANCEL_STOP_BOUND_MS, hostName = "", multiHost = false, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
190
194
|
// The lost-lock gate's rejection lines, said ONCE per job id, stall count and attempt. The gate runs before every
|
|
191
195
|
// deferral (pause window, wait, scope or endpoint busy), and BullMQ never resets a job's stall count, so a stalled
|
|
192
196
|
// scheduled job whose record is refused meets the gate again on every deferred pickup: a 30-minute scope-busy hold
|
|
@@ -199,7 +203,12 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
199
203
|
if (rejectedSeen.size > REJECTED_SEEN_MAX) rejectedSeen.delete(rejectedSeen.values().next().value);
|
|
200
204
|
return true;
|
|
201
205
|
};
|
|
202
|
-
|
|
206
|
+
// THE HOST BUDGET'S WAITER BOOKKEEPING (issue #596, phase 2), in ONE place around the whole pickup rather than at each
|
|
207
|
+
// exit: a job the budget deferred keeps its waiter (and so its hold); a job another gate deferred (a pause window, a
|
|
208
|
+
// wait, a full scope, a busy endpoint) keeps its waiter SUSPENDED, so it holds no room it could not use; a job that
|
|
209
|
+
// ended any other way (it ran, it was refused, it failed) is no waiter at all. Listing this per exit is how one of
|
|
210
|
+
// the many deferral sites would come to leave a live hold behind.
|
|
211
|
+
const pickup = async (job, token, signal, budgetState) => {
|
|
203
212
|
// THE LOST-LOCK GATE, first because it is free and because it can only ever stop a run (CONST-RETRY-INFRA-ONLY).
|
|
204
213
|
// BullMQ hands a job to the processor again after its stall check took it back. That happens to a job whose
|
|
205
214
|
// processor FINISHED when Valkey was unreachable for longer than the lock renewal window: the record was
|
|
@@ -240,6 +249,68 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
240
249
|
throw new DelayedError();
|
|
241
250
|
}
|
|
242
251
|
|
|
252
|
+
// The limits snapshot, the project and the size, read here, right after the pause gate and ABOVE the wait gate, so
|
|
253
|
+
// the never-fits check below can refuse a size before a job waits (issue #596, gate round 1 of phase 2).
|
|
254
|
+
const limits = scopedLimits();
|
|
255
|
+
// The job's project (issue #499, INT-PROJECTS-FILE-CONTRACT), resolved ONCE here from one read of the projects ref,
|
|
256
|
+
// beside the limits snapshot and for its reason: the gate, the ledger and the record agree for this attempt,
|
|
257
|
+
// whatever an operator does to projects.json mid-run. A retry or a deferral is a new pickup and resolves again.
|
|
258
|
+
// Every record from the never-fits check on carries it (the refusal there by hand, every one past the wait gate
|
|
259
|
+
// through `recordAfterGate`); the wait gate's refusals carry none and are resolved from the live ref (start.mjs).
|
|
260
|
+
// An id or null, never a name.
|
|
261
|
+
const pickupProjects = projects();
|
|
262
|
+
const project = projectOf(job.data, pickupProjects);
|
|
263
|
+
// THE JOB'S SIZE (issue #596, `job-size.mjs`), resolved ONCE here from the same limits snapshot and pickup project
|
|
264
|
+
// every gate below reads: the project row's memory and CPUs, else the deployment's PI_JOB_MEMORY and PI_JOB_CPUS
|
|
265
|
+
// (`jobSizeEnv`, validated at boot), else 4g and 2. It reaches the container as an ARGUMENT (runJob's `jobSize`,
|
|
266
|
+
// then `runContainer`'s `size`), never through `job.data`, so nothing queued can choose its own size, and every
|
|
267
|
+
// record below carries it beside the project.
|
|
268
|
+
const size = resolveJobSize({ project, limits, env: jobSizeEnv });
|
|
269
|
+
|
|
270
|
+
// THE NEVER-FITS CHECK (issue #596, phase 2, DES-HOST-BUDGET), right after the size and ABOVE the wait gate (gate
|
|
271
|
+
// round 1): a size that can never start here is known now, and a job must not hold for a day on a wait and
|
|
272
|
+
// THEN be told so (the wait gate's own determinate-refusals-then-holds rule). Nothing is held yet, so nothing is
|
|
273
|
+
// given back.
|
|
274
|
+
//
|
|
275
|
+
// A job that can run on NO OTHER HOST is refused: a size larger than this host's budget, or than its project's
|
|
276
|
+
// `hostShare` of it, is a determinate POLICY refusal, RETURNED before anything is spent (CONST-BUDGET-BEFORE-TOKENS,
|
|
277
|
+
// CONST-RETRY-INFRA-ONLY): `job-size-exceeds-host` or `job-size-exceeds-share`. The log line and the record name
|
|
278
|
+
// both sizes; the forge comment names neither (an issue author can act on neither). Two such jobs: one on THIS
|
|
279
|
+
// HOST'S OWN QUEUE (`pi-jobs@<name>`), and EVERY job on a worker with no host queue (`multiHost` false, no
|
|
280
|
+
// `PI_WORKER_NAME`): there the shared queue is this host's alone in all but name, since no other host that
|
|
281
|
+
// declared a fleet drains it, and deferring a never-fits job there (gate round 2 of phase 2) re-asked it
|
|
282
|
+
// every 60 s forever with no record, which is the silent no-op this project refuses.
|
|
283
|
+
//
|
|
284
|
+
// A job on the SHARED queue of a MULTI-HOST worker is NEVER refused for its size, local or forge: another
|
|
285
|
+
// host draining the queue may have a larger budget, or give the project a larger share of it. It is deferred for
|
|
286
|
+
// `NEVER_FITS_RECHECK_MS` with a named line carrying both sizes, and doctor names a project that fits no live
|
|
287
|
+
// host. There used to be a fleet refusal after two registry reads agreed that no live host fits, and the registry
|
|
288
|
+
// cannot carry that verdict: a host's row is deleted while it restarts (a clean stop) or expires after a crash, and a
|
|
289
|
+
// read whose HGETALL times out drops a row, so a job a restarting host would have run was refused for good.
|
|
290
|
+
if (hostBudget) {
|
|
291
|
+
await hostBudget.ready;
|
|
292
|
+
const misfit = hostBudget.neverFits(size, project, limits);
|
|
293
|
+
if (misfit !== null) {
|
|
294
|
+
const budgetNow = hostBudget.current();
|
|
295
|
+
const share = hostBudget.shareOf(project, limits);
|
|
296
|
+
const sizeFields = { memMiB: size.memMiB, cpuCenti: size.cpuCenti, budgetMemMiB: budgetNow.memMiB, budgetCpuCenti: budgetNow.cpuCenti, hostShare: share };
|
|
297
|
+
if (!multiHost || (job.queueName ?? QUEUE) !== QUEUE) {
|
|
298
|
+
const reason = `job-size-exceeds-${misfit}`;
|
|
299
|
+
deps?.log?.(reason.replaceAll("-", "_"), { jobId: job.id, project, ...sizeFields });
|
|
300
|
+
if (deps?.comment) await Promise.resolve(deps.comment(job.data, SIZE_REFUSAL_COMMENTS[reason])).catch(() => {});
|
|
301
|
+
const at = new Date(now()).toISOString();
|
|
302
|
+
const result = { outcome: "policy", reason, exitCode: null, turns: null, tokens: null, budgetReserved: false, hostBudget: { memMiB: budgetNow.memMiB, cpuCenti: budgetNow.cpuCenti, hostShare: share } };
|
|
303
|
+
// Above the wait gate, so through the recorder's own arguments rather than the bound one below it: the
|
|
304
|
+
// same pickup project and size, by hand once.
|
|
305
|
+
recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString(), project, size });
|
|
306
|
+
return result;
|
|
307
|
+
}
|
|
308
|
+
deps?.log?.("job_size_never_fits_here_deferred", { jobId: job.id, project, misfit, delayMs: NEVER_FITS_RECHECK_MS, ...sizeFields });
|
|
309
|
+
await job.moveToDelayed(nowMs + NEVER_FITS_RECHECK_MS, token);
|
|
310
|
+
throw new DelayedError();
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
|
|
243
314
|
// The wait gate (issue #230, REQ-WAIT-FOR). THIRD: after the pause gate, because a paused job must
|
|
244
315
|
// not burn a wait evaluation any more than it burns a scope re-check, and BEFORE the scope acquire,
|
|
245
316
|
// because a job that is going to sit until tomorrow morning must not hold the folder mutex while it
|
|
@@ -583,7 +654,62 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
583
654
|
// Before the scope acquire, so a job that cannot run on this machine at all never takes a folder
|
|
584
655
|
// mutex it would immediately have to give back, and so the two releases nest rather than interleave.
|
|
585
656
|
let hostHeld = false;
|
|
586
|
-
|
|
657
|
+
// THE ONE RELEASE (issue #596, phase 2). Every hold this pickup takes (the host slot, the repo or folder slot and
|
|
658
|
+
// the project slot with their fleet claims, the endpoint slots, and the host budget's hold) is given back here and
|
|
659
|
+
// nowhere else, at every exit: each gate's deferral, the setup guard and the finally. A release site that lists
|
|
660
|
+
// holds by hand is the one that forgets the hold added after it was written, and a bolt test refuses a bare
|
|
661
|
+
// release anywhere else in this function. Last taken, first given back: the endpoint holds (issue #503), then the
|
|
662
|
+
// scope holds (project, then repo), then the host slot, then the budget.
|
|
663
|
+
//
|
|
664
|
+
// `orphan` is the finally's: when the container's stop did not take (`stop_did_not_take`), the container may still
|
|
665
|
+
// run, so the budget hold becomes an ORPHAN that keeps its room until the runtime says the container is gone
|
|
666
|
+
// (`host-budget.mjs` `sweep`; the boot reaper is the backstop). The count slots OUTSIDE the budget (the scope, the
|
|
667
|
+
// project and the endpoint slots, and the host slot when there is no budget) still go back: they bound starts, and
|
|
668
|
+
// the 30-minute bound already ended this job. The `PI_CONCURRENCY` slot does NOT: with a budget the count is the
|
|
669
|
+
// budget's third dimension, so an orphan, like a seeded survivor, holds one job slot until its container
|
|
670
|
+
// is gone (gate round 2 of phase 2).
|
|
671
|
+
//
|
|
672
|
+
// The in-process halves go back synchronously, before the first await, so a caller that cannot await (the setup
|
|
673
|
+
// guard) still frees every local slot before it rethrows; the fleet halves are release-if-mine and awaited where
|
|
674
|
+
// the caller can. Draining, so a second call releases nothing: the in-process map's release is not idempotent.
|
|
675
|
+
let budgetHeld = false;
|
|
676
|
+
let stopDidNotTake = false;
|
|
677
|
+
let name;
|
|
678
|
+
let venue;
|
|
679
|
+
// THE SCOPE HOLDS, in acquire order (issue #499 part B): the repo or folder slot, then the project slot. Each is
|
|
680
|
+
// `{ key, fleet }`: the in-process slot under `key` (the row scope) and its fleet claim, or null.
|
|
681
|
+
const scopeHolds = [];
|
|
682
|
+
const releaseScopeHolds = () => {
|
|
683
|
+
const taken = scopeHolds.splice(0).reverse();
|
|
684
|
+
for (const hold of taken) inFlight.release(hold.key);
|
|
685
|
+
return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
|
|
686
|
+
};
|
|
687
|
+
// THE ENDPOINT HOLDS (issue #503), `{ id, fleet }` each, in id order.
|
|
688
|
+
const endpointHolds = [];
|
|
689
|
+
const releaseEndpointHolds = () => {
|
|
690
|
+
const taken = endpointHolds.splice(0).reverse();
|
|
691
|
+
for (const hold of taken) endpointSlots.release(hold.id);
|
|
692
|
+
return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
|
|
693
|
+
};
|
|
694
|
+
const releaseAllHolds = ({ orphan = false } = {}) => {
|
|
695
|
+
const endpointsReleased = releaseEndpointHolds();
|
|
696
|
+
const scopesReleased = releaseScopeHolds();
|
|
697
|
+
if (hostHeld) {
|
|
698
|
+
hostBound.slots.release(HOST_SLOT_KEY);
|
|
699
|
+
hostHeld = false;
|
|
700
|
+
}
|
|
701
|
+
if (budgetHeld) {
|
|
702
|
+
budgetHeld = false;
|
|
703
|
+
if (orphan) hostBudget.orphan(job.id, { name, venue, ticket: budgetState.ticket });
|
|
704
|
+
else hostBudget.release(job.id, { ticket: budgetState.ticket });
|
|
705
|
+
}
|
|
706
|
+
return Promise.all([endpointsReleased, scopesReleased]);
|
|
707
|
+
};
|
|
708
|
+
// With a host budget the count is the budget's own third dimension (issue #596, gate round 1 of phase 2),
|
|
709
|
+
// judged LAST with the memory and the CPU, so a waiting job's hold keeps a job slot too. A host slot taken here, first,
|
|
710
|
+
// deferred a big shared-queue job at the slot while every small that ended was replaced at once from the host queue,
|
|
711
|
+
// and a job deferred here never reached the budget, so it held nothing and never ran while the flood lasted.
|
|
712
|
+
if (hostBound && !hostBudget) {
|
|
587
713
|
if (!hostBound.slots.tryAcquire(HOST_SLOT_KEY, hostBound.limit())) {
|
|
588
714
|
deps?.log?.("host_busy_deferred", { jobId: job.id, delayMs: SCOPE_BUSY_RECHECK_MS });
|
|
589
715
|
await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
|
|
@@ -592,46 +718,22 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
592
718
|
hostHeld = true;
|
|
593
719
|
}
|
|
594
720
|
|
|
595
|
-
const limits = scopedLimits();
|
|
596
|
-
// The job's project (issue #499, INT-PROJECTS-FILE-CONTRACT), resolved ONCE here from one read of the projects ref,
|
|
597
|
-
// beside the limits snapshot and for its reason: the gate, the ledger and the record agree for this attempt,
|
|
598
|
-
// whatever an operator does to projects.json mid-run. A retry or a deferral is a new pickup and resolves again.
|
|
599
|
-
// Every record below this line carries it (through `recordAfterGate`); a record written before this gate carries
|
|
600
|
-
// none and is resolved from the live ref (start.mjs). An id or null, never a name.
|
|
601
|
-
const pickupProjects = projects();
|
|
602
|
-
const project = projectOf(job.data, pickupProjects);
|
|
603
721
|
// THE ONE RECORDER BELOW THE GATE, bound once, so the pickup project is a property of the path and not of each call
|
|
604
722
|
// site: every record from here on goes through it, and none can drop the field and fall back to the live ref in
|
|
605
723
|
// start.mjs, which would disagree with the pickup value exactly when projects.json was edited mid-run. A bolt in
|
|
606
724
|
// project-pickup.test.mjs refuses a bare `recordRun(` call below this line.
|
|
607
|
-
const recordAfterGate = (args) => recordRun({ ...args, project });
|
|
725
|
+
const recordAfterGate = (args) => recordRun({ ...args, project, size });
|
|
726
|
+
|
|
608
727
|
// The MATCHED ROW's scope keys both the in-process slot and the fleet lease (issue #498), the same string
|
|
609
728
|
// `budgetCapsFor` hashes below and the boot sweeper hashes from the file: a qualified `github:acme/web` row holds
|
|
610
729
|
// GitHub jobs only, a bare `acme/web` row holds every forge's under the key it always had. With no row it is the
|
|
611
730
|
// job's canonical scope, so the folder mutex is keyed exactly as before.
|
|
612
731
|
const scope = rowScopeFor(job.data, limits);
|
|
613
|
-
// THE SCOPE HOLDS, in acquire order (issue #499 part B): the repo or folder slot, then the project slot. Each is
|
|
614
|
-
// `{ key, fleet }`: the in-process slot under `key` (the row scope) and its fleet claim, or null. ONE drain gives
|
|
615
|
-
// every hold back, last first, at every exit (a deferral, the setup guard, the finally), the endpoint holds' shape:
|
|
616
|
-
// a release site that lists slots by hand is the one that forgets the slot added after it was written.
|
|
617
|
-
const scopeHolds = [];
|
|
618
|
-
// Drains the holds, so a second call releases nothing: the in-process map's release is not idempotent. The
|
|
619
|
-
// in-process half goes back synchronously, so a caller that cannot await (the setup guard) still frees every local
|
|
620
|
-
// slot before it rethrows; the fleet half is release-if-mine and awaited where it can be.
|
|
621
|
-
const releaseScopeHolds = () => {
|
|
622
|
-
const taken = scopeHolds.splice(0).reverse();
|
|
623
|
-
for (const hold of taken) inFlight.release(hold.key);
|
|
624
|
-
return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
|
|
625
|
-
};
|
|
626
732
|
// Give every hold back (scope and host), then defer. Every scope-gate deferral goes through here.
|
|
627
733
|
const deferScope = async (fields) => {
|
|
628
|
-
|
|
629
|
-
//
|
|
630
|
-
|
|
631
|
-
if (hostHeld) {
|
|
632
|
-
hostBound.slots.release(HOST_SLOT_KEY);
|
|
633
|
-
hostHeld = false;
|
|
634
|
-
}
|
|
734
|
+
// Every hold goes back before we defer, the host slot too: `makeInFlight().release` is not idempotent, so a
|
|
735
|
+
// slot held across a deferral would be a slot this machine never gets back.
|
|
736
|
+
await releaseAllHolds();
|
|
635
737
|
deps?.log?.(fields.event, { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS, ...fields.extra });
|
|
636
738
|
await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
|
|
637
739
|
throw new DelayedError();
|
|
@@ -729,16 +831,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
729
831
|
// One snapshot per pickup (the endpoints, the overlay models, the derived set), handed to runJob as
|
|
730
832
|
// `modelEndpoints` so a later gate reads the same declaration this one leased against. With no endpoints
|
|
731
833
|
// declared the overlay is not even read, and the job touches nothing new: no command, no key, no field.
|
|
732
|
-
const endpointHolds = [];
|
|
733
834
|
let endpointSnapshot = null;
|
|
734
|
-
// Drains the holds, so a second call releases nothing: the in-process map's release is not idempotent.
|
|
735
|
-
// The in-process half goes back synchronously, so a caller that cannot await (the setup guard) still frees
|
|
736
|
-
// every local slot before it rethrows; the fleet half is release-if-mine and awaited where it can be.
|
|
737
|
-
const releaseEndpointHolds = () => {
|
|
738
|
-
const taken = endpointHolds.splice(0).reverse();
|
|
739
|
-
for (const hold of taken) endpointSlots.release(hold.id);
|
|
740
|
-
return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
|
|
741
|
-
};
|
|
742
835
|
if (modelEndpoints && !settingsThrew && !settings?.invalid) {
|
|
743
836
|
let endpoints = [];
|
|
744
837
|
let models = null;
|
|
@@ -790,19 +883,15 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
790
883
|
// fleet half; the in-process bound above is exact for a server only this host reaches.
|
|
791
884
|
fleet = await endpointLease.acquire(job.id, { slots: endpoint.slots, keyArgs: [hash16(endpoint.id)] });
|
|
792
885
|
if (!fleet) {
|
|
793
|
-
|
|
886
|
+
// Taken in process above, so it goes back through the one release with every other hold.
|
|
887
|
+
endpointHolds.push({ id: endpoint.id, fleet: null });
|
|
794
888
|
where = "fleet";
|
|
795
889
|
} else if (fleet.degraded) {
|
|
796
890
|
deps?.log?.("endpoint_lease_degraded", { jobId: job.id, endpoint: endpoint.id });
|
|
797
891
|
}
|
|
798
892
|
}
|
|
799
893
|
if (where !== null) {
|
|
800
|
-
await
|
|
801
|
-
await releaseScopeHolds();
|
|
802
|
-
if (hostHeld) {
|
|
803
|
-
hostBound.slots.release(HOST_SLOT_KEY);
|
|
804
|
-
hostHeld = false;
|
|
805
|
-
}
|
|
894
|
+
await releaseAllHolds();
|
|
806
895
|
deps?.log?.("endpoint_busy_deferred", { jobId: job.id, endpoint: endpoint.id, where, delayMs: ENDPOINT_BUSY_RECHECK_MS });
|
|
807
896
|
await job.moveToDelayed(nowMs + ENDPOINT_BUSY_RECHECK_MS, token);
|
|
808
897
|
throw new DelayedError();
|
|
@@ -811,8 +900,40 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
811
900
|
}
|
|
812
901
|
}
|
|
813
902
|
|
|
903
|
+
// THE HOST BUDGET GATE (issue #596, phase 2, DES-HOST-BUDGET), LAST of the gates, so a job waits on the budget only
|
|
904
|
+
// when the budget is its only obstacle: a job a scope, its project's `concurrent` or an endpoint deferred never got
|
|
905
|
+
// here, so its hold (if it had one) is suspended rather than kept (`processor` above). Synchronous: the ledger is
|
|
906
|
+
// read, decided on and written with no await between, so two pickups on this host never both take the same room.
|
|
907
|
+
// A deferral, never a refusal: a full host is transient state (CONST-RETRY-INFRA-ONLY), and it is free. Two whys make
|
|
908
|
+
// no waiter: `running-here` (the ledger already holds this job id: another pickup of it still runs here, or its
|
|
909
|
+
// container outlived it as an orphan) and `unseeded` (the job containers left from before this worker started are
|
|
910
|
+
// not listed yet, so nothing is admitted).
|
|
911
|
+
// Skipped when the settings are unreadable or invalid: that job is refused or retried below without a container.
|
|
912
|
+
//
|
|
913
|
+
// The gate is told the job's VENUE (the name it names, null for the default) and the container NAME this pickup
|
|
914
|
+
// will use (gate round 2 of phase 2): an unread boot listing blocks only its own venue's jobs, and a
|
|
915
|
+
// seeded survivor or orphan whose container carries this name defers the job `running-here`, since its
|
|
916
|
+
// `docker run` would create a container of that very name, which the sweep would take for the survivor. A name the
|
|
917
|
+
// venue cannot build is null here; the registry's refusal below records that job.
|
|
918
|
+
if (hostBudget && !settingsThrew && !settings?.invalid) {
|
|
919
|
+
let budgetName = null;
|
|
920
|
+
try {
|
|
921
|
+
budgetName = containerName({ ...job.data, id: job.id });
|
|
922
|
+
} catch {
|
|
923
|
+
budgetName = null;
|
|
924
|
+
}
|
|
925
|
+
const verdict = hostBudget.gate({ id: job.id, ticket: budgetState.ticket, project, size, venue: job.data?.backend ?? null, name: typeof budgetName === "string" ? budgetName : null, getState: typeof job.getState === "function" ? () => job.getState() : null, limits });
|
|
926
|
+
if (!verdict.admitted) {
|
|
927
|
+
await releaseAllHolds();
|
|
928
|
+
budgetState.budgetDeferred = true;
|
|
929
|
+
deps?.log?.("host_budget_deferred", { jobId: job.id, project, why: verdict.why, rank: verdict.rank, memMiB: size.memMiB, cpuCenti: size.cpuCenti, delayMs: BUDGET_RECHECK_MS });
|
|
930
|
+
await job.moveToDelayed(nowMs + BUDGET_RECHECK_MS, token);
|
|
931
|
+
throw new DelayedError();
|
|
932
|
+
}
|
|
933
|
+
budgetHeld = true;
|
|
934
|
+
}
|
|
935
|
+
|
|
814
936
|
let startedAt;
|
|
815
|
-
let name;
|
|
816
937
|
let timer;
|
|
817
938
|
let cancelPoll;
|
|
818
939
|
let cancelPolling = false;
|
|
@@ -854,7 +975,6 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
854
975
|
}
|
|
855
976
|
};
|
|
856
977
|
let onAbort;
|
|
857
|
-
let venue;
|
|
858
978
|
try {
|
|
859
979
|
// Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
|
|
860
980
|
// finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
|
|
@@ -968,12 +1088,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
968
1088
|
} catch (error) {
|
|
969
1089
|
// Release and DRAIN: this throw never reaches the main finally below, but a shared scope must never be
|
|
970
1090
|
// releasable twice -- a double release frees another holder's slot. Last taken, first given back.
|
|
971
|
-
void
|
|
972
|
-
void releaseScopeHolds();
|
|
973
|
-
if (hostHeld) {
|
|
974
|
-
hostBound.slots.release(HOST_SLOT_KEY);
|
|
975
|
-
hostHeld = false;
|
|
976
|
-
}
|
|
1091
|
+
void releaseAllHolds();
|
|
977
1092
|
clearTimeout(timer);
|
|
978
1093
|
clearInterval(cancelPoll);
|
|
979
1094
|
throw error;
|
|
@@ -1062,6 +1177,10 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
1062
1177
|
// snapshot, and refuses before any reserve when it is another project's.
|
|
1063
1178
|
pickupProject: project,
|
|
1064
1179
|
folderProject: (folder) => projectOf({ kind: "local", folder }, pickupProjects),
|
|
1180
|
+
// Issue #596: the size resolved at pickup above, for the retained run's manifest and the container's argv.
|
|
1181
|
+
jobSize: size,
|
|
1182
|
+
// Issue #596, phase 2: the host's CPU budget, every job's `--cpus` while it is a number (off or unknown: null).
|
|
1183
|
+
...(hostBudget && Number.isSafeInteger(hostBudget.current().cpuCenti) ? { cpuBudgetCenti: hostBudget.current().cpuCenti } : {}),
|
|
1065
1184
|
// Issue #504 part B: `_other`'s ledger, the deployment cap's source and the envelope verdict, under an envelope;
|
|
1066
1185
|
// absent without one, so the processor's defaults keep such a deployment byte-identical.
|
|
1067
1186
|
...(dollarInputs.otherDollars ? { otherDollars: dollarInputs.otherDollars } : {}),
|
|
@@ -1114,7 +1233,9 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
1114
1233
|
// promised: the run is recorded as that cancel (the existing path, "partial work may exist") even when the
|
|
1115
1234
|
// container happened to exit on its own, because the operator was told the record would say operator-cancel.
|
|
1116
1235
|
runContainer: (ctx) =>
|
|
1117
|
-
boundAfterAbort(deps.runContainer({ ...ctx, name, signal }), signal, job, deps.log ?? (() => {})
|
|
1236
|
+
boundAfterAbort(deps.runContainer({ ...ctx, name, signal }), signal, job, deps.log ?? (() => {}), abortGraceMs, () => {
|
|
1237
|
+
stopDidNotTake = true;
|
|
1238
|
+
}).then(async (r) => {
|
|
1118
1239
|
await stopCancelPoll();
|
|
1119
1240
|
const raced = r && r.aborted !== true && signal.aborted === true && signal.reason === "operator-cancel";
|
|
1120
1241
|
if (raced) deps.log?.("cancel_acked_as_container_exited", { jobId: job.id, exitCode: r.code ?? null });
|
|
@@ -1226,7 +1347,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
1226
1347
|
const endCancelled = async ({ retryable }) => {
|
|
1227
1348
|
const spent = error?.budgetReserved === true;
|
|
1228
1349
|
const beforeStart = retryable && !spent;
|
|
1229
|
-
const result = { outcome: "policy", reason: "operator-cancel", exitCode: spent ? (error.exitCode ?? null) : null, turns: spent ? (error.turns ?? null) : null, tokens: spent ? (error.tokens ?? null) : null, ...(spent && error.usage ? { usage: error.usage } : {}), provider: error?.provider ?? null, model: error?.model ?? null, session: error?.session ?? null, budgetReserved: retryable ? spent : (error?.budgetReserved ?? null), ...(error?.dollars ? { dollars: error.dollars } : {}) };
|
|
1350
|
+
const result = { outcome: "policy", reason: "operator-cancel", exitCode: spent ? (error.exitCode ?? null) : null, turns: spent ? (error.turns ?? null) : null, tokens: spent ? (error.tokens ?? null) : null, ...(spent && error.usage ? { usage: error.usage } : {}), provider: error?.provider ?? null, model: error?.model ?? null, session: error?.session ?? null, budgetReserved: retryable ? spent : (error?.budgetReserved ?? null), ...(error?.dollars ? { dollars: error.dollars } : {}), ...(spent && error?.resources ? { resources: error.resources } : {}) };
|
|
1230
1351
|
deps?.log?.("job_cancelled_instead_of_retry", { jobId: job.id, spent, retryable, ...(retryable ? {} : { failure: scrubCredentials(String(error?.message ?? error)).slice(0, 300) }) });
|
|
1231
1352
|
if (deps?.comment) await Promise.resolve(deps.comment(job.data, beforeStart ? CANCELLED_BEFORE_START_COMMENT : TERMINAL_COMMENTS["operator-cancel"])).catch(() => {});
|
|
1232
1353
|
recordAfterGate({ job, result, startedAt, endedAt: new Date().toISOString() });
|
|
@@ -1323,16 +1444,32 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
1323
1444
|
// full re-check interval while the slot it wanted went free behind it. The finally is already inside
|
|
1324
1445
|
// an async function, and `release` never throws. Last taken, first given back: the endpoint holds
|
|
1325
1446
|
// (issue #503), then the scope holds (project, then repo), then the host slot.
|
|
1326
|
-
|
|
1327
|
-
const scopesReleased = releaseScopeHolds();
|
|
1328
|
-
if (hostHeld) hostBound.slots.release(HOST_SLOT_KEY);
|
|
1329
|
-
await endpointsReleased;
|
|
1330
|
-
await scopesReleased;
|
|
1447
|
+
await releaseAllHolds({ orphan: stopDidNotTake });
|
|
1331
1448
|
clearTimeout(timer);
|
|
1332
1449
|
clearInterval(cancelPoll);
|
|
1333
1450
|
signal.removeEventListener("abort", onAbort);
|
|
1334
1451
|
}
|
|
1335
1452
|
};
|
|
1453
|
+
return async function processor(job, token, signal) {
|
|
1454
|
+
if (!hostBudget) return pickup(job, token, signal, { budgetDeferred: false });
|
|
1455
|
+
// The pickup's TICKET: the ledger entry this pickup takes carries it, and only this pickup's release or
|
|
1456
|
+
// orphan can act on that entry, so a second pickup of the same job id (a stalled scheduled job handed back while
|
|
1457
|
+
// its first attempt still runs here) can never give back the first's hold.
|
|
1458
|
+
const budgetState = { budgetDeferred: false, ticket: hostBudget.enter(job.id) };
|
|
1459
|
+
try {
|
|
1460
|
+
const result = await pickup(job, token, signal, budgetState);
|
|
1461
|
+
hostBudget.forget(job.id);
|
|
1462
|
+
return result;
|
|
1463
|
+
} catch (error) {
|
|
1464
|
+
if (!budgetState.budgetDeferred) {
|
|
1465
|
+
if (error instanceof DelayedError) hostBudget.suspend(job.id);
|
|
1466
|
+
else hostBudget.forget(job.id);
|
|
1467
|
+
}
|
|
1468
|
+
throw error;
|
|
1469
|
+
} finally {
|
|
1470
|
+
hostBudget.leave(job.id);
|
|
1471
|
+
}
|
|
1472
|
+
};
|
|
1336
1473
|
}
|
|
1337
1474
|
|
|
1338
1475
|
/**
|
|
@@ -1341,6 +1478,17 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
1341
1478
|
*/
|
|
1342
1479
|
export const CANCEL_STOP_BOUND_MS = 1_000;
|
|
1343
1480
|
|
|
1481
|
+
/**
|
|
1482
|
+
* The forge comments for the two never-fits refusals (issue #596, phase 2). GENERIC on purpose: never the size, the
|
|
1483
|
+
* budget or the project, which are operator configuration an issue author can act on none of. The worker log, the run
|
|
1484
|
+
* record and doctor name both sizes. There is no fleet refusal (gate round 1 of phase 2): a job on the shared
|
|
1485
|
+
* queue waits for a host it fits on.
|
|
1486
|
+
*/
|
|
1487
|
+
export const SIZE_REFUSAL_COMMENTS = Object.freeze({
|
|
1488
|
+
"job-size-exceeds-host": "Refused: this job's size is larger than the worker host's job budget, so it could never start there. No container was started and nothing was spent. Ask the operator to lower this project's job size or raise the host budget. Not run.",
|
|
1489
|
+
"job-size-exceeds-share": "Refused: this job's size is larger than the share of the worker host's job budget its project may use, so it could never start there. No container was started and nothing was spent. Ask the operator to lower this project's job size or raise its host share. Not run.",
|
|
1490
|
+
});
|
|
1491
|
+
|
|
1344
1492
|
/** The comment for a job the operator cancelled before it started, where it would have been held or retried (gate of PR #479). */
|
|
1345
1493
|
export const CANCELLED_BEFORE_START_COMMENT = "Stopped: the operator cancelled this run before it started. Nothing was spent. Not retried.";
|
|
1346
1494
|
|
|
@@ -1369,7 +1517,7 @@ export function keeperHoldState(job, at) {
|
|
|
1369
1517
|
return { at, since, startedMs };
|
|
1370
1518
|
}
|
|
1371
1519
|
|
|
1372
|
-
export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, settledRecord = null, limiter, pauseUntil, scopedLimits, projects, allocation = null, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels, extraClosers = [] }) {
|
|
1520
|
+
export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, settledRecord = null, limiter, pauseUntil, scopedLimits, projects, jobSizeEnv = {}, allocation = null, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels, extraClosers = [], hostBudget: hostBudgetOptions = null }) {
|
|
1373
1521
|
// One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
|
|
1374
1522
|
// check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
|
|
1375
1523
|
// because BullMQ has no selective pop and the put-it-back alternative does not work: promotion out of
|
|
@@ -1395,6 +1543,20 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
1395
1543
|
const hostBound = hostQueue ? { slots: hostSlots, limit: () => liveConcurrency() } : null;
|
|
1396
1544
|
const liveConcurrency = () => workers[0]?.concurrency ?? concurrency;
|
|
1397
1545
|
|
|
1546
|
+
// THE HOST BUDGET (issue #596, phase 2, DES-HOST-BUDGET), built ONCE here and handed to every processor, for the host
|
|
1547
|
+
// slot's reason: it bounds the MACHINE, so two queues with two ledgers would each admit a full budget. Unlike the host
|
|
1548
|
+
// slot it is armed with ONE queue too, because a single Worker's count still cannot tell a 20g job from a 2g one.
|
|
1549
|
+
// `hostBudgetOptions` is start.mjs's (the settings, the default size, the facts reader, the orphan check); a bare
|
|
1550
|
+
// wiring passes none and keeps today's behaviour. The tick re-reads the facts, verifies stale holds and sweeps
|
|
1551
|
+
// orphans, off every job path, unref'd and cleared on stop like the registry's beat.
|
|
1552
|
+
// The live `PI_CONCURRENCY` is the budget's third dimension, so with a budget the host slot above is not taken.
|
|
1553
|
+
const hostBudget = hostBudgetOptions ? makeHostBudget({ scopedLimits, countLimit: () => liveConcurrency(), ...hostBudgetOptions }) : null;
|
|
1554
|
+
let budgetTick = null;
|
|
1555
|
+
if (hostBudget) {
|
|
1556
|
+
budgetTick = setInterval(() => void hostBudget.tick(), hostBudgetOptions.tickMs ?? HOST_BUDGET_TICK_MS);
|
|
1557
|
+
budgetTick.unref?.();
|
|
1558
|
+
}
|
|
1559
|
+
|
|
1398
1560
|
for (const queueName of names) {
|
|
1399
1561
|
let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
|
|
1400
1562
|
const processor = makeProcessor({
|
|
@@ -1413,6 +1575,11 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
1413
1575
|
containerName,
|
|
1414
1576
|
redis,
|
|
1415
1577
|
getSettings,
|
|
1578
|
+
// Issue #596, phase 2: the ONE host budget. `multiHost` is whether this worker declared a fleet (a host queue):
|
|
1579
|
+
// without one it declares no fleet and no other host is there to wait for, so a size that never fits here is refused rather than
|
|
1580
|
+
// deferred for a host that does not exist (gate round 2 of phase 2).
|
|
1581
|
+
hostBudget,
|
|
1582
|
+
multiHost: hostQueue !== null,
|
|
1416
1583
|
// Late-bound over EVERY worker: an overlay concurrency change re-binds the live slot count at the
|
|
1417
1584
|
// next job start, and with two queues both have to move or the host bound and the queue bounds
|
|
1418
1585
|
// stop agreeing. Guarded so only an integer that actually differs touches the property.
|
|
@@ -1426,6 +1593,8 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
1426
1593
|
// independent maps would double every one of them exactly as two Workers double concurrency.
|
|
1427
1594
|
scopedLimits,
|
|
1428
1595
|
projects,
|
|
1596
|
+
// Issue #596: the deployment's default job size settings, read beside the limits snapshot at every pickup.
|
|
1597
|
+
jobSizeEnv,
|
|
1429
1598
|
allocation,
|
|
1430
1599
|
inFlight,
|
|
1431
1600
|
hostBound,
|
|
@@ -1481,12 +1650,15 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
1481
1650
|
// The host-queue worker, for the caller that must register listeners on both. Attached rather than
|
|
1482
1651
|
// returned as a pair so every existing caller keeps receiving exactly what it received before.
|
|
1483
1652
|
primary.hostWorker = workers[1] ?? null;
|
|
1653
|
+
// Issue #596, phase 2: the host budget, for the registry beat's thunks (start.mjs) and doctor-facing snapshots.
|
|
1654
|
+
primary.hostBudget = hostBudget;
|
|
1484
1655
|
|
|
1485
1656
|
const stop = async () => {
|
|
1486
1657
|
// Abort active jobs (=> docker stop via onAbort), then close. Without the cancel,
|
|
1487
1658
|
// worker.close() would wait up to 30 minutes for the container. ONE shutdown for every queue: two
|
|
1488
1659
|
// registrations would mean two `process.exit(0)` racing, and the second worker's containers would
|
|
1489
1660
|
// outlive the handler that was meant to stop them.
|
|
1661
|
+
if (budgetTick) clearInterval(budgetTick);
|
|
1490
1662
|
for (const w of workers) await Promise.resolve(w.cancelAllJobs?.("shutdown")).catch(() => {});
|
|
1491
1663
|
for (const w of workers) await w.close().catch(() => {});
|
|
1492
1664
|
// Close auxiliary resources (a cron scheduler, the live-edit file watchers) after the worker drains.
|